Skip to main content

spg_storage/
lib.rs

1//! In-memory storage primitives.
2//!
3//! v0.3 is intentionally simple: a flat catalog of tables, each holding rows
4//! as `Vec<Value>` (positional, matching the table's `TableSchema`). No MVCC,
5//! no on-disk format — those land in later milestones.
6#![no_std]
7// v3.3.2 NEON path for l2_distance_sq (aarch64 only). Scoped allow:
8// `unsafe_code = "deny"` at workspace level stays in force for every
9// other crate.
10#![cfg_attr(target_arch = "aarch64", allow(unsafe_code))]
11
12extern crate alloc;
13
14pub mod bignum;
15pub mod bloom;
16mod codec;
17pub mod fts_simple;
18pub mod halfvec;
19pub mod jsonb_gin;
20mod nsw;
21pub mod persistent;
22pub mod persistent_btree;
23pub mod posting;
24pub mod quantize;
25pub mod row_header;
26pub mod row_locator;
27pub mod segment;
28pub mod snapshot;
29mod table;
30pub mod trgm;
31pub mod vacuum;
32
33pub use self::bloom::{BloomError, BloomFilter};
34// v7.31 monster tier-3 cut 3 — on-disk codec moved to `codec`; the
35// public dense-row surface keeps its `spg_storage::*` paths, and the
36// low-level write/read primitives stay crate-visible for the
37// `Catalog::serialize`/`deserialize` methods that remain in this file.
38pub(crate) use self::codec::*;
39pub use self::codec::{
40    decode_row_body_dense, decode_row_body_dense_pruned, encode_row_body_dense,
41    encode_row_body_dense_into, encode_row_body_dense_masked_into, row_body_encoded_len,
42};
43// v7.31 monster tier-3 cut 2 — HNSW algorithms moved to `nsw`; the
44// public vector-search surface keeps its `spg_storage::*` paths via
45// these re-exports, and `nsw_insert_at` stays crate-visible for the
46// `Table` insert paths in the `table` module.
47pub(crate) use self::nsw::nsw_insert_at;
48pub use self::nsw::{NswMetric, cosine_dot_norms_f32, inner_product_f32, nsw_index_on, nsw_query};
49pub use self::posting::PostingList;
50
51/// The list handed back for an absent key, so callers cannot tell an
52/// absent key from an empty posting list — the property the old
53/// `&[][..]` return had, kept.
54static EMPTY_POSTINGS: crate::posting::PostingList = crate::posting::PostingList::new();
55pub use self::row_locator::{RowLocator, RowLocatorError};
56pub use self::segment::{
57    BRIN_SIDECAR_MAGIC, BrinSummary, OwnedSegment, SEGMENT_COMPRESS_ALGO_LZSS,
58    SEGMENT_COMPRESS_ALGO_NONE, SEGMENT_MAGIC, SEGMENT_MAGIC_V2, SEGMENT_PAGE_BYTES, SegmentError,
59    SegmentMeta, SegmentReader, derive_brin_summaries, encode_segment, wrap_v2_envelope,
60    wrap_v2_envelope_with_brin,
61};
62
63use alloc::borrow::Cow;
64use alloc::boxed::Box;
65use alloc::collections::{BTreeMap, BTreeSet};
66use alloc::format;
67use alloc::string::{String, ToString};
68use alloc::sync::Arc;
69use alloc::vec::Vec;
70use core::fmt;
71
72use self::persistent::PersistentVec;
73use self::persistent_btree::PersistentBTreeMap;
74
75/// In-cell encoding for `DataType::Vector`. Mirrors
76/// `spg_sql::ast::VecEncoding` — kept here so storage stays
77/// dep-free of `spg-sql`. The engine bridges between the two
78/// at DDL-execution time.
79///
80/// `F32` is the pre-v6 default: each cell holds a raw `Vec<f32>`.
81/// `Sq8` (v6.0.1) stores `Sq8Vector { min, max, bytes: Vec<u8> }`
82/// per cell; 4× compression vs `F32` with recall@10 ≥ 0.95 on
83/// natural embeddings (Gaussian / unit-sphere corpora).
84/// `F16` (v6.0.3, DDL keyword `HALF`) stores each element as
85/// IEEE-754 binary16; 2× compression and bit-exact dequantise.
86#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
87pub enum VecEncoding {
88    #[default]
89    F32,
90    Sq8,
91    F16,
92}
93
94impl fmt::Display for VecEncoding {
95    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
96        match self {
97            Self::F32 => f.write_str("F32"),
98            Self::Sq8 => f.write_str("SQ8"),
99            Self::F16 => f.write_str("HALF"),
100        }
101    }
102}
103
104/// Runtime type tags. `Vector { dim, encoding }` / `Varchar(max)` /
105/// `Char(size)` are parameterised; the parameter travels with both
106/// the column schema and the on-wire serialised representation.
107#[derive(Debug, Clone, Copy, PartialEq, Eq)]
108pub enum DataType {
109    /// 16-bit signed. Backed by `Value::SmallInt(i16)`; arithmetic that
110    /// would overflow surfaces as a type error at INSERT time.
111    SmallInt,
112    Int,    // 32-bit signed
113    BigInt, // 64-bit signed
114    Float,  // f64 (PG double precision)
115    /// v7.38 (read01, T-float4) — `real` / `float4`: 32-bit IEEE float (PG
116    /// `real`). Backed by `Value::Real(f32)`; behaves like `Float` for most
117    /// dispatch but renders / stores at f32 precision.
118    Real,
119    Text,
120    /// `VARCHAR(n)` — same byte representation as `Text`, but INSERT
121    /// rejects values longer than `n` Unicode characters.
122    Varchar(u32),
123    /// `CHAR(n)` — same representation as `Text`, but INSERT right-pads
124    /// with U+0020 to exactly `n` Unicode characters (or rejects when
125    /// the input is already longer).
126    Char(u32),
127    Bool,
128    /// pgvector-style fixed-dimension vector. `encoding` selects
129    /// the in-cell representation (`F32` = pre-v6 raw f32 buffer;
130    /// `Sq8` = v6.0.1 8-bit scalar-quantised). The DDL grammar
131    /// surfaces encoding via the optional `USING <encoding>`
132    /// clause: `VECTOR(128) USING SQ8`.
133    Vector {
134        dim: u32,
135        encoding: VecEncoding,
136    },
137    /// `NUMERIC(precision, scale)` — exact fixed-point decimal stored as
138    /// a scaled `i128`. `precision` caps total decimal digits, `scale`
139    /// fixes digits after the decimal point. v1.12 supports up to
140    /// precision 38 (the i128-safe ceiling). `NUMERIC` and `NUMERIC(p)`
141    /// surface as `Numeric { precision: p, scale: 0 }`.
142    Numeric {
143        /// v7.39 (round 272) — widened from u8. PG's declared precision
144        /// runs to 1000; at u8 it could not even be spelled, and the
145        /// parser rejected anything past 38 (i128's width) outright.
146        precision: u16,
147        /// v7.39 (round 271) — widened alongside the value's scale.
148        /// v7.39 (round 273) — and signed: PG's DECLARED scale runs
149        /// -1000..=1000, where a negative one rounds to tens / hundreds.
150        /// A VALUE's display scale is always non-negative.
151        scale: i16,
152    },
153    /// `DATE` — calendar date with day precision, stored as `i32` days
154    /// since the Unix epoch (1970-01-01).
155    Date,
156    /// `TIMESTAMP` (a.k.a. `MySQL` `DATETIME`) — instant with microsecond
157    /// precision, stored as `i64` microseconds since the Unix epoch.
158    Timestamp,
159    /// v7.9.2 `TIMESTAMPTZ` — bit-identical to `Timestamp` on disk
160    /// (i64 microseconds, UTC by convention). Carried as a distinct
161    /// type tag so the PG-wire layer can advertise OID 1184 (PG's
162    /// `timestamp with time zone`) and `sqlx`/`pgx`/JDBC clients
163    /// decode into their TZ-aware datetime types. The internal
164    /// semantics are unchanged: SPG never stored per-row offsets,
165    /// and neither did PG — `TIMESTAMPTZ` in PG is also UTC i64.
166    Timestamptz,
167    /// v7.39 (round 291) — PG's `name`: the type its catalogs use for
168    /// identifiers. Text truncated to NAMEDATALEN-1 (63) bytes, with
169    /// its own type identity — `pg_typeof('abc'::name)` is `name`, and
170    /// `CREATE TABLE t (a name)` is legal SQL that SPG rejected.
171    Name,
172    /// v7.39 (round 640) — PG's `xid`: a transaction id. [`Value::Xid`]
173    /// has existed since round 512, so a `'5'::xid` literal already knew
174    /// what it was; this is the DECLARED half, which nothing had. Without
175    /// it `pg_typeof(NULL::xid)` answered `bigint`, `pg_type` could not
176    /// list oid 28 — leaving the 48 `pg_attribute` rows that describe
177    /// `xmin` / `xmax` pointing at a type no catalog carried — and
178    /// `CREATE TABLE t (a xid)` was refused as an unknown type.
179    ///
180    /// On disk it is the 8-byte body its BIGINT sibling writes, and it
181    /// reads back as a `Value::Xid`, so a stored column and a literal are
182    /// the same thing to everything downstream.
183    ///
184    /// What is NOT yet true of the identity: PG gives `xid` equality and
185    /// hashing and no ordering operator at all, so `min` / `max` /
186    /// `count(DISTINCT …)` / `<=` all error there and all answer here.
187    /// Measured, not assumed — and left for the operator surface rather
188    /// than claimed by this comment.
189    Xid,
190    /// v7.39 (round 640) — PG's `xid8`: the same transaction id, 64 bits
191    /// wide and monotonic. Unlike [`DataType::Xid`] it has no value of
192    /// its own; a cell is a `Value::BigInt` and only the declared type
193    /// witnesses it. That is enough for `pg_typeof`, the catalogs and
194    /// the wire OID, and not enough to refuse a bigint where PG refuses
195    /// one. `pg_current_xact_id()` returns this type on PG.
196    Xid8,
197    /// v7.39 (round 667) — PG's `oid`: an unsigned 32-bit object
198    /// identifier. Modelled exactly like [`DataType::Xid8`] above: it has
199    /// no value of its own, a cell is a `Value::BigInt`, and only the
200    /// declared type witnesses it.
201    ///
202    /// That deliberately buys less than a full value type. What it buys:
203    /// `CREATE TABLE t(o OID)` is accepted (it was rejected outright with
204    /// `type "oid" does not exist`, while the neighbouring `XID` worked),
205    /// `pg_typeof` answers `oid` rather than `bigint`, and the catalogs
206    /// report their own key columns honestly. What it does NOT buy is
207    /// refusing a bigint where PG refuses an oid — `sum(oid)` and
208    /// `avg(oid)` still answer here and error on PG, because at runtime
209    /// the cell is indistinguishable from a bigint. Round 664 tried to
210    /// close those two by name and withdrew: a guard keyed on the name
211    /// would have caught `sum(bigint)` with it.
212    ///
213    /// The cast itself was already right before this — `4294967296::oid`
214    /// and `'abc'::oid` produce PG's errors word for word, and `(-1)::oid`
215    /// wraps to 4294967295 as PG does. Only the resulting type was lost,
216    /// because `conversions.rs` mapped the target to `BigInt`.
217    Oid,
218    /// `INTERVAL` — calendar-aware span (months + microseconds). v2.11
219    /// supports INTERVAL only as a runtime intermediate (literals,
220    /// arithmetic results); on-disk encoding is rejected so this branch
221    /// can't appear in a `ColumnSchema`.
222    Interval,
223    /// v4.9: `JSON` — text-backed JSON document. We don't parse
224    /// the content (no path operators or jsonb functions yet) —
225    /// the column accepts any TEXT-compatible value and round-trips
226    /// it verbatim. PG OID 114 on the wire.
227    Json,
228    /// v7.9.0: `JSONB` — semantically identical to `Json` on
229    /// the storage side (same `Value::Json` cells, same
230    /// row codec), but advertised as PG OID 3802 on the wire
231    /// so `sqlx`-style clients that bind `jsonb` columns
232    /// decode correctly. mailrs migration blocker #3.
233    Jsonb,
234    /// v7.10.4: `BYTES` / `BYTEA` — variable-length raw binary.
235    /// Backed by `Value::Bytes(Vec<u8>)`. PG wire OID 17. Literal
236    /// forms accepted by parser/engine: PG hex form `'\xDEADBEEF'`
237    /// (case-insensitive hex pairs) and escape form
238    /// `'foo\\000bar'` (the latter decoded at coercion time when
239    /// the target column is BYTEA — TEXT columns leave the
240    /// backslash sequence verbatim).
241    Bytes,
242    /// v7.10.9: `TEXT[]` — single-dimension TEXT array. Elements
243    /// may be NULL (PG semantics). PG wire OID 1009. Literal
244    /// forms: `ARRAY['a', 'b', NULL]` and the PG external form
245    /// `'{a,b,NULL}'::TEXT[]`. Engine implements `= ANY(arr)`,
246    /// `<> ALL(arr)`, and 1-based indexing `arr[i]`. Catalog
247    /// FILE_VERSION 18+; older snapshots reject this DataType
248    /// (forward-only by design — TEXT[] columns aren't readable
249    /// on a pre-v7.10 binary).
250    TextArray,
251    /// v7.11.12: `INT[]` — single-dimension i32 array. PG wire
252    /// OID 1007 (_int4). Same `ARRAY[...]` / `'{1,2,3}'::INT[]`
253    /// literal surface as TEXT[]. Catalog FILE_VERSION 19+.
254    IntArray,
255    /// v7.11.12: `BIGINT[]` — single-dimension i64 array. PG
256    /// wire OID 1016 (_int8). Catalog FILE_VERSION 19+.
257    BigIntArray,
258    /// v7.39 (round 694) — `oid[]`. It exists for the reason
259    /// [`DataType::Oid`] does: mapping it onto `BigIntArray` answers
260    /// `pg_typeof('{1,2}'::oid[])` with `bigint[]`, which is the defect
261    /// round 667 closed for the scalar.
262    OidArray,
263    /// v7.37.5 β-P4 — `INTERVAL[]` — single-dimension array of
264    /// `IntervalSpan { months, days, micros }`. PG wire OID 1187
265    /// (`_interval`). Catalog tag 35 + per-cell body
266    /// `[u16 count][per elem: u8 null + (if non-null) 16-byte
267    /// interval body in LE PG-byte-equal field order]`.
268    /// FILE_VERSION 48+.
269    IntervalArray,
270    /// v7.37.5 γ — full PG array-of-scalar family. Catalog tags
271    /// 36..48; wire OIDs from PG `pg_type.dat`. Per-element body
272    /// uses the scalar's existing `write_value_body` shape.
273    /// FILE_VERSION 48+ (same window as β; no separate bump).
274    BoolArray, // PG `_bool`        OID 1000, tag 36
275    SmallIntArray,    // PG `_int2`        OID 1005, tag 37
276    FloatArray,       // PG `_float8`      OID 1022, tag 38
277    NumericArray,     // PG `_numeric`     OID 1231, tag 39
278    DateArray,        // PG `_date`        OID 1182, tag 40
279    TimestampArray,   // PG `_timestamp`   OID 1115, tag 41
280    TimestamptzArray, // PG `_timestamptz` OID 1185, tag 42
281    UuidArray,        // PG `_uuid`        OID 2951, tag 43
282    JsonArray,        // PG `_json`        OID 199,  tag 44
283    JsonbArray,       // PG `_jsonb`       OID 3807, tag 45
284    BytesArray,       // PG `_bytea`       OID 1001, tag 46
285    VarcharArray,     // PG `_varchar`     OID 1015, tag 47
286    CharArray,        // PG `_bpchar`      OID 1014, tag 48
287    /// v7.37.5 δ — PG 14+ multirange types. A multirange is an
288    /// ordered collection of non-overlapping ranges of the same
289    /// element kind (e.g. `int4multirange(int4range(1,5),
290    /// int4range(10,15))` → `{[1,5),[10,15)}`). The same DataType
291    /// variant covers all six builtin multiranges; `RangeKind`
292    /// pins the element type so encode/decode/display can route
293    /// off one switch (parallel to `Range(RangeKind)`).
294    /// Wire OIDs: int4multirange=4451, int8multirange=4537,
295    /// nummultirange=4536, tsmultirange=4533, tstzmultirange=4534,
296    /// datemultirange=4535. Catalog tag 49 + 1-byte RangeKind on
297    /// the dense type-tag side. FILE_VERSION 48+ (same window as
298    /// β/γ, no separate bump).
299    Multirange(RangeKind),
300    /// v7.37.5 ε — PG geometry scalar family. Mirrors PG's seven
301    /// builtin geometric types one-for-one. Body shapes (LE):
302    ///   Point   = 16 B fixed (f64 x + f64 y)            OID 600
303    ///   Lseg    = 32 B fixed (Point p1 + Point p2)      OID 601
304    ///   Path    = varlena ([u8 closed][u32 n][Point*n]) OID 602
305    ///   Box     = 32 B fixed (Point ur + Point ll)      OID 603
306    ///   Polygon = varlena ([u32 n][Point*n])            OID 604
307    ///   Line    = 24 B fixed (f64 a + f64 b + f64 c)    OID 628
308    ///   Circle  = 24 B fixed (Point center + f64 r)     OID 718
309    /// Catalog tags 50..56. FILE_VERSION 48+ (same window as β/γ/δ;
310    /// no separate bump). Geometric operators (`<->` / `@>` / `&&`
311    /// / `<<` / `>>` / `~=`) are a planner-integration follow-up,
312    /// parallel to the Range operator defer in e2e_pg_range.rs.
313    Point,
314    Lseg,
315    Path,
316    PgBox,
317    Polygon,
318    Line,
319    Circle,
320    /// v7.37.5 ζ-A — PG network address family. Body shapes (LE):
321    ///   Inet     = 18 B fixed (u8 family + u8 bits + 16 B addr)  OID 869
322    ///   Cidr     = 18 B fixed (same shape as Inet; CIDR rejects
323    ///                          host bits at parse / coerce)       OID 650
324    ///   Macaddr  = 6 B fixed                                      OID 829
325    ///   Macaddr8 = 8 B fixed (EUI-64)                             OID 774
326    /// Catalog tags 57-60. FILE_VERSION 48+. `family = 4` is IPv4
327    /// (uses the first 4 bytes of the 16-B addr slot, rest 0);
328    /// `family = 6` is IPv6 (full 16 B).
329    Inet,
330    Cidr,
331    Macaddr,
332    Macaddr8,
333    /// v7.39 (read01 pg_lsn.c) — PG `pg_lsn` (WAL location). 8 bytes,
334    /// rendered `%X/%X`. Catalog tag 66. OID 3220.
335    PgLsn,
336    /// v7.37.5 ζ-A — PG bit string. Body = `[u32 nbits][ceil(nbits/8) bytes]`,
337    /// big-endian within each byte (matches PG binary).
338    ///   Bit         OID 1560 (fixed-length, but SPG carries the
339    ///                         length per cell — column declaration
340    ///                         `BIT(n)` constrains at coerce time)
341    ///   BitVarying  OID 1562 (variable-length, declared as `VARBIT`)
342    /// Catalog tags 61-62.
343    /// v7.39 (round 281) — `BIT(n)`: a FIXED-length bit string. `0`
344    /// means the type was written without a typmod, which PG treats as
345    /// `bit(1)`. Column assignment requires the length to match
346    /// exactly; an explicit cast pads or truncates instead.
347    Bit(u32),
348    /// v7.39 (round 281) — `BIT VARYING(n)`: `n` is a MAXIMUM, and `0`
349    /// means unbounded (`varbit` with no typmod).
350    BitVarying(u32),
351    /// v7.37.5 ζ-A — PG `xml`. Body identical to TEXT (storage is
352    /// the verbatim XML string; no parse-time validation). Only
353    /// the wire OID (142) differs. Catalog tag 63.
354    Xml,
355    /// v7.37.5 ζ-A — PG `"char"` (the internal single-byte type,
356    /// distinct from `CHAR(n)` / `BPCHAR`). Body = 1 byte raw.
357    /// OID 18. Catalog tag 64.
358    Char1,
359    /// v7.37.5 ζ-A — `MONEY[]`. Body = `[u16 count][per elem: u8 null
360    /// + (non-null) i64 LE cents]`. OID 791. Catalog tag 65.
361    MoneyArray,
362    /// v7.12.0: PG `tsvector` — ordered, deduplicated set of
363    /// `(lexeme, positions, weight)` tuples. PG wire OID 3614.
364    /// Catalog FILE_VERSION 20+. Storage shape is row-codec
365    /// tag 22; the schema-agnostic `write_value` path emits tag
366    /// 18. Literal: `'foo:1 bar:2,3'::tsvector` (PG external
367    /// form). G-CRIT-3 entry — v7.12.0 only ships the type +
368    /// codec; matching `@@` lands in v7.12.2.
369    TsVector,
370    /// v7.12.0: PG `tsquery` — parse tree of lexemes joined by
371    /// `&` `|` `!` and phrase operators. PG wire OID 3615.
372    /// Catalog FILE_VERSION 20+.
373    TsQuery,
374    /// v7.17.0: PG `uuid` — 128-bit identifier stored as
375    /// `Value::Uuid([u8; 16])`. PG wire OID 2950. Canonical
376    /// text form is lowercase 8-4-4-4-12 hyphenated; input
377    /// also accepts uppercase, unhyphenated, and brace-wrapped
378    /// forms (`{xxxx…}`). Catalog FILE_VERSION 36+; tag 24 on
379    /// the dense type-tag side, tag 20 on the schema-agnostic
380    /// value side. The drop-in PG/MySQL surface for Django /
381    /// Rails / Hibernate "id UUID PRIMARY KEY DEFAULT
382    /// gen_random_uuid()" default-PK pattern.
383    Uuid,
384    /// v7.17.0 Phase 3.P0-32: PG `time` (without time zone) — i64
385    /// microseconds since 00:00:00. PG wire OID 1083. Display:
386    /// canonical zero-padded `HH:MM:SS` when fractional is zero,
387    /// `HH:MM:SS.ffffff` otherwise. Catalog FILE_VERSION 37+;
388    /// tag 25 on the dense type-tag side, tag 21 on the schema-
389    /// agnostic value side. The wall-clock-of-day half of PG's
390    /// date/time triplet (date / time / timestamp).
391    Time,
392    /// v7.17.0 Phase 3.P0-33: MySQL `YEAR` — u16 in range
393    /// 1901..=2155 plus the special zero-year sentinel 0. No
394    /// dedicated PG OID (advertised as INT4 / OID 23 on the wire
395    /// — psql renders integers, MySQL CLI renders 4-digit
396    /// zero-padded text). Display always 4 digits: `0000` for the
397    /// zero-year, `1985` / `2007` / etc otherwise. Catalog
398    /// FILE_VERSION 38+; tag 26 on the dense type-tag side, tag
399    /// 22 on the schema-agnostic value side.
400    Year,
401    /// v7.17.0 Phase 3.P0-34: PG `time with time zone` (TIMETZ) —
402    /// i64 microseconds since 00:00:00 in the local wall clock
403    /// PLUS i32 offset-from-UTC in seconds. PG wire OID 1266.
404    /// Display: `HH:MM:SS[.ffffff]±HH[:MM]` (PG `timetz_out`).
405    /// Range: offset in ±50400 seconds (±14 hours). Catalog
406    /// FILE_VERSION 39+; tag 27 on the dense type-tag side, tag
407    /// 23 on the schema-agnostic value side.
408    TimeTz,
409    /// v7.17.0 Phase 3.P0-35: PG `money` — i64 cents (locale-
410    /// independent storage). PG wire OID 790. Display: en_US
411    /// locale (`$N,NNN.CC`, negative → `-$1.23`). Input accepts
412    /// `$N.NN`, `$N,NNN.NN`, bare integer (treated as major
413    /// units), optional leading `-`. Range: full i64. Catalog
414    /// FILE_VERSION 40+; tag 28 on the dense type-tag side, tag
415    /// 24 on the schema-agnostic value side.
416    Money,
417    /// v7.17.0 Phase 3.P0-38: PG range type. The same DataType
418    /// variant covers all six builtin ranges (int4range,
419    /// int8range, numrange, tsrange, tstzrange, daterange) —
420    /// `RangeKind` pins the element type so encode / decode /
421    /// display can route off one switch. Catalog FILE_VERSION
422    /// 43+; tag 29 + a 1-byte RangeKind on the dense type-tag
423    /// side, tag 25 on the schema-agnostic value side.
424    Range(RangeKind),
425    /// v7.17.0 Phase 3.P0-39: PG `hstore` extension type — flat
426    /// `text => text` map with NULL value support. Catalog
427    /// FILE_VERSION 44+; tag 30 on the dense type-tag side, tag
428    /// 26 on the schema-agnostic value side. The contrib OID is
429    /// installation-dependent in real PG; SPG advertises it via
430    /// dynamic lookup, falling back to TEXT (OID 25) on the wire
431    /// when the installed `hstore` extension hasn't claimed an
432    /// OID yet.
433    Hstore,
434    /// v7.17.0 Phase 3.P0-40: PG `int[][]` — 2-dimensional INT
435    /// matrix. Storage: row-major Vec<Vec<Option<i32>>>. All
436    /// rows must share the same column count. Wire OID 1007
437    /// (same as INT[]; the dimension count travels in the data
438    /// header, not the OID). Catalog FILE_VERSION 45+; tag 31
439    /// on the dense type-tag side, tag 27 on the schema-agnostic
440    /// value side.
441    IntArray2D,
442    /// v7.17.0 Phase 3.P0-40: PG `bigint[][]` — 2-dimensional
443    /// BIGINT matrix. Storage / OID / tags mirror IntArray2D.
444    /// Tag 32 dense, tag 28 schema-agnostic.
445    BigIntArray2D,
446    /// v7.17.0 Phase 3.P0-40: PG `text[][]` — 2-dimensional TEXT
447    /// matrix. Storage: row-major Vec<Vec<Option<String>>>.
448    /// Tag 33 dense, tag 29 schema-agnostic.
449    TextArray2D,
450    /// v7.39 (read01 round 75) — `bool[][]`. BOOL is the ONE element type whose
451    /// ARRAY rendering differs from its scalar one (`t` vs `true`), so a
452    /// text-backed 2-D cannot be PG-faithful for it: rendering the whole array
453    /// wants `t`, and subscripting a cell to text wants `false`. Every other
454    /// element type renders the same either way, which is why this is the only
455    /// typed 2-D variant SPG needs.
456    BoolArray2D,
457}
458
459/// v7.17.0 Phase 3.P0-38 — pins the element type of a range value
460/// or column. Wire OIDs: Int4=3904, Int8=3926, Num=3906,
461/// Ts=3908, TsTz=3910, Date=3912.
462#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, PartialOrd, Ord)]
463pub enum RangeKind {
464    Int4,
465    Int8,
466    Num,
467    Ts,
468    TsTz,
469    Date,
470}
471
472impl RangeKind {
473    pub const fn tag(self) -> u8 {
474        match self {
475            Self::Int4 => 0,
476            Self::Int8 => 1,
477            Self::Num => 2,
478            Self::Ts => 3,
479            Self::TsTz => 4,
480            Self::Date => 5,
481        }
482    }
483    pub const fn from_tag(t: u8) -> Option<Self> {
484        Some(match t {
485            0 => Self::Int4,
486            1 => Self::Int8,
487            2 => Self::Num,
488            3 => Self::Ts,
489            4 => Self::TsTz,
490            5 => Self::Date,
491            _ => return None,
492        })
493    }
494    pub const fn keyword(self) -> &'static str {
495        match self {
496            Self::Int4 => "INT4RANGE",
497            Self::Int8 => "INT8RANGE",
498            Self::Num => "NUMRANGE",
499            Self::Ts => "TSRANGE",
500            Self::TsTz => "TSTZRANGE",
501            Self::Date => "DATERANGE",
502        }
503    }
504}
505
506impl fmt::Display for DataType {
507    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
508        match self {
509            Self::SmallInt => f.write_str("SMALLINT"),
510            Self::Int => f.write_str("INT"),
511            Self::BigInt => f.write_str("BIGINT"),
512            Self::Xid => f.write_str("XID"),
513            Self::Xid8 => f.write_str("XID8"),
514            Self::Oid => f.write_str("OID"),
515            Self::OidArray => f.write_str("OID[]"),
516            Self::Float => f.write_str("FLOAT"),
517            Self::Real => f.write_str("REAL"),
518            Self::Text => f.write_str("TEXT"),
519            Self::Varchar(n) => write!(f, "VARCHAR({n})"),
520            Self::Char(n) => write!(f, "CHAR({n})"),
521            Self::Bool => f.write_str("BOOL"),
522            Self::Vector { dim, encoding } => match encoding {
523                VecEncoding::F32 => write!(f, "VECTOR({dim})"),
524                VecEncoding::Sq8 => write!(f, "VECTOR({dim}) USING SQ8"),
525                VecEncoding::F16 => write!(f, "VECTOR({dim}) USING HALF"),
526            },
527            Self::Numeric { precision, scale } => {
528                if *scale == 0 {
529                    write!(f, "NUMERIC({precision})")
530                } else {
531                    write!(f, "NUMERIC({precision}, {scale})")
532                }
533            }
534            Self::Date => f.write_str("DATE"),
535            Self::Timestamp => f.write_str("TIMESTAMP"),
536            Self::Timestamptz => f.write_str("TIMESTAMPTZ"),
537            Self::Name => f.write_str("NAME"),
538            Self::Interval => f.write_str("INTERVAL"),
539            Self::Json => f.write_str("JSON"),
540            Self::Jsonb => f.write_str("JSONB"),
541            Self::Bytes => f.write_str("BYTEA"),
542            Self::TextArray => f.write_str("TEXT[]"),
543            Self::IntArray => f.write_str("INT[]"),
544            Self::BigIntArray => f.write_str("BIGINT[]"),
545            Self::IntervalArray => f.write_str("INTERVAL[]"),
546            Self::BoolArray => f.write_str("BOOL[]"),
547            Self::SmallIntArray => f.write_str("SMALLINT[]"),
548            Self::FloatArray => f.write_str("FLOAT[]"),
549            Self::NumericArray => f.write_str("NUMERIC[]"),
550            Self::DateArray => f.write_str("DATE[]"),
551            Self::TimestampArray => f.write_str("TIMESTAMP[]"),
552            Self::TimestamptzArray => f.write_str("TIMESTAMPTZ[]"),
553            Self::UuidArray => f.write_str("UUID[]"),
554            Self::JsonArray => f.write_str("JSON[]"),
555            Self::JsonbArray => f.write_str("JSONB[]"),
556            Self::BytesArray => f.write_str("BYTEA[]"),
557            Self::VarcharArray => f.write_str("VARCHAR[]"),
558            Self::CharArray => f.write_str("CHAR[]"),
559            Self::Multirange(k) => f.write_str(match k {
560                RangeKind::Int4 => "INT4MULTIRANGE",
561                RangeKind::Int8 => "INT8MULTIRANGE",
562                RangeKind::Num => "NUMMULTIRANGE",
563                RangeKind::Ts => "TSMULTIRANGE",
564                RangeKind::TsTz => "TSTZMULTIRANGE",
565                RangeKind::Date => "DATEMULTIRANGE",
566            }),
567            Self::Point => f.write_str("POINT"),
568            Self::Lseg => f.write_str("LSEG"),
569            Self::Path => f.write_str("PATH"),
570            Self::PgBox => f.write_str("BOX"),
571            Self::Polygon => f.write_str("POLYGON"),
572            Self::Line => f.write_str("LINE"),
573            Self::Circle => f.write_str("CIRCLE"),
574            Self::Inet => f.write_str("INET"),
575            Self::Cidr => f.write_str("CIDR"),
576            Self::Macaddr => f.write_str("MACADDR"),
577            Self::Macaddr8 => f.write_str("MACADDR8"),
578            Self::PgLsn => f.write_str("PG_LSN"),
579            Self::Bit(0) => f.write_str("BIT"),
580            Self::Bit(n) => write!(f, "BIT({n})"),
581            Self::BitVarying(0) => f.write_str("VARBIT"),
582            Self::BitVarying(n) => write!(f, "VARBIT({n})"),
583            Self::Xml => f.write_str("XML"),
584            Self::Char1 => f.write_str("\"char\""),
585            Self::MoneyArray => f.write_str("MONEY[]"),
586            Self::TsVector => f.write_str("TSVECTOR"),
587            Self::TsQuery => f.write_str("TSQUERY"),
588            Self::Uuid => f.write_str("UUID"),
589            Self::Time => f.write_str("TIME"),
590            Self::Year => f.write_str("YEAR"),
591            Self::TimeTz => f.write_str("TIMETZ"),
592            Self::Money => f.write_str("MONEY"),
593            Self::Range(k) => f.write_str(k.keyword()),
594            Self::Hstore => f.write_str("HSTORE"),
595            Self::IntArray2D => f.write_str("INT[][]"),
596            Self::BigIntArray2D => f.write_str("BIGINT[][]"),
597            Self::TextArray2D => f.write_str("TEXT[][]"),
598            Self::BoolArray2D => f.write_str("BOOL[][]"),
599        }
600    }
601}
602
603/// v7.12.0 — one entry in a `Value::TsVector`. The lexeme is the
604/// (already-tokenised + stemmed in v7.12.1+) word; `positions` is
605/// a strictly-ascending list of 1-based positions; `weight` is the
606/// PG weight letter (A=3, B=2, C=1, D=0) — v7.12.0 defaults every
607/// lexeme to D, the v7.12.2 ranking path consumes the weight.
608#[derive(Debug, Clone, PartialEq, Eq)]
609pub struct TsLexeme {
610    pub word: String,
611    pub positions: Vec<u16>,
612    pub weight: u8,
613}
614
615/// v7.12.0 — parse tree for a PG `tsquery`. v7.12.0 ships the
616/// type + codec only; the `to_tsquery` / `plainto_tsquery` lexer
617/// lands in v7.12.1 and the `@@` evaluator in v7.12.2.
618#[derive(Debug, Clone, PartialEq, Eq)]
619pub enum TsQueryAst {
620    /// Single lexeme term. The `weight_mask` is the PG-style
621    /// bitmask of accepted weights (`A=1<<3`, `B=1<<2`, `C=1<<1`,
622    /// `D=1<<0`); `0` = any weight. v7.12.0 always sets it to 0.
623    Term {
624        word: String,
625        weight_mask: u8,
626    },
627    And(Box<TsQueryAst>, Box<TsQueryAst>),
628    Or(Box<TsQueryAst>, Box<TsQueryAst>),
629    Not(Box<TsQueryAst>),
630    /// `phrase <distance> phrase`. v7.12.0 only persists this; the
631    /// match semantics arrive in v7.12.2 alongside `@@`.
632    Phrase {
633        left: Box<TsQueryAst>,
634        right: Box<TsQueryAst>,
635        distance: u16,
636    },
637}
638
639/// v7.38.19 — whether an `interval` is finite, and if not, which way.
640///
641/// PostgreSQL has no NaN interval — measured, not assumed: `'nan'::interval`
642/// is a syntax error on 18.4 while `'infinity'` and `'-infinity'` parse —
643/// so this carries three states where `NumericKind` carries four.
644#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Default)]
645pub enum IntervalKind {
646    #[default]
647    Finite,
648    NegInf,
649    PosInf,
650}
651
652impl IntervalKind {
653    /// PostgreSQL's own representation of the two infinities, measured
654    /// off the wire rather than read out of its source.
655    ///
656    /// ```text
657    /// COPY (SELECT 'infinity'::interval)  TO STDOUT (FORMAT binary)
658    ///   … 7fffffffffffffff 7fffffff 7fffffff
659    /// COPY (SELECT '-infinity'::interval) TO STDOUT (FORMAT binary)
660    ///   … 8000000000000000 80000000 80000000
661    /// COPY (SELECT '1 day'::interval)     TO STDOUT (FORMAT binary)
662    ///   … 0000000000000000 00000001 00000000
663    /// ```
664    ///
665    /// All three fields at their extreme, which is why SPG can carry an
666    /// explicit `kind` in memory -- so the compiler names every site
667    /// that has to decide what infinity means there -- and still write
668    /// sixteen bytes on disk and on the wire. No finite interval reaches
669    /// the triple: PostgreSQL reserves it, so no value PostgreSQL ever
670    /// produced holds it either, and a file written before this version
671    /// cannot contain one.
672    #[must_use]
673    pub const fn from_fields(months: i32, days: i32, micros: i64) -> Self {
674        if micros == i64::MAX && days == i32::MAX && months == i32::MAX {
675            Self::PosInf
676        } else if micros == i64::MIN && days == i32::MIN && months == i32::MIN {
677            Self::NegInf
678        } else {
679            Self::Finite
680        }
681    }
682
683    /// The three fields this kind is written as. `Finite` hands back
684    /// what it was given.
685    #[must_use]
686    pub const fn to_fields(self, months: i32, days: i32, micros: i64) -> (i32, i32, i64) {
687        match self {
688            Self::Finite => (months, days, micros),
689            Self::PosInf => (i32::MAX, i32::MAX, i64::MAX),
690            Self::NegInf => (i32::MIN, i32::MIN, i64::MIN),
691        }
692    }
693
694    #[must_use]
695    pub const fn is_finite(self) -> bool {
696        matches!(self, Self::Finite)
697    }
698
699    /// Where this kind sits in the total order.
700    ///
701    /// v7.38.19 — PostgreSQL 18.4, measured: `'-infinity' < '-100 years'`
702    /// and `'infinity' > '100 years'` are both true, and `'infinity' =
703    /// 'infinity'` is true. So the rank decides first and the numbers
704    /// only speak between two finite values.
705    ///
706    /// Every comparison of two intervals asks THIS -- the ordering
707    /// comparator, the value comparator and the binary operators each
708    /// had their own copy of the span arithmetic, and three copies of a
709    /// question is how they come to disagree.
710    #[must_use]
711    pub const fn rank(self) -> i8 {
712        match self {
713            Self::NegInf => -1,
714            Self::Finite => 0,
715            Self::PosInf => 1,
716        }
717    }
718}
719
720/// A row-cell value, including SQL `NULL`. `Float` uses `f64`; NaN compares
721/// non-equal to itself (PG behaviour) — `PartialEq` is derived so callers
722/// must opt into NaN-aware comparison if they need stronger guarantees.
723///
724/// v7.37.42-arena Phase 1: parameterised on `'arena` so heap-bearing
725/// variants (Text/Json/Xml/Bytes/Vector/BitString.bytes) can borrow from
726/// a per-query bump arena (`Cow::Borrowed(&'arena ...)`). Persistent /
727/// catalog Values use `Value<'static>` (alias `ValueOwned`) with
728/// `Cow::Owned(...)`. Phase 1 keeps Range/Multirange recursive `Box<Value>`
729/// at `'static` (owned) — arena migration deferred to a later phase.
730/// Array-of-Option<String> variants (TextArray etc.) also stay owned in
731/// Phase 1; their nested shape is awkward for the simple Cow lift and the
732/// SCALARSQ hot path doesn't touch them.
733/// v7.38 (read01, T6) — the IEEE-style class of a NUMERIC value. `Finite` is the
734/// ordinary fixed-point case; the specials mirror PG's `'NaN'` / `'Infinity'` /
735/// `'-Infinity'`. Derived `PartialEq` gives `NaN == NaN` — correct for NUMERIC
736/// (unlike float's NaN ≠ NaN); the total order (`-Inf < finite < +Inf < NaN`)
737/// lives in the comparison paths, not in `Ord`.
738#[derive(Debug, Clone, Copy, PartialEq, Eq, Default, Hash)]
739pub enum NumericKind {
740    #[default]
741    Finite,
742    NaN,
743    PosInf,
744    NegInf,
745}
746
747#[derive(Debug, Clone, PartialEq)]
748#[non_exhaustive]
749pub enum Value<'arena> {
750    SmallInt(i16),
751    Int(i32),
752    BigInt(i64),
753    Float(f64),
754    /// v7.38 (read01, T-float4) — PG `real` (32-bit IEEE float).
755    Real(f32),
756    Text(Cow<'arena, str>),
757    Bool(bool),
758    Vector(Cow<'arena, [f32]>),
759    /// v6.0.1: 8-bit scalar-quantised vector cell. Lives in
760    /// columns declared `VECTOR(N) USING SQ8`. Layout per cell:
761    /// `Sq8Vector { min: f32, max: f32, bytes: Vec<u8> }` —
762    /// 4× compression vs `Vector(Vec<f32>)`. The wire layer
763    /// dequantises to `f32` on SELECT; INSERT path quantises
764    /// incoming `Vector(Vec<f32>)` cells into this variant.
765    Sq8Vector(crate::quantize::Sq8Vector),
766    /// v6.0.3: IEEE-754 binary16 vector cell. Lives in columns
767    /// declared `VECTOR(N) USING HALF`. Stores raw u16 LE bits
768    /// (2× compression vs `Vector(Vec<f32>)`). Wire / display
769    /// paths dequantise to f32 bit-exactly; INSERT path converts
770    /// incoming f32 vectors at the engine boundary.
771    HalfVector(crate::halfvec::HalfVector),
772    /// Exact fixed-point decimal. `scaled` holds the value as
773    /// `actual * 10^scale` so the storage type is always integral —
774    /// arithmetic never falls back to floating-point. v7.38 (read01, T6) —
775    /// `kind` classifies the value as finite (the common case, using
776    /// `scaled`/`scale`) or one of PG's NUMERIC specials (NaN / ±Infinity),
777    /// which ignore `scaled`/`scale` (canonicalized to 0).
778    Numeric {
779        scaled: i128,
780        /// v7.39 (round 271) — widened from u8. PG's numeric carries a
781        /// display scale up to 16383; at u8 a literal with 256 decimal
782        /// places could not be represented at all, and the conversion
783        /// aborted the query with an internal error.
784        scale: u16,
785        kind: NumericKind,
786    },
787    /// v7.38 (read01, T3) — an exact NUMERIC whose mantissa overflows `i128`
788    /// (PG's NUMERIC is unbounded). Boxed so the common finite case keeps its
789    /// small footprint; specials never take this form (they stay `Numeric`).
790    NumericBig(alloc::boxed::Box<crate::bignum::BigNumeric>),
791    /// Days since the Unix epoch (1970-01-01). Negative for earlier dates.
792    Date(i32),
793    /// Microseconds since the Unix epoch (1970-01-01T00:00:00Z).
794    Timestamp(i64),
795    /// Calendar span: `months` + `days` + `micros`. Three fields are
796    /// required for PG byte-equal: `'1 day'` ≠ `'24 hours'` (DST,
797    /// month-boundary, and the on-wire `pg_type` `interval` are all
798    /// `i64 micros + i32 days + i32 months`). v7.37.5 β widened from
799    /// `{months, micros}`; column storage lands in the same window.
800    Interval {
801        months: i32,
802        days: i32,
803        micros: i64,
804        /// v7.38.19 — finite, or one of the two infinities.
805        ///
806        /// PostgreSQL 17 gave `interval` an infinite value and SPG had
807        /// none, so `'infinity'::interval` was refused outright and the
808        /// subtraction error the ledger described was one symptom of
809        /// that, not the defect.
810        ///
811        /// A field beside the numbers rather than a sentinel inside
812        /// them, which is the shape `Value::Numeric` already uses for
813        /// exactly this question — and a field on THIS variant rather
814        /// than a new one, so the compiler names every site that has to
815        /// decide what infinity means there. A new variant would have
816        /// compiled everywhere on the first try and let a `_` arm
817        /// answer for it at one of a hundred and five of them.
818        kind: IntervalKind,
819    },
820    /// v4.9 `JSON` — raw JSON text. No structural validation
821    /// happens at the storage layer; whatever the parser hands us
822    /// round-trips verbatim. Equality is byte-wise.
823    Json(Cow<'arena, str>),
824    /// v7.10.4 `BYTEA` — raw binary blob. Equality is byte-wise.
825    /// Layout matches `Text`'s length-prefixed shape (`[u32 LE
826    /// len][bytes]`) under tag 18; the engine accepts PG hex
827    /// literals (`'\xDEADBEEF'`) and escape literals at the
828    /// coercion boundary.
829    Bytes(Cow<'arena, [u8]>),
830    /// v7.10.9 `TEXT[]` — single-dimension TEXT array with
831    /// optional NULL elements. Equality is element-wise. PG's
832    /// NULL-element comparison semantics: NULL ≠ NULL inside
833    /// arrays under `=`, so `[NULL] != [NULL]` (the engine
834    /// honours this).
835    TextArray(Vec<Option<String>>),
836    /// v7.11.12 `INT[]` — single-dimension i32 array with optional
837    /// NULL elements. Codec mirrors TextArray with i32 LE per
838    /// element instead of length-prefixed UTF-8.
839    IntArray(Vec<Option<i32>>),
840    /// v7.11.12 `BIGINT[]` — single-dimension i64 array with optional
841    /// NULL elements.
842    BigIntArray(Vec<Option<i64>>),
843    /// v7.37.5 β-P4 `INTERVAL[]` — single-dimension array of
844    /// `IntervalSpan { months, days, micros }` with optional NULL
845    /// elements. PG external form quotes each non-NULL element
846    /// (`{"1 day","24:00:00",NULL}`) because interval text contains
847    /// spaces and colons. Storage codec follows the BigIntArray
848    /// shape with a 16-byte per-element body.
849    IntervalArray(Vec<Option<IntervalSpan>>),
850    /// v7.37.5 γ — single-dimension arrays of the remaining PG
851    /// scalar types. Each carries `Vec<Option<T>>` with the
852    /// scalar's natural Rust shape; element NULLs are first-class
853    /// (per PG: `{1,NULL,3}` is a 3-element array, not a 2-element
854    /// one). Codec follows the IntervalArray shape — `[u16 count]
855    /// [per elem: u8 null + (non-null) scalar body]`.
856    BoolArray(Vec<Option<bool>>),
857    SmallIntArray(Vec<Option<i16>>),
858    FloatArray(Vec<Option<f64>>),
859    /// PG `NUMERIC[]` — `(scaled: i128, scale: u16)` per element.
860    NumericArray(Vec<Option<(i128, u16)>>),
861    DateArray(Vec<Option<i32>>),
862    TimestampArray(Vec<Option<i64>>),
863    TimestamptzArray(Vec<Option<i64>>),
864    UuidArray(Vec<Option<[u8; 16]>>),
865    JsonArray(Vec<Option<String>>),
866    JsonbArray(Vec<Option<String>>),
867    BytesArray(Vec<Option<Vec<u8>>>),
868    VarcharArray(Vec<Option<String>>),
869    CharArray(Vec<Option<String>>),
870    /// v7.37.5 δ — PG 14+ multirange. `ranges` is a Vec of
871    /// non-overlapping bounds spans of the shared `kind`. PG's
872    /// canonical text form is `{[a,b),[c,d),...}` (comma-separated
873    /// ranges in braces; `{}` for the empty multirange). SPG's
874    /// constructor enforces no overlap/coalescing — for now the
875    /// engine trusts the caller (mirrors PG's `_construct_array`
876    /// pattern). Catalog tag 49 + 1-byte RangeKind on the dense
877    /// type-tag side; schema-less path is unreachable (multirange
878    /// is column-typed only).
879    Multirange {
880        kind: RangeKind,
881        ranges: Vec<RangeSpan>,
882    },
883    /// v7.37.5 ε — PG geometry scalars. Per-type Vec/struct shape;
884    /// codec body shape is described on the matching DataType
885    /// variant. PG canonical text forms:
886    ///   Point   `(x,y)`
887    ///   Lseg    `[(x1,y1),(x2,y2)]`
888    ///   Path    open `[(x,y),(x,y),...]` / closed `((x,y),(x,y),...)`
889    ///   Box     `(ux,uy),(lx,ly)` (PG normalises to upper-right + lower-left)
890    ///   Polygon `((x,y),(x,y),...)` (implicit closed)
891    ///   Line    `{a,b,c}` (Ax + By + C = 0)
892    ///   Circle  `<(x,y),r>`
893    Point(Point2D),
894    Lseg(Point2D, Point2D),
895    /// `closed = true` is `((p,p,...))`; `false` is `[(p,p,...)]`.
896    Path {
897        points: Vec<Point2D>,
898        closed: bool,
899    },
900    /// PG `box` — stored as `(upper_right, lower_left)` (PG's
901    /// normalised order). The engine accepts both endpoint
902    /// orderings at parse time and normalises here.
903    PgBox(Point2D, Point2D),
904    Polygon(Vec<Point2D>),
905    Line {
906        a: f64,
907        b: f64,
908        c: f64,
909    },
910    Circle {
911        center: Point2D,
912        radius: f64,
913    },
914    /// v7.37.5 ζ-A — PG `inet`. `family = 4` (IPv4) or `6` (IPv6).
915    /// `bits` is the netmask bit count (0..=32 for IPv4, 0..=128
916    /// for IPv6). `addr` is right-padded with zeros when family=4
917    /// (first 4 bytes are the address).
918    Inet {
919        family: u8,
920        bits: u8,
921        addr: [u8; 16],
922    },
923    /// v7.37.5 ζ-A — PG `cidr`. Same shape as Inet; CIDR's
924    /// invariant (host bits zero) is enforced at parse / coerce.
925    Cidr {
926        family: u8,
927        bits: u8,
928        addr: [u8; 16],
929    },
930    /// v7.37.5 ζ-A — PG `macaddr`. 6 bytes (XX:XX:XX:XX:XX:XX).
931    Macaddr([u8; 6]),
932    /// v7.37.5 ζ-A — PG `macaddr8`. 8 bytes (EUI-64).
933    Macaddr8([u8; 8]),
934    /// v7.39 (read01 pg_lsn.c) — PG `pg_lsn`, a 64-bit WAL location.
935    PgLsn(u64),
936    /// v7.39 (read01 ruleutils.c) — PG `regclass`: an OID-typed relation
937    /// reference that renders as the relation name. SPG carries BOTH
938    /// (the synthetic oid for catalog joins, the name for display) so
939    /// `conrelid = 't'::regclass` and `'t'::regclass::text` agree.
940    /// Eval-only (no column storage).
941    RegClass(i64, alloc::boxed::Box<str>),
942    /// v7.39 (round 342, V65) — PG `regproc`: an OID-typed FUNCTION
943    /// reference that renders as the function name. Same dual shape
944    /// [`Value::RegClass`] carries, and for the same reason: without the
945    /// oid half, `pg_proc.oid = 'f'::regproc` cannot join, and a callee
946    /// cannot tell `pg_get_functiondef('f'::regproc)` — which PG answers
947    /// — from `pg_get_functiondef('f')` — which PG rejects.
948    /// Eval-only (no column storage).
949    RegProc(i64, alloc::boxed::Box<str>),
950    /// v7.39 (round 648) — PG `regtype`: an OID-typed TYPE reference
951    /// that renders as the type name. The third of the shape
952    /// [`Value::RegClass`] and [`Value::RegProc`] carry, and the one
953    /// that was missing it: `::regtype` produced a plain `Value::Text`
954    /// holding the canonical name, so `'text'::regtype::oid` tried to
955    /// parse the NAME as a number and answered `invalid input syntax
956    /// for type oid: "text"` where PG answers 25. `pg_typeof` on one
957    /// said `text` rather than `regtype` for the same reason.
958    ///
959    /// Eval-only (no column storage).
960    RegType(i64, alloc::boxed::Box<str>),
961    /// v7.39 (round 512) — PG `xid` and `cid`, the transaction and command
962    /// ids the `xmin` / `xmax` / `cmin` / `cmax` system columns carry.
963    ///
964    /// Their own types rather than integers, because PG deliberately gives
965    /// them almost no operators: measured on PG18, `xmin + 1` is "operator
966    /// does not exist: xid + integer", `xmin > 0` likewise, `xmin::bigint`
967    /// is "cannot cast type xid to bigint", and there is no `max(xid)`.
968    /// Carrying them as BigInt would quietly allow all four.
969    ///
970    /// Eval-only (no column storage).
971    Xid(u32),
972    Cid(u32),
973    /// v7.39 (round 511) — PG `tid`, the physical row identity `ctid`
974    /// carries: a block number and a one-based offset inside it, rendered
975    /// `(block,offset)`.
976    ///
977    /// It is a real type rather than a two-field record because the idiom
978    /// that makes `ctid` worth having — `DELETE … WHERE ctid NOT IN (SELECT
979    /// min(ctid) … GROUP BY key)` — needs `min()` over it, and PG has no
980    /// `min(record)`. Ordering is by block then offset, so `(0,2) < (0,9) <
981    /// (0,10)`; a text form would order those `(0,10) < (0,2) < (0,9)` and
982    /// the dedup would keep the wrong row.
983    ///
984    /// Eval-only (no column storage).
985    Tid(u32, u32),
986    /// v7.37.5 ζ-A — PG `bit` / `bit varying`. `nbits` is the
987    /// actual bit count; `bytes` is the packed representation
988    /// (big-endian within each byte; final byte right-padded
989    /// with 0s if `nbits % 8 != 0`).
990    BitString {
991        nbits: u32,
992        bytes: Cow<'arena, [u8]>,
993    },
994    /// v7.37.5 ζ-A — PG `xml`. Stored verbatim as a string; no
995    /// parse-time validation (matches the SPG JSON convention).
996    Xml(Cow<'arena, str>),
997    /// v7.37.5 ζ-A — PG `"char"` (internal single-byte type,
998    /// distinct from CHAR(n)).
999    Char1(u8),
1000    /// v7.38 (read01, T11) — PG `bpchar` / CHAR(n): blank-padded fixed-length
1001    /// string. Stored space-padded to the declared width (as PG does + for wire
1002    /// display); length / comparison / ::text / concat all ignore the trailing
1003    /// blanks (handled at those sites).
1004    BpChar(Cow<'arena, str>),
1005    /// v7.37.5 ζ-A — PG `money[]`.
1006    MoneyArray(Vec<Option<i64>>),
1007    /// v7.12.0 `tsvector` — sorted-by-word, deduped lexeme set with
1008    /// positions + weights. The engine enforces sort/dedup on
1009    /// construction; consumers can rely on `lexemes.windows(2)`
1010    /// being strictly ascending by `word`.
1011    TsVector(Vec<TsLexeme>),
1012    /// v7.12.0 `tsquery` — boolean / phrase parse tree over
1013    /// lexemes. Engine builds via `to_tsquery` family.
1014    TsQuery(TsQueryAst),
1015    /// v7.17.0 `uuid` — 128-bit identifier. Stored as 16 bytes
1016    /// (big-endian / network-byte order, same as RFC 4122).
1017    /// Display normalises to canonical lowercase 8-4-4-4-12
1018    /// hyphenated form. Equality is byte-wise.
1019    Uuid([u8; 16]),
1020    /// v7.17.0 Phase 3.P0-32 — PG `time` (without time zone) —
1021    /// i64 microseconds since 00:00:00. Range 0..86_400_000_000.
1022    /// Display: `HH:MM:SS` zero-padded, with optional `.ffffff`
1023    /// suffix when fractional is non-zero.
1024    Time(i64),
1025    /// v7.17.0 Phase 3.P0-33 — MySQL `YEAR` — u16 in range
1026    /// 1901..=2155 plus the special zero-year sentinel 0.
1027    /// Display always 4 digits zero-padded (`0000` for the
1028    /// sentinel; `1985`/`2007` otherwise).
1029    Year(u16),
1030    /// v7.17.0 Phase 3.P0-34 — PG `time with time zone` — i64
1031    /// microseconds since 00:00:00 in the LOCAL wall clock PLUS
1032    /// an i32 offset-from-UTC in seconds. PG preserves the
1033    /// offset on output, so the wall-clock value is NOT shifted
1034    /// to UTC at storage time. Offset range: ±50400 seconds
1035    /// (±14 hours).
1036    TimeTz {
1037        us: i64,
1038        offset_secs: i32,
1039    },
1040    /// v7.17.0 Phase 3.P0-35 — PG `money` — i64 cents
1041    /// (locale-independent storage; the en_US locale renders on
1042    /// display via `$N,NNN.CC`).
1043    Money(i64),
1044    /// v7.17.0 Phase 3.P0-39 — PG `hstore` value: flat
1045    /// `text => text` map with NULL value support. Insertion
1046    /// order preserved on input; duplicate keys take last-write-
1047    /// wins at parse time.
1048    Hstore(Vec<(String, Option<String>)>),
1049    /// v7.17.0 Phase 3.P0-40 — 2D INT matrix (row-major).
1050    IntArray2D(Vec<Vec<Option<i32>>>),
1051    /// v7.17.0 Phase 3.P0-40 — 2D BIGINT matrix (row-major).
1052    BigIntArray2D(Vec<Vec<Option<i64>>>),
1053    /// v7.17.0 Phase 3.P0-40 — 2D TEXT matrix (row-major).
1054    TextArray2D(Vec<Vec<Option<String>>>),
1055    /// v7.39 (read01 round 75) — see `DataType::BoolArray2D`.
1056    BoolArray2D(Vec<Vec<Option<bool>>>),
1057    /// v7.17.0 Phase 3.P0-38 — PG range value. One shape covers
1058    /// all six builtin range types; `kind` pins the element type
1059    /// (must match the column's `DataType::Range(kind)`).
1060    /// `lower` / `upper` are `None` for the unbounded sides;
1061    /// `lower_inc` / `upper_inc` mirror the canonical PG
1062    /// `[` / `(` / `]` / `)` bracket inclusivity. `empty=true`
1063    /// supersedes all other fields (the empty range has no
1064    /// bounds).
1065    Range {
1066        kind: RangeKind,
1067        // v7.37.42-arena Phase 1: Range bounds stay owned ('static).
1068        // Recursive arena lifetimes are awkward to migrate at this
1069        // phase and the SCALARSQ hot path doesn't construct ranges.
1070        lower: Option<alloc::boxed::Box<Value<'static>>>,
1071        upper: Option<alloc::boxed::Box<Value<'static>>>,
1072        lower_inc: bool,
1073        upper_inc: bool,
1074        empty: bool,
1075    },
1076    /// v7.38 (read01, T9) — a composite / record value (a `row(...)`
1077    /// constructor or a whole-row reference). Fields are `(name, value)`; the
1078    /// names are `f1..fN` for an anonymous `row(...)` or the source column
1079    /// names for a table row. Transient — flows through row_to_json / to_json
1080    /// and the composite text form `(a,b)`; not a storable column type here.
1081    Composite(alloc::vec::Vec<(alloc::string::String, Value<'static>)>),
1082    Null,
1083}
1084
1085/// Owned `Value` — heap-bearing variants are `Cow::Owned`. Used everywhere
1086/// a Value must outlive a query-scoped arena (catalog defaults, persistent
1087/// storage, public APIs).
1088pub type ValueOwned = Value<'static>;
1089
1090/// v7.37.5 ε — PG `point` building block. Shared by every other
1091/// geometric type (lseg / path / box / polygon / circle all
1092/// reduce to compositions of `Point2D`). Packed `{x: f64, y: f64}`,
1093/// 16 B, on-disk LE field order matches the PG binary point
1094/// format byte-for-byte (so a future binary BIND path lands
1095/// without rearrangement).
1096#[derive(Debug, Clone, Copy, PartialEq)]
1097pub struct Point2D {
1098    pub x: f64,
1099    pub y: f64,
1100}
1101
1102/// v7.37.5 δ — single-range bounds without the kind tag. Used as
1103/// the element type of `Value::Multirange { kind, ranges }` so a
1104/// multirange carries one shared `RangeKind` plus N bounds-only
1105/// spans (saves 1 byte/elem vs duplicating the kind). The five
1106/// other fields mirror `Value::Range` exactly.
1107#[derive(Debug, Clone, PartialEq)]
1108pub struct RangeSpan {
1109    // v7.37.42-arena Phase 1: stays owned ('static) — same rationale as
1110    // Range bounds above.
1111    pub lower: Option<alloc::boxed::Box<Value<'static>>>,
1112    pub upper: Option<alloc::boxed::Box<Value<'static>>>,
1113    pub lower_inc: bool,
1114    pub upper_inc: bool,
1115    pub empty: bool,
1116}
1117
1118/// v7.37.5 β-P4 — element type for `Value::IntervalArray`. Mirrors
1119/// the `{months, days, micros}` shape of scalar `Value::Interval`,
1120/// broken out as a named struct so `IntervalArray`'s element type
1121/// is concrete (24 bytes, packed) instead of an enum-boxed Value.
1122/// All three dimensions are independent — `IntervalSpan { days: 1,
1123/// .. }` is distinct from `IntervalSpan { micros: 86_400_000_000,
1124/// .. }` per PG byte-equal.
1125#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1126pub struct IntervalSpan {
1127    pub months: i32,
1128    pub days: i32,
1129    pub micros: i64,
1130    /// v7.38.19 — see [`IntervalKind`].
1131    pub kind: IntervalKind,
1132}
1133
1134impl<'arena> Value<'arena> {
1135    /// Type tag, or `None` for `NULL` (unknown at value level).
1136    pub fn data_type(&self) -> Option<DataType> {
1137        match self {
1138            Self::SmallInt(_) => Some(DataType::SmallInt),
1139            Self::Int(_) => Some(DataType::Int),
1140            Self::BigInt(_) => Some(DataType::BigInt),
1141            Self::Float(_) => Some(DataType::Float),
1142            Self::Real(_) => Some(DataType::Real),
1143            // `Text` covers both unbounded TEXT and bounded VARCHAR/CHAR
1144            // — the constraint lives on the column schema, not the value.
1145            Self::Text(_) => Some(DataType::Text),
1146            Self::Bool(_) => Some(DataType::Bool),
1147            Self::Vector(v) => Some(DataType::Vector {
1148                dim: u32::try_from(v.len()).expect("vector dim ≤ u32"),
1149                encoding: VecEncoding::F32,
1150            }),
1151            Self::Sq8Vector(q) => Some(DataType::Vector {
1152                dim: u32::try_from(q.bytes.len()).expect("vector dim ≤ u32"),
1153                encoding: VecEncoding::Sq8,
1154            }),
1155            Self::HalfVector(h) => Some(DataType::Vector {
1156                dim: u32::try_from(h.dim()).expect("vector dim ≤ u32"),
1157                encoding: VecEncoding::F16,
1158            }),
1159            // `Value::Numeric` doesn't carry its precision (the column
1160            // schema does); we surface precision=0 as "unknown" and let
1161            // the engine reconcile against the column type at coercion
1162            // time.
1163            // v7.39 (round 273) — a VALUE's display scale is unsigned and
1164            // never exceeds PG's 16383 ceiling, so it always fits the
1165            // signed declared-scale field this describes itself with.
1166            Self::Numeric { scale, .. } => Some(DataType::Numeric {
1167                precision: 0,
1168                scale: i16::try_from(*scale).unwrap_or(i16::MAX),
1169            }),
1170            Self::NumericBig(b) => Some(DataType::Numeric {
1171                precision: 0,
1172                scale: i16::try_from(b.scale()).unwrap_or(i16::MAX),
1173            }),
1174            Self::Date(_) => Some(DataType::Date),
1175            Self::Timestamp(_) => Some(DataType::Timestamp),
1176            Self::Interval { .. } => Some(DataType::Interval),
1177            Self::Json(_) => Some(DataType::Json),
1178            Self::Bytes(_) => Some(DataType::Bytes),
1179            Self::TextArray(_) => Some(DataType::TextArray),
1180            Self::IntArray(_) => Some(DataType::IntArray),
1181            Self::BigIntArray(_) => Some(DataType::BigIntArray),
1182            Self::IntervalArray(_) => Some(DataType::IntervalArray),
1183            Self::BoolArray(_) => Some(DataType::BoolArray),
1184            Self::SmallIntArray(_) => Some(DataType::SmallIntArray),
1185            Self::FloatArray(_) => Some(DataType::FloatArray),
1186            Self::NumericArray(_) => Some(DataType::NumericArray),
1187            Self::DateArray(_) => Some(DataType::DateArray),
1188            Self::TimestampArray(_) => Some(DataType::TimestampArray),
1189            Self::TimestamptzArray(_) => Some(DataType::TimestamptzArray),
1190            Self::UuidArray(_) => Some(DataType::UuidArray),
1191            Self::JsonArray(_) => Some(DataType::JsonArray),
1192            Self::JsonbArray(_) => Some(DataType::JsonbArray),
1193            Self::BytesArray(_) => Some(DataType::BytesArray),
1194            Self::VarcharArray(_) => Some(DataType::VarcharArray),
1195            Self::CharArray(_) => Some(DataType::CharArray),
1196            Self::Multirange { kind, .. } => Some(DataType::Multirange(*kind)),
1197            Self::Point(_) => Some(DataType::Point),
1198            Self::Lseg(_, _) => Some(DataType::Lseg),
1199            Self::Path { .. } => Some(DataType::Path),
1200            Self::PgBox(_, _) => Some(DataType::PgBox),
1201            Self::Polygon(_) => Some(DataType::Polygon),
1202            Self::Line { .. } => Some(DataType::Line),
1203            Self::Circle { .. } => Some(DataType::Circle),
1204            Self::Inet { .. } => Some(DataType::Inet),
1205            Self::Cidr { .. } => Some(DataType::Cidr),
1206            Self::Macaddr(_) => Some(DataType::Macaddr),
1207            Self::Macaddr8(_) => Some(DataType::Macaddr8),
1208            Self::PgLsn(_) => Some(DataType::PgLsn),
1209            // BitString could be either Bit or BitVarying; column
1210            // schema decides. Default to BitVarying when called
1211            // schema-less (rare; storage path is always
1212            // schema-aware so this only matters for diagnostics).
1213            Self::BitString { .. } => Some(DataType::BitVarying(0)),
1214            Self::Xml(_) => Some(DataType::Xml),
1215            Self::Char1(_) => Some(DataType::Char1),
1216            // BpChar reports its declared width from the padded length.
1217            Self::BpChar(s) => Some(DataType::Char(
1218                u32::try_from(s.chars().count()).unwrap_or(0),
1219            )),
1220            Self::MoneyArray(_) => Some(DataType::MoneyArray),
1221            Self::TsVector(_) => Some(DataType::TsVector),
1222            Self::TsQuery(_) => Some(DataType::TsQuery),
1223            Self::Uuid(_) => Some(DataType::Uuid),
1224            Self::Time(_) => Some(DataType::Time),
1225            Self::Year(_) => Some(DataType::Year),
1226            Self::TimeTz { .. } => Some(DataType::TimeTz),
1227            Self::Money(_) => Some(DataType::Money),
1228            Self::Range { kind, .. } => Some(DataType::Range(*kind)),
1229            Self::Hstore(_) => Some(DataType::Hstore),
1230            Self::IntArray2D(_) => Some(DataType::IntArray2D),
1231            Self::BigIntArray2D(_) => Some(DataType::BigIntArray2D),
1232            Self::TextArray2D(_) => Some(DataType::TextArray2D),
1233            Self::BoolArray2D(_) => Some(DataType::BoolArray2D),
1234            // v7.38 (read01, T9) — a transient composite/record has no storable
1235            // column DataType (it flows through row_to_json / to_json).
1236            Self::Composite(_) => None,
1237            // v7.39 (read01 ruleutils.c) — regclass is eval-only (dual
1238            // oid+name shape); no column storage type.
1239            // v7.39 (round 640) — `xid` became a column type, so its value
1240            // has a DataType to answer with. `cid` and `tid` are equally
1241            // legal column types on PG (measured: `CREATE TABLE t (a cid,
1242            // b tid)` is accepted), but SPG's grammar has no keyword for
1243            // them yet; they stay eval-only rather than half-declared.
1244            Self::Xid(_) => Some(DataType::Xid),
1245            Self::RegClass(..)
1246            | Self::RegProc(..)
1247            | Self::RegType(..)
1248            | Self::Tid(..)
1249            | Self::Cid(_) => None,
1250            Self::Null => None,
1251        }
1252    }
1253
1254    pub const fn is_null(&self) -> bool {
1255        matches!(self, Self::Null)
1256    }
1257
1258    /// v7.37.42-arena Phase 1: lift any `Value<'arena>` (possibly
1259    /// borrowing from a bump arena) into a fully-owned `Value<'static>`.
1260    /// Used at boundaries that must outlive the per-query arena
1261    /// (catalog write, public QueryResult emit, sqlx materialise).
1262    ///
1263    /// For the recursive Range/Multirange variants — bounds are already
1264    /// `Box<Value<'static>>` per Phase 1 design, so we just rebuild the
1265    /// outer enum at `'static`.
1266    pub fn into_owned(self) -> Value<'static> {
1267        match self {
1268            Value::SmallInt(n) => Value::SmallInt(n),
1269            Value::Int(n) => Value::Int(n),
1270            Value::BigInt(n) => Value::BigInt(n),
1271            Value::Float(f) => Value::Float(f),
1272            Value::Real(f) => Value::Real(f),
1273            Value::Text(s) => Value::Text(Cow::Owned(s.into_owned())),
1274            Value::Bool(b) => Value::Bool(b),
1275            Value::Vector(v) => Value::Vector(Cow::Owned(v.into_owned())),
1276            Value::Sq8Vector(q) => Value::Sq8Vector(q),
1277            Value::HalfVector(h) => Value::HalfVector(h),
1278            Value::Numeric {
1279                scaled,
1280                scale,
1281                kind,
1282            } => Value::Numeric {
1283                scaled,
1284                scale,
1285                kind,
1286            },
1287            Value::NumericBig(b) => Value::NumericBig(b),
1288            Value::Date(d) => Value::Date(d),
1289            Value::Timestamp(t) => Value::Timestamp(t),
1290            Value::Interval {
1291                months,
1292                days,
1293                micros,
1294                kind,
1295            } => Value::Interval {
1296                months,
1297                days,
1298                micros,
1299                kind,
1300            },
1301            Value::Json(s) => Value::Json(Cow::Owned(s.into_owned())),
1302            Value::Bytes(b) => Value::Bytes(Cow::Owned(b.into_owned())),
1303            Value::TextArray(v) => Value::TextArray(v),
1304            Value::IntArray(v) => Value::IntArray(v),
1305            Value::BigIntArray(v) => Value::BigIntArray(v),
1306            Value::IntervalArray(v) => Value::IntervalArray(v),
1307            Value::BoolArray(v) => Value::BoolArray(v),
1308            Value::SmallIntArray(v) => Value::SmallIntArray(v),
1309            Value::FloatArray(v) => Value::FloatArray(v),
1310            Value::NumericArray(v) => Value::NumericArray(v),
1311            Value::DateArray(v) => Value::DateArray(v),
1312            Value::TimestampArray(v) => Value::TimestampArray(v),
1313            Value::TimestamptzArray(v) => Value::TimestamptzArray(v),
1314            Value::UuidArray(v) => Value::UuidArray(v),
1315            Value::JsonArray(v) => Value::JsonArray(v),
1316            Value::JsonbArray(v) => Value::JsonbArray(v),
1317            Value::BytesArray(v) => Value::BytesArray(v),
1318            Value::VarcharArray(v) => Value::VarcharArray(v),
1319            Value::CharArray(v) => Value::CharArray(v),
1320            Value::Multirange { kind, ranges } => Value::Multirange { kind, ranges },
1321            // v7.38 (read01, T9) — Composite fields are already `Value<'static>`.
1322            Value::Composite(fields) => Value::Composite(fields),
1323            Value::RegClass(oid, name) => Value::RegClass(oid, name),
1324            Value::Tid(b, o) => Value::Tid(b, o),
1325            Value::Xid(x) => Value::Xid(x),
1326            Value::Cid(c) => Value::Cid(c),
1327            Value::RegProc(oid, name) => Value::RegProc(oid, name),
1328            Value::RegType(oid, name) => Value::RegType(oid, name),
1329            Value::Point(p) => Value::Point(p),
1330            Value::Lseg(a, b) => Value::Lseg(a, b),
1331            Value::Path { points, closed } => Value::Path { points, closed },
1332            Value::PgBox(a, b) => Value::PgBox(a, b),
1333            Value::Polygon(p) => Value::Polygon(p),
1334            Value::Line { a, b, c } => Value::Line { a, b, c },
1335            Value::Circle { center, radius } => Value::Circle { center, radius },
1336            Value::Inet { family, bits, addr } => Value::Inet { family, bits, addr },
1337            Value::Cidr { family, bits, addr } => Value::Cidr { family, bits, addr },
1338            Value::Macaddr(m) => Value::Macaddr(m),
1339            Value::Macaddr8(m) => Value::Macaddr8(m),
1340            Value::PgLsn(l) => Value::PgLsn(l),
1341            Value::BitString { nbits, bytes } => Value::BitString {
1342                nbits,
1343                bytes: Cow::Owned(bytes.into_owned()),
1344            },
1345            Value::Xml(s) => Value::Xml(Cow::Owned(s.into_owned())),
1346            Value::Char1(c) => Value::Char1(c),
1347            Value::BpChar(s) => Value::BpChar(Cow::Owned(s.into_owned())),
1348            Value::MoneyArray(v) => Value::MoneyArray(v),
1349            Value::TsVector(v) => Value::TsVector(v),
1350            Value::TsQuery(q) => Value::TsQuery(q),
1351            Value::Uuid(u) => Value::Uuid(u),
1352            Value::Time(t) => Value::Time(t),
1353            Value::Year(y) => Value::Year(y),
1354            Value::TimeTz { us, offset_secs } => Value::TimeTz { us, offset_secs },
1355            Value::Money(m) => Value::Money(m),
1356            Value::Range {
1357                kind,
1358                lower,
1359                upper,
1360                lower_inc,
1361                upper_inc,
1362                empty,
1363            } => Value::Range {
1364                kind,
1365                lower,
1366                upper,
1367                lower_inc,
1368                upper_inc,
1369                empty,
1370            },
1371            Value::Hstore(h) => Value::Hstore(h),
1372            Value::IntArray2D(a) => Value::IntArray2D(a),
1373            Value::BigIntArray2D(a) => Value::BigIntArray2D(a),
1374            Value::TextArray2D(a) => Value::TextArray2D(a),
1375            Value::BoolArray2D(a) => Value::BoolArray2D(a),
1376            Value::Null => Value::Null,
1377        }
1378    }
1379
1380    /// v7.37.42-arena Phase 4 — copy heap payloads into the supplied
1381    /// bump arena, yielding a `Value<'a>` whose Cow-variant payloads
1382    /// are arena-borrowed (or stay as small owned scalars for the
1383    /// `Copy`-able variants).
1384    ///
1385    /// Used at the catalog ↔ ephemeral boundary: a `ColumnSchema.default`
1386    /// is `Value<'static>` but INSERT-time eval may want it stamped into
1387    /// the per-statement arena alongside other arena-built scalars.
1388    ///
1389    /// Allocates only into the supplied arena; the input `&self` keeps
1390    /// its own storage. For `Copy`-able / nested-owned variants the
1391    /// implementation falls back to `clone()` (the nested heap blocks
1392    /// stay on the global allocator, which is fine — the boundary
1393    /// requirement is just "no aliasing of caller-owned strings").
1394    pub fn clone_into<'a>(&self, arena: &'a bumpalo::Bump) -> Value<'a> {
1395        match self {
1396            Value::Text(s) => Value::Text(Cow::Borrowed(arena.alloc_str(s))),
1397            Value::Json(s) => Value::Json(Cow::Borrowed(arena.alloc_str(s))),
1398            Value::Xml(s) => Value::Xml(Cow::Borrowed(arena.alloc_str(s))),
1399            Value::BpChar(s) => Value::BpChar(Cow::Borrowed(arena.alloc_str(s))),
1400            Value::Bytes(b) => {
1401                let slot = arena.alloc_slice_copy::<u8>(b);
1402                Value::Bytes(Cow::Borrowed(slot))
1403            }
1404            Value::Vector(v) => {
1405                let slot = arena.alloc_slice_copy::<f32>(v);
1406                Value::Vector(Cow::Borrowed(slot))
1407            }
1408            Value::BitString { nbits, bytes } => {
1409                let slot = arena.alloc_slice_copy::<u8>(bytes);
1410                Value::BitString {
1411                    nbits: *nbits,
1412                    bytes: Cow::Borrowed(slot),
1413                }
1414            }
1415            // Copy-able scalars + variants whose nested heap blocks are
1416            // `'static` regardless of `'arena` (TextArray, JsonArray,
1417            // Hstore, TsVector, Range bounds, …). Clone the heap block
1418            // via the standard `into_owned()` path then lift the
1419            // resulting `Value<'static>` to `Value<'a>` via the Cow
1420            // variance — `'static` covers any lifetime.
1421            other => other.clone().into_owned(),
1422        }
1423    }
1424}
1425
1426impl Value<'static> {
1427    /// v7.37.42-arena Phase 1 — owned-Text constructor. The variant now
1428    /// holds `Cow<'arena, str>`, so the previous `Value::Text(String)`
1429    /// shape no longer compiles directly. This helper preserves the
1430    /// historical ergonomics: `Value::text("foo")` or
1431    /// `Value::text(String::from("foo"))`.
1432    pub fn text<S: Into<String>>(s: S) -> Self {
1433        Value::Text(Cow::Owned(s.into()))
1434    }
1435
1436    /// v7.38 (read01, T6) — a finite NUMERIC from its fixed-point parts.
1437    pub const fn numeric(scaled: i128, scale: u16) -> Self {
1438        Value::Numeric {
1439            scaled,
1440            scale,
1441            kind: NumericKind::Finite,
1442        }
1443    }
1444
1445    /// v7.38 (read01, T6) — a special NUMERIC (NaN / ±Infinity). The fixed-point
1446    /// fields are canonicalized to 0 so equal specials compare byte-identical.
1447    pub const fn numeric_special(kind: NumericKind) -> Self {
1448        Value::Numeric {
1449            scaled: 0,
1450            scale: 0,
1451            kind,
1452        }
1453    }
1454
1455    /// v7.37.42-arena Phase 1 — owned-Json constructor (mirrors `text`).
1456    pub fn json<S: Into<String>>(s: S) -> Self {
1457        Value::Json(Cow::Owned(s.into()))
1458    }
1459
1460    /// v7.37.42-arena Phase 1 — owned-Xml constructor.
1461    pub fn xml<S: Into<String>>(s: S) -> Self {
1462        Value::Xml(Cow::Owned(s.into()))
1463    }
1464
1465    /// v7.37.42-arena Phase 1 — owned-Bytes constructor.
1466    pub fn bytes<B: Into<Vec<u8>>>(b: B) -> Self {
1467        Value::Bytes(Cow::Owned(b.into()))
1468    }
1469
1470    /// v7.37.42-arena Phase 1 — owned-Vector constructor.
1471    pub fn vector<V: Into<Vec<f32>>>(v: V) -> Self {
1472        Value::Vector(Cow::Owned(v.into()))
1473    }
1474
1475    /// v7.37.42-arena Phase 1 — owned-BitString constructor.
1476    pub fn bit_string<B: Into<Vec<u8>>>(nbits: u32, bytes: B) -> Self {
1477        Value::BitString {
1478            nbits,
1479            bytes: Cow::Owned(bytes.into()),
1480        }
1481    }
1482}
1483
1484/// One table row — values are positional and must match
1485/// `TableSchema.columns` in length and (modulo NULL) in `DataType`.
1486///
1487/// v7.37.42-arena Phase 1: parameterised on `'arena` so per-query rows
1488/// can borrow from a bump arena. The owned shape (`Row<'static>`, alias
1489/// `RowOwned`) is what catalog storage, public APIs, and tests use.
1490#[derive(Debug, Clone, PartialEq)]
1491pub struct Row<'arena> {
1492    pub values: Vec<Value<'arena>>,
1493}
1494
1495/// Owned `Row` — values are `Value<'static>`. Used everywhere a row must
1496/// outlive a query-scoped arena.
1497pub type RowOwned = Row<'static>;
1498
1499impl<'arena> Row<'arena> {
1500    pub const fn new(values: Vec<Value<'arena>>) -> Self {
1501        Self { values }
1502    }
1503
1504    pub fn len(&self) -> usize {
1505        self.values.len()
1506    }
1507
1508    pub fn is_empty(&self) -> bool {
1509        self.values.is_empty()
1510    }
1511}
1512
1513impl<'arena> Row<'arena> {
1514    /// v7.37.42-arena Phase 4 — copy every cell into the supplied bump
1515    /// arena, yielding a `Row<'a>` whose Cow-payloads are arena-borrowed.
1516    /// Boundary helper for catalog defaults → DML eval handoff and
1517    /// arena-local row scratch.
1518    pub fn clone_into<'a>(&self, arena: &'a bumpalo::Bump) -> Row<'a> {
1519        Row {
1520            values: self.values.iter().map(|v| v.clone_into(arena)).collect(),
1521        }
1522    }
1523
1524    /// v7.37.42-arena Phase 4 — lift this `Row<'arena>` to a fully-owned
1525    /// `Row<'static>` for catalog write / WAL serialisation. Equivalent
1526    /// to `Row::from_arena(self)` but consumes by value at any lifetime
1527    /// (callers can write `row.into_owned()` mirroring `Value::into_owned`).
1528    pub fn into_owned(self) -> Row<'static> {
1529        Row {
1530            values: self.values.into_iter().map(Value::into_owned).collect(),
1531        }
1532    }
1533}
1534
1535impl Row<'static> {
1536    /// v7.37.42-arena Phase 1 — lift any `Row<'arena>` (possibly arena-
1537    /// borrowed) into a fully-owned `Row<'static>`. Mirrors
1538    /// `Value::into_owned`.
1539    pub fn from_arena(row: Row<'_>) -> Self {
1540        Self {
1541            values: row.values.into_iter().map(Value::into_owned).collect(),
1542        }
1543    }
1544}
1545
1546/// Each bool is an independent, separately-persisted column attribute
1547/// (`nullable`, `auto_increment`, `is_unsigned`, `identity_always`) that the
1548/// catalog appendix reads and writes by name. Packing them into a bitflags
1549/// word would buy nothing and would put a decoding step between the on-disk
1550/// format and every reader of the schema.
1551#[allow(clippy::struct_excessive_bools)]
1552#[derive(Debug, Clone, PartialEq)]
1553pub struct ColumnSchema {
1554    pub name: String,
1555    pub ty: DataType,
1556    pub nullable: bool,
1557    /// Optional `DEFAULT` value, frozen at CREATE TABLE time. `None`
1558    /// means "no default" (so omitted columns become NULL, or error
1559    /// out when the column is NOT NULL). Literal defaults take this
1560    /// path.
1561    ///
1562    /// v7.37.42-arena Phase 1: explicitly `Value<'static>` — catalog
1563    /// defaults must outlive any per-query arena.
1564    pub default: Option<Value<'static>>,
1565    /// v7.9.21 — for DEFAULT expressions that need INSERT-time
1566    /// evaluation (e.g. `DEFAULT now()`, `DEFAULT CURRENT_TIMESTAMP`),
1567    /// the Display form of the expression. The engine re-parses
1568    /// it on each INSERT default-fill, evaluates against an empty
1569    /// row context, and coerces to the column type. mailrs G4.
1570    /// Persisted in catalog FILE_VERSION 15+; older catalogs
1571    /// deserialise with None.
1572    pub runtime_default: Option<String>,
1573    /// MySQL-style `AUTO_INCREMENT`. When set, an INSERT that leaves
1574    /// this column unbound (or sets it to NULL) gets the next integer
1575    /// computed from the column's current max + 1.
1576    /// v7.39 (round 676) — the collation NAME as written, when the column
1577    /// carried an explicit `COLLATE`.
1578    ///
1579    /// `spg_sql::Collation` cannot carry it: it is a two-variant MySQL enum
1580    /// and `from_collation_name` folds `C`, `POSIX`, `en_US` and `default`
1581    /// all into `Binary`. Without the name `pg_attribute.attcollation` can
1582    /// only ever report the type's default, which is what F36 records as
1583    /// "the declaration is taken and ignored".
1584    ///
1585    /// None means the column was written without a `COLLATE` clause and
1586    /// takes its type's collation. Persisted through the v88 appendix,
1587    /// which costs two bytes for a table that declares none.
1588    pub collation_name: Option<String>,
1589    pub auto_increment: bool,
1590    /// v7.17.0 Phase 1.4 — when the column is bound to a user-
1591    /// defined ENUM type (the parser saw an unknown type ident
1592    /// and the engine resolved it against `catalog.enum_types`),
1593    /// this carries the enum name so INSERT/UPDATE can validate
1594    /// the cell value against the enum's labels. `ty` is
1595    /// `DataType::Text` in that case. Persisted in catalog
1596    /// FILE_VERSION 29+; older catalogs deserialise with None.
1597    pub user_enum_type: Option<String>,
1598    /// v7.17.0 Phase 1.5 — when the column is bound to a user-
1599    /// defined DOMAIN (the parser saw an unknown type ident and
1600    /// the engine resolved it against `catalog.domain_types`),
1601    /// this carries the domain name. `ty` is the domain's base
1602    /// type; INSERT/UPDATE re-evaluates the domain's CHECK list
1603    /// + NOT NULL against the cell value. Persisted in catalog
1604    /// FILE_VERSION 30+; older catalogs deserialise with None.
1605    pub user_domain_type: Option<String>,
1606    /// v7.39 (read01 round 56) — when the column is bound to a user-defined
1607    /// COMPOSITE type. `ty` stays `DataType::Jsonb` (the on-disk form), but the
1608    /// engine REHYDRATES the stored JSON into a `Value::Composite` on read, so
1609    /// field access `(p).x`, `= ROW(…)`, ordering and the canonical `(2,b)`
1610    /// text form all work — they were already implemented on Value::Composite;
1611    /// what was missing was that the column never recorded WHICH composite type
1612    /// it holds (this field's doc comment existed for two releases, the field
1613    /// itself did not). Persisted in the composite-column appendix
1614    /// (FILE_VERSION 63+); older catalogs deserialise with None.
1615    pub user_composite_type: Option<String>,
1616    /// v7.39 (read01 round 59) — column-level privileges (PG
1617    /// `pg_attribute.attacl`). `GRANT SELECT (pub) ON t TO dan` lands here and
1618    /// does NOT touch the table's `relacl`. Empty = no column grant, which is
1619    /// every column until one is made.
1620    pub acl: Vec<AclItem>,
1621    /// v7.17.0 Phase 2.1 — MySQL `ON UPDATE CURRENT_TIMESTAMP`
1622    /// column attribute. When `Some(expr_src)`, an UPDATE that
1623    /// does NOT bind this column overrides the new value with
1624    /// the engine-evaluated expression (always `now()` in
1625    /// v7.17.0). Stored as Display-form source so storage
1626    /// stays free of spg-sql; the engine re-parses at UPDATE
1627    /// time. Persisted in catalog FILE_VERSION 32+; older
1628    /// catalogs deserialise with None — preserves the existing
1629    /// "silent ignore" behaviour for snapshots written before
1630    /// the upgrade.
1631    pub on_update_runtime: Option<String>,
1632    /// v7.17.0 Phase 2.5 — text collation. Pre-2.5 SPG accepted
1633    /// `COLLATE <name>` clauses but discarded the name, so a
1634    /// column declared `COLLATE "case_insensitive"` (or any
1635    /// MySQL `_ci` collation) still compared byte-wise — a
1636    /// Tier-S silent failure where `WHERE name = 'foo'` never
1637    /// matched stored `'Foo'`. This carries the parser-derived
1638    /// classification so the engine's WHERE evaluator can route
1639    /// text equality through a case-aware compare. `Binary` (the
1640    /// default) preserves the prior byte-wise behaviour. Only
1641    /// CaseInsensitive lands in the catalog appendix — Binary
1642    /// columns stay implicit, keeping snapshots compact.
1643    /// Persisted in catalog FILE_VERSION 34+; older catalogs
1644    /// deserialise every column as `Binary`.
1645    pub collation: Collation,
1646    /// v7.17.0 Phase 4.4 — MySQL `UNSIGNED` modifier flag. Drives
1647    /// engine-side INSERT / UPDATE range enforcement (rejects
1648    /// negative values on UNSIGNED int columns). Pre-4.4 the
1649    /// parser consumed and discarded the keyword silently, so
1650    /// every UNSIGNED column quietly accepted negatives — a
1651    /// Tier-A correctness drift. Sparse: only UNSIGNED columns
1652    /// land in the catalog appendix; the default `false` keeps
1653    /// snapshots compact for the common signed-int path.
1654    /// Persisted in catalog FILE_VERSION 35+; older catalogs
1655    /// deserialise every column as `is_unsigned = false`.
1656    pub is_unsigned: bool,
1657    /// v7.17.0 Phase 3.P0-36 — MySQL inline `ENUM('a','b','c')`
1658    /// value list. Distinct from `user_enum_type` (which points
1659    /// to a separately CREATE TYPE'd PG enum); this carries the
1660    /// column-local list MySQL DDL declares inline. When `Some`,
1661    /// `ty` is `DataType::Text` and INSERT/UPDATE validates the
1662    /// cell value against this list. Variant ORDER is preserved
1663    /// (MySQL uses it for `ORDER BY col`). Sparse: only ENUM
1664    /// columns land in the catalog appendix.
1665    /// Persisted in catalog FILE_VERSION 41+; older catalogs
1666    /// deserialise with None — preserves silent-drop behaviour
1667    /// for snapshots written before P0-36.
1668    pub inline_enum_variants: Option<Vec<String>>,
1669    /// v7.17.0 Phase 3.P0-37 — MySQL inline `SET('a','b','c')`
1670    /// variant list. Storage is TEXT (canonical comma-joined in
1671    /// definition order, de-duplicated). INSERT/UPDATE validates
1672    /// every comma-separated token against this list. Sparse:
1673    /// only SET columns land in the catalog appendix.
1674    /// Persisted in catalog FILE_VERSION 42+; older catalogs
1675    /// deserialise with None.
1676    pub inline_set_variants: Option<Vec<String>>,
1677    /// v7.37.7(sentori Epic 3 P1)— `GENERATED ALWAYS AS (<expr>)
1678    /// STORED` computed-column source. When `Some`, INSERT / UPDATE
1679    /// recompute the cell against the candidate row(re-parse the
1680    /// stored Display form and evaluate)and overwrite any
1681    /// user-supplied value, matching PG's stored-generated-column
1682    /// semantics. `None` (the default) preserves the regular
1683    /// "column value is whatever the caller passed" path.
1684    /// Persisted in catalog FILE_VERSION 50+; older catalogs
1685    /// deserialise with None.
1686    pub generated_stored_expr: Option<String>,
1687    /// v7.38 (read01) — `GENERATED ALWAYS AS IDENTITY`. Both identity
1688    /// flavours set `auto_increment`; this additionally marks the ALWAYS
1689    /// flavour, whose explicit INSERT value PG rejects ("cannot insert a
1690    /// non-DEFAULT value into column …") unless `OVERRIDING SYSTEM VALUE`.
1691    /// `false` (serial / `BY DEFAULT`) keeps the permissive path. In-memory
1692    /// only for now — not yet in the catalog appendix, so a reloaded table
1693    /// deserialises as `false` (the pre-existing permissive behaviour).
1694    pub identity_always: bool,
1695    /// v7.38 (read01) — the DEFAULT expression's source text, deparsed to
1696    /// PG-compatible form at CREATE TABLE time (e.g. `0`, `(3 + 4)`,
1697    /// `'hi'::text`, `now()`, `CURRENT_DATE`). Distinct from `default`
1698    /// (the coerced value the INSERT path fills) and `runtime_default`
1699    /// (the recompute-per-row Display form): those lose the source
1700    /// spelling, so `information_schema.columns.column_default` /
1701    /// `pg_attrdef` / `pg_get_expr` reported the coerced render
1702    /// (`0.00` for `numeric(10,2) DEFAULT 0`) instead of PG's `0`.
1703    /// `None` for a column with no explicit default. Persisted in catalog
1704    /// FILE_VERSION 58+; older catalogs deserialise with None.
1705    pub default_text: Option<String>,
1706    /// v7.39 (round 220) — `ALTER TABLE … ALTER COLUMN … RESTART [WITH n]`
1707    /// on an identity column. SPG's identity allocation is a max+1 scan;
1708    /// this floor lifts the next allocated value to at least `n`
1709    /// (`max(max+1, n)`) — exactly what a dump-restore RESTART needs, and
1710    /// safer than PG for a backward RESTART (no duplicate-key landmine).
1711    /// Persisted in the FILE_VERSION 73+ sparse appendix; older catalogs
1712    /// deserialise with None.
1713    pub auto_restart: Option<i64>,
1714    /// v7.39 (read01 round 78) — this column is the ONLY column of a FROM item
1715    /// that calls a function returning a BASE type, so the item's row type IS
1716    /// this column: a whole-row reference collapses to the value
1717    /// (`SELECT j FROM jsonb_array_elements('[1]') AS j` → `1`, PG). Runtime
1718    /// only — a catalogued table column is never one, and it is not persisted.
1719    pub scalar_row_source: bool,
1720    /// v7.39 (round 386, type-fidelity epic P1) — the declared MySQL narrow
1721    /// integer width (TINYINT / MEDIUMINT) whose range the storage `ty`
1722    /// (SmallInt / Int) is too wide to enforce. `None` for every other
1723    /// column. Drives the epic-P2 write-path range check. Persisted in the
1724    /// FILE_VERSION 81+ sparse appendix; older catalogs deserialise as None.
1725    pub mysql_int_width: Option<MysqlIntWidth>,
1726    /// v7.39 (round 424, type-fidelity epic) — the declared MySQL
1727    /// fractional-seconds precision of a temporal column: `DATETIME(3)` is
1728    /// `Some(3)`, a BARE `DATETIME` / `TIME` / `TIMESTAMP` is `Some(0)`
1729    /// (MySQL's default is zero — the fraction is dropped on write), and
1730    /// `None` means "not a MySQL-declared temporal column", which is every
1731    /// PG column and leaves microsecond behaviour untouched.
1732    ///
1733    /// Drives write-path truncation (toward zero) and render padding
1734    /// (exactly this many digits, `.000` when the fraction is zero).
1735    /// Persisted in the FILE_VERSION 82+ sparse appendix; older catalogs
1736    /// deserialise as None.
1737    pub mysql_fsp: Option<u8>,
1738    /// v7.39.2 — this column was DECLARED `TIMESTAMP` in a MySQL
1739    /// session.
1740    ///
1741    /// MySQL and MariaDB both keep `timestamp` and `datetime` apart in
1742    /// `SHOW CREATE TABLE`, `SHOW COLUMNS` and `information_schema`
1743    /// (measured on 9.7.2 and 12.3.3); SPG stores both as
1744    /// `DataType::Timestamp` and so reported `datetime` for both. A
1745    /// client dumping and reloading had the column's declared type
1746    /// SILENTLY CHANGED — and MySQL's TIMESTAMP is not DATETIME: it has
1747    /// a different range and converts to and from UTC.
1748    ///
1749    /// What this records is the SPELLING, which is the half a dump
1750    /// round-trips. The storage and the semantics are unchanged, and
1751    /// that gap is written down rather than papered over.
1752    ///
1753    /// Persisted in the FILE_VERSION 93+ sparse appendix; older
1754    /// catalogs deserialise as `false`.
1755    pub mysql_declared_timestamp: bool,
1756}
1757
1758/// v7.17.0 Phase 2.5 — column-level text collation. Drives the
1759/// engine's WHERE / GROUP BY equality routing for `Value::Text`.
1760/// Only two variants are modelled in v7.17:
1761///   * `Binary`  — byte-wise comparison (the SPG default;
1762///                 matches PG `COLLATE "C"` / `pg_catalog.default`
1763///                 and MySQL `*_bin`).
1764///   * `CaseInsensitive` — ASCII case-folded comparison (like
1765///                 MySQL `*_ci` collations; PG has NO built-in
1766///                 collation of this name — round-761 audit: a
1767///                 nondeterministic ICU collation must be CREATEd
1768///                 there first). Non-ASCII bytes
1769///                 still compare byte-wise; full ICU folding is
1770///                 out of v7.17 scope.
1771/// New variants append at the end — older catalogs read missing
1772/// columns as `Binary`.
1773#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1774pub enum Collation {
1775    Binary,
1776    CaseInsensitive,
1777}
1778
1779/// v7.39 (round 386, type-fidelity epic P1) — the declared MySQL narrow
1780/// integer type for a column whose storage `DataType` cannot express it.
1781/// MySQL `TINYINT` (i8, -128..127) collapses to `DataType::SmallInt` (i16)
1782/// and `MEDIUMINT` (24-bit) to `DataType::Int` (i32) — both wider than the
1783/// declared type, so a range check against `ty` alone accepts out-of-range
1784/// values (`INSERT 128 INTO TINYINT` is stored silently where MariaDB
1785/// strict raises ERROR 1264). This annotation records the lost width so the
1786/// write path (epic P2) can enforce the real bounds. `SMALLINT` / `INT` /
1787/// `BIGINT` need no marker — their storage `DataType` is already faithful.
1788/// Sparse: only TINYINT / MEDIUMINT columns carry it; persisted in the
1789/// FILE_VERSION 81+ appendix, older catalogs deserialise as None.
1790#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1791pub enum MysqlIntWidth {
1792    /// MySQL `TINYINT` — signed -128..127, unsigned 0..255. Storage i16.
1793    Tiny,
1794    /// MySQL `SMALLINT UNSIGNED` — 0..65535. Storage widened to i32 (a
1795    /// signed SMALLINT keeps `DataType::SmallInt` and carries no marker).
1796    Small,
1797    /// MySQL `MEDIUMINT` — signed -8388608..8388607, unsigned 0..16777215.
1798    /// Storage i32.
1799    Medium,
1800    /// MySQL `INT UNSIGNED` — 0..4294967295. Storage widened to i64 (a
1801    /// signed INT keeps `DataType::Int` and carries no marker).
1802    Int,
1803    /// v7.39 (round 471, epic P4b) — MySQL `BIGINT UNSIGNED` —
1804    /// 0..18446744073709551615. i64 stops at 2^63-1, so the storage tag is
1805    /// widened to `Numeric` (i128-backed, scale 0), which already compares,
1806    /// orders, indexes and renders as an exact integer. A signed BIGINT
1807    /// keeps `DataType::BigInt` and carries no marker.
1808    Big,
1809}
1810
1811/// v7.39 (round 363, M4 P1) — MySQL's default accent- and
1812/// case-insensitive fold (`utf8mb4_uca1400_ai_ci`).
1813///
1814/// This is the primitive M4 rests on: a session on the MySQL dialect
1815/// compares, groups, sorts and de-duplicates text by its FOLDED form, so
1816/// `Foo` = `foo` = `FOO` and, because the default collation is accent-
1817/// insensitive too, `Bär` = `bar`. The later stages (read path, then the
1818/// UNIQUE / index write path) all route through here so they cannot fold
1819/// differently from one another.
1820///
1821/// The fold is more than case + strip-combining: MariaDB EXPANDS some
1822/// letters — `ß` → `ss`, `æ` → `ae`, `œ` → `oe` — which is why the result
1823/// is built as a `String` rather than mapped char-for-char. Every mapping
1824/// below was measured on MariaDB 11 (`'Bär'='bar'` is 1, `'straße'=
1825/// 'strasse'` is 1, `'a'='æ'` is 0, `'s'='ß'` is 0). Characters with no
1826/// entry keep their lower-cased self, so ASCII and unknown scripts pass
1827/// through unchanged.
1828#[must_use]
1829pub fn mysql_ci_fold(s: &str) -> String {
1830    let mut out = String::with_capacity(s.len());
1831    for ch in s.chars() {
1832        // Lower-case first (`À` → `à`, `Æ` → `æ`), then fold the base.
1833        for lc in ch.to_lowercase() {
1834            match fold_latin_base(lc) {
1835                Some(base) => out.push_str(base),
1836                None => out.push(lc),
1837            }
1838        }
1839    }
1840    out
1841}
1842
1843/// The fold used to COMPARE / GROUP / de-dup text on the MySQL dialect:
1844/// case- and accent-insensitive, and **trailing spaces significant**.
1845///
1846/// v7.38.17 — this used to strip trailing spaces first, and its comment
1847/// said why: "measured on MariaDB 11". MariaDB's default collation is
1848/// PAD SPACE, so that measurement was right about MariaDB. SPG
1849/// advertises `8.0.0-spg-v…` on the MySQL wire, and MySQL 8.0's default
1850/// `utf8mb4_0900_ai_ci` is **NO PAD**. The rule had been calibrated
1851/// against the engine we do not claim to be.
1852///
1853/// Measured today, MySQL 9.7.2 against MariaDB 12.3.2, each in its own
1854/// default collation, over rows `'alpha'` and `'alpha  '`:
1855///
1856/// | | MySQL | MariaDB |
1857/// |---|---|---|
1858/// | `WHERE s = 'alpha'` | 1 | 1,2 |
1859/// | `s IN ('alpha','beta')` | 1,3,4 | 1,2,3,4 |
1860/// | `COUNT(DISTINCT s)` | 3 | 2 |
1861/// | `GROUP BY s` groups | 3 | 2 |
1862/// | `JOIN ON v.s = r.s` | 1/10, 2/20 | all four pairs |
1863///
1864/// SPG answered MariaDB's four and MySQL's join — the same question
1865/// decided differently by two paths, which is the shape v7.38.13,
1866/// v7.38.14 and v7.38.16 were each spent on.
1867///
1868/// `CHAR(n)` is a separate question and keeps its old answer: BOTH
1869/// engines ignore a CHAR's trailing spaces, because that is a property
1870/// of the TYPE rather than of the collation. Use
1871/// [`mysql_compare_fold_char`] for a `BpChar` cell.
1872///
1873/// Only literal spaces ever padded — a tab is significant either way —
1874/// and neither function is used by `LIKE`, whose pattern treats a
1875/// trailing space literally.
1876/// Whether a collation of this NAME orders by bytes.
1877///
1878/// v7.38.18 (S0) — pure string classification, and it lives here because
1879/// storage has to ask it: an index whose column collates by a locale
1880/// cannot key on the raw text, and the write path is here. The engine's
1881/// `collate::is_byte_wise` delegates to this one, for the reason the SQL
1882/// type spellings have one owner.
1883///
1884/// `C`, `POSIX`, MySQL's `binary` and every `_bin` family member. The
1885/// encoding suffix rides along: PG publishes `C.utf8` beside `C`.
1886pub fn collation_is_byte_wise(collation: &str) -> bool {
1887    let name = collation.trim();
1888    let base = name.split(['.', '@']).next().unwrap_or(name);
1889    base.eq_ignore_ascii_case("C")
1890        || base.eq_ignore_ascii_case("POSIX")
1891        || base.eq_ignore_ascii_case("binary")
1892        || base
1893            .rsplit_once('_')
1894            .is_some_and(|(_, tail)| tail.eq_ignore_ascii_case("bin"))
1895}
1896
1897/// v7.38.18 (S0/S2) — does an index on a column of this collation key
1898/// by an ICU SORT KEY rather than by the raw text?
1899///
1900/// True for a locale collation (`en_US.utf8`, `de_DE`), which orders by
1901/// rules a byte comparison cannot express.
1902///
1903/// False for byte-wise names, and false for MySQL's folding collations
1904/// (`utf8mb4_0900_ai_ci` and family). Those fold rather than collate,
1905/// and the engine has folded them since v7.37 — routing them here made
1906/// an indexed `s = 'ALPHA'` over the MySQL wire answer nothing where
1907/// MySQL 9.7.1 answers one row, because ICU at PG's strength does not
1908/// call `ALPHA` and `alpha` equal.
1909///
1910/// One owner for the same reason the byte-wise question has one: the
1911/// engine builds the PROBE and this crate builds the ENTRIES, and a
1912/// probe built in another space finds nothing — which reads exactly
1913/// like "no matching rows".
1914pub fn collation_uses_sort_key(collation: &str) -> bool {
1915    if collation_is_byte_wise(collation) {
1916        return false;
1917    }
1918    let name = collation.trim();
1919    let base = name.split(['.', '@']).next().unwrap_or(name);
1920    let lower = base.to_ascii_lowercase();
1921    !(lower.ends_with("_ci") || lower.ends_with("_cs"))
1922}
1923
1924pub fn mysql_compare_fold(s: &str) -> String {
1925    mysql_ci_fold(s)
1926}
1927
1928/// The comparison form of one text value under the MySQL default
1929/// collation, or `None` for a value that is not text.
1930///
1931/// v7.38.18 — one function, applied to each side SEPARATELY, because
1932/// the pair is not the unit. Several sites matched
1933/// `(Text, Text) | (BpChar, BpChar)` and folded a pair; a CHAR compared
1934/// against a VARCHAR or against a literal is neither shape, so it fell
1935/// through and was compared by bytes — with the CHAR still carrying its
1936/// padding. `CASE c WHEN 'ALPHA'` on a `CHAR(8)` holding `'alpha'`
1937/// answered ELSE where MySQL 9.7.2 answers the branch.
1938///
1939/// Folding per value also states the rule correctly: whether trailing
1940/// spaces count is a property of EACH side's own type, so a pair whose
1941/// sides differ has two answers rather than one.
1942pub fn mysql_fold_value(v: &Value<'_>) -> Option<String> {
1943    match v {
1944        Value::BpChar(s) => Some(mysql_compare_fold_char(s)),
1945        Value::Text(s) => Some(mysql_compare_fold(s)),
1946        _ => None,
1947    }
1948}
1949
1950/// [`mysql_compare_fold`] for a `CHAR(n)` cell, whose trailing spaces
1951/// are padding rather than data.
1952///
1953/// Measured on both engines: over `'alpha'` and `'alpha  '` in a
1954/// `CHAR(8)`, `WHERE s = 'alpha'` returns both rows and
1955/// `COUNT(DISTINCT s)` is 2 (four rows folding to two values) — MySQL
1956/// 9.7.2 and MariaDB 12.3.2 agree, unlike the VARCHAR case above.
1957pub fn mysql_compare_fold_char(s: &str) -> String {
1958    mysql_ci_fold(s.trim_end_matches(' '))
1959}
1960
1961/// The base letter(s) a lower-cased Latin character folds to, or `None`
1962/// when it is already a base / has no fold. Expansions (`ß` → `ss`) are
1963/// why this returns a string.
1964fn fold_latin_base(c: char) -> Option<&'static str> {
1965    Some(match c {
1966        'à' | 'á' | 'â' | 'ã' | 'ä' | 'å' | 'ā' | 'ă' | 'ą' => "a",
1967        'æ' => "ae",
1968        'ç' | 'ć' | 'č' | 'ĉ' | 'ċ' => "c",
1969        'ð' | 'ď' | 'đ' => "d",
1970        'è' | 'é' | 'ê' | 'ë' | 'ē' | 'ĕ' | 'ė' | 'ę' | 'ě' => "e",
1971        'ĝ' | 'ğ' | 'ġ' | 'ģ' => "g",
1972        'ì' | 'í' | 'î' | 'ï' | 'ĩ' | 'ī' | 'ĭ' | 'į' => "i",
1973        'ĵ' => "j",
1974        'ķ' => "k",
1975        'ł' | 'ĺ' | 'ļ' | 'ľ' => "l",
1976        'ñ' | 'ń' | 'ņ' | 'ň' => "n",
1977        'ò' | 'ó' | 'ô' | 'õ' | 'ö' | 'ø' | 'ō' | 'ŏ' | 'ő' => "o",
1978        'œ' => "oe",
1979        'ŕ' | 'ŗ' | 'ř' => "r",
1980        'ś' | 'š' | 'ŝ' | 'ş' => "s",
1981        'ß' => "ss",
1982        'ţ' | 'ť' | 'ŧ' => "t",
1983        'ù' | 'ú' | 'û' | 'ü' | 'ũ' | 'ū' | 'ŭ' | 'ů' | 'ű' | 'ų' => "u",
1984        'ý' | 'ÿ' => "y",
1985        'ź' | 'ž' | 'ż' => "z",
1986        _ => return None,
1987    })
1988}
1989
1990#[allow(clippy::derivable_impls)]
1991impl Default for Collation {
1992    fn default() -> Self {
1993        Self::Binary
1994    }
1995}
1996
1997impl Collation {
1998    /// Wire tag persisted in the FILE_VERSION 34+ catalog appendix.
1999    /// Stable: future variants append above the recognised range
2000    /// and unknown tags read back as `Binary` for forward-compat
2001    /// on rollback.
2002    pub const TAG_BINARY: u8 = 0;
2003    pub const TAG_CASE_INSENSITIVE: u8 = 1;
2004}
2005
2006/// v7.39 (RLS) — the command a policy applies to. `ALL` is the default and
2007/// covers every command; the others scope the policy to one statement kind.
2008/// Persisted as a single byte in the policy appendix (FILE_VERSION 59+).
2009#[derive(Debug, Clone, Copy, PartialEq, Eq)]
2010pub enum PolicyCmd {
2011    All,
2012    Select,
2013    Insert,
2014    Update,
2015    Delete,
2016}
2017
2018impl PolicyCmd {
2019    /// PG `pg_policy.polcmd` single-char encoding.
2020    #[must_use]
2021    pub const fn as_pg_char(self) -> char {
2022        match self {
2023            Self::All => '*',
2024            Self::Select => 'r',
2025            Self::Insert => 'a',
2026            Self::Update => 'w',
2027            Self::Delete => 'd',
2028        }
2029    }
2030
2031    /// PG `pg_policies.cmd` word form.
2032    #[must_use]
2033    pub const fn as_pg_word(self) -> &'static str {
2034        match self {
2035            Self::All => "ALL",
2036            Self::Select => "SELECT",
2037            Self::Insert => "INSERT",
2038            Self::Update => "UPDATE",
2039            Self::Delete => "DELETE",
2040        }
2041    }
2042
2043    #[must_use]
2044    pub const fn to_wire_byte(self) -> u8 {
2045        match self {
2046            Self::All => 0,
2047            Self::Select => 1,
2048            Self::Insert => 2,
2049            Self::Update => 3,
2050            Self::Delete => 4,
2051        }
2052    }
2053
2054    #[must_use]
2055    pub const fn from_wire_byte(b: u8) -> Option<Self> {
2056        match b {
2057            0 => Some(Self::All),
2058            1 => Some(Self::Select),
2059            2 => Some(Self::Insert),
2060            3 => Some(Self::Update),
2061            4 => Some(Self::Delete),
2062            _ => None,
2063        }
2064    }
2065}
2066
2067/// v7.39 (RLS) — one `CREATE POLICY` object, stored per table. The `using_expr`
2068/// / `with_check_expr` hold the qualifying expression's `Display` form
2069/// (re-parsed and evaluated per row at enforcement time, exactly like
2070/// `TableSchema.checks`); `None` means the clause was absent. `roles` empty =
2071/// PUBLIC. Persisted in the policy appendix (FILE_VERSION 59+).
2072#[derive(Debug, Clone, PartialEq)]
2073pub struct PolicyDef {
2074    pub name: String,
2075    pub cmd: PolicyCmd,
2076    /// `true` = PERMISSIVE (default, OR-combined), `false` = RESTRICTIVE
2077    /// (AND-combined).
2078    pub permissive: bool,
2079    pub roles: Vec<String>,
2080    pub using_expr: Option<String>,
2081    pub with_check_expr: Option<String>,
2082}
2083
2084#[derive(Debug, Clone, PartialEq)]
2085pub struct TableSchema {
2086    pub name: String,
2087    pub columns: Vec<ColumnSchema>,
2088    /// v6.7.2 — per-table hot-tier byte budget override. `None`
2089    /// falls through to the global `SPG_HOT_TIER_BYTES` setting;
2090    /// `Some(n)` overrides it for this specific table. Set via
2091    /// `ALTER TABLE t SET hot_tier_bytes = X`. Persisted in
2092    /// catalog FILE_VERSION 11+.
2093    pub hot_tier_bytes: Option<u64>,
2094    /// v7.6.1 — FOREIGN KEY constraints declared on this table.
2095    /// Engine maintains this in lock-step with `spg-sql`'s parser
2096    /// AST; the storage layer carries the on-disk shape so a
2097    /// catalog snapshot round-trips without external mapping.
2098    /// Persisted in catalog FILE_VERSION 13+. Older catalogs
2099    /// deserialise with an empty vec.
2100    pub foreign_keys: Vec<ForeignKeyConstraint>,
2101    /// v7.9.19 — composite UNIQUE / PRIMARY KEY constraints
2102    /// declared at the table level. Each entry's leading column
2103    /// has a BTree index (created via the constraint), and INSERT
2104    /// path enforces the full-tuple uniqueness via a scan keyed
2105    /// by the leading column. Persisted in catalog FILE_VERSION
2106    /// 15+. Older catalogs (≤ 14) deserialise with an empty vec.
2107    pub uniqueness_constraints: Vec<UniquenessConstraint>,
2108    /// v7.39 (round 210) — `EXCLUDE` constraints declared at the table level.
2109    /// Enforced on INSERT/UPDATE by a full live-row scan re-checking each
2110    /// element's operator (no equality index can answer overlap). Persisted
2111    /// in catalog FILE_VERSION 72+; older catalogs deserialise with an empty
2112    /// vec.
2113    pub exclusion_constraints: Vec<ExclusionConstraint>,
2114    /// v7.13.0 — `CHECK (<expr>)` predicates declared on this
2115    /// table. Both column-level inline `CHECK (…)` and
2116    /// table-level `CHECK (…)` fold into this list. Each entry
2117    /// is the AST Expr's `Display` form, re-parsed on every
2118    /// INSERT/UPDATE and evaluated against the candidate row.
2119    /// A false / NULL result rejects the mutation (PG semantics).
2120    /// Persisted in catalog FILE_VERSION 23+. Older catalogs
2121    /// deserialise with an empty vec. v7.39 (read01 round 48) — each entry
2122    /// now carries the user's constraint name too (FILE_VERSION 60+).
2123    pub checks: Vec<CheckConstraint>,
2124    /// v7.37.6-B — declarative partition role(sentori Epic 2 P0).
2125    /// `None` = 普通表(后向兼容,< v49 catalog 默认 None)。
2126    /// `Some(Parent { … })` = `CREATE TABLE p (...) PARTITION BY RANGE (key_col)` 父表 —
2127    /// 父表自己 `rows` 永远空,INSERT 在引擎层路由到命中的 child。
2128    /// `Some(Range { … })` = `CREATE TABLE c PARTITION OF p FOR VALUES FROM (a) TO (b)` 范围子表。
2129    /// `Some(Default { … })` = `CREATE TABLE c PARTITION OF p DEFAULT` 兜底子表。
2130    /// 持久化于 FILE_VERSION 49+。
2131    pub partition_role: Option<PartitionRole>,
2132    /// v7.39 (RLS) — `CREATE POLICY` objects on this table, independent of the
2133    /// `row_security` flag (PG stores policies even on non-RLS tables; they
2134    /// only take effect once RLS is enabled). Persisted in the policy appendix
2135    /// (FILE_VERSION 59+). Older catalogs deserialise with an empty vec.
2136    pub policies: Vec<PolicyDef>,
2137    /// v7.39 (RLS) — `ALTER TABLE … ENABLE ROW LEVEL SECURITY`
2138    /// (PG `pg_class.relrowsecurity`). Fresh table = `false`.
2139    pub row_security: bool,
2140    /// v7.39 (RLS) — `ALTER TABLE … FORCE ROW LEVEL SECURITY`
2141    /// (PG `pg_class.relforcerowsecurity`); subjects the table owner to RLS
2142    /// too. Fresh table = `false`.
2143    pub force_row_security: bool,
2144    /// v7.39 (read01 round 57, ACL) — the role that owns this table: whoever
2145    /// ran CREATE TABLE (PG `pg_class.relowner`). The owner holds every
2146    /// privilege implicitly and is the only role that may ALTER / DROP it.
2147    /// `None` = an image written before FILE_VERSION 64, which predates roles
2148    /// entirely; those tables read back as owned by the login role.
2149    pub owner: Option<String>,
2150    /// v7.39 (read01 round 57, ACL) — explicit GRANTs on this table
2151    /// (PG `pg_class.relacl`). EMPTY means "never granted": PG leaves relacl
2152    /// NULL while only the owner's implicit privileges apply, and materialises
2153    /// the whole list — owner's default entry included — on the first GRANT.
2154    /// Once materialised it stays, even after every grant is revoked.
2155    pub acl: Vec<AclItem>,
2156}
2157
2158/// v7.39 (read01 round 57) — one PG `aclitem`: what `grantee` may do to a
2159/// table, and who granted it. Renders as `grantee=privs/grantor`, with an
2160/// EMPTY grantee meaning PUBLIC (`=r/owner`).
2161#[derive(Debug, Clone, PartialEq, Eq)]
2162pub struct AclItem {
2163    /// The role the privileges are held by. Empty string = PUBLIC.
2164    pub grantee: String,
2165    /// Bitmask over `priv_bits`: which privileges are held.
2166    pub privs: u16,
2167    /// Bitmask over `priv_bits`: which of them carry WITH GRANT OPTION
2168    /// (PG renders those with a trailing `*` — `r*`).
2169    pub grantable: u16,
2170    /// The role that ran the GRANT.
2171    pub grantor: String,
2172}
2173
2174/// v7.39 (read01 round 57) — the table-privilege bits, in PG's `aclitem`
2175/// rendering order (`arwdDxtm`). The order matters: `relacl` output is
2176/// byte-compared against PG.
2177pub mod priv_bits {
2178    pub const INSERT: u16 = 1 << 0; // a
2179    pub const SELECT: u16 = 1 << 1; // r
2180    pub const UPDATE: u16 = 1 << 2; // w
2181    pub const DELETE: u16 = 1 << 3; // d
2182    pub const TRUNCATE: u16 = 1 << 4; // D
2183    pub const REFERENCES: u16 = 1 << 5; // x
2184    pub const TRIGGER: u16 = 1 << 6; // t
2185    pub const MAINTAIN: u16 = 1 << 7; // m
2186    /// v7.39 (read01 round 60) — the non-table privileges. They share the
2187    /// bitmask because an aclitem is an aclitem whatever it hangs off; which
2188    /// bits are MEANINGFUL depends on the object (a sequence has r / w / U, a
2189    /// schema has U / C, a database has C / c / T).
2190    pub const USAGE: u16 = 1 << 8; // U
2191    pub const CREATE: u16 = 1 << 9; // C
2192    pub const CONNECT: u16 = 1 << 10; // c
2193    pub const TEMPORARY: u16 = 1 << 11; // T
2194    pub const EXECUTE: u16 = 1 << 12; // X
2195    /// Every TABLE privilege — what `GRANT ALL ON <table>` grants and what a
2196    /// table's owner holds.
2197    pub const ALL: u16 =
2198        INSERT | SELECT | UPDATE | DELETE | TRUNCATE | REFERENCES | TRIGGER | MAINTAIN;
2199    /// `GRANT ALL ON SEQUENCE` — PG renders a sequence owner's default as `rwU`.
2200    pub const ALL_SEQUENCE: u16 = SELECT | UPDATE | USAGE;
2201    /// `GRANT ALL ON SCHEMA` — `UC`.
2202    pub const ALL_SCHEMA: u16 = USAGE | CREATE;
2203    /// `GRANT ALL ON DATABASE` — `CTc`.
2204    pub const ALL_DATABASE: u16 = CREATE | CONNECT | TEMPORARY;
2205    /// `GRANT ALL ON FUNCTION` — just `X`.
2206    pub const ALL_FUNCTION: u16 = EXECUTE;
2207}
2208
2209/// v7.37.6-B — partition 三态(parent / range child / default child)。
2210#[derive(Debug, Clone, PartialEq, Eq)]
2211pub enum PartitionRole {
2212    Parent {
2213        kind: PartitionKind,
2214        /// 父表 columns 中 key 列的下标(单列 v7.37.6-B,
2215        /// `Vec` 为将来扩多列预留)。
2216        key_column_positions: Vec<usize>,
2217        /// `CREATE INDEX ON parent (…)` 的 Display-form 源串。
2218        /// child 创建时再 parse + 在 child 上 execute,这样 future
2219        /// child 也自动继承父表索引。fan-out 实施在引擎层。
2220        index_template_sources: Vec<String>,
2221    },
2222    Range {
2223        parent_name: String,
2224        /// 半开区间下界(`>=`,SQL `FROM (lower)`).
2225        lower: PartitionBound,
2226        /// 半开区间上界(`<`,SQL `TO (upper)`).
2227        upper: PartitionBound,
2228    },
2229    /// v7.37.16 (16.1) — LIST child:行属于本 child iff key ∈ values。
2230    /// `values` 在 child 创建时从 SQL `FOR VALUES IN (lit, …)` 求值;
2231    /// 跟 PG 一样,显式 NULL ∈ values 由 caller 单独处理(不在
2232    /// PartitionBound 内表达 NULL)。
2233    List {
2234        parent_name: String,
2235        values: Vec<PartitionBound>,
2236    },
2237    /// v7.39 (round 645) — PG 表继承的 CHILD:`CREATE TABLE c (…)
2238    /// INHERITS (p1, p2)`。跟分区 child 的三个本质区别(实测 PG18):
2239    ///   * 父表**自己有行**(分区父表永远空),所以父表的联合体要含自身;
2240    ///   * `INSERT INTO 父表` **不路由**到 child(分区会路由);
2241    ///   * `DROP TABLE 父表` 不带 CASCADE **报错**(分区父表连子表一起删)。
2242    /// 多父继承合法,故 `parent_names` 是 Vec;`pg_inherits.inhseqno`
2243    /// 正是父表在这个列表里的位置(1-based)。
2244    Inherits {
2245        parent_names: Vec<String>,
2246    },
2247    /// v7.37.16 (16.2) — HASH child:行属于本 child iff
2248    /// `pg_compatible_hash(key) mod modulus == remainder`。
2249    /// PG 强制 `0 ≤ remainder < modulus`;parser/DDL 层先 gate。
2250    Hash {
2251        parent_name: String,
2252        modulus: u32,
2253        remainder: u32,
2254    },
2255    Default {
2256        parent_name: String,
2257    },
2258}
2259
2260/// v7.37.6-B — 分区策略。
2261///
2262/// - `Range`:半开区间 `[lower, upper)`(v7.37.6-B 初始)
2263/// - `List` (v7.37.16):枚举集合 — 行属于 partition iff key ∈ children list
2264/// - `Hash` (v7.37.16):`hash(key) mod modulus == remainder`
2265#[derive(Debug, Clone, Copy, PartialEq, Eq)]
2266pub enum PartitionKind {
2267    Range,
2268    List,
2269    Hash,
2270}
2271
2272/// v7.37.6-B — partition 边界 literal。
2273///
2274/// v7.37.6-B 仅 `TimestampTz`(i64 microseconds since epoch);
2275/// v7.37.16 (16.6) 加全 PG 内建可比类型,匹配 `Value` 的对应 variant
2276/// 以避免 LIST membership 比较时的类型转换。
2277///
2278/// `MinValue` / `MaxValue` 对应 SQL `MINVALUE` / `MAXVALUE`,仅
2279/// Range 策略有意义(LIST 无 minvalue/maxvalue 概念,HASH 不
2280/// 使用 PartitionBound)。
2281#[derive(Debug, Clone, PartialEq, Eq)]
2282pub enum PartitionBound {
2283    MinValue,
2284    MaxValue,
2285    TimestampTz(i64),
2286    /// v7.37.16 (16.6) — BIGINT partition key.
2287    BigInt(i64),
2288    /// v7.37.16 (16.6) — INTEGER partition key (also covers
2289    /// `SERIAL` since SPG decomposes it to INTEGER + sequence).
2290    Int(i32),
2291    /// v7.37.16 (16.6) — SMALLINT partition key.
2292    SmallInt(i16),
2293    /// v7.37.16 (16.6) — DATE partition key. Stored as days
2294    /// since the Unix epoch (matches `Value::Date`).
2295    Date(i32),
2296    /// v7.37.16 (16.6) — TEXT / VARCHAR partition key.
2297    Text(alloc::string::String),
2298}
2299
2300impl PartitionBound {
2301    /// v7.37.16 (16.6) — true iff this bound's underlying value
2302    /// equals `other`'s. Used for LIST partition membership
2303    /// checks. Returns false for `MinValue` / `MaxValue`
2304    /// (sentinels — never literal equality).
2305    #[must_use]
2306    pub fn equals_value(&self, other: &Value<'_>) -> bool {
2307        match (self, other) {
2308            (PartitionBound::TimestampTz(a), Value::Timestamp(b)) => a == b,
2309            (PartitionBound::BigInt(a), Value::BigInt(b)) => a == b,
2310            (PartitionBound::Int(a), Value::Int(b)) => a == b,
2311            (PartitionBound::SmallInt(a), Value::SmallInt(b)) => a == b,
2312            (PartitionBound::Date(a), Value::Date(b)) => a == b,
2313            (PartitionBound::Text(a), Value::Text(b)) => a.as_str() == b.as_ref(),
2314            _ => false,
2315        }
2316    }
2317}
2318
2319/// v7.9.19 — composite UNIQUE / PRIMARY KEY constraint persisted
2320/// on the table schema. The leading column always has a BTree
2321/// index (created at CREATE TABLE time); INSERT enforcement
2322/// scans that index for collisions on the full column tuple.
2323/// v7.39 (read01 round 48) — a `CHECK` constraint: the SQL name the user
2324/// gave it (via `ADD CONSTRAINT <name> CHECK (...)` or the inline
2325/// `CONSTRAINT <name> CHECK (...)` form) plus the predicate source. `None`
2326/// name = unnamed, in which case `pg_constraint` synthesises PG's
2327/// `<table>_<col>_check` form. Names are persisted in the constraint-name
2328/// appendix (FILE_VERSION 60+); older catalogs deserialise with `None`.
2329#[derive(Debug, Clone, PartialEq, Eq)]
2330pub struct CheckConstraint {
2331    pub name: Option<String>,
2332    /// The AST Expr's `Display` form, re-parsed on every INSERT/UPDATE.
2333    pub expr: String,
2334    /// v7.39 (round 652) — `false` for a constraint added `NOT VALID`: the
2335    /// rows already in the table were never scanned against it, and
2336    /// `pg_constraint.convalidated` says so. It does NOT weaken the check on
2337    /// new rows — INSERT and UPDATE enforce it either way, as in PG.
2338    /// `VALIDATE CONSTRAINT` does the deferred scan and flips it. Persisted
2339    /// by the FILE_VERSION 87 appendix; older catalogs deserialise as `true`,
2340    /// which is what every constraint they could hold actually was.
2341    pub validated: bool,
2342}
2343
2344#[derive(Debug, Clone, PartialEq, Eq)]
2345pub struct UniquenessConstraint {
2346    /// `true` when this constraint was declared as `PRIMARY KEY`
2347    /// (vs `UNIQUE`). Semantically PK implies NOT NULL on all
2348    /// referenced columns; the engine enforces that at CREATE
2349    /// TABLE time.
2350    pub is_primary_key: bool,
2351    /// Column positions on the parent table. ≥ 1 element. For
2352    /// single-column UNIQUE this is exactly one position; the
2353    /// BTree index alone enforces it.
2354    pub columns: Vec<usize>,
2355    /// v7.13.0 — `UNIQUE NULLS NOT DISTINCT` modifier
2356    /// (mailrs round-5 G10; PG 15+ surface). When `true`, two
2357    /// rows whose constrained columns are all NULL collide on
2358    /// the constraint. Default (`false`) is the SQL-standard
2359    /// `NULLS DISTINCT` behaviour where any NULL passes.
2360    /// Persisted in catalog FILE_VERSION 23+.
2361    pub nulls_not_distinct: bool,
2362    /// v7.39 (read01 round 48) — the constraint's SQL name when the user
2363    /// supplied one (`ADD CONSTRAINT <name> PRIMARY KEY/UNIQUE (...)`, or
2364    /// the inline `CONSTRAINT <name>` form). `None` = unnamed, in which
2365    /// case `pg_constraint` synthesises PG's `<table>_pkey` /
2366    /// `<table>_<col>_key` form. DROP CONSTRAINT resolves the stored name
2367    /// first and falls back to the synthesised one, so catalogs written
2368    /// before this field (< FILE_VERSION 60) keep working unchanged.
2369    pub name: Option<String>,
2370    /// v7.39 (round 711) — `[NOT] DEFERRABLE`. Round 621 taught the parser
2371    /// to CONSUME the clause on PK/UNIQUE (the FK path had stored it since
2372    /// round 288); this is the storing half. Persisted in the v89 timing
2373    /// appendix.
2374    pub deferrable: bool,
2375    /// `INITIALLY DEFERRED`: the check belongs to COMMIT, not the
2376    /// statement, unless `SET CONSTRAINTS … IMMEDIATE` pulls it in.
2377    pub initially_deferred: bool,
2378}
2379
2380/// v7.39 (round 210) — an `EXCLUDE` constraint. Forbids two distinct live
2381/// rows from satisfying, for EVERY element, `new.col <op> existing.col`
2382/// (e.g. `EXCLUDE USING gist (during WITH &&)` = no two `during` ranges
2383/// overlap). Unlike a uniqueness constraint the operator is not equality,
2384/// so enforcement is a full live-row scan re-checking the operator (a real
2385/// GiST index that answers overlap in O(log n) is a later perf phase). A
2386/// NULL in any element column exempts the row (matching PG / UNIQUE NULL
2387/// semantics). Persisted in catalog FILE_VERSION 72+.
2388#[derive(Debug, Clone, PartialEq, Eq)]
2389pub struct ExclusionConstraint {
2390    /// The constraint's SQL name. PG auto-names an unnamed EXCLUDE
2391    /// `<table>_<leading-col>_excl`; the engine synthesises that at CREATE
2392    /// TABLE time so this is always populated.
2393    pub name: String,
2394    /// Access method spelled after `USING` (`gist`, `spgist`, …), lower-cased.
2395    /// `None` = no `USING` clause. Purely cosmetic for enforcement; it round-
2396    /// trips into `pg_get_constraintdef`.
2397    pub method: Option<String>,
2398    /// One `(column-position, operator-spelling)` pair per element, in
2399    /// declaration order. The operator spelling is the wire token (`&&`,
2400    /// `=`, `@>`, `<@`, `&<`, `&>`) evaluated against each existing row.
2401    pub elements: Vec<(usize, String)>,
2402}
2403
2404/// v7.6.1 — Storage-layer mirror of `spg_sql::ast::ForeignKeyConstraint`.
2405/// The engine's CREATE TABLE path translates between the two; keeping
2406/// them separate preserves the no-deps boundary between
2407/// `spg-storage` and `spg-sql`.
2408#[derive(Debug, Clone, PartialEq, Eq)]
2409pub struct ForeignKeyConstraint {
2410    /// Optional user-supplied constraint name (`CONSTRAINT <name>`
2411    /// prefix). Used by `ALTER TABLE DROP CONSTRAINT <name>` in
2412    /// v7.6.8; ignored by enforcement.
2413    pub name: Option<String>,
2414    /// Positions of local columns in this table's column list.
2415    /// Same arity as `parent_columns`.
2416    pub local_columns: Vec<usize>,
2417    /// Referenced parent table name.
2418    pub parent_table: String,
2419    /// Positions of parent columns in the parent's column list.
2420    /// Engine resolves these at CREATE TABLE time (after the parent
2421    /// schema is known) so enforcement paths can skip the name
2422    /// lookup on every row.
2423    pub parent_columns: Vec<usize>,
2424    /// Referential action when a parent row is deleted.
2425    pub on_delete: FkAction,
2426    /// Referential action when a parent row's referenced columns
2427    /// are updated.
2428    pub on_update: FkAction,
2429    /// v7.38 (read01, T29) — `MATCH SIMPLE | FULL`. Defaults to `Simple`.
2430    pub match_type: MatchType,
2431    /// v7.39 (round 288) — `[NOT] DEFERRABLE`.
2432    pub deferrable: bool,
2433    /// `INITIALLY DEFERRED`: the check runs at COMMIT rather than at
2434    /// the statement, unless `SET CONSTRAINTS … IMMEDIATE` pulls it in.
2435    pub initially_deferred: bool,
2436}
2437
2438/// v7.38 (read01, T29) — FK MATCH type. Mirrors `spg_sql::ast::MatchType`.
2439#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
2440pub enum MatchType {
2441    #[default]
2442    Simple,
2443    Full,
2444}
2445
2446impl MatchType {
2447    /// On-disk tag byte (catalog appendix, `FILE_VERSION` 55+).
2448    pub const fn tag(self) -> u8 {
2449        match self {
2450            Self::Simple => 0,
2451            Self::Full => 1,
2452        }
2453    }
2454    pub const fn from_tag(b: u8) -> Option<Self> {
2455        Some(match b {
2456            0 => Self::Simple,
2457            1 => Self::Full,
2458            _ => return None,
2459        })
2460    }
2461}
2462
2463/// v7.6.1 — referential action tag. Mirrors `spg_sql::ast::FkAction`.
2464#[derive(Debug, Clone, Copy, PartialEq, Eq)]
2465pub enum FkAction {
2466    Restrict,
2467    Cascade,
2468    SetNull,
2469    SetDefault,
2470    NoAction,
2471}
2472
2473impl FkAction {
2474    /// On-disk tag byte (v13 catalog appendix).
2475    pub const fn tag(self) -> u8 {
2476        match self {
2477            Self::Restrict => 0,
2478            Self::Cascade => 1,
2479            Self::SetNull => 2,
2480            Self::SetDefault => 3,
2481            Self::NoAction => 4,
2482        }
2483    }
2484    pub const fn from_tag(b: u8) -> Option<Self> {
2485        Some(match b {
2486            0 => Self::Restrict,
2487            1 => Self::Cascade,
2488            2 => Self::SetNull,
2489            3 => Self::SetDefault,
2490            4 => Self::NoAction,
2491            _ => return None,
2492        })
2493    }
2494}
2495
2496impl TableSchema {
2497    pub fn column_position(&self, name: &str) -> Option<usize> {
2498        self.columns.iter().position(|c| c.name == name)
2499    }
2500}
2501
2502/// Key type accepted by secondary indices. Float / NULL / Vector values
2503/// can't participate in a B-tree index — `f64` is only `PartialOrd`, NULL
2504/// has SQL-three-valued semantics, and Vector belongs to the (future) HNSW
2505/// path. Index lookups on those columns fall back to full scan.
2506#[derive(Debug, Clone, PartialEq, Eq, PartialOrd, Ord)]
2507pub enum IndexKey {
2508    Int(i64),
2509    Text(String),
2510    Bool(bool),
2511    /// v7.17.0 — `Value::Uuid` index key. Comparison is byte-wise
2512    /// (RFC 4122 byte order) so PRIMARY KEY UUID lookups land on
2513    /// the same fast-path as Int / Text.
2514    Uuid([u8; 16]),
2515    /// r1039 — `Value::Bytes` (bytea). PG orders bytea by plain byte
2516    /// comparison, shorter-prefix first (`'' < \x00 < \x0000 < \x01ff <
2517    /// \xff`, measured on 18.4), which is exactly `Vec<u8>`'s `Ord`.
2518    Bytes(Vec<u8>),
2519    /// r1039 — exact decimal, in the canonical form described on
2520    /// [`NumericKey`].
2521    ///
2522    /// r1040 — BOXED, and the box is load-bearing for every OTHER index.
2523    /// A `NumericKey` is 48 bytes against `Text(String)`'s 24, so inline
2524    /// it set the size of the whole enum and every B-tree node in every
2525    /// index grew with it: 32 bytes per key to 48, align 8 to 16.
2526    /// Measured through the release sweep, `SELECT pad FROM t ORDER BY
2527    /// id` over 400,000 rows — a walk of the primary key's index — went
2528    /// 39.4-40.6 ms to 42.3-44.1, in both leg orders. The indirection is
2529    /// charged to numeric keys, which are new, instead of to every index
2530    /// that existed already.
2531    Numeric(alloc::boxed::Box<NumericKey>),
2532    /// v7.38.1 (L12) — a NULL component INSIDE a composite key, and
2533    /// nothing else. `IndexKey::from_value(Value::Null)` still returns
2534    /// `None`, so single-column B-trees never hold one, and no probe
2535    /// path ever BUILDS one (`col = NULL` is not a match in SQL) — the
2536    /// variant is only reachable through a composite key's component
2537    /// list, where it exists so that a row like `(2, 3, NULL)` stays
2538    /// findable by a PREFIX probe on `(w, d)`. Declared last: slice
2539    /// `Ord` then sorts NULL components after every value, PG's
2540    /// NULLS LAST.
2541    Null,
2542}
2543
2544/// r1039 — an exact-decimal index key, canonical so that representation
2545/// equality IS value equality.
2546///
2547/// That property is the whole reason this is a struct rather than the
2548/// `(scaled, scale)` pair the value carries. `1.5` and `1.50` are the
2549/// same NUMERIC (PG18.4: `1.5::numeric = 1.50::numeric` is true) and
2550/// arrive here as `(15, 1)` and `(150, 2)`. A B-tree keyed on the raw
2551/// pair would file them apart, so `WHERE n = 1.5` would miss a row stored
2552/// as `1.50` — an index changing the answer, which is the one thing an
2553/// index may never do. `BigNumeric::cmp` carries the same warning and
2554/// declines to implement `Ord` for exactly this reason; a KEY cannot
2555/// decline, so it normalizes instead.
2556///
2557/// Canonical form: significant decimal digits with no leading and no
2558/// trailing zeros, most significant first, plus the decimal exponent of
2559/// the leading digit. Zero is the empty digit vector with `neg == false`
2560/// and `exp == 0`, so there is no `-0`.
2561///
2562/// Ordering is PG's, measured: `-Infinity < -1 < 0 < 1 < Infinity < NaN`,
2563/// and `NaN = NaN`.
2564#[derive(Debug, Clone, PartialEq, Eq)]
2565pub struct NumericKey {
2566    /// 0 = -Infinity, 1 = finite, 2 = +Infinity, 3 = NaN. Ordering the
2567    /// classes by this byte is what puts NaN on top, where PG keeps it.
2568    class: u8,
2569    /// Finite only, and never set for zero.
2570    neg: bool,
2571    /// Decimal exponent of the leading significant digit; 0 for zero.
2572    exp: i32,
2573    /// r1040 — the first [`HEAD_DIGITS`] significant digits, LEFT-ALIGNED
2574    /// (multiplied up so the leading digit always sits at 10^36). That
2575    /// alignment is what makes an integer comparison of two heads the same
2576    /// answer as a digit-by-digit one: `12` and `1` become 1.2e36 and
2577    /// 1.0e36, which order the way the digit strings do, where the bare
2578    /// integers 12 and 1 would not.
2579    ///
2580    /// Zero for the value zero and for every special.
2581    ///
2582    /// This started as a `Vec<u8>` of digits, which is correct and cost
2583    /// an allocation per key and a slice comparison per sort comparison.
2584    /// `ORDER BY <numeric>` builds one key per row and compares n log n
2585    /// times: 200,000 rows measured 65.4 ms against 39.6 for the f64
2586    /// projection that had been returning rows in the wrong order.
2587    head: u128,
2588    /// Significant digits past the 37th, one per byte, no trailing zeros.
2589    /// Empty for everything an `i128` mantissa can hold with room to
2590    /// spare — and an empty `Vec` does not allocate, which is the point.
2591    tail: Vec<u8>,
2592}
2593
2594/// Significant digits carried in [`NumericKey::head`]. 37 is the most
2595/// that can be left-aligned inside a `u128`: the largest such value is
2596/// 9.99…e36, and `u128::MAX` is 3.4e38.
2597const HEAD_DIGITS: u32 = 37;
2598/// `10^36` — where a left-aligned leading digit sits.
2599const HEAD_SCALE: u128 = 1_000_000_000_000_000_000_000_000_000_000_000_000;
2600
2601/// The `class` byte of [`NumericKey`], in PG's order.
2602const NUM_CLASS_NEG_INF: u8 = 0;
2603const NUM_CLASS_FINITE: u8 = 1;
2604const NUM_CLASS_POS_INF: u8 = 2;
2605const NUM_CLASS_NAN: u8 = 3;
2606
2607impl NumericKey {
2608    /// The key for a `Value::Numeric`'s three fields.
2609    ///
2610    /// Public because the ORDER BY key wants the same canonical form the
2611    /// index key uses: two sort keys that disagree about which of two
2612    /// NUMERICs is larger is the same class of defect as an index that
2613    /// disagrees with a scan, and one definition is how they stay honest.
2614    #[must_use]
2615    pub fn from_numeric(scaled: i128, scale: u16, kind: NumericKind) -> Self {
2616        match kind {
2617            NumericKind::Finite => {
2618                let mut buf = [0u8; 40];
2619                let n = digits_of_u128(scaled.unsigned_abs(), &mut buf);
2620                Self::finite(scaled < 0, &buf[..n], i32::from(scale))
2621            }
2622            NumericKind::NaN => Self::special(NUM_CLASS_NAN),
2623            NumericKind::PosInf => Self::special(NUM_CLASS_POS_INF),
2624            NumericKind::NegInf => Self::special(NUM_CLASS_NEG_INF),
2625        }
2626    }
2627
2628    /// The key for an exact integer — no scale, so no rounding.
2629    #[must_use]
2630    pub fn from_i128(n: i128) -> Self {
2631        let mut buf = [0u8; 40];
2632        let len = digits_of_u128(n.unsigned_abs(), &mut buf);
2633        Self::finite(n < 0, &buf[..len], 0)
2634    }
2635
2636    /// The key for a mantissa that overflowed `i128`. The two
2637    /// representations of one value land on one key.
2638    #[must_use]
2639    pub fn from_big(b: &crate::bignum::BigNumeric) -> Self {
2640        let (neg, limbs, scale) = b.parts();
2641        Self::finite(neg, &digits_of_limbs(limbs), i32::from(scale))
2642    }
2643
2644    /// The `f64` this key means, for the one comparison PG defines that
2645    /// way: `numeric` against `float8` demotes the numeric.
2646    ///
2647    /// Lossy by construction — that is the point, and it is why nothing
2648    /// else uses it.
2649    #[must_use]
2650    #[allow(clippy::cast_precision_loss)]
2651    pub fn to_f64(&self) -> f64 {
2652        match self.class {
2653            NUM_CLASS_NAN => return f64::NAN,
2654            NUM_CLASS_POS_INF => return f64::INFINITY,
2655            NUM_CLASS_NEG_INF => return f64::NEG_INFINITY,
2656            _ => {}
2657        }
2658        if self.head == 0 {
2659            return 0.0;
2660        }
2661        // `head` is `d.ddd… × 10^36`; the value is that leading digit and
2662        // its followers at `exp`. The tail is below f64's resolution by
2663        // construction (it starts at the 38th significant digit).
2664        let mantissa = self.head as f64 / HEAD_SCALE as f64;
2665        let out = mantissa * pow10_f64(self.exp);
2666        if self.neg { -out } else { out }
2667    }
2668
2669    /// The significant decimal digits, most significant first — the form
2670    /// the catalog codec writes, and the one `from_parts` reads back.
2671    #[must_use]
2672    pub fn digits(&self) -> Vec<u8> {
2673        let mut out = Vec::new();
2674        if self.head != 0 {
2675            let mut h = self.head;
2676            for _ in 0..HEAD_DIGITS {
2677                let d = u8::try_from(h / HEAD_SCALE).unwrap_or(0);
2678                out.push(d);
2679                h = (h % HEAD_SCALE) * 10;
2680            }
2681            while out.last() == Some(&0) {
2682                out.pop();
2683            }
2684        }
2685        out.extend_from_slice(&self.tail);
2686        out
2687    }
2688
2689    /// The wire parts, for the catalog codec.
2690    #[must_use]
2691    pub fn parts(&self) -> (u8, bool, i32) {
2692        (self.class, self.neg, self.exp)
2693    }
2694
2695    /// Rebuild from the wire parts. Returns `None` on parts that are not
2696    /// canonical, so a corrupt catalog cannot smuggle in a key whose `Eq`
2697    /// and `Ord` disagree.
2698    #[must_use]
2699    pub fn from_parts(class: u8, neg: bool, exp: i32, digits: &[u8]) -> Option<Self> {
2700        if class > NUM_CLASS_NAN || digits.iter().any(|d| *d > 9) {
2701            return None;
2702        }
2703        if class != NUM_CLASS_FINITE && (neg || exp != 0 || !digits.is_empty()) {
2704            return None;
2705        }
2706        if digits.is_empty() {
2707            if neg || exp != 0 {
2708                return None;
2709            }
2710            return Some(Self::special(class));
2711        }
2712        if digits[0] == 0 || digits[digits.len() - 1] == 0 {
2713            return None;
2714        }
2715        Some(Self {
2716            class,
2717            neg,
2718            exp,
2719            head: head_of(digits),
2720            tail: digits.iter().skip(HEAD_DIGITS as usize).copied().collect(),
2721        })
2722    }
2723
2724    /// Canonicalize `(-1)^neg · <digits as an integer> · 10^-scale`.
2725    ///
2726    /// `digits` is most-significant-first and may carry leading and
2727    /// trailing zeros; both are stripped, which is what makes `1.5` and
2728    /// `1.50` land on the same key.
2729    fn finite(neg: bool, digits: &[u8], scale: i32) -> Self {
2730        let lead = digits.iter().position(|d| *d != 0).unwrap_or(digits.len());
2731        let digits = &digits[lead..];
2732        if digits.is_empty() {
2733            return Self::special(NUM_CLASS_FINITE);
2734        }
2735        // The leading digit's exponent, taken BEFORE trailing zeros go:
2736        // dropping low-order digits does not move the leading one.
2737        let exp = i32::try_from(digits.len()).unwrap_or(i32::MAX) - 1 - scale;
2738        let mut end = digits.len();
2739        while end > 0 && digits[end - 1] == 0 {
2740            end -= 1;
2741        }
2742        let digits = &digits[..end];
2743        Self {
2744            class: NUM_CLASS_FINITE,
2745            neg,
2746            exp,
2747            head: head_of(digits),
2748            tail: digits.iter().skip(HEAD_DIGITS as usize).copied().collect(),
2749        }
2750    }
2751
2752    fn special(class: u8) -> Self {
2753        Self {
2754            class,
2755            neg: false,
2756            exp: 0,
2757            head: 0,
2758            tail: Vec::new(),
2759        }
2760    }
2761}
2762
2763/// The first [`HEAD_DIGITS`] of `digits`, left-aligned so the leading one
2764/// sits at `10^36`.
2765fn head_of(digits: &[u8]) -> u128 {
2766    let mut head: u128 = 0;
2767    let take = (HEAD_DIGITS as usize).min(digits.len());
2768    for d in &digits[..take] {
2769        head = head * 10 + u128::from(*d);
2770    }
2771    for _ in take..HEAD_DIGITS as usize {
2772        head *= 10;
2773    }
2774    head
2775}
2776
2777/// Decimal digits of `mag` into `buf`, most significant first; returns how
2778/// many were written. Zero writes none.
2779///
2780/// r1040 — split at `u64` on purpose. A `u128` divide is a called routine,
2781/// not an instruction, and this loop runs once per digit per key.
2782fn digits_of_u128(mag: u128, buf: &mut [u8; 40]) -> usize {
2783    if mag == 0 {
2784        return 0;
2785    }
2786    let mut rev = [0u8; 40];
2787    let mut n = 0usize;
2788    let mut big = mag;
2789    // Peel nineteen digits at a time — the most a `u64` holds — so the
2790    // wide divide runs at most twice.
2791    while big > u128::from(u64::MAX) {
2792        let mut chunk = u64::try_from(big % 10_000_000_000_000_000_000_u128).unwrap_or(0);
2793        big /= 10_000_000_000_000_000_000_u128;
2794        for _ in 0..19 {
2795            rev[n] = u8::try_from(chunk % 10).unwrap_or(0);
2796            chunk /= 10;
2797            n += 1;
2798        }
2799    }
2800    let mut small = u64::try_from(big).unwrap_or(0);
2801    while small > 0 {
2802        rev[n] = u8::try_from(small % 10).unwrap_or(0);
2803        small /= 10;
2804        n += 1;
2805    }
2806    for i in 0..n {
2807        buf[i] = rev[n - 1 - i];
2808    }
2809    n
2810}
2811
2812/// Decimal digits of a base-10^9 little-endian limb vector, most
2813/// significant first. Every limb but the leading one is padded to its
2814/// full nine digits — that padding is the whole point, since a limb of 5
2815/// in the middle of a number means `000000005`.
2816fn digits_of_limbs(limbs: &[u32]) -> Vec<u8> {
2817    let mut out = Vec::new();
2818    let mut buf = [0u8; 40];
2819    for (i, limb) in limbs.iter().enumerate().rev() {
2820        let n = digits_of_u128(u128::from(*limb), &mut buf);
2821        if i + 1 == limbs.len() {
2822            out.extend_from_slice(&buf[..n]);
2823        } else {
2824            out.extend(core::iter::repeat_n(0u8, 9 - n));
2825            out.extend_from_slice(&buf[..n]);
2826        }
2827    }
2828    out
2829}
2830
2831/// `10^e` as an `f64`, for any `e` a canonical key can carry.
2832#[allow(clippy::cast_precision_loss)]
2833fn pow10_f64(e: i32) -> f64 {
2834    let mut out = 1.0_f64;
2835    let mag = e.unsigned_abs();
2836    for _ in 0..mag {
2837        out *= 10.0;
2838    }
2839    if e < 0 { 1.0 / out } else { out }
2840}
2841
2842impl Ord for NumericKey {
2843    fn cmp(&self, other: &Self) -> core::cmp::Ordering {
2844        use core::cmp::Ordering;
2845        if self.class != other.class {
2846            return self.class.cmp(&other.class);
2847        }
2848        if self.class != NUM_CLASS_FINITE {
2849            // Each of the three specials is a single value, and PG holds
2850            // `'NaN'::numeric = 'NaN'::numeric` true.
2851            return Ordering::Equal;
2852        }
2853        // Zero first: it is stored with `neg == false` and `exp == 0`, so
2854        // the magnitude comparison below would put it above every value
2855        // smaller than 1 rather than between the negatives and positives.
2856        match (self.head == 0, other.head == 0) {
2857            (true, true) => return Ordering::Equal,
2858            (true, false) => {
2859                return if other.neg {
2860                    Ordering::Greater
2861                } else {
2862                    Ordering::Less
2863                };
2864            }
2865            (false, true) => {
2866                return if self.neg {
2867                    Ordering::Less
2868                } else {
2869                    Ordering::Greater
2870                };
2871            }
2872            (false, false) => {}
2873        }
2874        match (self.neg, other.neg) {
2875            (false, true) => return Ordering::Greater,
2876            (true, false) => return Ordering::Less,
2877            _ => {}
2878        }
2879        // Same sign, both non-zero: more integer digits is bigger, and at
2880        // equal exponent the left-aligned heads compare as one integer —
2881        // the alignment is what makes that the same answer as comparing
2882        // the digit strings. The tail only speaks when the first 37
2883        // significant digits are identical.
2884        let mag = self
2885            .exp
2886            .cmp(&other.exp)
2887            .then_with(|| self.head.cmp(&other.head))
2888            .then_with(|| self.tail.cmp(&other.tail));
2889        if self.neg { mag.reverse() } else { mag }
2890    }
2891}
2892
2893impl PartialOrd for NumericKey {
2894    fn partial_cmp(&self, other: &Self) -> Option<core::cmp::Ordering> {
2895        Some(self.cmp(other))
2896    }
2897}
2898
2899impl IndexKey {
2900    /// v7.37.43 (INSUBQ B-4) — inline-friendly BigInt fast path.
2901    /// `try_count_star_pk_in_subquery_fast` (and any other hot loop
2902    /// probing an integer PK) already holds an `i64`; this builds the
2903    /// `IndexKey` without going through the generic `from_value`
2904    /// dispatch tree.
2905    #[inline]
2906    pub fn from_i64(n: i64) -> Self {
2907        Self::Int(n)
2908    }
2909
2910    /// r1039 — the key a value takes when the INDEXED COLUMN is `ty`, or
2911    /// `None` when it takes none (→ the caller falls back to a scan).
2912    ///
2913    /// Every key under one index comes from one column, so they all live
2914    /// in one key SPACE. A probe built in a different space finds nothing
2915    /// — and "nothing" is indistinguishable from "no matching rows",
2916    /// which is how round 564 and r1037 both turned an index into a wrong
2917    /// answer (a TEXT key sought against a DATE-keyed and a UUID-keyed
2918    /// index).
2919    ///
2920    /// The two spaces this round adds make that trap reachable again from
2921    /// a new direction: `WHERE n = 2` on a NUMERIC column produces
2922    /// `Value::Int`, and an integer key would look in a space nothing
2923    /// lives in. So NUMERIC columns take integers by converting them
2924    /// exactly, and refuse anything they cannot convert; BYTEA columns
2925    /// take only `Value::Bytes`; and no other column may be keyed in
2926    /// either of the two new spaces.
2927    ///
2928    /// Use this wherever the key comes from a LITERAL or from another
2929    /// table's value. [`IndexKey::from_value`] stays right for building
2930    /// the index itself, where the value is the column's own.
2931    pub fn from_value_for_column(v: &Value<'_>, ty: DataType) -> Option<Self> {
2932        match ty {
2933            DataType::Numeric { .. } => match v {
2934                Value::SmallInt(n) => Some(Self::exact_int_key(i128::from(*n))),
2935                Value::Int(n) => Some(Self::exact_int_key(i128::from(*n))),
2936                Value::BigInt(n) => Some(Self::exact_int_key(i128::from(*n))),
2937                Value::Numeric { .. } | Value::NumericBig(_) => Self::from_value(v),
2938                // Float included: `2.0::float8` and `2.0::numeric` are not
2939                // the same value to a B-tree, and rounding one into the
2940                // other's space is how a seek reaches the wrong row.
2941                _ => None,
2942            },
2943            DataType::Bytes => match v {
2944                Value::Bytes(b) => Some(Self::Bytes(b.to_vec())),
2945                _ => None,
2946            },
2947            _ => match Self::from_value(v) {
2948                Some(Self::Numeric(_) | Self::Bytes(_)) => None,
2949                other => other,
2950            },
2951        }
2952    }
2953
2954    /// An integer as a NUMERIC key. Exact by construction — no scale, no
2955    /// rounding — which is why the conversion is allowed at all.
2956    fn exact_int_key(n: i128) -> Self {
2957        Self::Numeric(alloc::boxed::Box::new(NumericKey::from_i128(n)))
2958    }
2959
2960    pub fn from_value(v: &Value<'_>) -> Option<Self> {
2961        match v {
2962            // v7.37.43 (INSUBQ B-4) — BigInt hits first (the dominant
2963            // INSUBQ shape probes PK as BigInt). Tiny micro-win.
2964            Value::BigInt(n) => Some(Self::Int(*n)),
2965            Value::SmallInt(n) => Some(Self::Int(i64::from(*n))),
2966            Value::Int(n) => Some(Self::Int(i64::from(*n))),
2967            Value::Text(s) => Some(Self::Text(s.clone().into_owned())),
2968            // v7.38 (read01, T11) — bpchar keys compare blank-insensitively.
2969            Value::BpChar(s) => Some(Self::Text(s.trim_end_matches(' ').to_string())),
2970            Value::Bool(b) => Some(Self::Bool(*b)),
2971            // Date/Timestamp use their integer storage repr as the
2972            // index key — same order semantics, same comparison.
2973            Value::Date(d) => Some(Self::Int(i64::from(*d))),
2974            Value::Timestamp(t) => Some(Self::Int(*t)),
2975            // v7.17.0: UUID indexable via byte-wise ordering. Lookup
2976            // on `id = '...'::uuid` resolves through the secondary
2977            // index rather than full-scan.
2978            Value::Uuid(b) => Some(Self::Uuid(*b)),
2979            // v7.17.0 Phase 3.P0-32: TIME indexable via i64 — same
2980            // order semantics as Date/Timestamp.
2981            Value::Time(us) => Some(Self::Int(*us)),
2982            // v7.17.0 Phase 3.P0-33: YEAR indexable as i64 — u16
2983            // widens losslessly and gives the natural calendar
2984            // ordering.
2985            Value::Year(y) => Some(Self::Int(i64::from(*y))),
2986            // v7.17.0 Phase 3.P0-34: TIMETZ indexable by its
2987            // UTC-equivalent microseconds (local wall - offset).
2988            // Without normalising, two values for the same
2989            // physical instant in different zones would sort
2990            // wrong. Matches PG's TIMETZ index behaviour.
2991            Value::TimeTz { us, offset_secs } => {
2992                Some(Self::Int(us - i64::from(*offset_secs) * 1_000_000))
2993            }
2994            // v7.17.0 Phase 3.P0-35: MONEY indexable as i64 cents
2995            // (no scaling needed — natural numeric ordering).
2996            Value::Money(c) => Some(Self::Int(*c)),
2997            // v7.17.0 Phase 3.P0-38: ranges are NOT indexable in
2998            // v7.17.0 — they'd need a custom comparator (PG uses
2999            // SP-GiST for this). Skip.
3000            Value::Range { .. } => None,
3001            // v7.17.0 Phase 3.P0-39: hstore is NOT indexable in
3002            // v7.17.0 — map columns need GIN with bespoke ops.
3003            Value::Hstore(_) => None,
3004            // r1039 — exact decimals index through the canonical
3005            // [`NumericKey`], which is what makes `1.5` and `1.50` one key.
3006            Value::NumericBig(b) => Some(Self::Numeric(alloc::boxed::Box::new(NumericKey::from_big(b)))),
3007            Value::Numeric {
3008                scaled,
3009                scale,
3010                kind,
3011            } => Some(Self::Numeric(alloc::boxed::Box::new(
3012                NumericKey::from_numeric(*scaled, *scale, *kind),
3013            ))),
3014            // r1039 — bytea orders by plain byte comparison, which is
3015            // `Vec<u8>`'s own.
3016            Value::Bytes(b) => Some(Self::Bytes(b.to_vec())),
3017            // v7.17.0 Phase 3.P0-40: 2D arrays aren't indexable.
3018            Value::IntArray2D(_)
3019            | Value::BigIntArray2D(_)
3020            | Value::TextArray2D(_)
3021            | Value::BoolArray2D(_) => None,
3022            // v7.37.5 β-P4: INTERVAL[] isn't indexable (PG uses
3023            // GIN/intarray for array-contains queries; SPG plans
3024            // that as a separate axis under v7.37.8 GIN-on-jsonb).
3025            Value::IntervalArray(_) => None,
3026            // v7.37.5 γ — none of the array-of-scalar family is
3027            // B-tree indexable. Same reason as IntervalArray: PG
3028            // serves array-contains / array-overlap queries via
3029            // GIN, and SPG's GIN axis lands in v7.37.8.
3030            Value::BoolArray(_)
3031            | Value::SmallIntArray(_)
3032            | Value::FloatArray(_)
3033            | Value::NumericArray(_)
3034            | Value::DateArray(_)
3035            | Value::TimestampArray(_)
3036            | Value::TimestamptzArray(_)
3037            | Value::UuidArray(_)
3038            | Value::JsonArray(_)
3039            | Value::JsonbArray(_)
3040            | Value::BytesArray(_)
3041            | Value::VarcharArray(_)
3042            | Value::CharArray(_)
3043            // v7.37.5 δ — multirange not indexable (PG uses GiST/
3044            // SP-GiST + a custom operator class; SPG plans the same
3045            // axis under v7.37.8 with ranges).
3046            | Value::Multirange { .. }
3047            // v7.37.5 ε — geometric scalars not B-tree indexable
3048            // (PG uses GiST/SP-GiST for these too; SPG plans the
3049            // same axis under v7.37.8).
3050            | Value::Point(_)
3051            | Value::Lseg(_, _)
3052            | Value::Path { .. }
3053            | Value::PgBox(_, _)
3054            | Value::Polygon(_)
3055            | Value::Line { .. }
3056            | Value::Circle { .. }
3057            // v7.37.5 ζ-A — network / bit / xml / "char" / money[].
3058            // INET / CIDR / MACADDR / MACADDR8 could be B-tree
3059            // indexable (PG does this), but the byte-wise compare
3060            // family-blind would mis-order IPv4 vs IPv6; left as
3061            // a follow-up under v7.37.8 GIN window.
3062            | Value::Inet { .. }
3063            | Value::Cidr { .. }
3064            | Value::Macaddr(_)
3065            | Value::Macaddr8(_)
3066            | Value::PgLsn(_)
3067            | Value::BitString { .. }
3068            | Value::Xml(_)
3069            | Value::Char1(_)
3070            | Value::MoneyArray(_)
3071            | Value::Composite(_)
3072            | Value::Tid(..)
3073            | Value::Xid(_)
3074            | Value::Cid(_)
3075            | Value::RegClass(..)
3076            | Value::RegProc(..)
3077            | Value::RegType(..) => None,
3078            // Interval isn't index-eligible (and can't reach this path
3079            // through column storage anyway). Float / Real stay out
3080            // because `f64` is only `PartialOrd`.
3081            Value::Null
3082            | Value::Float(_)
3083            | Value::Vector(_)
3084            | Value::Sq8Vector(_)
3085            | Value::HalfVector(_)
3086            | Value::Interval { .. }
3087            | Value::Json(_)
3088            | Value::TextArray(_)
3089            | Value::IntArray(_)
3090            | Value::BigIntArray(_)
3091            | Value::TsVector(_)
3092            | Value::TsQuery(_)
3093            | Value::Real(_) => None,
3094        }
3095    }
3096}
3097
3098/// A single-column secondary index. v2.0 carries either a B-tree map
3099/// (the default — used for equality / range lookups on scalar columns)
3100/// or a navigable-small-world graph (used for kNN over vector
3101/// columns).
3102#[derive(Debug, Clone)]
3103pub struct Index {
3104    pub name: String,
3105    pub column_position: usize,
3106    pub kind: IndexKind,
3107    /// v6.8.0 — column positions of `INCLUDE (col1, col2, …)`
3108    /// non-key columns. Carries the planner's "this query is
3109    /// covered by the index" signal; lookup paths still resolve
3110    /// via the `RowLocator` to fetch the row body, but EXPLAIN
3111    /// surfaces the covered-scan annotation so operators can
3112    /// confirm the planner sees the coverage.
3113    ///
3114    /// Empty `Vec` = no `INCLUDE` clause (the legacy shape). v12
3115    /// catalog snapshots deserialise with an empty vec.
3116    pub included_columns: Vec<usize>,
3117    /// v6.8.1 — partial-index predicate stored as its canonical
3118    /// Display form (the engine re-parses it on the maintenance
3119    /// path). `None` = unconditional index (the legacy shape).
3120    /// Persisted as `[u8 has_pred][u16 LE len][bytes]` on the
3121    /// catalog snapshot (FILE_VERSION 12, appended after
3122    /// `included_columns`).
3123    pub partial_predicate: Option<String>,
3124    /// v6.8.2 — expression-index key, stored as the expression's
3125    /// canonical Display form. `None` = bare column-reference
3126    /// index (the legacy shape). Persisted alongside
3127    /// `partial_predicate` on the v12 catalog snapshot.
3128    pub expression: Option<String>,
3129    /// v7.39 (read01 round 52) — `CREATE UNIQUE INDEX … NULLS NOT DISTINCT`
3130    /// (PG 15+): a NULL in the key no longer exempts the row, so two
3131    /// all-NULL keys collide. Default `false` = SQL-standard NULLS DISTINCT.
3132    /// Persisted in the index appendix (FILE_VERSION 62+); older catalogs
3133    /// deserialise with `false`.
3134    pub nulls_not_distinct: bool,
3135    /// v7.39 (round 537) — the key column's ordering clause, as written.
3136    ///
3137    /// SPG's index does not scan in a direction, so this changes no
3138    /// lookup; `pg_indexes.indexdef` is a reproduction of the DDL and
3139    /// dropping the clause made `CREATE INDEX i ON t (a DESC NULLS
3140    /// LAST)` read back as `(a)` — a dump lost it and a schema diff saw
3141    /// drift every run. `nulls_first` is `None` when the statement did
3142    /// not say, in which case PG's default applies and neither word is
3143    /// rendered.
3144    pub descending: bool,
3145    pub nulls_first: Option<bool>,
3146    /// v7.39 (round 538) — an explicit `COLLATE` on the key, as written.
3147    /// SPG orders text by bytes, so it changes no comparison; PG prints
3148    /// it because a named collation and an inherited one are different
3149    /// objects even where they sort identically.
3150    pub collation: Option<String>,
3151    /// v7.9.29 — `CREATE UNIQUE INDEX …`. When true the engine
3152    /// rejects INSERTs whose key already appears in this index
3153    /// (combined with `partial_predicate` when present — only
3154    /// rows matching the predicate enter the uniqueness check).
3155    /// Catalog FILE_VERSION 16+; older snapshots deserialise
3156    /// with `false`. mailrs K1.
3157    pub is_unique: bool,
3158    /// v7.9.29 — extra (non-leading) column positions for
3159    /// multi-column indexes (`CREATE INDEX … (a, b, c)`). The
3160    /// planner today still only uses the leading
3161    /// `column_position` for index seeks, but UNIQUE INDEX
3162    /// enforcement walks the full tuple so partial-unique
3163    /// invariants like CalDAV `(calendar_id, uid,
3164    /// recurrence_id)` are enforced correctly. Catalog
3165    /// FILE_VERSION 16+; older snapshots deserialise empty.
3166    pub extra_column_positions: Vec<usize>,
3167}
3168
3169/// Default neighbor degree (M) for the NSW graph. Picked at construction
3170/// time and persisted with the index.
3171pub const NSW_DEFAULT_M: usize = 16;
3172
3173/// v5.2.2: outcome of a successful [`Catalog::freeze_oldest_to_cold`]
3174/// call. The catalog state has already been mutated by the time this
3175/// is returned (hot rows dropped + segment registered + Cold locators
3176/// flipped). The caller's only remaining concern is `segment_bytes` —
3177/// persist them to disk under `<db>.spg/segments/seg_<id>.spg` so a
3178/// future restart can reload via the v5.1 `SPG_PRELOAD_COLD_SEGMENT`
3179/// path. (v5.3's manifest will subsume this manual step.)
3180#[derive(Debug, Clone)]
3181pub struct FreezeReport {
3182    /// Id allocated by [`Catalog::load_segment_bytes`] for the new
3183    /// cold-tier segment. Stable across the call's success path.
3184    pub segment_id: u32,
3185    /// Number of rows that moved hot → cold. Equals the `max_rows`
3186    /// the caller asked for (the API is strict on the count).
3187    pub frozen_rows: usize,
3188    /// Hot-tier bytes reclaimed by the freeze — the
3189    /// [`Table::hot_bytes`] delta before vs after. Useful to feed
3190    /// back into the freezer's budget check on the next tick.
3191    pub bytes_freed: u64,
3192    /// Encoded segment bytes, byte-identical to what
3193    /// [`encode_segment`] produced. The catalog already owns a
3194    /// copy inside `cold_segments`; this hand-off lets the caller
3195    /// persist them without re-encoding.
3196    pub segment_bytes: Vec<u8>,
3197}
3198
3199/// v6.7.4 — read-only output of [`Catalog::prepare_freeze_slice`].
3200/// Carries every row body + key in a contiguous hot-row range,
3201/// already encoded and sorted by PK so the coordinator's merge
3202/// step is a k-way merge over already-sorted streams.
3203///
3204/// `Vec<FreezeSlice>` from N independent workers feeds
3205/// [`Catalog::commit_freeze_slices`], which concats + encodes the
3206/// merged segment + atomically swaps the catalog state.
3207#[derive(Debug, Clone)]
3208pub struct FreezeSlice {
3209    /// Hot-row index range this slice covered (half-open, in the
3210    /// table's `rows: PersistentVec` ordering at call time). The
3211    /// commit step uses this to compute the union range that
3212    /// gets passed to [`Table::delete_rows`].
3213    pub row_range: core::ops::Range<usize>,
3214    /// `(pk_u64, encoded_row_body, IndexKey)` triples, sorted
3215    /// ascending by `pk_u64`. Per-slice sort happens inside
3216    /// `prepare_freeze_slice`; the coordinator does only a
3217    /// k-way merge to reach the global PK ordering
3218    /// [`encode_segment`] requires.
3219    pub rows: Vec<(u64, Vec<u8>, IndexKey)>,
3220}
3221
3222/// v6.7.3 — outcome of a [`Catalog::compact_cold_segments`] call.
3223/// The catalog state has already been mutated when this is returned:
3224/// the merged segment is loaded into `cold_segments`, the source
3225/// segment slots are tombstoned (`None`), and every BTree-index
3226/// `RowLocator::Cold` that previously pointed at a source now
3227/// points at the merged segment. The caller's remaining job is to
3228/// persist `merged_segment_bytes` under
3229/// `<db>.spg/segments/seg_<merged_segment_id>.spg` and update the
3230/// in-memory `segment_id → path` map (remove the source ids, add
3231/// the merged id) so the next CHECKPOINT writes a manifest that
3232/// no longer lists the retired sources.
3233///
3234/// On a no-op (fewer than 2 candidate segments under the threshold),
3235/// `merged_segment_id` is `None` and `sources` is empty; the
3236/// catalog was not mutated.
3237#[derive(Debug, Clone)]
3238pub struct CompactReport {
3239    /// Source segment ids that were merged + tombstoned.
3240    pub sources: Vec<u32>,
3241    /// Id allocated for the merged segment. `None` on no-op.
3242    pub merged_segment_id: Option<u32>,
3243    /// Encoded merged-segment bytes (empty on no-op).
3244    pub merged_segment_bytes: Vec<u8>,
3245    /// Number of rows that landed in the merged segment.
3246    pub merged_rows: usize,
3247    /// `Σ source.num_rows − merged_rows`. Rows present in source
3248    /// segment payloads but unreferenced by any live BTree
3249    /// `Cold` locator — DELETE'd-but-still-frozen rows that
3250    /// compaction GC'd during the merge.
3251    pub deleted_rows_pruned: usize,
3252    /// `Σ source.bytes() − merged.bytes()`. Estimate of on-disk
3253    /// space the merge will reclaim once the source segment files
3254    /// are GC'd. Saturating subtract — never negative.
3255    pub bytes_reclaimed_estimate: u64,
3256}
3257
3258#[derive(Debug, Clone)]
3259pub enum IndexKind {
3260    /// v4.40: structural-sharing B-tree over `IndexKey`. Replaces the v0.8
3261    /// `BTreeMap<IndexKey, Vec<usize>>` — `Index::clone` is now an `Arc`
3262    /// bump regardless of index size, so `Catalog::clone` inside the
3263    /// v4.34 auto-commit wrap stays O(1) even for tables with secondary
3264    /// indices (the case that bottlenecked v4.39 at 1M rows in the
3265    /// sweep).
3266    ///
3267    /// v5.1: value type widened from `Vec<usize>` to `Vec<RowLocator>` so
3268    /// a single key can point to a mix of hot-tier rows (`RowLocator::Hot`,
3269    /// equivalent to the pre-v5 `usize` row index) and cold-tier rows
3270    /// (`RowLocator::Cold { segment_id, page_offset }`) once the v5.2
3271    /// freezer starts producing them. Pre-v5.2 only `Hot` entries appear
3272    /// — the on-disk encoding stays at `FILE_VERSION` 8 (raw u64 row index)
3273    /// because every locator round-trips through `RowLocator::from_legacy_v8_u64`
3274    /// without information loss. `FILE_VERSION` 9 with tagged encoding lands
3275    /// alongside the first freezer commit (v5.1 step 2b / v5.2).
3276    BTree(PersistentBTreeMap<IndexKey, crate::posting::PostingList>),
3277    /// Navigable-small-world graph for vector kNN search.
3278    Nsw(NswGraph),
3279    /// v6.7.1 — BRIN (Block Range INdex). Pure metadata: BRIN
3280    /// indexes carry NO in-memory key→locator map. The (min,
3281    /// max) summaries live in each cold-tier segment's v2
3282    /// envelope sidecar; the BRIN entry in `Table.indices` only
3283    /// records THAT a BRIN index exists on this column so the
3284    /// segment encoder + planner can opt into the summary path.
3285    Brin {
3286        /// The cell type at `column_position` at CREATE INDEX time.
3287        /// Used by the planner to type-check WHERE-clause range
3288        /// predicates against the BRIN-indexed column.
3289        column_type: DataType,
3290        /// v7.38.11 — one `(min, max)` per [`BRIN_RANGE_ROWS`] slots of
3291        /// the hot tier, so a range predicate can skip the ranges that
3292        /// cannot contain a match.
3293        ///
3294        /// Maintenance is WIDEN-ONLY and that is the whole safety
3295        /// argument: an insert widens its range, an update widens, and
3296        /// a delete leaves the range alone. A range left wider than the
3297        /// rows it now covers is correct and merely less selective —
3298        /// which is exactly PG's contract for a lossy index, since the
3299        /// predicate is re-checked on every row the summary lets
3300        /// through. A summary may over-report; it can never
3301        /// under-report, so no matching row can be skipped.
3302        ///
3303        /// `None` for a range whose rows carry no comparable key (all
3304        /// NULL, say), and such a range is never skipped.
3305        summaries: alloc::vec::Vec<Option<(i64, i64)>>,
3306    },
3307    /// v7.12.3 — GIN inverted index over a `tsvector` column.
3308    ///
3309    /// Storage shape: `lexeme word → Vec<RowLocator>`. The posting
3310    /// list per word is appended in row-order, so range scans are
3311    /// O(matching rows) once the per-word lookup is done. Multi-
3312    /// term queries intersect / union posting lists.
3313    ///
3314    /// `IndexKey::from_value(TsVector)` returns `None` — GIN doesn't
3315    /// participate in `try_index_seek` (which is BTree-equality-keyed).
3316    /// The engine consults this index through `try_gin_lookup` on
3317    /// `WHERE col @@ tsquery` predicates instead.
3318    ///
3319    /// Backed by a `PersistentBTreeMap` so `Catalog::clone` (the
3320    /// per-write snapshot) stays O(1) — same structural-sharing
3321    /// invariant as BTree.
3322    Gin(PersistentBTreeMap<alloc::string::String, crate::posting::PostingList>),
3323    /// v7.15.0 — `USING gin (col gin_trgm_ops)` over a `TEXT`
3324    /// column. Posting lists map `trigram` (PG-compatible 3-byte
3325    /// shingle on the lower-cased + space-padded input) to row
3326    /// locators. The planner uses this index to accelerate
3327    /// `WHERE col LIKE '…'` / `ILIKE '…'` / `similarity(col, q) >
3328    /// t` — every literal run of length ≥ 1 in the pattern
3329    /// produces a trigram set, the engine intersects the posting
3330    /// lists, and the LIKE / similarity predicate is re-evaluated
3331    /// per candidate row to filter the over-approximation.
3332    /// Persisted via tag-4 index payload in `FILE_VERSION` 24+.
3333    GinTrgm(PersistentBTreeMap<alloc::string::String, crate::posting::PostingList>),
3334    /// v7.17.0 Phase 2.2 — MySQL `FULLTEXT KEY (col)` over a
3335    /// `TEXT` / `VARCHAR` column. Posting lists map
3336    /// `tsvector('simple') lexeme` to row locators. At insert /
3337    /// build time the engine derives the lexemes from the cell
3338    /// via the same lower-case tokenisation rule as
3339    /// `to_tsvector('simple', ...)` — the column itself stays a
3340    /// plain text type on disk (mysqldump round-trips would be
3341    /// broken otherwise). The planner uses this index to
3342    /// accelerate MySQL-shape `MATCH(col) AGAINST('term')`
3343    /// queries by mapping them onto the existing tsquery `@@`
3344    /// walker. Persisted via tag-5 index payload in
3345    /// `FILE_VERSION` 33+.
3346    GinFulltext(PersistentBTreeMap<alloc::string::String, crate::posting::PostingList>),
3347    /// v7.37.8(sentori Epic 5 P2)— `USING gin (col)` over a
3348    /// `JSON` / `JSONB` column. Posting lists map a canonical
3349    /// `(path, leaf)` token(see [`crate::jsonb_gin::extract_tokens`])
3350    /// to row locators so the planner can resolve
3351    /// `<col> @> <jsonb_literal>` to a candidate row set via
3352    /// posting-list intersection + per-row `json::contains`
3353    /// re-verification. Pre-7.37.8 the same DDL loaded as a
3354    /// BTree fallback so `pg_dump` JSONB-GIN scripts kept loading
3355    /// without query-time acceleration. Persisted via tag-6 index
3356    /// payload in `FILE_VERSION` 51+.
3357    GinJsonb(PersistentBTreeMap<alloc::string::String, crate::posting::PostingList>),
3358    /// v7.38.1 (L12) — a REAL multi-column B-tree: the key is the whole
3359    /// column tuple, `[leading, extras…]`, ordered lexicographically by
3360    /// slice `Ord`. That ordering is the entire design: every key
3361    /// sharing a prefix is contiguous, so an equality on a PREFIX of
3362    /// the columns is one `O(log N)` descent plus a bounded walk, and a
3363    /// full-tuple equality is a point `get`. The single-column `BTree`
3364    /// kind used to stand in for multi-column DDL by keying on the
3365    /// leading column only and carrying the rest as metadata — TPC-C's
3366    /// `customer (c_w_id, c_d_id, c_last, c_first)` then answered a
3367    /// three-column equality with every row of one warehouse and a
3368    /// per-row filter over 30 000 candidates.
3369    ///
3370    /// Rows where any component column is NULL (or of an unkeyable
3371    /// type) are NOT entered: this index serves `=` probes, and in SQL
3372    /// `col = v` never selects a NULL. Uniqueness keeps its own
3373    /// full-tuple walk with NULLS-DISTINCT semantics on the
3374    /// enforcement path, exactly as before.
3375    ///
3376    /// Persisted via tag-7 index payload in `FILE_VERSION` 91+.
3377    BTreeMulti(PersistentBTreeMap<alloc::boxed::Box<[IndexKey]>, crate::posting::PostingList>),
3378}
3379
3380impl IndexKind {
3381    /// v7.31 (memory campaign, C2) — bytes this index variant holds
3382    /// resident in RAM, computed by walking its OWN structure rather
3383    /// than a parametric guess made by the engine. Replaces the old
3384    /// `spg_admin::memory_stats` inline match, which charged NSW with
3385    /// a stale `m_max_0 * 8` per node (neighbour slots are `u32` = 4 B
3386    /// since v6.1.x, and most nodes never fill `m_max_0`) and lumped
3387    /// every GIN family index into a flat 1 KiB token — a gross
3388    /// undercount for the text-heavy posting lists that dominate
3389    /// mailrs' footprint. Per-entry container overhead uses the
3390    /// 3-word (24 B on 64-bit) `Vec`/`String` header as the charge.
3391    ///
3392    /// O(index entries): operator/monitoring surface (`memory_stats` /
3393    /// `spg_memory_stats`), not a query path.
3394    #[must_use]
3395    pub fn approx_resident_bytes(&self) -> u64 {
3396        const HEADER: usize = 24; // Vec/String 3-word header on 64-bit.
3397        let loc = core::mem::size_of::<RowLocator>();
3398        match self {
3399            IndexKind::BTree(map) => {
3400                let key = core::mem::size_of::<IndexKey>();
3401                map.iter()
3402                    .map(|(_, locs)| (key + HEADER + locs.len() * loc) as u64)
3403                    .sum()
3404            }
3405            // v7.38.1 (L12) — multi keys own a boxed slice of components.
3406            IndexKind::BTreeMulti(map) => {
3407                let key = core::mem::size_of::<IndexKey>();
3408                map.iter()
3409                    .map(|(k, locs)| (HEADER + k.len() * key + HEADER + locs.len() * loc) as u64)
3410                    .sum()
3411            }
3412            IndexKind::Nsw(g) => {
3413                // `levels` is one byte per node; each layer's adjacency
3414                // is a `Vec<u32>` per node whose actual length we walk
3415                // (the dense layer-0 list dominates, but upper layers
3416                // are sparse — the old estimate ignored that).
3417                let mut b = g.levels.len() as u64;
3418                for layer in &g.layers {
3419                    for nbrs in layer.iter() {
3420                        b += (HEADER + nbrs.len() * core::mem::size_of::<u32>()) as u64;
3421                    }
3422                }
3423                b
3424            }
3425            // BRIN carries NO in-memory key→locator map (the (min,max)
3426            // summaries live in cold-segment sidecars on disk); the
3427            // resident footprint is just the column-type token.
3428            IndexKind::Brin { .. } => core::mem::size_of::<DataType>() as u64,
3429            IndexKind::Gin(map)
3430            | IndexKind::GinTrgm(map)
3431            | IndexKind::GinFulltext(map)
3432            | IndexKind::GinJsonb(map) => map
3433                .iter()
3434                .map(|(word, postings)| {
3435                    (word.len() + HEADER + HEADER + postings.len() * loc) as u64
3436                })
3437                .sum(),
3438        }
3439    }
3440}
3441
3442/// Multi-layer HNSW graph (v2.13). Each node is assigned a `top_level`;
3443/// it appears in layers `0..=top_level`. Higher layers are sparser, so
3444/// search starts from the entry at the top layer, greedy-descends to
3445/// layer 0, and beam-searches there. Layer 0 keeps a larger neighbour
3446/// budget (`m_max_0 = 2 * m` per the HNSW paper); upper layers cap at
3447/// `m`. The struct name stays `NswGraph` so external users / on-disk
3448/// callers don't have to track a rename — the algorithm changed, the
3449/// data slot didn't.
3450#[derive(Debug, Clone)]
3451pub struct NswGraph {
3452    /// Max neighbours per node on layers ≥ 1.
3453    pub m: usize,
3454    /// Max neighbours on layer 0 (the dense bottom layer). HNSW
3455    /// convention: `m_max_0 = 2 * m`.
3456    pub m_max_0: usize,
3457    /// Entry point — the node that sits on the topmost layer. Search
3458    /// always starts here.
3459    pub entry: Option<usize>,
3460    /// Top layer of the entry node (== `layers.len() - 1` when populated).
3461    pub entry_level: u8,
3462    /// `levels[i]` = top layer of node `i`. Nodes whose vector cell is
3463    /// NULL / non-Vector have `levels[i] = 0` and no neighbour entries.
3464    ///
3465    /// v5.5.0: backed by `PersistentVec` so `NswGraph::clone` (and the
3466    /// `Catalog::clone` on every group-commit write that contains it) is O(1)
3467    /// structural-sharing instead of an O(N) element copy.
3468    pub levels: PersistentVec<u8>,
3469    /// `layers[l][i]` = neighbours of node `i` at layer `l`. Inner vec
3470    /// is empty when node `i` doesn't reach layer `l`.
3471    ///
3472    /// v5.5.0: the per-node middle dimension (the O(N) one) is a
3473    /// `PersistentVec`; the outer layer dimension stays a plain `Vec`
3474    /// (layer count ≤ 8, so its clone is O(1) in practice) and the inner
3475    /// neighbour list stays a `Vec` (bounded by `m_max_0`).
3476    ///
3477    /// v6.1.x: neighbour slot widened from `usize` (8 B on 64-bit) to
3478    /// `u32` (4 B). Row indices are catalog-bounded by `u32::MAX` (4G
3479    /// rows per table); the cast at the NSW boundary asserts this. At
3480    /// 1M dim-128 SQ8, layer 0 adjacency alone shrinks by ~128 MiB
3481    /// — the largest single contribution to the v6.0.5-measured
3482    /// 624 MiB ambition gap. On-disk format already used u32 LE, so
3483    /// this is a pure in-memory layout change; no `FILE_VERSION` bump.
3484    pub layers: Vec<PersistentVec<Vec<u32>>>,
3485}
3486
3487impl NswGraph {
3488    fn new(m: usize) -> Self {
3489        Self {
3490            m,
3491            m_max_0: m.saturating_mul(2),
3492            entry: None,
3493            entry_level: 0,
3494            levels: PersistentVec::new(),
3495            layers: alloc::vec![PersistentVec::new()],
3496        }
3497    }
3498
3499    /// Max-neighbour budget for layer `l`.
3500    pub const fn cap_for_layer(&self, layer: u8) -> usize {
3501        if layer == 0 { self.m_max_0 } else { self.m }
3502    }
3503}
3504
3505/// Deterministic level assignment, seeded on the row index so the same
3506/// insert order reproduces the same topology. Distribution is roughly
3507/// HNSW-flavoured with `mL ≈ 1/ln(M) ≈ 0.36` for M=16: each 4-bit
3508/// chunk that comes up zero promotes the node one layer (so P(level ≥
3509/// L) ≈ (1/16)^L).
3510#[allow(clippy::verbose_bit_mask)] // clippy suggests trailing_zeros(); we need an explicit MAX cap and a stable distribution shape.
3511pub fn nsw_assign_level(row_idx: usize) -> u8 {
3512    const MAX_LEVEL: u8 = 7; // 7 ⇒ ~16^7 ≈ 2.7e8 expected nodes between promotions; ample.
3513    // SplitMix-style mixer — cheap and seedable.
3514    let mut x = (row_idx as u64).wrapping_mul(0x9E37_79B9_7F4A_7C15);
3515    x ^= x >> 30;
3516    x = x.wrapping_mul(0xBF58_476D_1CE4_E5B9);
3517    x ^= x >> 27;
3518    x = x.wrapping_mul(0x94D0_49BB_1331_11EB);
3519    x ^= x >> 31;
3520    // Count contiguous low-end zero nibbles (4-bit chunks). Each zero
3521    // nibble has probability 1/16, mirroring HNSW's `mL ≈ 1/ln(M)` for
3522    // M=16. `trailing_zeros / 4` would lose the ordering when x = 0, so
3523    // a plain loop with a cap is clearer.
3524    let mut level: u8 = 0;
3525    while x & 0xF == 0 && level < MAX_LEVEL {
3526        level += 1;
3527        x >>= 4;
3528    }
3529    level
3530}
3531
3532/// v7.38.1 (L12) — the composite key `values` takes in a multi-column
3533/// B-tree over `[lead, extras…]`. A NULL component keys as
3534/// [`IndexKey::Null`] (declared to sort last, PG's NULLS LAST) so the
3535/// row stays findable by prefix probes on the columns before it. `None`
3536/// = some non-null component has no key form; the row is then not
3537/// entered, which is why creation gates every component column's type
3538/// through [`multi_component_type_ok`].
3539pub(crate) fn compose_multi_key(
3540    values: &[Value<'_>],
3541    lead: usize,
3542    extras: &[usize],
3543) -> Option<alloc::boxed::Box<[IndexKey]>> {
3544    let mut comps: Vec<IndexKey> = Vec::with_capacity(1 + extras.len());
3545    for pos in core::iter::once(lead).chain(extras.iter().copied()) {
3546        let v = values.get(pos)?;
3547        if matches!(v, Value::Null) {
3548            comps.push(IndexKey::Null);
3549        } else {
3550            comps.push(IndexKey::from_value(v)?);
3551        }
3552    }
3553    Some(comps.into_boxed_slice())
3554}
3555
3556/// v7.38.1 (L12) — component-type gate for multi-column B-trees: every
3557/// NON-NULL value of these types keys through `IndexKey::from_value`,
3558/// so a row can only be absent from the index when creation raced a
3559/// type this list does not name. Deliberately conservative — a type
3560/// outside the list simply keeps its index on the leading-column path.
3561pub(crate) fn multi_component_type_ok(ty: DataType) -> bool {
3562    matches!(
3563        ty,
3564        DataType::SmallInt
3565            | DataType::Int
3566            | DataType::BigInt
3567            | DataType::Text
3568            | DataType::Varchar(_)
3569            | DataType::Char(_)
3570            | DataType::Bool
3571            | DataType::Uuid
3572            | DataType::Date
3573            | DataType::Timestamp
3574    )
3575}
3576
3577impl Index {
3578    /// Any key this B-tree currently holds, or `None` if it holds none.
3579    ///
3580    /// A probe built from a query literal has to be the same SHAPE as the
3581    /// keys the maintenance side made, or `lookup_eq` misses every row and
3582    /// the caller reads the empty answer as "no rows match". One stored
3583    /// key settles it: an index keys one expression, whose values are one
3584    /// type.
3585    pub fn sample_key(&self) -> Option<&IndexKey> {
3586        match &self.kind {
3587            IndexKind::BTree(map) => map.iter().next().map(|(k, _)| k),
3588            _ => None,
3589        }
3590    }
3591
3592    /// v7.38.19 — the largest integer key this index holds.
3593    ///
3594    /// For the one question it answers — what number comes next for a
3595    /// `serial` column — a tree already knows, and knew all along.
3596    /// [`Table::next_auto_value`] read every row instead:
3597    ///
3598    /// ```text
3599    ///   rows in the table    one INSERT      PostgreSQL 18
3600    ///      1,000              1.831 ms          1.245
3601    ///     10,000              1.814             1.289
3602    ///     50,000              2.703             1.386
3603    ///    200,000              3.666             1.375
3604    /// ```
3605    ///
3606    /// Theirs is flat because a sequence is a counter. Ours grew with
3607    /// the table, so an ingest workload got slower the longer it ran.
3608    ///
3609    /// A dead row version's key is still in the tree, so this can be
3610    /// HIGHER than the maximum over live rows. That is the safe
3611    /// direction — it hands out a value no row has ever held — and it
3612    /// is the direction PostgreSQL goes too, which never reuses a
3613    /// number a deleted row was given.
3614    ///
3615    /// `None` = no B-tree, or its keys are not integers, and the caller
3616    /// falls back to the scan.
3617    pub fn max_int_key(&self) -> Option<i64> {
3618        let IndexKind::BTree(map) = &self.kind else {
3619            return None;
3620        };
3621        match map.iter_rev().next()? {
3622            (IndexKey::Int(n), _) => Some(*n),
3623            _ => None,
3624        }
3625    }
3626
3627    fn new_btree(name: String, column_position: usize) -> Self {
3628        Self {
3629            name,
3630            column_position,
3631            kind: IndexKind::BTree(PersistentBTreeMap::new()),
3632            included_columns: Vec::new(),
3633            partial_predicate: None,
3634            expression: None,
3635            is_unique: false,
3636            nulls_not_distinct: false,
3637            descending: false,
3638            nulls_first: None,
3639            collation: None,
3640            extra_column_positions: Vec::new(),
3641        }
3642    }
3643
3644    /// v7.38.1 (L12) — a real multi-column B-tree shell. The caller
3645    /// sets `extra_column_positions` before the first row enters; the
3646    /// key arity is `1 + extras` from then on.
3647    fn new_btree_multi(name: String, column_position: usize) -> Self {
3648        Self {
3649            kind: IndexKind::BTreeMulti(PersistentBTreeMap::new()),
3650            ..Self::new_btree(name, column_position)
3651        }
3652    }
3653
3654    /// v7.38.1 (L12) — the composite key this row takes in a
3655    /// [`IndexKind::BTreeMulti`] index. NULL components key as
3656    /// [`IndexKey::Null`] so prefix probes still find the row; `None`
3657    /// only when a non-null component produces no key, which creation's
3658    /// component-type gate makes unreachable for well-formed indexes.
3659    pub fn multi_key_for_row(&self, values: &[Value<'_>]) -> Option<alloc::boxed::Box<[IndexKey]>> {
3660        compose_multi_key(values, self.column_position, &self.extra_column_positions)
3661    }
3662
3663    fn new_nsw(name: String, column_position: usize, m: usize) -> Self {
3664        Self {
3665            name,
3666            column_position,
3667            kind: IndexKind::Nsw(NswGraph::new(m)),
3668            included_columns: Vec::new(),
3669            partial_predicate: None,
3670            expression: None,
3671            is_unique: false,
3672            nulls_not_distinct: false,
3673            descending: false,
3674            nulls_first: None,
3675            collation: None,
3676            extra_column_positions: Vec::new(),
3677        }
3678    }
3679
3680    /// v6.7.1 — BRIN index constructor. BRIN carries no in-memory
3681    /// data; the `column_type` snapshot is used by the segment
3682    /// encoder + planner for type-checking range predicates.
3683    fn new_brin(name: String, column_position: usize, column_type: DataType) -> Self {
3684        Self {
3685            name,
3686            column_position,
3687            kind: IndexKind::Brin {
3688                column_type,
3689                summaries: alloc::vec::Vec::new(),
3690            },
3691            included_columns: Vec::new(),
3692            partial_predicate: None,
3693            expression: None,
3694            is_unique: false,
3695            nulls_not_distinct: false,
3696            descending: false,
3697            nulls_first: None,
3698            collation: None,
3699            extra_column_positions: Vec::new(),
3700        }
3701    }
3702
3703    /// v7.12.3 — GIN inverted-index constructor. Empty posting-list
3704    /// map; caller (typically [`Table::add_gin_index`] or
3705    /// [`Table::restore_gin_index`]) populates it from existing rows
3706    /// or from a deserialised snapshot.
3707    fn new_gin(name: String, column_position: usize) -> Self {
3708        Self {
3709            name,
3710            column_position,
3711            kind: IndexKind::Gin(PersistentBTreeMap::new()),
3712            included_columns: Vec::new(),
3713            partial_predicate: None,
3714            expression: None,
3715            is_unique: false,
3716            nulls_not_distinct: false,
3717            descending: false,
3718            nulls_first: None,
3719            collation: None,
3720            extra_column_positions: Vec::new(),
3721        }
3722    }
3723
3724    /// v7.15.0 — `gin_trgm_ops`-flavoured GIN constructor. Same
3725    /// shape as `new_gin` but the posting-list keys are 3-byte
3726    /// trigram shingles (`pg_trgm`-compatible) and the column
3727    /// type is `TEXT` / `VARCHAR` (not `TSVECTOR`).
3728    fn new_gin_trgm(name: String, column_position: usize) -> Self {
3729        Self {
3730            name,
3731            column_position,
3732            kind: IndexKind::GinTrgm(PersistentBTreeMap::new()),
3733            included_columns: Vec::new(),
3734            partial_predicate: None,
3735            expression: None,
3736            is_unique: false,
3737            nulls_not_distinct: false,
3738            descending: false,
3739            nulls_first: None,
3740            collation: None,
3741            extra_column_positions: Vec::new(),
3742        }
3743    }
3744
3745    /// v7.17.0 Phase 2.2 — MySQL `FULLTEXT KEY` GIN constructor.
3746    /// Same shape as `new_gin_trgm` but the posting-list keys
3747    /// are lower-cased word lexemes (`to_tsvector('simple', col)`
3748    /// equivalent) instead of trigrams, and the column type is
3749    /// `TEXT` / `VARCHAR` (not `TSVECTOR`).
3750    fn new_gin_fulltext(name: String, column_position: usize) -> Self {
3751        Self {
3752            name,
3753            column_position,
3754            kind: IndexKind::GinFulltext(PersistentBTreeMap::new()),
3755            included_columns: Vec::new(),
3756            partial_predicate: None,
3757            expression: None,
3758            is_unique: false,
3759            nulls_not_distinct: false,
3760            descending: false,
3761            nulls_first: None,
3762            collation: None,
3763            extra_column_positions: Vec::new(),
3764        }
3765    }
3766
3767    /// v7.37.8(sentori Epic 5 P2)— JSONB-GIN constructor. Same
3768    /// shape as the other GIN-family indexes; posting-list keys
3769    /// are the canonical `(path, leaf)` tokens emitted by
3770    /// `crate::jsonb_gin::extract_tokens`. Maintains posting
3771    /// lists from `Value::Json` cells(JSONB is a synonym for the
3772    /// same in-memory string-backed Value).
3773    fn new_gin_jsonb(name: String, column_position: usize) -> Self {
3774        Self {
3775            name,
3776            column_position,
3777            kind: IndexKind::GinJsonb(PersistentBTreeMap::new()),
3778            included_columns: Vec::new(),
3779            partial_predicate: None,
3780            expression: None,
3781            is_unique: false,
3782            nulls_not_distinct: false,
3783            descending: false,
3784            nulls_first: None,
3785            collation: None,
3786            extra_column_positions: Vec::new(),
3787        }
3788    }
3789
3790    /// v7.34.4 — descending-order iterator over `(IndexKey, locators)`
3791    /// pairs for a BTree index, with O(log N) descent to the rightmost
3792    /// leaf and lazy emission thereafter. Returns an empty iterator
3793    /// for non-BTree index kinds — callers handle both uniformly.
3794    /// Used by the ORDER BY `<indexed col>` DESC + LIMIT N executor
3795    /// path: walking only the first N matches off the rightmost leaf
3796    /// avoids the per-row materialisation + partial-sort cost on
3797    /// large tables (mailrs `content_worker` at 250 k rows).
3798    pub fn iter_desc(
3799        &self,
3800    ) -> alloc::boxed::Box<dyn Iterator<Item = (&IndexKey, &crate::posting::PostingList)> + '_>
3801    {
3802        match &self.kind {
3803            IndexKind::BTree(m) => alloc::boxed::Box::new(m.iter_rev()),
3804            // v7.38.1 (L12) — projecting the leading component of a
3805            // composite key preserves order: keys sort by the whole
3806            // tuple, so the leading component is non-increasing here
3807            // (non-decreasing in iter_asc), exactly what an ORDER BY
3808            // on the leading column needs.
3809            IndexKind::BTreeMulti(m) => {
3810                alloc::boxed::Box::new(m.iter_rev().map(|(k, l)| (&k[0], l)))
3811            }
3812            IndexKind::Nsw(_)
3813            | IndexKind::Brin { .. }
3814            | IndexKind::Gin(_)
3815            | IndexKind::GinTrgm(_)
3816            | IndexKind::GinFulltext(_)
3817            | IndexKind::GinJsonb(_) => alloc::boxed::Box::new(core::iter::empty()),
3818        }
3819    }
3820
3821    /// v7.34.4 — ascending-order iterator over `(IndexKey, locators)`
3822    /// pairs. Mirror of `iter_desc` for ORDER BY ... ASC + LIMIT N.
3823    pub fn iter_asc(
3824        &self,
3825    ) -> alloc::boxed::Box<dyn Iterator<Item = (&IndexKey, &crate::posting::PostingList)> + '_>
3826    {
3827        match &self.kind {
3828            IndexKind::BTree(m) => alloc::boxed::Box::new(m.iter()),
3829            // v7.38.1 (L12) — see iter_desc: the leading component of
3830            // a tuple-sorted walk is itself in order.
3831            IndexKind::BTreeMulti(m) => alloc::boxed::Box::new(m.iter().map(|(k, l)| (&k[0], l))),
3832            IndexKind::Nsw(_)
3833            | IndexKind::Brin { .. }
3834            | IndexKind::Gin(_)
3835            | IndexKind::GinTrgm(_)
3836            | IndexKind::GinFulltext(_)
3837            | IndexKind::GinJsonb(_) => alloc::boxed::Box::new(core::iter::empty()),
3838        }
3839    }
3840
3841    /// Look up the locators stored under `key` (B-tree only). Returns
3842    /// an empty slice when the key is absent or the index isn't a
3843    /// BTree — callers can treat both cases uniformly.
3844    ///
3845    /// v5.1: return type widened from `&[usize]` to `&[RowLocator]`.
3846    /// Pre-v5.2 callers can read the slice and `.as_hot().unwrap()`
3847    /// each entry (no `Cold` variants exist until the freezer lands);
3848    /// post-v5.2 callers dispatch hot vs. cold per locator.
3849    pub fn lookup_eq(&self, key: &IndexKey) -> &crate::posting::PostingList {
3850        match &self.kind {
3851            IndexKind::BTree(m) => m.get(key).map_or(&EMPTY_POSTINGS, |l| l),
3852            // BRIN / NSW / GIN / trigram-GIN / fulltext-GIN have
3853            // no IndexKey-keyed map; lookup is a no-op. GIN uses
3854            // [`Index::gin_lookup_word`] instead.
3855            IndexKind::Nsw(_)
3856            | IndexKind::Brin { .. }
3857            | IndexKind::Gin(_)
3858            | IndexKind::GinTrgm(_)
3859            | IndexKind::GinFulltext(_)
3860            | IndexKind::GinJsonb(_)
3861            | IndexKind::BTreeMulti(_) => &EMPTY_POSTINGS,
3862        }
3863    }
3864
3865    /// v7.37.43 (INSUBQ B-2) — specialised lookup for integer-PK probes.
3866    /// `try_count_star_pk_in_subquery_fast` already holds an `i64` (the
3867    /// inner survivor key); skip the `IndexKey::from_value` enum-dispatch
3868    /// trip and build the key inline. ~20 ns × N_survivors saved on
3869    /// the INSUBQ hot loop.
3870    #[inline]
3871    pub fn lookup_eq_i64(&self, n: i64) -> &crate::posting::PostingList {
3872        match &self.kind {
3873            IndexKind::BTree(m) => m.get(&IndexKey::Int(n)).map_or(&EMPTY_POSTINGS, |l| l),
3874            IndexKind::Nsw(_)
3875            | IndexKind::Brin { .. }
3876            | IndexKind::Gin(_)
3877            | IndexKind::GinTrgm(_)
3878            | IndexKind::GinFulltext(_)
3879            | IndexKind::GinJsonb(_)
3880            | IndexKind::BTreeMulti(_) => &EMPTY_POSTINGS,
3881        }
3882    }
3883
3884    /// v7.38 (perf, index range scan) — flatten the row locators for every key
3885    /// in `[lo, hi]` (bounds per `core::ops::Bound`) via the BTree's `O(log N +
3886    /// k)` range walk. Returns `None` once more than `cap` locators accumulate
3887    /// — a "this range isn't selective enough, seq-scan instead" signal that
3888    /// stops a wide range from materialising a near-full table's worth of rows
3889    /// through the index. BTree only (other kinds → None).
3890    pub fn lookup_range_capped(
3891        &self,
3892        lo: core::ops::Bound<&IndexKey>,
3893        hi: core::ops::Bound<&IndexKey>,
3894        cap: usize,
3895    ) -> Option<Vec<RowLocator>> {
3896        self.lookup_range_capped_by(lo, hi, cap, |_| true)
3897    }
3898
3899    /// v7.39 (round 490) — the same range walk, but the caller decides
3900    /// which locators are worth carrying, and the cap counts only those.
3901    ///
3902    /// A BTree index holds one locator per row VERSION. On a churned table
3903    /// the dead versions are still in there: round 490 measured a
3904    /// 1000-row range handing back 61 000 locators after 60
3905    /// delete-and-reinsert cycles with the background vacuum switched off.
3906    /// Every caller then dropped the dead ones — the mutation paths and the
3907    /// SELECT range path all test `is_row_visible` and `continue` — but only
3908    /// after they had been collected into a `Vec`, sorted, and walked.
3909    ///
3910    /// Handing the predicate down means the walk keeps ~1000, and the cap
3911    /// (which exists so an index walk never costs more than the scan it
3912    /// replaces) is once again measured in rows a caller will actually look
3913    /// at. Round 461 had to add the dead count to the budget to stop the
3914    /// seek being refused outright; with the filter here that compensation
3915    /// is no longer needed.
3916    pub fn lookup_range_capped_by(
3917        &self,
3918        lo: core::ops::Bound<&IndexKey>,
3919        hi: core::ops::Bound<&IndexKey>,
3920        cap: usize,
3921        keep: impl Fn(RowLocator) -> bool,
3922    ) -> Option<Vec<RowLocator>> {
3923        match &self.kind {
3924            IndexKind::BTree(m) => {
3925                let mut out: Vec<RowLocator> = Vec::new();
3926                for (_, locs) in m.range(lo, hi) {
3927                    out.extend(locs.iter().copied().filter(|l| keep(*l)));
3928                    if out.len() > cap {
3929                        return None;
3930                    }
3931                }
3932                Some(out)
3933            }
3934            IndexKind::Nsw(_)
3935            | IndexKind::Brin { .. }
3936            | IndexKind::Gin(_)
3937            | IndexKind::GinTrgm(_)
3938            | IndexKind::GinFulltext(_)
3939            | IndexKind::GinJsonb(_)
3940            | IndexKind::BTreeMulti(_) => None,
3941        }
3942    }
3943
3944    /// v7.38.1 (L12) — full-tuple point lookup on a [`IndexKind::BTreeMulti`]
3945    /// index. `key` must carry exactly as many components as the index
3946    /// has columns; anything else (including a probe against a
3947    /// non-multi index) finds nothing, and "nothing" here is safe
3948    /// because the caller falls back to a scan, never to an answer.
3949    pub fn lookup_eq_multi(&self, key: &[IndexKey]) -> &crate::posting::PostingList {
3950        match &self.kind {
3951            IndexKind::BTreeMulti(m) if key.len() == 1 + self.extra_column_positions.len() => {
3952                m.get_by(key).map_or(&EMPTY_POSTINGS, |l| l)
3953            }
3954            _ => &EMPTY_POSTINGS,
3955        }
3956    }
3957
3958    /// v7.38.1 (L12) — locators for every key whose leading components
3959    /// equal `prefix`, on a [`IndexKind::BTreeMulti`] index. Slice
3960    /// ordering keeps a prefix's keys contiguous, so this is one
3961    /// descent to `[prefix]` and a walk that stops at the first key
3962    /// leaving the prefix. Same cap/keep contract as
3963    /// [`Index::lookup_range_capped_by`]: `None` = not selective
3964    /// enough (or not a multi index), fall back.
3965    pub fn lookup_prefix_capped_by(
3966        &self,
3967        prefix: &[IndexKey],
3968        cap: usize,
3969        keep: impl Fn(RowLocator) -> bool,
3970    ) -> Option<Vec<RowLocator>> {
3971        let IndexKind::BTreeMulti(m) = &self.kind else {
3972            return None;
3973        };
3974        if prefix.is_empty() || prefix.len() > 1 + self.extra_column_positions.len() {
3975            return None;
3976        }
3977        let lo: alloc::boxed::Box<[IndexKey]> = prefix.to_vec().into_boxed_slice();
3978        let mut out: Vec<RowLocator> = Vec::new();
3979        for (k, locs) in m.range(core::ops::Bound::Included(&lo), core::ops::Bound::Unbounded) {
3980            if k.len() < prefix.len() || k[..prefix.len()] != *prefix {
3981                break;
3982            }
3983            out.extend(locs.iter().copied().filter(|l| keep(*l)));
3984            if out.len() > cap {
3985                return None;
3986            }
3987        }
3988        Some(out)
3989    }
3990
3991    /// v7.38.19 — a RANGE on the composite tree's leading column.
3992    ///
3993    /// Tuples order lexicographically, so every key whose first
3994    /// component is `x` sorts at or after the one-element tuple `[x]`
3995    /// and before `[x']` for any larger `x'`. That makes a leading-
3996    /// column range one contiguous run, walked exactly like the
3997    /// single-column range walk — the only difference is that the
3998    /// comparison is against `k[0]` rather than the whole key.
3999    ///
4000    /// Without this, `WHERE project_id > 90` on a table whose only
4001    /// index was `(project_id, kind)` read every row: 4.067 ms against
4002    /// PostgreSQL 18's 0.220, on a predicate matching nothing. The same
4003    /// query with a single-column index took 0.165, which is what says
4004    /// the range was never the problem.
4005    pub fn lookup_leading_range_capped_by(
4006        &self,
4007        lo: core::ops::Bound<&IndexKey>,
4008        hi: core::ops::Bound<&IndexKey>,
4009        cap: usize,
4010        keep: impl Fn(RowLocator) -> bool,
4011    ) -> Option<Vec<RowLocator>> {
4012        let IndexKind::BTreeMulti(m) = &self.kind else {
4013            return None;
4014        };
4015        // The start of the run. An EXCLUDED lower bound cannot be
4016        // handed to the map as-is: `[x]` sorts BEFORE `[x, y]`, so
4017        // excluding `[x]` would still admit every tuple that begins
4018        // with `x`. Start at `[x]` included and drop those tuples by
4019        // the per-key test below, which compares the component.
4020        let lo_key: Option<alloc::boxed::Box<[IndexKey]>> = match lo {
4021            core::ops::Bound::Included(k) | core::ops::Bound::Excluded(k) => {
4022                Some(alloc::vec![k.clone()].into_boxed_slice())
4023            }
4024            core::ops::Bound::Unbounded => None,
4025        };
4026        let start = match &lo_key {
4027            Some(k) => core::ops::Bound::Included(k),
4028            None => core::ops::Bound::Unbounded,
4029        };
4030        let mut out: Vec<RowLocator> = Vec::new();
4031        for (k, locs) in m.range(start, core::ops::Bound::Unbounded) {
4032            let Some(first) = k.first() else { continue };
4033            match lo {
4034                core::ops::Bound::Excluded(b) if first == b => continue,
4035                _ => {}
4036            }
4037            match hi {
4038                core::ops::Bound::Included(b) if first > b => break,
4039                core::ops::Bound::Excluded(b) if first >= b => break,
4040                _ => {}
4041            }
4042            out.extend(locs.iter().copied().filter(|l| keep(*l)));
4043            if out.len() > cap {
4044                return None;
4045            }
4046        }
4047        Some(out)
4048    }
4049
4050    /// v7.39 (round 560) — the index range as (key, locator) pairs.
4051    ///
4052    /// `lookup_range_capped_by` throws the KEY away and returns only
4053    /// locators, so a query whose projection is exactly the indexed
4054    /// column still goes to the row store for a value the walk already
4055    /// had in hand — paying per row for something the index knows.
4056    ///
4057    /// Uncapped on purpose: an index-only walk touches no row, so the
4058    /// selectivity ceiling that keeps a seek from being worse than the
4059    /// scan it replaces does not apply to it.
4060    ///
4061    /// v7.39 (round 562) — and it does not collect, either. This
4062    /// returned a `Vec<(IndexKey, RowLocator)>`: for a 100k-row range,
4063    /// 100k key clones into a `Vec::new()` that doubles its way up to
4064    /// several MB, all to be walked once and dropped. A profile of the
4065    /// server serving that query put 20% of the connection thread's CPU
4066    /// on the collect alone, with another 18% in the allocator beside
4067    /// it. The caller consumes the pairs in order and needs the key
4068    /// only by reference, so it can have the walk itself.
4069    pub fn range_keyed(
4070        &self,
4071        lo: core::ops::Bound<&IndexKey>,
4072        hi: core::ops::Bound<&IndexKey>,
4073    ) -> Option<impl Iterator<Item = (&IndexKey, RowLocator)> + '_> {
4074        match &self.kind {
4075            IndexKind::BTree(m) => Some(
4076                m.range(lo, hi)
4077                    .flat_map(|(k, locs)| locs.iter().map(move |l| (k, *l))),
4078            ),
4079            IndexKind::Nsw(_)
4080            | IndexKind::Brin { .. }
4081            | IndexKind::Gin(_)
4082            | IndexKind::GinTrgm(_)
4083            | IndexKind::GinFulltext(_)
4084            | IndexKind::GinJsonb(_)
4085            | IndexKind::BTreeMulti(_) => None,
4086        }
4087    }
4088
4089    /// v7.12.3 — GIN posting-list lookup. Returns the row locators
4090    /// whose `tsvector` cell contains `word`. Empty when the word is
4091    /// absent from the index or this isn't a GIN index.
4092    pub fn gin_lookup_word(&self, word: &str) -> &crate::posting::PostingList {
4093        match &self.kind {
4094            // v7.17.0 Phase 2.2 — fulltext-GIN shares the same
4095            // lexeme-keyed posting list shape as the
4096            // tsvector-typed GIN, so the same lookup applies.
4097            IndexKind::Gin(m) | IndexKind::GinFulltext(m) => {
4098                m.get(&String::from(word)).map_or(&EMPTY_POSTINGS, |l| l)
4099            }
4100            IndexKind::BTree(_)
4101            | IndexKind::Nsw(_)
4102            | IndexKind::Brin { .. }
4103            | IndexKind::GinTrgm(_)
4104            | IndexKind::GinJsonb(_)
4105            | IndexKind::BTreeMulti(_) => &EMPTY_POSTINGS,
4106        }
4107    }
4108
4109    /// v7.15.0 — trigram-GIN posting-list lookup. Returns the row
4110    /// locators whose indexed `TEXT` cell contains the trigram
4111    /// `tri`. Empty when the trigram is absent or this isn't a
4112    /// trigram-GIN index.
4113    pub fn gin_trgm_lookup(&self, tri: &str) -> &crate::posting::PostingList {
4114        match &self.kind {
4115            IndexKind::GinTrgm(m) => m.get(&String::from(tri)).map_or(&EMPTY_POSTINGS, |l| l),
4116            IndexKind::BTree(_)
4117            | IndexKind::Nsw(_)
4118            | IndexKind::Brin { .. }
4119            | IndexKind::Gin(_)
4120            | IndexKind::GinFulltext(_)
4121            | IndexKind::GinJsonb(_)
4122            | IndexKind::BTreeMulti(_) => &EMPTY_POSTINGS,
4123        }
4124    }
4125
4126    /// v7.37.8(sentori Epic 5 P2)— JSONB-GIN posting-list lookup.
4127    /// Returns the row locators whose indexed JSONB cell carries
4128    /// the canonical `token`(see [`crate::jsonb_gin::extract_tokens`]).
4129    /// Empty when the token is absent or this isn't a JSONB-GIN
4130    /// index. Planners drive `<col> @> <jsonb_literal>` through here.
4131    pub fn gin_jsonb_lookup(&self, token: &str) -> &crate::posting::PostingList {
4132        match &self.kind {
4133            IndexKind::GinJsonb(m) => m.get(&String::from(token)).map_or(&EMPTY_POSTINGS, |l| l),
4134            IndexKind::BTree(_)
4135            | IndexKind::Nsw(_)
4136            | IndexKind::Brin { .. }
4137            | IndexKind::Gin(_)
4138            | IndexKind::GinTrgm(_)
4139            | IndexKind::GinFulltext(_)
4140            | IndexKind::BTreeMulti(_) => &EMPTY_POSTINGS,
4141        }
4142    }
4143
4144    /// Borrow the NSW graph (if this is an NSW index). Callers that need
4145    /// the graph for a kNN search go through here.
4146    pub const fn nsw(&self) -> Option<&NswGraph> {
4147        match &self.kind {
4148            IndexKind::Nsw(g) => Some(g),
4149            IndexKind::BTree(_)
4150            | IndexKind::Brin { .. }
4151            | IndexKind::Gin(_)
4152            | IndexKind::GinTrgm(_)
4153            | IndexKind::GinFulltext(_)
4154            | IndexKind::GinJsonb(_)
4155            | IndexKind::BTreeMulti(_) => None,
4156        }
4157    }
4158
4159    /// v6.7.1 — true when this index is a BRIN (block range) index.
4160    /// Used by the segment encoder to opt into BRIN sidecar emission
4161    /// at freeze time, and by the planner to opt into page-skipping
4162    /// on range predicates.
4163    pub const fn is_brin(&self) -> bool {
4164        matches!(self.kind, IndexKind::Brin { .. })
4165    }
4166
4167    /// v7.15.0 — true when this index is a trigram GIN
4168    /// (`gin_trgm_ops`-flavoured). Used by the LIKE planner to
4169    /// opt into trigram acceleration.
4170    pub const fn is_gin_trgm(&self) -> bool {
4171        matches!(self.kind, IndexKind::GinTrgm(_))
4172    }
4173
4174    /// v7.12.3 — true when this index is a GIN inverted index.
4175    /// Used by the planner to opt into posting-list acceleration on
4176    /// `WHERE col @@ tsquery` predicates.
4177    pub const fn is_gin(&self) -> bool {
4178        matches!(self.kind, IndexKind::Gin(_))
4179    }
4180
4181    /// v7.17.0 Phase 2.2 — true when this index is a fulltext
4182    /// GIN over a TEXT / VARCHAR column (MySQL `FULLTEXT KEY`
4183    /// surface). Used by the planner to opt the FULLTEXT-indexed
4184    /// column into MATCH AGAINST acceleration.
4185    pub const fn is_gin_fulltext(&self) -> bool {
4186        matches!(self.kind, IndexKind::GinFulltext(_))
4187    }
4188
4189    /// v7.37.8(sentori Epic 5 P2)— true when this index is a
4190    /// real JSONB-GIN(posting-list backed). Used by the planner
4191    /// to opt `<col> @> <jsonb_literal>` into posting-list seek.
4192    pub const fn is_gin_jsonb(&self) -> bool {
4193        matches!(self.kind, IndexKind::GinJsonb(_))
4194    }
4195}
4196
4197/// In-memory table: schema + a persistent row vector + secondary indices.
4198///
4199/// v4.39: `rows` is a [`PersistentVec`] (Bitmapped Vector Trie, 32-way) so
4200/// `Table::clone()` is `O(1)` — the whole reason for v4.39's existence is
4201/// to make `Catalog::clone()` cheap inside the v4.34 auto-commit wrap.
4202///
4203/// v5.2.1: `hot_bytes` tracks the encoded byte size of every row currently
4204/// in [`Self::rows`], summed over rows. Updated incrementally by `insert`
4205/// (+= encoded row size), `delete_rows` (-= removed rows' encoded sizes),
4206/// and `update_row` (-= old size, += new size). The value is what the
4207/// v5.2 freezer reads to decide when to demote cold rows — when the
4208/// catalog-wide sum crosses `SPG_HOT_TIER_BYTES` (default 4 GiB) the
4209/// freezer thread wakes. v5.2.1 ships measurement only; the freezer
4210/// itself lands in v5.2.2. Stored as `u64` so a single field clone in
4211/// `Catalog::clone` stays at the O(1) invariant v4.39 built.
4212/// v7.34 (crash-recovery P0 #2) — one row-level physical redo record.
4213/// Row-level redo replaces statement-based WAL replay (which re-executes
4214/// each SQL through the full engine — O(records × catalog_rows), the
4215/// superlinear recovery hang root-caused on the mailrs crash-recovery
4216/// P0). A `RowChange` is the exact storage mutation the engine applied
4217/// (`Table::insert` / `update_row` / `delete_rows`); replaying it on a
4218/// catalog restored from the matching checkpoint reproduces the state
4219/// WITHOUT re-validating uniqueness/FK/parse/plan — O(changed rows).
4220///
4221/// Positions are physical, not key-based: `serialize`/`deserialize`
4222/// preserve row order exactly (rows written + read back in `self.rows`
4223/// order) and the mutation ops are deterministic, so the same op sequence
4224/// replayed from the same checkpoint reproduces the same positions. This
4225/// matches PostgreSQL's physical redo and supports tables with no primary
4226/// key. (Caveat handled at replay integration: a post-checkpoint cold-tier
4227/// freeze shifts hot positions and must itself be logged or fenced by a
4228/// checkpoint — see `row-level-redo-design`.)
4229/// ## v7.37.15 (Epic W slice 1) — additive MVCC identity metadata
4230///
4231/// Each variant now also carries, additively, the stable
4232/// [`RowId`](row_header::RowId) of the affected row(s) and the
4233/// **writer version** (`xmin` for an insert, `xmax` for a
4234/// delete/update). This is the codec foundation for making
4235/// in-place MVCC tombstones durable across crash/upgrade recovery.
4236///
4237/// Two important properties for the durability path:
4238///
4239/// 1. **Replay resolution is UNCHANGED.** `apply_redo_run_on_table`
4240///    still resolves every change by physical `pos`/`positions`
4241///    exactly as before. The new metadata is *carried but unused*
4242///    by replay in this slice; resolving-by-`RowId` and
4243///    header-preserving replay are later slices.
4244/// 2. **Backward compatibility.** A redo payload written by
4245///    pre-Epic-W code carries no metadata; [`decode_redo_log`]
4246///    fills `rowid`/`rowids` with [`RowId::UNASSIGNED`](row_header::RowId::UNASSIGNED)
4247///    (empty for `Delete`) and `writer_version` with `0`. See the
4248///    codec version gate in [`encode_redo_log`]/[`decode_redo_log`].
4249///
4250/// The `writer_version` is captured as `0` at the storage layer
4251/// (`Table::insert`/`delete_rows`/`update_row` don't have the
4252/// committing `TxId`), then **stamped with the real committing
4253/// version by the engine** after it drains the statement's changes
4254/// (Epic W slice 2 — [`RowChange::set_writer_version`], driven from
4255/// `Engine::writer_version_for_current_stmt`). All changes from one
4256/// statement share the one version. Replay still resolves by
4257/// physical position and does not read `writer_version` — that is a
4258/// later slice (header-preserving replay).
4259#[derive(Debug, Clone, PartialEq)]
4260pub enum RowChange {
4261    /// Append `row` to `table`.
4262    Insert {
4263        table: String,
4264        row: Row<'static>,
4265        /// Epic W: stable id the appended row will receive.
4266        /// [`RowId::UNASSIGNED`](row_header::RowId::UNASSIGNED) when
4267        /// decoded from a pre-Epic-W redo payload.
4268        rowid: row_header::RowId,
4269        /// Epic W: writer version (`xmin`). `0` until the writing
4270        /// `TxId` is threaded to the storage layer (later slice).
4271        writer_version: u64,
4272    },
4273    /// Replace the row at physical `pos` in `table` with `new_row`.
4274    Update {
4275        table: String,
4276        pos: usize,
4277        new_row: Vec<Value<'static>>,
4278        /// Epic W: stable id of the row at `pos`.
4279        /// [`RowId::UNASSIGNED`](row_header::RowId::UNASSIGNED) when
4280        /// decoded from a pre-Epic-W redo payload.
4281        rowid: row_header::RowId,
4282        /// Epic W: writer version (`xmax` of the superseded tuple).
4283        /// `0` until the writing `TxId` is threaded (later slice).
4284        writer_version: u64,
4285    },
4286    /// Remove the rows at the given physical `positions` from `table`.
4287    Delete {
4288        table: String,
4289        positions: Vec<usize>,
4290        /// Epic W: stable ids parallel to `positions` (same length,
4291        /// [`RowId::UNASSIGNED`](row_header::RowId::UNASSIGNED) for an
4292        /// out-of-bounds input position). **Empty** when decoded from
4293        /// a pre-Epic-W redo payload (no metadata was recorded).
4294        rowids: Vec<row_header::RowId>,
4295        /// Epic W: writer version (`xmax`). `0` until the writing
4296        /// `TxId` is threaded to the storage layer (later slice).
4297        writer_version: u64,
4298    },
4299    /// v7.37.15 (Epic W durable-tombstone slice) — an **in-place MVCC
4300    /// delete**: the row(s) named by `rowids` are NOT physically
4301    /// removed; their header `xmax` is stamped so newer snapshots stop
4302    /// seeing them (vacuum reclaims later). This is the redo shape of
4303    /// the gate-on (`SPG_MVCC_INPLACE`) DELETE / UPDATE-old-version /
4304    /// ON-CONFLICT paths, which call [`Table::mark_row_deleted`]
4305    /// instead of `delete_rows`.
4306    ///
4307    /// Unlike `Delete`, the target is named by **stable `RowId`**, not
4308    /// physical position: a tombstone keeps the slot, so position would
4309    /// be ambiguous after later compaction, and the header-preserving
4310    /// replay must re-find the exact row the writer tombstoned. On
4311    /// replay the id is matched against the ids the same redo run
4312    /// produced (an `Insert`'s `rowid`, or the table's ids snapshotted
4313    /// at run start); an id that cannot be resolved is skipped and
4314    /// counted (see `apply_redo_run_on_table`) — this is the documented
4315    /// cross-checkpoint limitation until the V6 envelope persists ids.
4316    Tombstone {
4317        table: String,
4318        /// Stable ids of the tombstoned rows (from `self.rowids()[pos]`
4319        /// at capture). Never empty for a recorded tombstone.
4320        rowids: Vec<row_header::RowId>,
4321        /// The version stamped into each target row's header `xmax`
4322        /// (the deleting statement's writer version).
4323        xmax: u64,
4324    },
4325}
4326
4327impl RowChange {
4328    /// v7.39 (round 736) — which table this change applies to.
4329    #[must_use]
4330    pub fn table_name(&self) -> &str {
4331        match self {
4332            Self::Insert { table, .. }
4333            | Self::Update { table, .. }
4334            | Self::Delete { table, .. }
4335            | Self::Tombstone { table, .. } => table,
4336        }
4337    }
4338
4339    /// v7.37.15 (Epic W slice 2) — stamp the committing writer
4340    /// version onto this change. Every change drained from a single
4341    /// statement shares one version (the statement's `xmin`/`xmax`),
4342    /// so the engine calls this on each drained change with the value
4343    /// from [`Engine::writer_version_for_current_stmt`]. Additive
4344    /// metadata only: replay still resolves by physical position and
4345    /// does not read `writer_version` (that is a later slice).
4346    pub fn set_writer_version(&mut self, v: u64) {
4347        match self {
4348            RowChange::Insert { writer_version, .. }
4349            | RowChange::Update { writer_version, .. }
4350            | RowChange::Delete { writer_version, .. } => *writer_version = v,
4351            // A tombstone captures `xmax` directly from the deleting
4352            // statement's version at record time (via
4353            // `mark_row_deleted`), so it already equals `v`. Keep the
4354            // "one statement, one version" invariant mechanical by
4355            // asserting agreement in debug builds rather than silently
4356            // overwriting a possibly-different value.
4357            RowChange::Tombstone { xmax, .. } => {
4358                debug_assert_eq!(
4359                    *xmax, v,
4360                    "tombstone xmax must match the statement writer version"
4361                );
4362                *xmax = v;
4363            }
4364        }
4365    }
4366}
4367
4368/// v7.37.15 (Epic W slice 1) — leading marker byte of the
4369/// metadata-carrying redo layout. A **pre-Epic-W** redo payload leads
4370/// with `FILE_VERSION` (8..=52 today, rising ~1 per release); this
4371/// marker is `0xFF` and can therefore never collide with a real
4372/// `FILE_VERSION`, so [`decode_redo_log`] tells the two layouts apart
4373/// by inspecting the first byte alone. The compile-time assertion
4374/// below makes the "never collide" invariant a hard build gate: if
4375/// `FILE_VERSION` ever climbs toward `0xFF` the build breaks and forces
4376/// a redesign long before an ambiguity could ship.
4377const REDO_META_MARKER: u8 = 0xFF;
4378/// v7.37.15 (Epic W slice 1) — version of the metadata-carrying redo
4379/// layout that follows [`REDO_META_MARKER`]. Bumped when the per-change
4380/// metadata shape changes; an unknown value is a hard decode error.
4381const REDO_META_VERSION: u8 = 1;
4382
4383/// v7.37.15 (Epic W durable-tombstone slice) — process-wide count of
4384/// [`RowChange::Tombstone`] targets that `apply_redo` could NOT resolve
4385/// to a row by `RowId`. A non-zero value is expected only across a
4386/// checkpoint boundary (the table's ids are reassigned on deserialize
4387/// and the V6 envelope does not yet persist them), where a tombstone
4388/// naming a pre-checkpoint row is left visible rather than mis-applied.
4389/// Surfaced for observability; never affects correctness of the resolved
4390/// tombstones. Read via [`unresolved_tombstone_count`].
4391static UNRESOLVED_TOMBSTONES: core::sync::atomic::AtomicU64 = core::sync::atomic::AtomicU64::new(0);
4392
4393/// v7.39 (flip crash-replay P0) — observability read for the replay
4394/// tombstones that could not be resolved to a row (each one is a
4395/// resurrected delete).
4396#[must_use]
4397pub fn unresolved_tombstones() -> u64 {
4398    UNRESOLVED_TOMBSTONES.load(core::sync::atomic::Ordering::Relaxed)
4399}
4400
4401/// v7.37.15 (Epic W durable-tombstone slice) — read the process-wide
4402/// count of redo tombstones that could not be resolved to a row by
4403/// `RowId` during `apply_redo`. See [`UNRESOLVED_TOMBSTONES`].
4404#[must_use]
4405pub fn unresolved_tombstone_count() -> u64 {
4406    UNRESOLVED_TOMBSTONES.load(core::sync::atomic::Ordering::Relaxed)
4407}
4408// Provably-unambiguous old/new distinction: the pre-Epic-W layout's
4409// first byte is `FILE_VERSION`, which must stay strictly below the
4410// marker forever.
4411const _: () = assert!(FILE_VERSION < REDO_META_MARKER);
4412
4413/// v7.34 (crash-recovery P0 #2), extended v7.37.15 (Epic W slice 1) —
4414/// encode a row-level redo log to bytes for a WAL record.
4415///
4416/// ## Layout (Epic W metadata-carrying form, always emitted now)
4417///
4418/// `[u8 REDO_META_MARKER=0xFF][u8 REDO_META_VERSION][u8 FILE_VERSION]
4419/// [u32 count]` then per change `[u8 op][str table]` and, per op:
4420/// - `Insert [u32 n][value×n][u64 rowid][u64 writer_version]`
4421/// - `Update [u32 pos][u32 n][value×n][u64 rowid][u64 writer_version]`
4422/// - `Delete [u32 n][u32 pos×n][u64 rowid×n][u64 writer_version]`
4423/// - `Tombstone [u32 n][u64 rowid×n][u64 xmax]` (op byte 3; only ever
4424///   emitted under the metadata-carrying layout — the pre-Epic-W layout
4425///   had no in-place tombstone, so a legacy stream can never carry it)
4426///
4427/// Positions are physical (u32 ≤ 4 G rows). The `FILE_VERSION` byte
4428/// still rides along (now the 3rd byte) so the value codec decodes
4429/// string / BYTEA escapes exactly as before.
4430///
4431/// ## Backward compatibility
4432///
4433/// The **pre-Epic-W** layout was `[u8 FILE_VERSION][u32 count]…` with
4434/// no per-change metadata. [`decode_redo_log`] still decodes that form
4435/// (first byte < `0xFF`) byte-for-byte identically — every WAL file
4436/// written by released code replays unchanged.
4437#[must_use]
4438pub fn encode_redo_log(changes: &[RowChange]) -> Vec<u8> {
4439    let mut out = Vec::new();
4440    out.push(REDO_META_MARKER);
4441    out.push(REDO_META_VERSION);
4442    out.push(FILE_VERSION);
4443    codec::write_u32(&mut out, changes.len() as u32);
4444    let write_values = |out: &mut Vec<u8>, vals: &[Value<'static>]| {
4445        codec::write_u32(out, vals.len() as u32);
4446        for v in vals {
4447            codec::write_value(out, v);
4448        }
4449    };
4450    for change in changes {
4451        match change {
4452            RowChange::Insert {
4453                table,
4454                row,
4455                rowid,
4456                writer_version,
4457            } => {
4458                out.push(0);
4459                codec::write_str(&mut out, table);
4460                write_values(&mut out, &row.values);
4461                codec::write_u64(&mut out, rowid.0);
4462                codec::write_u64(&mut out, *writer_version);
4463            }
4464            RowChange::Update {
4465                table,
4466                pos,
4467                new_row,
4468                rowid,
4469                writer_version,
4470            } => {
4471                out.push(1);
4472                codec::write_str(&mut out, table);
4473                codec::write_u32(&mut out, *pos as u32);
4474                write_values(&mut out, new_row);
4475                codec::write_u64(&mut out, rowid.0);
4476                codec::write_u64(&mut out, *writer_version);
4477            }
4478            RowChange::Delete {
4479                table,
4480                positions,
4481                rowids,
4482                writer_version,
4483            } => {
4484                out.push(2);
4485                codec::write_str(&mut out, table);
4486                codec::write_u32(&mut out, positions.len() as u32);
4487                for p in positions {
4488                    codec::write_u32(&mut out, *p as u32);
4489                }
4490                // Epic W: one RowId per position (parallel). Capture
4491                // sites always produce `rowids.len() == positions.len()`;
4492                // this assertion pins that invariant at encode time so a
4493                // mismatch is a loud bug, not a silently short payload.
4494                debug_assert_eq!(
4495                    rowids.len(),
4496                    positions.len(),
4497                    "redo Delete: rowids must be parallel to positions"
4498                );
4499                for rid in rowids {
4500                    codec::write_u64(&mut out, rid.0);
4501                }
4502                codec::write_u64(&mut out, *writer_version);
4503            }
4504            RowChange::Tombstone {
4505                table,
4506                rowids,
4507                xmax,
4508            } => {
4509                out.push(3);
4510                codec::write_str(&mut out, table);
4511                codec::write_u32(&mut out, rowids.len() as u32);
4512                for rid in rowids {
4513                    codec::write_u64(&mut out, rid.0);
4514                }
4515                codec::write_u64(&mut out, *xmax);
4516            }
4517        }
4518    }
4519    out
4520}
4521
4522/// v7.34, extended v7.37.15 (Epic W slice 1) — decode a row-level redo
4523/// log written by [`encode_redo_log`].
4524///
4525/// Decodes **both** the Epic W metadata-carrying layout (first byte
4526/// `REDO_META_MARKER = 0xFF`) and the pre-Epic-W layout (first byte is
4527/// `FILE_VERSION`, always `< 0xFF`). For the old layout the per-change
4528/// metadata is absent, so `rowid`/`rowids` come back
4529/// [`RowId::UNASSIGNED`](row_header::RowId::UNASSIGNED) (empty for
4530/// `Delete`) and `writer_version` comes back `0`.
4531///
4532/// A truncated / corrupt buffer is a hard error — never a panic — the
4533/// embedding layer frames each record with its own length + CRC, so a
4534/// frame that decodes short is corruption, not a torn tail.
4535pub fn decode_redo_log(bytes: &[u8]) -> Result<Vec<RowChange>, StorageError> {
4536    let first = *bytes
4537        .first()
4538        .ok_or_else(|| StorageError::Corrupt("redo log: empty".into()))?;
4539    // Epic W: `0xFF` marker ⇒ metadata-carrying layout; anything else
4540    // is a pre-Epic-W `FILE_VERSION` byte (old layout, no metadata).
4541    let has_meta = first == REDO_META_MARKER;
4542    let (codec_version, header_len) = if has_meta {
4543        let meta_version = *bytes
4544            .get(1)
4545            .ok_or_else(|| StorageError::Corrupt("redo log: short header".into()))?;
4546        if meta_version != REDO_META_VERSION {
4547            return Err(StorageError::Corrupt(alloc::format!(
4548                "redo log: unknown metadata version {meta_version}"
4549            )));
4550        }
4551        let file_version = *bytes
4552            .get(2)
4553            .ok_or_else(|| StorageError::Corrupt("redo log: short header".into()))?;
4554        // header = [marker][meta_version][file_version]
4555        (file_version, 3usize)
4556    } else {
4557        // Old layout: the first byte IS the FILE_VERSION.
4558        (first, 1usize)
4559    };
4560    let mut cur = codec::Cursor::new(bytes).with_codec_version(codec_version);
4561    for _ in 0..header_len {
4562        cur.read_u8()?;
4563    }
4564    let count = cur.read_u32()? as usize;
4565    let mut read_values =
4566        |cur: &mut codec::Cursor<'_>| -> Result<Vec<Value<'static>>, StorageError> {
4567            let n = cur.read_u32()? as usize;
4568            let mut vals = Vec::with_capacity(n);
4569            for _ in 0..n {
4570                vals.push(cur.read_value()?);
4571            }
4572            Ok(vals)
4573        };
4574    let mut changes = Vec::with_capacity(count);
4575    for _ in 0..count {
4576        let op = cur.read_u8()?;
4577        let table = cur.read_str()?;
4578        let change = match op {
4579            0 => {
4580                let row = Row::new(read_values(&mut cur)?);
4581                let (rowid, writer_version) = if has_meta {
4582                    (row_header::RowId(cur.read_u64()?), cur.read_u64()?)
4583                } else {
4584                    (row_header::RowId::UNASSIGNED, 0)
4585                };
4586                RowChange::Insert {
4587                    table,
4588                    row,
4589                    rowid,
4590                    writer_version,
4591                }
4592            }
4593            1 => {
4594                let pos = cur.read_u32()? as usize;
4595                let new_row = read_values(&mut cur)?;
4596                let (rowid, writer_version) = if has_meta {
4597                    (row_header::RowId(cur.read_u64()?), cur.read_u64()?)
4598                } else {
4599                    (row_header::RowId::UNASSIGNED, 0)
4600                };
4601                RowChange::Update {
4602                    table,
4603                    pos,
4604                    new_row,
4605                    rowid,
4606                    writer_version,
4607                }
4608            }
4609            2 => {
4610                let n = cur.read_u32()? as usize;
4611                let mut positions = Vec::with_capacity(n);
4612                for _ in 0..n {
4613                    positions.push(cur.read_u32()? as usize);
4614                }
4615                let (rowids, writer_version) = if has_meta {
4616                    let mut rowids = Vec::with_capacity(n);
4617                    for _ in 0..n {
4618                        rowids.push(row_header::RowId(cur.read_u64()?));
4619                    }
4620                    (rowids, cur.read_u64()?)
4621                } else {
4622                    // Old layout carried no RowId metadata.
4623                    (Vec::new(), 0)
4624                };
4625                RowChange::Delete {
4626                    table,
4627                    positions,
4628                    rowids,
4629                    writer_version,
4630                }
4631            }
4632            // Op 3 is the Epic W in-place tombstone — it only exists in
4633            // the metadata-carrying layout. Guarding on `has_meta` means
4634            // a legacy stream that happens to contain a `3` byte here is
4635            // reported as an unknown op (corruption), never mis-decoded.
4636            3 if has_meta => {
4637                let n = cur.read_u32()? as usize;
4638                let mut rowids = Vec::with_capacity(n);
4639                for _ in 0..n {
4640                    rowids.push(row_header::RowId(cur.read_u64()?));
4641                }
4642                let xmax = cur.read_u64()?;
4643                RowChange::Tombstone {
4644                    table,
4645                    rowids,
4646                    xmax,
4647                }
4648            }
4649            other => {
4650                return Err(StorageError::Corrupt(alloc::format!(
4651                    "redo log: unknown op {other}"
4652                )));
4653            }
4654        };
4655        changes.push(change);
4656    }
4657    Ok(changes)
4658}
4659
4660/// v7.39 (pg_stat knife B) — per-table scan counters, bumped from
4661/// `&self` read paths. Clone (tx shadow catalogs clone tables) copies
4662/// the current values; the counters are volatile like PG's cumulative
4663/// stats.
4664#[derive(Debug, Default)]
4665pub struct ScanStats {
4666    pub seq_scan: core::sync::atomic::AtomicU64,
4667    pub seq_tup_read: core::sync::atomic::AtomicU64,
4668    pub idx_scan: core::sync::atomic::AtomicU64,
4669    pub idx_tup_fetch: core::sync::atomic::AtomicU64,
4670}
4671
4672impl Clone for ScanStats {
4673    fn clone(&self) -> Self {
4674        use core::sync::atomic::{AtomicU64, Ordering};
4675        Self {
4676            seq_scan: AtomicU64::new(self.seq_scan.load(Ordering::Relaxed)),
4677            seq_tup_read: AtomicU64::new(self.seq_tup_read.load(Ordering::Relaxed)),
4678            idx_scan: AtomicU64::new(self.idx_scan.load(Ordering::Relaxed)),
4679            idx_tup_fetch: AtomicU64::new(self.idx_tup_fetch.load(Ordering::Relaxed)),
4680        }
4681    }
4682}
4683
4684/// v7.39 (round 215) — the lower-bound sort key for a range value, used by
4685/// the range-exclusion index. The bound as an `i128` (unbounded lower =
4686/// `i128::MIN`, sorting first) plus an inclusivity rank (inclusive lower
4687/// sorts before exclusive at the same value, `[3` before `(3`). Returns
4688/// `None` for range kinds whose bound isn't an integer scalar (numrange's
4689/// numeric/bignum), for empty ranges, and for non-range values — the caller
4690/// then keeps the O(n) scan rather than risk an unsound order. Int4/Int8/
4691/// Date/Ts/TsTz all reduce here (tstzrange bounds are `Value::Timestamp`).
4692/// Maintenance (index build) and query (overlap probe) MUST agree on this
4693/// key, so both sides call exactly this function.
4694#[must_use]
4695pub fn range_excl_index_key(v: &Value<'_>) -> Option<(i128, u8)> {
4696    let Value::Range {
4697        lower,
4698        lower_inc,
4699        empty,
4700        ..
4701    } = v
4702    else {
4703        return None;
4704    };
4705    if *empty {
4706        return None;
4707    }
4708    let key = match lower {
4709        None => i128::MIN,
4710        Some(b) => match b.as_ref() {
4711            Value::SmallInt(n) => i128::from(*n),
4712            Value::Int(n) => i128::from(*n),
4713            Value::BigInt(n) => i128::from(*n),
4714            Value::Date(n) => i128::from(*n),
4715            Value::Timestamp(n) => i128::from(*n),
4716            _ => return None,
4717        },
4718    };
4719    Some((key, u8::from(!*lower_inc)))
4720}
4721
4722/// v7.39 (round 215) — a per-table range-exclusion index: an incrementally
4723/// maintained map from a range column's lower-bound key
4724/// ([`range_excl_index_key`]) to the physical row locators carrying that
4725/// bound. Lets EXCLUDE enforcement find the few candidate rows a new range
4726/// might overlap in O(log n) instead of scanning every row (measured O(N²),
4727/// r213). Because the stored ranges under a valid `EXCLUDE (col WITH &&)`
4728/// are pairwise disjoint, a candidate overlaps only its predecessor or the
4729/// successors whose lower bound precedes its upper — a handful of probes.
4730///
4731/// NOT persisted: rebuilt from the (persisted) exclusion constraints + rows
4732/// on catalog load, exactly like BRIN re-derives. Backed by a
4733/// `PersistentBTreeMap` so `Table::clone` (the per-write snapshot) stays
4734/// O(1). Locators to tombstoned rows are left in place and filtered by the
4735/// consumer via `is_deleted()` at query time — the established index pattern.
4736#[derive(Debug, Clone)]
4737pub struct ExclRangeIndex {
4738    /// The constrained range column's position in the table.
4739    pub column_position: usize,
4740    /// Lower-bound key → row locators. A key maps to a `Vec` because a
4741    /// tombstoned-then-reinserted bound can transiently collide; live rows
4742    /// under the constraint are disjoint so each key has one live locator.
4743    pub map: PersistentBTreeMap<(i128, u8), crate::posting::PostingList>,
4744}
4745
4746/// v7.38.2 (R2) — see [`Table::tx_write_track`]. Positions are the
4747/// insert-time slots (verified against the header's version at
4748/// extraction, so a shifted slot falls back to the scan); tombstones
4749/// carry the stable RowId, which is what the write-set wants anyway.
4750#[derive(Debug, Clone, Default)]
4751struct TxWriteTrack {
4752    version: u64,
4753    inserted: Vec<(usize, row_header::RowId)>,
4754    tombstoned: Vec<row_header::RowId>,
4755}
4756
4757/// v7.38.11 — hot-tier BRIN granularity: slots per summarised range.
4758///
4759/// 1024 keeps the summary vector three orders of magnitude smaller
4760/// than the table while staying fine enough that a one-day window over
4761/// a 90-day table skips ~99 % of it. A tuning constant, not a format:
4762/// summaries are rebuilt from the rows on load, so changing it costs
4763/// nothing on disk.
4764pub const BRIN_RANGE_ROWS: usize = 1024;
4765
4766/// The comparable scalar a BRIN summary tracks, or `None` for a value
4767/// with no ordering this index can use.
4768///
4769/// Deliberately narrow: only types whose ordering IS the i64 ordering
4770/// of this number. A type added here whose comparison is not that —
4771/// text under a collation, say — would make the summary under-report
4772/// and skip matching rows, which is the one failure this design must
4773/// not have.
4774#[must_use]
4775pub fn brin_scalar(v: &Value<'_>) -> Option<i64> {
4776    match v {
4777        Value::SmallInt(n) => Some(i64::from(*n)),
4778        Value::Int(n) => Some(i64::from(*n)),
4779        Value::BigInt(n) | Value::Timestamp(n) => Some(*n),
4780        Value::Date(d) => Some(i64::from(*d)),
4781        Value::Bool(b) => Some(i64::from(*b)),
4782        _ => None,
4783    }
4784}
4785
4786#[derive(Debug, Clone)]
4787pub struct Table {
4788    schema: TableSchema,
4789    /// v7.38.18 (S2) — the DATABASE's collation, copied in by the
4790    /// catalog that owns this table.
4791    ///
4792    /// A text column that declares no collation inherits it, which is
4793    /// what PostgreSQL does and what `information_schema.columns`
4794    /// reports as NULL. Runtime only, never serialised: it belongs to
4795    /// the catalog, and a table that has been handed around outside one
4796    /// falls back to `C`, which is the answer for every database written
4797    /// before this existed.
4798    db_collation: Option<String>,
4799    /// v7.38.16 — names of the expression indexes whose B-tree currently
4800    /// holds keys derived from the EXPRESSION.
4801    ///
4802    /// Every catalog written before this version stored, under an
4803    /// expression index, the values of its leading column — keys no
4804    /// lookup could ever match, which is why every read path guarded
4805    /// itself with `expression.is_none()` and the index bought nothing
4806    /// while costing 1.9x a plain insert to maintain.
4807    ///
4808    /// Deliberately NOT persisted: a table read off disk starts with the
4809    /// set empty, so those old wrong keys can never answer a query. The
4810    /// engine, which owns the expression evaluator, refills it.
4811    expr_index_complete: alloc::collections::BTreeSet<String>,
4812    /// v7.37.15 (Phase C.1) — stable per-catalog relation identity.
4813    /// [`RelId::UNASSIGNED`](row_header::RelId::UNASSIGNED) until
4814    /// `Catalog::create_table` (or the deserialize dense-assign pass)
4815    /// stamps a real id. Keys the Phase C.4 row-lock table and the
4816    /// Phase C.5 `RelationStore`; survives `DROP TABLE` slot shifts.
4817    rel_id: row_header::RelId,
4818    rows: PersistentVec<Row<'static>>,
4819    /// v7.37.15 (Phase A.2) — per-row MVCC visibility headers
4820    /// parallel to `rows`. `headers.len() == rows.len()` is the
4821    /// load-bearing invariant; debug builds assert it on every
4822    /// scan boundary, release builds rely on it from
4823    /// disciplined insert / delete / update paths.
4824    ///
4825    /// Pre-v7.37.15-loaded tables (every row currently in the
4826    /// fleet) start as `RowHeader::frozen()` — `is_all_visible_fast()`
4827    /// returns `true`, so the per-row visibility gate Phase B
4828    /// adds is a no-op against any snapshot.
4829    ///
4830    /// Headers are NOT yet serialised into the envelope at this
4831    /// commit — on snapshot deserialize every row gets a fresh
4832    /// `RowHeader::frozen()`. Phase D adds the visibility-map
4833    /// + segment-freeze story which makes serialisation
4834    /// meaningful; until then the on-disk story is "the catalog
4835    /// is the set of visible rows."
4836    headers: PersistentVec<row_header::RowHeader>,
4837    /// v7.37.15 (Phase C.1) — stable per-relation row identity
4838    /// parallel to `rows` / `headers`. `rowids[i]` is the never-
4839    /// reused [`RowId`](row_header::RowId) of the row physically at
4840    /// slot `i`; `rowids.len() == rows.len()` joins the same load-
4841    /// bearing lock-step invariant as `headers`. Compaction (delete
4842    /// / vacuum) rebuilds all three vecs together so the id travels
4843    /// with the row while the slot shifts.
4844    ///
4845    /// Introduced additively: allocated + kept lock-step, but index
4846    /// locators still address rows by physical slot at this commit.
4847    /// Later phases migrate the lock table (C.4), HOT chains (D),
4848    /// and the WAL (Epic W) to address by `RowId`.
4849    ///
4850    /// Not yet serialised into the envelope — on load every row is
4851    /// assigned a fresh dense id `1..=len` (see `next_rowid`), which
4852    /// is sufficient while the id is process-local bookkeeping. The
4853    /// V6 envelope (Phase C.6) will persist ids so a WAL redo can
4854    /// name a row across restart.
4855    rowids: PersistentVec<row_header::RowId>,
4856    /// v7.37.15 (Phase C.1) — per-relation monotonic allocator for
4857    /// `rowids`. Starts at 1 (0 is the `RowId::UNASSIGNED` sentinel);
4858    /// every append takes `next_rowid` then increments. Never reused
4859    /// even after the row is deleted / vacuumed, so a stale lock /
4860    /// redo reference can be detected rather than silently aliasing a
4861    /// later row that reused the slot.
4862    ///
4863    /// 7.38.1 (S2.4, MATRIX #20 root cause) — the allocator is SHARED
4864    /// across every `clone()` of the relation (`Arc`), because the
4865    /// monotonic-never-reused promise is a LINEAGE invariant: each
4866    /// open transaction's shadow catalog is a clone, and when clones
4867    /// carried private counters two concurrent shadows minted the
4868    /// same id — duplicate rids in the base after both committed,
4869    /// aliasing every rid-addressed mechanism (locks, tombstones,
4870    /// redo, the rebase unique pre-check).
4871    next_rowid: alloc::sync::Arc<core::sync::atomic::AtomicU64>,
4872    /// v7.37.16 (autovacuum) — live count of tombstoned-but-present hot
4873    /// rows (`headers[i].xmax != XMAX_ALIVE`). Maintained incrementally:
4874    /// `mark_row_deleted` / `mark_rows_deleted` increment (the only
4875    /// tombstone producers), `delete_rows_no_index` recomputes over the
4876    /// survivors (it is the compaction hub every physical removal —
4877    /// including vacuum — flows through), and the v53 snapshot loader
4878    /// recounts verbatim-restored headers. Drives the engine's
4879    /// autovacuum threshold; not persisted (recomputed on load).
4880    dead_rows: u64,
4881    /// v7.39 (pg_stat knife A) — volatile per-table write counters
4882    /// backing `pg_stat_user_tables.n_tup_ins/upd/del`. Not persisted
4883    /// (PG's cumulative stats are shared-memory-volatile too — a
4884    /// restart zeroes them).
4885    stat_tup_ins: u64,
4886    stat_tup_upd: u64,
4887    stat_tup_del: u64,
4888    /// v7.39 (pg_stat knife B) — volatile scan counters
4889    /// (`seq_scan/seq_tup_read/idx_scan/idx_tup_fetch`). Atomics: the
4890    /// read paths that bump them hold only `&Table`.
4891    scan_stats: ScanStats,
4892    /// v7.39 (pg_stat knife C) — wall-clock stamps (unix µs, from the
4893    /// host ClockFn) for pg_stat_user_tables' last_autovacuum /
4894    /// last_analyze. Volatile, like PG's cumulative stats. SPG has no
4895    /// manual-VACUUM statement semantics, so last_vacuum stays NULL.
4896    last_autovacuum_us: Option<i64>,
4897    last_analyze_us: Option<i64>,
4898    indices: Vec<Index>,
4899    hot_bytes: u64,
4900    /// v6.7.0 — cached count of rows currently materialised in the
4901    /// cold tier via `RowLocator::Cold` entries across THIS table's
4902    /// indices. Populated by `ANALYZE` (walks every BTree index and
4903    /// counts Cold locators); the count survives until the next
4904    /// ANALYZE recomputes it. Surfaced via `spg_statistic.cold_row_count`
4905    /// and `spg_stat_segment.table_name`.
4906    ///
4907    /// Honest scope: this is a CACHED count, not a live one.
4908    /// Freezer / promote / DELETE don't currently update the cache
4909    /// incrementally — they invalidate it by setting the
4910    /// `cold_row_count_stale` flag, and the next ANALYZE re-walks.
4911    /// Incremental maintenance is a v6.7.x candidate if observation
4912    /// shows the ANALYZE walk cost dominates.
4913    cold_row_count: u64,
4914    /// v6.7.0 — set when the cached `cold_row_count` may be wrong
4915    /// because rows moved into / out of the cold tier since the last
4916    /// ANALYZE. The virtual-table surface reports the cached value
4917    /// regardless (operators run ANALYZE to refresh).
4918    cold_row_count_stale: bool,
4919    /// v7.34 (crash-recovery P0 #2) — row-level redo capture buffer.
4920    /// `None` (default, in-memory mode) captures nothing — zero overhead.
4921    /// `Some` (set by the engine when persistence is on, before a
4922    /// mutating call) makes `insert` / `update_row` / `delete_rows`
4923    /// record the physical [`RowChange`] they applied, which the engine
4924    /// drains after the statement and writes to the WAL in place of the
4925    /// SQL text. Transient: never serialized; a `Catalog::clone` between
4926    /// enable and drain copies it (cheap — empty in the steady state).
4927    redo_log: Option<Vec<RowChange>>,
4928    /// v7.39 (round 215) — per-`EXCLUDE`-constraint range-overlap indexes,
4929    /// one per single-`&&` constraint on an integer-keyable range column.
4930    /// Maintained incrementally on insert / update / rebuild (mirroring the
4931    /// BTree secondary indexes); NOT serialized — rebuilt from the schema's
4932    /// exclusion constraints on load. Empty for tables with no EXCLUDE
4933    /// constraint (the common case), so `Table::clone` pays nothing.
4934    excl_indexes: Vec<ExclRangeIndex>,
4935    /// v7.38.2 (R2) — incremental write-set track for the RC rebase.
4936    /// `extract_tx_writeset` used to full-scan every header per call —
4937    /// ~200 µs on a 20k-row table, per in-transaction statement, every
4938    /// time a concurrent COMMIT moved the epoch; on tpcb's 100k-row
4939    /// accounts that scan was the c2 concurrency cliff itself. The
4940    /// three version-marking funnels (`insert_with_xmin`,
4941    /// `mark_row_deleted`, `mark_rows_deleted`) record here instead.
4942    ///
4943    /// One track per table, keyed by the LAST writer version: a shadow
4944    /// belongs to one transaction, so a different version claiming the
4945    /// table simply replaces the track (on the committed base that
4946    /// makes memory bounded by the last writer's footprint). Extraction
4947    /// verifies every recorded position still carries the version —
4948    /// any mismatch (compaction, inherited track, pre-track rows)
4949    /// falls back to the full scan, so the fast path can be wrong
4950    /// about NOTHING, only slow.
4951    tx_write_track: Option<TxWriteTrack>,
4952    /// v7.39 (round 493) — the snapshot floor below which a deleted row
4953    /// version is invisible to everyone, as of the statement now running.
4954    ///
4955    /// Runtime only: never serialised, and `0` (the default) prunes
4956    /// nothing, so any path that forgets to set it is merely slower, not
4957    /// wrong. The engine sets it from `vacuum_oldest_active()` — the same
4958    /// floor `vacuum` itself takes — before the statement's inserts.
4959    prune_horizon: u64,
4960}
4961
4962/// Catalog: insertion-ordered `Vec<Table>` for stable iter / serialize,
4963/// plus a `BTreeMap<String, usize>` sidecar index so `get` / `get_mut`
4964/// run in O(log n) instead of the old linear scan with per-element
4965/// string compares.
4966///
4967/// A pure `BTreeMap<String, Table>` was tried in an interim version
4968/// of v3.1.2 and regressed the single-table catalog benches by ~10%
4969/// (the per-element `BTreeMap` overhead outweighs the lookup win
4970/// when n is small). The sidecar shape preserves the insertion-order
4971/// iteration the on-disk encoding relies on and keeps `last_mut`
4972/// (used by the deserialize hot path) cheap.
4973/// v7.39 (pg_stat blks knife) — catalog-wide cold-tier read counter
4974/// backing pg_stat_database.blks_read. Row-granular (SPG has no 8 KB
4975/// page notion): one cold-segment row resolution = one "block read",
4976/// one hot row access = one "block hit" — the hit RATIO monitoring
4977/// dashboards compute keeps its meaning. Volatile like PG's stats.
4978#[derive(Debug, Default)]
4979pub struct ColdReadStats {
4980    pub cold_reads: core::sync::atomic::AtomicU64,
4981}
4982
4983impl Clone for ColdReadStats {
4984    fn clone(&self) -> Self {
4985        Self {
4986            cold_reads: core::sync::atomic::AtomicU64::new(
4987                self.cold_reads.load(core::sync::atomic::Ordering::Relaxed),
4988            ),
4989        }
4990    }
4991}
4992
4993/// 7.38.1 S3.1 (D4) — the non-table catalog families that carry a
4994/// per-transaction dirty window (see `Catalog::dirty_nontable`). One
4995/// entry class per side-map the poisoned-commit merge reconciles.
4996#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord)]
4997pub enum NonTableKind {
4998    Sequence,
4999    View,
5000    MaterializedView,
5001    EnumType,
5002    DomainType,
5003    CompositeType,
5004}
5005
5006#[derive(Debug, Clone, Default)]
5007pub struct Catalog {
5008    /// v7.39 (pg_stat blks knife) — see [`ColdReadStats`].
5009    pub cold_read_stats: ColdReadStats,
5010    tables: Vec<Table>,
5011    /// `name → tables[index]`. Kept in lock-step with `tables`.
5012    /// `create_table` is the only write path.
5013    by_name: BTreeMap<String, usize>,
5014    /// v7.39 (round 436) — the current session's temporary-table namespace.
5015    /// A temp table is stored under `<prefix><name>`, and every lookup tries
5016    /// that first: exactly PG's `pg_temp` search-path rule, and MySQL's
5017    /// "a TEMPORARY table shadows a permanent one of the same name".
5018    ///
5019    /// Process-local, never serialised: the engine sets it per session, and
5020    /// a catalog read back from disk starts with none. Kept here rather than
5021    /// at each of the ~170 engine call sites because `by_name` is private —
5022    /// this is the ONE place a table name becomes an index.
5023    temp_prefix: Option<String>,
5024    /// v7.39.2 — see [`Catalog::set_case_insensitive_names`].
5025    case_insensitive_names: bool,
5026    /// v7.39 (round 496) — the names of tables this catalog handle has had
5027    /// changed since the set was last cleared.
5028    ///
5029    /// Runtime only, never serialised. A transaction's shadow catalog
5030    /// clears it at BEGIN, so at COMMIT the set is exactly the tables the
5031    /// transaction changed — which is what lets a commit that cannot use
5032    /// the row-level merge install only those tables instead of the whole
5033    /// catalog, leaving another session's concurrent work in place.
5034    ///
5035    /// Recorded where the change actually happens (`get_mut`,
5036    /// `create_table`, `drop_table`) rather than from the statement
5037    /// classifier: round 494 tried classification for a correctness gate
5038    /// and it was wrong, because `SELECT lo_write(…)` reads as read-only.
5039    dirty_tables: alloc::collections::BTreeSet<String>,
5040    /// 7.38.1 S3.1 (D4) — the non-table twin of `dirty_tables`: which
5041    /// sequences / views / matviews / enum / domain / composite types
5042    /// THIS window created, altered, renamed or dropped. Counter
5043    /// advances (`nextval`) deliberately do NOT record — counter
5044    /// values merge via `sequence_counters` / `restore_sequence_
5045    /// counters`, and a tx that only consumed ids must not shadow a
5046    /// neighbour's ALTER SEQUENCE. Cleared by `clear_dirty_tables`
5047    /// (one window, both records).
5048    dirty_nontable: alloc::collections::BTreeSet<(NonTableKind, String)>,
5049    /// v7.37.15 (Phase C.1) — monotonic allocator for stable
5050    /// [`RelId`](row_header::RelId)s. Pre-incremented on each
5051    /// `create_table` so real ids start at 1 (0 is `UNASSIGNED`);
5052    /// never reused even after `DROP TABLE`, so a stale lock / redo
5053    /// reference is detectable. Process-local bookkeeping — not yet
5054    /// serialised; `deserialize` re-assigns dense ids on load (the
5055    /// V6 envelope, Phase C.6, will round-trip real ids).
5056    next_rel_id: u64,
5057    /// v5.1: in-memory cold-tier segments. Side-loaded via
5058    /// [`Catalog::load_segment_bytes`] — they live outside the
5059    /// catalog snapshot (caller persists them as separate files
5060    /// and re-loads on boot, until v5.3's `CatalogManifest` makes
5061    /// that wiring automatic). `RowLocator::Cold { segment_id, .. }`
5062    /// indexes this `Vec`. Cleared on `Catalog::new` / fresh
5063    /// `deserialize`.
5064    ///
5065    /// `Arc` wrap keeps `Catalog::clone` at O(N segments) bumps
5066    /// (rather than O(total segment bytes) memcpy) so the v4.42
5067    /// group-commit pre-image rollback invariant — clone is
5068    /// effectively free — survives the cold-tier addition.
5069    ///
5070    /// v6.7.3 — slots became `Option<…>` so cold-segment compaction
5071    /// can tombstone merged sources without breaking the
5072    /// `segment_id = index_into_vec` contract that on-disk
5073    /// `RowLocator::Cold { segment_id }` already serialized.
5074    /// `None` slot = the segment was retired by compaction; the
5075    /// physical file may still be on disk (next CHECKPOINT writes
5076    /// a manifest that no longer lists it, and the file becomes
5077    /// an orphan eligible for offline cleanup).
5078    cold_segments: Vec<Option<Arc<OwnedSegment>>>,
5079    /// v7.12.4 — user-defined functions (PL/pgSQL + SQL).
5080    /// Keyed by function name (PG overloading is out of scope).
5081    /// Bodies are stored as the raw source text the parser saw
5082    /// between `$$ ... $$`; the engine re-parses on each
5083    /// invocation. This keeps `spg-storage` free of `spg-sql`
5084    /// dependency — same pattern as partial-index predicates.
5085    functions: BTreeMap<String, FunctionDef>,
5086    /// v7.12.4 — triggers in insertion order. PG18-measured (round
5087    /// 753): PG fires same-event triggers in NAME order (a_trig
5088    /// before z_trig regardless of creation order); SPG fires in
5089    /// insertion order — a real divergence, ledgered as F31-B2.
5090    triggers: Vec<TriggerDef>,
5091    /// v7.39 (round 139) — query-rewrite RULEs, flat like triggers.
5092    rules: Vec<RuleDef>,
5093    /// v7.39 (round 280) — extended-statistics objects. Recorded so a
5094    /// pg_dump restores them and reflection reports them; the planner
5095    /// does not consult them yet.
5096    statistics_ext: Vec<StatisticsExtDef>,
5097    /// v7.39 (round 287) — server-side large objects, keyed by OID.
5098    /// PG stores them as 2 KB pages in `pg_largeobject`; the page split
5099    /// is a storage detail of ITS heap, so SPG holds the whole byte
5100    /// string and renders the pages on read. What must match is the
5101    /// observable surface: the OIDs, the bytes, and the page rows.
5102    large_objects: alloc::collections::BTreeMap<u32, Vec<u8>>,
5103    /// v7.17.0 — catalogued SEQUENCE objects (Phase 1.1). Each
5104    /// `nextval(name)` reaches in here, atomically increments
5105    /// `last_value` / flips `is_called`, returns the new value.
5106    /// Persisted in catalog FILE_VERSION 26+; older catalogs
5107    /// deserialise with an empty map.
5108    sequences: BTreeMap<String, SequenceDef>,
5109    /// v7.39 (read01 round 60) — the `public` schema's ACL (PG
5110    /// `pg_namespace.nspacl`). EMPTY = PG's default, which is not "nothing":
5111    /// PUBLIC holds USAGE and the owner holds USAGE + CREATE. Materialised on
5112    /// the first GRANT / REVOKE, exactly like a table's relacl.
5113    schema_acl: Vec<AclItem>,
5114    /// v7.39 (read01 round 60) — the database's ACL. EMPTY = PG's default:
5115    /// PUBLIC holds CONNECT + TEMPORARY, the owner holds all three.
5116    database_acl: Vec<AclItem>,
5117    /// v7.17.0 — catalogued VIEW objects (Phase 1.2). Each
5118    /// `SELECT FROM v` at engine exec-time looks up `v` here and
5119    /// prepends the view body as a synthetic CTE. Persisted in
5120    /// catalog FILE_VERSION 27+; older catalogs deserialise with
5121    /// an empty map.
5122    views: BTreeMap<String, ViewDef>,
5123    /// v7.17.0 — catalogued MATERIALIZED VIEW source registry
5124    /// (Phase 1.3). Maps name → SELECT source. The materialised
5125    /// rows themselves live as a regular `Table` with the same
5126    /// name; REFRESH re-parses + re-executes the source against
5127    /// the table. Persisted in catalog FILE_VERSION 28+;
5128    /// older catalogs deserialise with an empty map.
5129    materialized_views: BTreeMap<String, String>,
5130    /// v7.17.0 — catalogued user-defined ENUM types (Phase 1.4).
5131    /// Maps name → label list. Columns reference these by name
5132    /// via `ColumnSchema.user_enum_type`. Persisted in catalog
5133    /// FILE_VERSION 29+; older catalogs deserialise with an empty
5134    /// map.
5135    enum_types: BTreeMap<String, EnumDef>,
5136    /// v7.17.0 — catalogued user-defined DOMAIN types (Phase 1.5).
5137    /// Maps name → base + CHECK constraints. Columns reference
5138    /// these by name via `ColumnSchema.user_domain_type`.
5139    /// Persisted in catalog FILE_VERSION 30+; older catalogs
5140    /// deserialise with an empty map.
5141    domain_types: BTreeMap<String, DomainDef>,
5142    /// v7.39 (read01 round 50) — `COMMENT ON <kind> <obj> IS '…'` store.
5143    /// Keyed by a canonical `"<kind>:<name>"` string (`"table:t"`,
5144    /// `"column:t.c"`, `"index:i"`, `"view:v"`, …) so a new commentable
5145    /// object kind needs no schema change. `COMMENT … IS NULL` removes the
5146    /// entry. Persisted in catalog FILE_VERSION 61+; older catalogs
5147    /// deserialise with an empty map. Read back by obj_description /
5148    /// col_description and the pg_description view.
5149    comments: BTreeMap<String, String>,
5150    /// v7.39 (round 547) — PG's `pg_db_role_setting`: the GUC defaults
5151    /// `ALTER ROLE … SET` / `ALTER DATABASE … SET` record, applied when
5152    /// a session starts.
5153    ///
5154    /// Keyed exactly as PG keys it — `(database, role)` where an empty
5155    /// name is PG's oid 0, meaning "all". So `ALTER ROLE ALL SET` is
5156    /// `("", "")`, `ALTER DATABASE d SET` is `(d, "")`, `ALTER ROLE r
5157    /// SET` is `("", r)` and `ALTER ROLE r IN DATABASE d SET` is
5158    /// `(d, r)`. The value is that scope's parameter list.
5159    db_role_settings: BTreeMap<(String, String), BTreeMap<String, String>>,
5160    /// v7.39 (round 550) — replication slots, by name.
5161    ///
5162    /// A slot in PG is two things: a named record, and a reservation
5163    /// that holds WAL back. SPG keeps the record — which is what every
5164    /// setup script and monitoring query reads — and reports
5165    /// `wal_status = 'unreserved'`, PG's own word for a slot that no
5166    /// longer holds WAL. The whole family used to answer NULL and
5167    /// report success, so `pg_drop_replication_slot('nosuchslot')` said
5168    /// it worked and a setup script created nothing.
5169    ///
5170    /// Value: (plugin, slot_type). `plugin` is empty for a physical slot.
5171    replication_slots: BTreeMap<String, (String, String)>,
5172    /// v7.38.18 (S1) — the collation this database was CREATED with, and
5173    /// the one every text column that declares none is compared under.
5174    ///
5175    /// `None` means `C`, which is what every database written by every
5176    /// earlier version was built with — so an upgrade changes no answer
5177    /// and rebuilds no index. That is the whole migration story, and it
5178    /// is why this is an `Option` rather than a `String` defaulting to
5179    /// `"C"`.
5180    ///
5181    /// Set once, at creation, and never after. PostgreSQL refuses
5182    /// `ALTER DATABASE … LC_COLLATE` and the reason is the one that
5183    /// matters here too: every index key in this database was built
5184    /// under this collation, so it cannot move out from under them.
5185    /// See `docs/DESIGN-2026-08-23-collation.md`.
5186    db_collation: Option<String>,
5187    /// v7.38.19 — every name a `CREATE DATABASE` has asked for.
5188    ///
5189    /// SPG serves one database and answers to any name, so the statement
5190    /// has always been a no-op for naming. `pg_database` then listed one
5191    /// row -- whatever name the current session connected with -- so a
5192    /// database that had just been created, and could be connected to,
5193    /// was absent from the catalogue. `psql \l`, a migration tool asking
5194    /// "does this database exist", and a backup script that enumerates
5195    /// all read that table.
5196    ///
5197    /// Reported by sentori against 7.38.18. Runtime only, like
5198    /// `db_collation`: the statement is audited whenever it records a
5199    /// name, so replay rebuilds the set.
5200    created_databases: alloc::collections::BTreeSet<String>,
5201    /// v7.37.42-T2 ζ-B — catalogued user-defined COMPOSITE types
5202    /// (`CREATE TYPE name AS (field_name field_type, …)`). Columns
5203    /// reference these by name via
5204    /// `ColumnSchema.user_composite_type` (parallel to
5205    /// `user_enum_type` / `user_domain_type`). Persisted in catalog
5206    /// FILE_VERSION 52+; older catalogs deserialise with an empty
5207    /// map.
5208    composite_types: BTreeMap<String, CompositeDef>,
5209    /// v7.17.0 — schema-namespace registry (Phase 1.6). Tracks
5210    /// which schemas exist. `public`, `pg_catalog`, and
5211    /// `information_schema` are built-in and always present.
5212    /// Schema-qualified table references still strip the prefix
5213    /// at lookup time per v7.16-and-earlier — full
5214    /// schema-as-isolation is v7.18+ scope. Persisted in catalog
5215    /// FILE_VERSION 31+; older catalogs deserialise with just
5216    /// the built-ins.
5217    schemas: alloc::collections::BTreeSet<String>,
5218}
5219
5220/// v7.12.4 — catalogued user-defined function. `body` is the raw
5221/// source text between `$$ ... $$`; the engine re-parses it on
5222/// invocation. This keeps the storage codec stable when the
5223/// PL/pgSQL surface grows (no breaking-change risk on the disk
5224/// format).
5225// v7.39 (round 322, V46) — no longer `Eq`: COST / ROWS are f64, as in PG.
5226#[derive(Debug, Clone, PartialEq)]
5227pub struct FunctionDef {
5228    pub name: String,
5229    /// Display form of the argument list, e.g.
5230    /// `"(name TEXT, ts TIMESTAMP)"`. Empty `"()"` for the trigger
5231    /// function shape. Parser-side canonicalised before storage.
5232    pub args_repr: String,
5233    /// Display form of the return type, e.g. `"TRIGGER"` /
5234    /// `"INT"` / `"SETOF text"`. The engine special-cases
5235    /// `"TRIGGER"` (case-insensitive) to gate trigger-only
5236    /// semantics (NEW/OLD).
5237    pub returns: String,
5238    /// `LANGUAGE` clause, lowercased. `"plpgsql"` / `"sql"`.
5239    pub language: String,
5240    /// Source body of the function. PL/pgSQL: includes the
5241    /// surrounding `BEGIN ... END;`. SQL: includes the
5242    /// statement(s). The engine re-parses on invocation; bad
5243    /// bodies surface as a parse error at CALL time, not CREATE.
5244    pub body: String,
5245    /// v7.39 (read01 round 61) — the role that ran CREATE FUNCTION.
5246    pub owner: Option<String>,
5247    /// v7.39 (read01 round 61) — explicit GRANTs (PG `pg_proc.proacl`). EMPTY
5248    /// is NOT "nobody may call it": PG grants EXECUTE to PUBLIC by default, and
5249    /// leaves proacl NULL to say so. The list materialises on the first
5250    /// GRANT / REVOKE.
5251    pub acl: Vec<AclItem>,
5252    /// v7.39 (round 322, V46) — `IMMUTABLE` / `STRICT` / `PARALLEL SAFE` /
5253    /// `SECURITY DEFINER` / `LEAKPROOF` / `COST` / `ROWS`. `strict` is the
5254    /// only one with execution semantics today (a NULL argument yields a
5255    /// NULL result without running the body); the rest are recorded so
5256    /// `pg_get_functiondef` and `pg_proc` report what was declared.
5257    pub volatility: u8,
5258    pub strict: bool,
5259    pub security_definer: bool,
5260    pub leakproof: bool,
5261    pub parallel: u8,
5262    pub cost: Option<f64>,
5263    pub rows: Option<f64>,
5264}
5265
5266/// v7.39 (round 322, V46) — `FunctionDef.volatility` codes: PG's
5267/// `pg_proc.provolatile` letters.
5268pub const FN_VOLATILE: u8 = b'v';
5269pub const FN_IMMUTABLE: u8 = b'i';
5270pub const FN_STABLE: u8 = b's';
5271
5272/// v7.39 (round 322, V46) — `FunctionDef.parallel` codes: PG's
5273/// `pg_proc.proparallel` letters.
5274pub const FN_PARALLEL_UNSAFE: u8 = b'u';
5275pub const FN_PARALLEL_RESTRICTED: u8 = b'r';
5276pub const FN_PARALLEL_SAFE: u8 = b's';
5277
5278/// v7.39 (round 315, V19) — which catalogued function does a persisted
5279/// ACL key refer to?
5280///
5281/// The key was computed by whichever formula was current when the image
5282/// was written, and the multi-word fix changed that formula for bare
5283/// types like `double precision`. A miss therefore does NOT mean "no
5284/// such function": an older image's key would land nowhere and its owner
5285/// and grants would be dropped in silence. Exact match first, then the
5286/// pre-fix formula.
5287#[must_use]
5288pub fn resolve_stored_function_key(
5289    functions: &BTreeMap<String, FunctionDef>,
5290    stored: &str,
5291) -> Option<String> {
5292    if functions.contains_key(stored) {
5293        return Some(stored.to_string());
5294    }
5295    functions
5296        .values()
5297        .find(|f| function_signature_key_legacy(&f.name, &f.args_repr) == stored)
5298        .map(|f| function_signature_key(&f.name, &f.args_repr))
5299}
5300
5301/// v7.39 (round 344, V49) — re-exported from [`spg_sql`], which owns the
5302/// SQL type spellings. This crate carried a byte-identical copy because
5303/// the two were siblings that did not depend on each other; spg-sql is a
5304/// dependency-free leaf, so the dependency is acyclic and the publish
5305/// order already puts it first. One list, one place to keep it right.
5306pub use spg_sql::parser::is_multiword_type_phrase;
5307
5308/// v7.39 (round 315, V19) — the signature key as computed BEFORE the
5309/// multi-word fix, used only to recognise what an older image wrote.
5310///
5311/// The function catalogue recomputes its keys from the stored name and
5312/// argument text on load, so it needs no migration. The ACL block does
5313/// not: it persists the computed key as a string and matches on it. A
5314/// key that changed shape would simply fail to match, and the owner and
5315/// grants would be dropped without a word — so the loader falls back to
5316/// this when the stored key finds nothing.
5317#[must_use]
5318pub fn function_signature_key_legacy(name: &str, args_repr: &str) -> String {
5319    let inner = args_repr
5320        .trim()
5321        .trim_start_matches('(')
5322        .trim_end_matches(')');
5323    let types: Vec<String> = if inner.trim().is_empty() {
5324        Vec::new()
5325    } else {
5326        inner
5327            .split(',')
5328            .map(|part| {
5329                let mut words: Vec<&str> = part.split_whitespace().collect();
5330                if !words.is_empty()
5331                    && (words[0].eq_ignore_ascii_case("OUT")
5332                        || words[0].eq_ignore_ascii_case("INOUT"))
5333                {
5334                    words.remove(0);
5335                }
5336                let ty = if words.len() >= 2 {
5337                    words[1..].join(" ")
5338                } else {
5339                    words.first().map_or(String::new(), |w| (*w).to_string())
5340                };
5341                normalize_type_name(&ty)
5342            })
5343            .collect()
5344    };
5345    format!("{}({})", name.to_ascii_lowercase(), types.join(","))
5346}
5347
5348pub fn function_signature_key(name: &str, args_repr: &str) -> String {
5349    let types = function_arg_types(args_repr);
5350    format!("{}({})", name.to_ascii_lowercase(), types.join(","))
5351}
5352
5353/// The declared argument TYPES of a function, out of its `args_repr`
5354/// (`"(x INT, y DOUBLE PRECISION)"` → `["int", "float"]`). An entry may be a
5355/// bare type with no name (`"(INT)"`).
5356#[must_use]
5357pub fn function_arg_types(args_repr: &str) -> Vec<String> {
5358    let inner = args_repr
5359        .trim()
5360        .trim_start_matches('(')
5361        .trim_end_matches(')');
5362    if inner.trim().is_empty() {
5363        return Vec::new();
5364    }
5365    inner
5366        .split(',')
5367        .map(|part| {
5368            let mut words: Vec<&str> = part.split_whitespace().collect();
5369            // `OUT x INT` / `INOUT x INT` — the mode is not part of the type.
5370            if !words.is_empty()
5371                && (words[0].eq_ignore_ascii_case("OUT") || words[0].eq_ignore_ascii_case("INOUT"))
5372            {
5373                words.remove(0);
5374            }
5375            // v7.39 (round 315, V19) — two or more words is USUALLY
5376            // `name TYPE`, but not when the type itself is spelled in
5377            // several words. `double precision` was read as a parameter
5378            // named "double" of type "precision", so it keyed differently
5379            // from `x double precision` — the same signature written two
5380            // ways did not resolve to the same function. Decide by asking
5381            // whether the whole phrase names a type first; only then is
5382            // the leading word a parameter name.
5383            let whole = words.join(" ");
5384            let ty = if words.len() >= 2 && !is_multiword_type_phrase(&whole) {
5385                words[1..].join(" ")
5386            } else {
5387                whole
5388            };
5389            normalize_type_name(&ty)
5390        })
5391        .collect()
5392}
5393
5394/// v7.39 (read01 round 65) — the declared argument NAMES of a function (`""` for
5395/// a bare type with no name).
5396#[must_use]
5397pub fn function_arg_names(args_repr: &str) -> Vec<String> {
5398    let inner = args_repr
5399        .trim()
5400        .trim_start_matches('(')
5401        .trim_end_matches(')');
5402    if inner.trim().is_empty() {
5403        return Vec::new();
5404    }
5405    inner
5406        .split(',')
5407        .map(|part| {
5408            let mut words: Vec<&str> = part.split_whitespace().collect();
5409            if !words.is_empty()
5410                && (words[0].eq_ignore_ascii_case("OUT") || words[0].eq_ignore_ascii_case("INOUT"))
5411            {
5412                words.remove(0);
5413            }
5414            if words.len() >= 2 {
5415                words[0].to_string()
5416            } else {
5417                String::new()
5418            }
5419        })
5420        .collect()
5421}
5422
5423/// Fold PG's type aliases so a signature key is stable across spellings.
5424/// Unknown names pass through lower-cased — consistency is what the key needs.
5425#[must_use]
5426pub fn normalize_type_name(ty: &str) -> String {
5427    let t = ty.trim().to_ascii_lowercase();
5428    // Peel a precision/length modifier: `numeric(10,2)`, `varchar(64)`.
5429    let base = t.split_once('(').map_or(t.as_str(), |(h, _)| h).trim();
5430    match base {
5431        "int" | "int4" | "integer" => "int",
5432        "bigint" | "int8" => "bigint",
5433        "smallint" | "int2" => "smallint",
5434        "text" | "varchar" | "character varying" | "char" | "character" | "bpchar" => "text",
5435        "bool" | "boolean" => "bool",
5436        "float" | "float8" | "double precision" => "float",
5437        "real" | "float4" => "real",
5438        "numeric" | "decimal" => "numeric",
5439        "timestamptz" | "timestamp with time zone" => "timestamptz",
5440        "timestamp" | "timestamp without time zone" => "timestamp",
5441        other => other,
5442    }
5443    .to_string()
5444}
5445
5446/// v7.12.4 — catalogued trigger. References its function by
5447/// name; the function must exist at TRIGGER creation time
5448/// (forward references are deferred to v7.12.5+).
5449#[derive(Debug, Clone, PartialEq, Eq)]
5450pub struct TriggerDef {
5451    pub name: String,
5452    /// Watched table. Trigger is dropped when the table drops.
5453    pub table: String,
5454    /// `"BEFORE"` / `"AFTER"` / `"INSTEAD OF"`. Stored as the
5455    /// uppercased keyword so deserialised catalogs round-trip
5456    /// without canonicalisation surprises.
5457    pub timing: String,
5458    /// Each entry is one of `"INSERT"` / `"UPDATE"` / `"DELETE"`
5459    /// / `"TRUNCATE"`. `INSERT OR UPDATE` parses to two entries.
5460    pub events: Vec<String>,
5461    /// `"ROW"` / `"STATEMENT"`. v7.12.4 ships `"ROW"` only;
5462    /// `"STATEMENT"` parses and persists but the executor
5463    /// refuses it at trigger fire time.
5464    pub for_each: String,
5465    /// Name of the PL/pgSQL function to invoke.
5466    pub function: String,
5467    /// v7.13.0 — `UPDATE OF col, col, …` column-list filter
5468    /// (mailrs round-5 G7). Non-empty means the trigger fires
5469    /// only when at least one of these columns appears in the
5470    /// UPDATE's SET list. Empty = no column filter. Stored in
5471    /// catalog FILE_VERSION 23+; older catalogs deserialise with
5472    /// an empty vec.
5473    pub update_columns: Vec<String>,
5474    /// v7.16.1 — whether the trigger fires when its watched
5475    /// event occurs. Toggled by `ALTER TABLE … { ENABLE |
5476    /// DISABLE } TRIGGER …`; pg_dump --disable-triggers wraps
5477    /// every data block with a DISABLE/ENABLE pair so the
5478    /// rows already-computed in prod don't get re-rewritten.
5479    /// Defaults to `true` at CREATE TRIGGER time. Stored in
5480    /// catalog FILE_VERSION 25+; older catalogs deserialise
5481    /// with `enabled = true`.
5482    pub enabled: bool,
5483    /// v7.39 (round 138) — the deparsed `WHEN ( condition )` predicate text
5484    /// (re-parsed at fire time to filter row triggers). Empty = no WHEN.
5485    /// Persisted from FILE_VERSION 70; older catalogs read back empty.
5486    pub when_condition: String,
5487}
5488
5489/// v7.39 (round 280) — one `CREATE STATISTICS` object.
5490#[derive(Debug, Clone, PartialEq, Eq)]
5491pub struct StatisticsExtDef {
5492    pub name: String,
5493    pub table: String,
5494    /// PG's single-letter kinds: `d` ndistinct, `f` dependencies,
5495    /// `m` mcv. PG's default set is all three.
5496    pub kinds: Vec<String>,
5497    pub columns: Vec<String>,
5498}
5499
5500/// v7.39 (round 139) — a catalogued query-rewrite RULE. Stored flat like
5501/// `TriggerDef`, keyed by `(name, table)`. Command / WHEN text is deparsed SQL
5502/// re-parsed at rewrite time (the same round-trip trick as
5503/// `TriggerDef.when_condition`). Persisted from FILE_VERSION 71.
5504#[derive(Debug, Clone, PartialEq, Eq)]
5505pub struct RuleDef {
5506    pub name: String,
5507    pub table: String,
5508    /// Event keyword, uppercased: `INSERT` / `UPDATE` / `DELETE` / `SELECT`.
5509    pub event: String,
5510    /// `true` = `DO INSTEAD`, `false` = `DO ALSO`.
5511    pub instead: bool,
5512    /// Deparsed `WHERE` predicate text; empty = unconditional.
5513    pub when_condition: String,
5514    /// Deparsed DO command statements; empty = `NOTHING`.
5515    pub commands: Vec<String>,
5516}
5517
5518/// v7.17.0 — catalogued SEQUENCE. PG semantics: a counter object
5519/// returning monotonically increasing values via `nextval(name)`.
5520/// `last_value` is the most recent value handed out; `is_called`
5521/// is false until the first `nextval`/`setval`. Stored separately
5522/// from tables in the catalog.
5523#[derive(Debug, Clone, PartialEq, Eq)]
5524pub struct SequenceDef {
5525    pub name: String,
5526    /// Data type — narrows the i64 range. PG default BIGINT.
5527    pub data_type: SequenceDataType,
5528    pub start: i64,
5529    pub increment: i64,
5530    pub min_value: i64,
5531    pub max_value: i64,
5532    pub cache: i64,
5533    pub cycle: bool,
5534    /// `OWNED BY` target — `(table, column)` or NONE.
5535    pub owned_by: Option<(String, String)>,
5536    /// Most recently handed-out value. Meaningless when
5537    /// `is_called == false`; in that case the NEXT `nextval`
5538    /// will return `start`.
5539    pub last_value: i64,
5540    pub is_called: bool,
5541    /// v7.39 (read01 round 60) — the role that ran CREATE SEQUENCE. `None` = an
5542    /// image written before FILE_VERSION 66, which predates sequence owners.
5543    pub owner: Option<String>,
5544    /// v7.39 (read01 round 60) — explicit GRANTs on this sequence. A sequence's
5545    /// meaningful privileges are SELECT (`currval`), UPDATE (`setval`) and
5546    /// USAGE (`nextval`).
5547    pub acl: Vec<AclItem>,
5548}
5549
5550/// v7.17.0 — sequence integer width.
5551#[derive(Debug, Clone, Copy, PartialEq, Eq)]
5552pub enum SequenceDataType {
5553    SmallInt,
5554    Int,
5555    BigInt,
5556}
5557
5558/// v7.17.0 Phase 1.6 — built-in schema names that every Catalog
5559/// understands without an explicit CREATE SCHEMA. Used by
5560/// [`Catalog::schema_exists`] and the engine's schema-qualified
5561/// lookup path.
5562#[must_use]
5563pub fn is_builtin_schema(name: &str) -> bool {
5564    name.eq_ignore_ascii_case("public")
5565        || name.eq_ignore_ascii_case("pg_catalog")
5566        || name.eq_ignore_ascii_case("information_schema")
5567}
5568
5569/// v7.17.0 — parse a PG-canonical UUID text representation into the
5570/// 16-byte network-order layout used by `Value::Uuid`. Accepted input
5571/// shapes (all case-insensitive):
5572///   * Canonical hyphenated 8-4-4-4-12 (`550e8400-e29b-41d4-a716-446655440000`)
5573///   * Unhyphenated 32-char hex (`550e8400e29b41d4a716446655440000`)
5574///   * Either form wrapped in `{ ... }`
5575///
5576/// Returns `None` for any malformed input (wrong length, non-hex
5577/// characters, misplaced hyphens). The caller surfaces a SQL error
5578/// at coercion time — silent acceptance of garbage would mask
5579/// application bugs and is exactly the divergence from PG that
5580/// breaks the 0-change cutover promise.
5581#[must_use]
5582pub fn parse_uuid_str(input: &str) -> Option<[u8; 16]> {
5583    let s = input.trim();
5584    // Strip surrounding braces if present.
5585    let s = if let Some(inner) = s.strip_prefix('{').and_then(|x| x.strip_suffix('}')) {
5586        inner
5587    } else {
5588        s
5589    };
5590    // Two valid shapes after braces are stripped: 32 hex chars or
5591    // the canonical 36-char hyphenated form.
5592    let hex: String = match s.len() {
5593        32 => s.to_ascii_lowercase(),
5594        36 => {
5595            // Hyphens must be exactly at positions 8, 13, 18, 23.
5596            let b = s.as_bytes();
5597            if b[8] != b'-' || b[13] != b'-' || b[18] != b'-' || b[23] != b'-' {
5598                return None;
5599            }
5600            let mut out = String::with_capacity(32);
5601            out.push_str(&s[0..8]);
5602            out.push_str(&s[9..13]);
5603            out.push_str(&s[14..18]);
5604            out.push_str(&s[19..23]);
5605            out.push_str(&s[24..36]);
5606            out.make_ascii_lowercase();
5607            out
5608        }
5609        _ => return None,
5610    };
5611    let bytes = hex.as_bytes();
5612    let mut out = [0u8; 16];
5613    for i in 0..16 {
5614        let hi = hex_nibble(bytes[i * 2])?;
5615        let lo = hex_nibble(bytes[i * 2 + 1])?;
5616        out[i] = (hi << 4) | lo;
5617    }
5618    Some(out)
5619}
5620
5621fn hex_nibble(b: u8) -> Option<u8> {
5622    match b {
5623        b'0'..=b'9' => Some(b - b'0'),
5624        b'a'..=b'f' => Some(10 + b - b'a'),
5625        b'A'..=b'F' => Some(10 + b - b'A'),
5626        _ => None,
5627    }
5628}
5629
5630/// v7.17.0 — render a `Value::Uuid` payload as the canonical
5631/// lowercase 8-4-4-4-12 hyphenated form PG `text` cast surfaces.
5632#[must_use]
5633pub fn format_uuid(b: &[u8; 16]) -> String {
5634    const HEX: &[u8; 16] = b"0123456789abcdef";
5635    let mut out = String::with_capacity(36);
5636    for (i, byte) in b.iter().enumerate() {
5637        if matches!(i, 4 | 6 | 8 | 10) {
5638            out.push('-');
5639        }
5640        out.push(HEX[(byte >> 4) as usize] as char);
5641        out.push(HEX[(byte & 0x0f) as usize] as char);
5642    }
5643    out
5644}
5645
5646/// v7.17.0 Phase 1.5 — catalogued user-defined DOMAIN. A domain
5647/// is a named CHECK-constrained alias over a built-in type;
5648/// columns bound to it inherit the base type plus the CHECK
5649/// predicates + NOT NULL + DEFAULT at INSERT/UPDATE time.
5650/// v7.37.17 (Phase E RC rebase) — the write-set one writer version left
5651/// on a table, addressed by stable [`row_header::RowId`]s so it can be
5652/// replayed onto a fresher clone of the relation whose physical slots
5653/// differ. Produced by [`Table::extract_tx_writeset`], consumed by
5654/// [`Table::replay_tx_writeset`].
5655#[derive(Debug, Clone, Default)]
5656pub struct TxWriteSet {
5657    /// INSERTs and UPDATE-new-versions (`header.xmin == v`).
5658    pub inserted: Vec<(row_header::RowId, Row<'static>)>,
5659    /// DELETE / UPDATE-old-version targets (`header.xmax == v`).
5660    pub tombstoned: Vec<row_header::RowId>,
5661}
5662
5663impl TxWriteSet {
5664    #[must_use]
5665    pub fn is_empty(&self) -> bool {
5666        self.inserted.is_empty() && self.tombstoned.is_empty()
5667    }
5668}
5669
5670/// v7.39 (round 260) — one named CHECK on a domain. PG auto-names an
5671/// unnamed one `<domain>_check`, then `_check1`, `_check2`, … (probed).
5672#[derive(Debug, Clone, PartialEq, Eq)]
5673pub struct DomainCheck {
5674    pub name: String,
5675    /// The predicate source, referencing the pseudo-column `VALUE`.
5676    pub expr: String,
5677}
5678
5679/// `default` / `checks` are stored as Display-form source so
5680/// `spg-storage` stays free of `spg-sql` dependency — same
5681/// pattern as FunctionDef / ViewDef.
5682#[derive(Debug, Clone, PartialEq, Eq)]
5683pub struct DomainDef {
5684    pub name: String,
5685    pub base_type: DataType,
5686    pub nullable: bool,
5687    pub default: Option<String>,
5688    /// v7.39 (round 260) — each CHECK carries its constraint NAME, so
5689    /// `ALTER DOMAIN … DROP CONSTRAINT <name>` can find it and the
5690    /// violation message can report the constraint that actually failed.
5691    /// PG's auto-naming for an unnamed check is `<domain>_check`, then
5692    /// `_check1`, `_check2`, … (probed).
5693    pub checks: Vec<DomainCheck>,
5694    /// v7.39 (round 258/259) — when this domain was declared over ANOTHER
5695    /// domain (`CREATE DOMAIN child AS parent CHECK (…)`), the parent's
5696    /// name. `base_type` is the ultimate scalar type either way, so
5697    /// without this the parent's constraints were invisible and a value
5698    /// violating them was silently accepted. PG checks the whole chain,
5699    /// base-first, and an `ALTER DOMAIN` on the parent takes effect for
5700    /// the child immediately (probed) — so the chain is walked at check
5701    /// time rather than copied at CREATE time. Catalog FILE_VERSION 74+.
5702    pub base_domain: Option<String>,
5703}
5704
5705/// v7.17.0 Phase 1.4 — catalogued user-defined ENUM type. The
5706/// label vector is order-preserving (PG enum ordering follows the
5707/// declared order). At INSERT/UPDATE on a column bound to this
5708/// enum, the engine looks up the value against `labels` and
5709/// rejects non-members.
5710#[derive(Debug, Clone, PartialEq, Eq)]
5711pub struct EnumDef {
5712    pub name: String,
5713    pub labels: Vec<String>,
5714}
5715
5716/// v7.37.42-T2 ζ-B — catalogued user-defined COMPOSITE type
5717/// (`CREATE TYPE name AS (field_name field_type, ...)`). Order
5718/// matters: PG composite literals are positional, and SPG mirrors
5719/// that. Stored as ordered `(name, DataType)` pairs to keep the
5720/// codec straightforward and to allow eventual `Value::Composite`
5721/// bodies to encode positionally. Persisted in catalog FILE_VERSION
5722/// 52+; older catalogs deserialise with an empty composite_types
5723/// map. Composite types can be used as a column type by spelling
5724/// the composite's name; the resolution from
5725/// `ColumnSchema.user_composite_type = Some(name)` happens at the
5726/// engine boundary (parallel to `user_enum_type` /
5727/// `user_domain_type`). The dense storage shape — JSON-text body
5728/// keyed by the composite's field list — keeps the codec free of
5729/// recursive `Value` bodies until the full Value::Composite arena
5730/// migration in a later phase.
5731#[derive(Debug, Clone, PartialEq, Eq)]
5732pub struct CompositeDef {
5733    pub name: String,
5734    /// Ordered `(field_name, field_type)` pairs. PG composite
5735    /// literals are positional, so order is part of the type's
5736    /// identity.
5737    pub fields: Vec<(String, DataType)>,
5738    /// v7.39 (round 264) — parallel to `fields`: the USER type name of
5739    /// each field when it is itself a composite (or another named user
5740    /// type). `DataType` has no room for one, so a nested composite
5741    /// field resolved to the parser's Text placeholder and the inner
5742    /// record stayed TEXT — `(x).inner.street` errored, `pg_typeof`
5743    /// said text, and `row_to_json` nested a string instead of an
5744    /// object. Same shape as `ColumnSchema.user_composite_type` and
5745    /// `DomainDef.base_domain`. Catalog FILE_VERSION 76+; an older
5746    /// catalog reads all-None, which is what it meant.
5747    pub field_user_types: Vec<Option<String>>,
5748}
5749
5750/// v7.17.0 Phase 1.2 — catalogued VIEW. The body is stored as the
5751/// raw source text the parser saw between `AS` and the statement
5752/// terminator; the engine re-parses on each invocation. Same
5753/// pattern as `FunctionDef` — keeps `spg-storage` free of
5754/// `spg-sql` dependency.
5755#[derive(Debug, Clone, PartialEq, Eq)]
5756pub struct ViewDef {
5757    pub name: String,
5758    /// Optional `(col, col, …)` rename list. Empty when the body's
5759    /// projected names are used directly.
5760    pub columns: Vec<String>,
5761    /// Raw SELECT source. Display-rendered at storage time so the
5762    /// catalog round-trips a deterministic form regardless of
5763    /// whitespace / comments in the original input. Re-parsed at
5764    /// SELECT-from-view time to materialise as a synthetic CTE.
5765    pub body: String,
5766    /// v7.39 (round 132) — `WITH CHECK OPTION`: 0 = none, 1 = LOCAL,
5767    /// 2 = CASCADED. A storage-local u8 (no dependency on the SQL AST).
5768    /// Persisted from FILE_VERSION 69; older catalogs read back as 0.
5769    pub check_option: u8,
5770}
5771
5772impl SequenceDataType {
5773    /// PG default min/max per AS clause.
5774    pub fn default_bounds(self, increment_positive: bool) -> (i64, i64) {
5775        match self {
5776            Self::SmallInt => {
5777                if increment_positive {
5778                    (1, i64::from(i16::MAX))
5779                } else {
5780                    (i64::from(i16::MIN), -1)
5781                }
5782            }
5783            Self::Int => {
5784                if increment_positive {
5785                    (1, i64::from(i32::MAX))
5786                } else {
5787                    (i64::from(i32::MIN), -1)
5788                }
5789            }
5790            Self::BigInt => {
5791                if increment_positive {
5792                    (1, i64::MAX)
5793                } else {
5794                    (i64::MIN, -1)
5795                }
5796            }
5797        }
5798    }
5799}
5800
5801impl Catalog {
5802    /// v7.37.15 (Phase D) — fleet-wide vacuum pass. Walks every
5803    /// user table and reclaims rows whose delete-commit version is
5804    /// older than `oldest_active_snapshot`. Returns an aggregated
5805    /// report with per-table breakdown so hosts can emit metrics.
5806    ///
5807    /// `dry_run = true` reports the work without doing it. Use it
5808    /// to estimate the cost before scheduling a real pass.
5809    pub fn vacuum_all(
5810        &mut self,
5811        oldest_active_snapshot: u64,
5812        dry_run: bool,
5813    ) -> vacuum::VacuumReport {
5814        let mut total = vacuum::VacuumReport::default();
5815        // Snapshot the table names so we don't hold an immutable
5816        // borrow during the get_mut loop.
5817        let names: Vec<String> = self
5818            .tables
5819            .iter()
5820            .map(|t| t.schema().name.clone())
5821            .collect();
5822        for name in names {
5823            let Some(t) = self.get_mut(&name) else {
5824                continue;
5825            };
5826            let r = t.vacuum(oldest_active_snapshot, dry_run);
5827            if r.rows_reclaimed > 0 {
5828                total.per_table.push((name, r.rows_reclaimed));
5829            }
5830            total.rows_reclaimed += r.rows_reclaimed;
5831            total.rows_examined += r.rows_examined;
5832        }
5833        total
5834    }
5835
5836    pub const fn new() -> Self {
5837        Self {
5838            cold_read_stats: ColdReadStats {
5839                cold_reads: core::sync::atomic::AtomicU64::new(0),
5840            },
5841            tables: Vec::new(),
5842            by_name: BTreeMap::new(),
5843            temp_prefix: None,
5844            case_insensitive_names: false,
5845            dirty_tables: alloc::collections::BTreeSet::new(),
5846            dirty_nontable: alloc::collections::BTreeSet::new(),
5847            next_rel_id: 0,
5848            cold_segments: Vec::new(),
5849            functions: BTreeMap::new(),
5850            triggers: Vec::new(),
5851            rules: Vec::new(),
5852            statistics_ext: Vec::new(),
5853            large_objects: alloc::collections::BTreeMap::new(),
5854            sequences: BTreeMap::new(),
5855            schema_acl: Vec::new(),
5856            database_acl: Vec::new(),
5857            views: BTreeMap::new(),
5858            materialized_views: BTreeMap::new(),
5859            enum_types: BTreeMap::new(),
5860            domain_types: BTreeMap::new(),
5861            comments: BTreeMap::new(),
5862            db_role_settings: BTreeMap::new(),
5863            replication_slots: BTreeMap::new(),
5864            db_collation: None,
5865            created_databases: alloc::collections::BTreeSet::new(),
5866            composite_types: BTreeMap::new(),
5867            schemas: alloc::collections::BTreeSet::new(),
5868        }
5869    }
5870
5871    /// v7.12.4 — read-only view of catalogued user-defined
5872    /// functions. Engine callers go through here to look up the
5873    /// function body before re-parsing it for invocation.
5874    pub const fn functions(&self) -> &BTreeMap<String, FunctionDef> {
5875        &self.functions
5876    }
5877
5878    /// v7.12.4 — register a new user-defined function. With
5879    /// `or_replace = false`, errors if the name is taken. The
5880    /// engine validates the body before passing it here.
5881    pub fn create_function(
5882        &mut self,
5883        def: FunctionDef,
5884        or_replace: bool,
5885    ) -> Result<(), StorageError> {
5886        // v7.39 (read01 round 62) — functions are keyed by SIGNATURE, not by
5887        // name: `f(int)` and `f(text)` are two functions, as in PG. Keying by
5888        // name alone made a second overload an "already exists" error — so a
5889        // pg_dump carrying an overload set could not restore — and, worse, a
5890        // call to one overload silently ran the other.
5891        let key = function_signature_key(&def.name, &def.args_repr);
5892        if !or_replace && self.functions.contains_key(&key) {
5893            return Err(StorageError::Corrupt(format!(
5894                "function {:?} already exists (drop or use CREATE OR REPLACE)",
5895                def.name
5896            )));
5897        }
5898        self.functions.insert(key, def);
5899        Ok(())
5900    }
5901
5902    /// v7.39 (read01 round 62) — every overload of `name`.
5903    #[must_use]
5904    pub fn functions_named(&self, name: &str) -> Vec<&FunctionDef> {
5905        self.functions
5906            .values()
5907            .filter(|f| f.name.eq_ignore_ascii_case(name))
5908            .collect()
5909    }
5910
5911    /// v7.39 (read01 round 62) — one overload, by its signature key.
5912    #[must_use]
5913    pub fn function_by_key(&self, key: &str) -> Option<&FunctionDef> {
5914        self.functions.get(key)
5915    }
5916
5917    /// v7.39 (read01 round 62) — drop ONE overload. `true` if it was there.
5918    pub fn drop_function_by_key(&mut self, key: &str) -> bool {
5919        self.functions.remove(key).is_some()
5920    }
5921
5922    /// v7.12.4 — remove a user-defined function by name. Returns
5923    /// `true` if a function was removed, `false` if none matched.
5924    /// Caller decides whether to surface `if_exists` semantics.
5925    /// v7.39 (read01 round 62) — with no signature, PG drops the function only
5926    /// when the name is unambiguous. SPG mirrors that: this removes EVERY
5927    /// overload of `name`, and the caller (ddl.rs) refuses the ambiguous case
5928    /// before getting here.
5929    pub fn drop_function(&mut self, name: &str) -> bool {
5930        let keys: Vec<String> = self
5931            .functions
5932            .iter()
5933            .filter(|(_, f)| f.name.eq_ignore_ascii_case(name))
5934            .map(|(k, _)| k.clone())
5935            .collect();
5936        let hit = !keys.is_empty();
5937        for k in keys {
5938            self.functions.remove(&k);
5939        }
5940        hit
5941    }
5942
5943    /// v7.17.0 — read-only handle to catalogued sequences.
5944    /// v7.39 (read01 round 60) — the `public` schema's ACL (PG nspacl).
5945    #[must_use]
5946    pub fn schema_acl(&self) -> &[AclItem] {
5947        &self.schema_acl
5948    }
5949
5950    pub fn schema_acl_mut(&mut self) -> &mut Vec<AclItem> {
5951        &mut self.schema_acl
5952    }
5953
5954    /// v7.39 (read01 round 60) — the database's ACL.
5955    #[must_use]
5956    pub fn database_acl(&self) -> &[AclItem] {
5957        &self.database_acl
5958    }
5959
5960    pub fn database_acl_mut(&mut self) -> &mut Vec<AclItem> {
5961        &mut self.database_acl
5962    }
5963
5964    /// v7.39 (read01 round 60) — mutable sequence access, for GRANT.
5965    /// v7.39 (round 469) — resolves the session's temporary sequence
5966    /// first, like its read-only twin. `nextval` and `setval` reach the
5967    /// map through here, so a temporary sequence shadowing a permanent one
5968    /// advances the temporary one — measured against PG18, where the
5969    /// permanent sequence's counter is untouched while the temp exists.
5970    pub fn sequence_mut(&mut self, name: &str) -> Option<&mut SequenceDef> {
5971        let key = self.sequence_key(name);
5972        self.sequences.get_mut(&key)
5973    }
5974
5975    /// v7.39 (read01 round 61) — mutable function access, for GRANT.
5976    pub fn function_mut(&mut self, name: &str) -> Option<&mut FunctionDef> {
5977        self.functions.get_mut(name)
5978    }
5979
5980    /// Every catalogued sequence, temp ones included under their mangled
5981    /// storage names. Listing code filters these through
5982    /// [`Self::listed_name`]; anything resolving ONE name by its logical
5983    /// spelling wants [`Self::sequence`] instead.
5984    pub const fn sequences_all(&self) -> &BTreeMap<String, SequenceDef> {
5985        &self.sequences
5986    }
5987
5988    /// v7.39 (round 469) — resolve one sequence by its logical name, the
5989    /// session's temporary one winning over a permanent one of the same
5990    /// name. The same rule [`Self::resolve_index`] applies to tables.
5991    #[must_use]
5992    pub fn sequence(&self, name: &str) -> Option<&SequenceDef> {
5993        if let Some(mangled) = self.temp_name_for(name)
5994            && let Some(def) = self.sequences.get(&mangled)
5995        {
5996            return Some(def);
5997        }
5998        self.sequences.get(name)
5999    }
6000
6001    /// Does a sequence of this logical name exist for this session?
6002    #[must_use]
6003    pub fn has_sequence(&self, name: &str) -> bool {
6004        self.sequence(name).is_some()
6005    }
6006
6007    /// The storage key a sequence of this logical name resolves to — the
6008    /// session's temp mangling when it has one, else the name itself.
6009    #[must_use]
6010    pub fn sequence_key(&self, name: &str) -> String {
6011        if let Some(mangled) = self.temp_name_for(name)
6012            && self.sequences.contains_key(&mangled)
6013        {
6014            return mangled;
6015        }
6016        name.into()
6017    }
6018
6019    /// v7.17.0 — register a new SEQUENCE. Errors if `name`
6020    /// collides with an existing sequence and `if_not_exists`
6021    /// is false.
6022    pub fn create_sequence(
6023        &mut self,
6024        def: SequenceDef,
6025        if_not_exists: bool,
6026    ) -> Result<(), StorageError> {
6027        if self.sequences.contains_key(&def.name) {
6028            if if_not_exists {
6029                return Ok(());
6030            }
6031            // v7.39 (read01 round 47) — a sequence is a relation to PG (42P07).
6032            return Err(StorageError::Corrupt(format!(
6033                "relation {:?} already exists",
6034                def.name
6035            )));
6036        }
6037        self.mark_nontable_dirty(NonTableKind::Sequence, &def.name);
6038        self.sequences.insert(def.name.clone(), def);
6039        Ok(())
6040    }
6041
6042    /// v7.17.0 — remove a SEQUENCE by name. Returns `true` if a
6043    /// sequence was removed, `false` if none matched. Caller
6044    /// surfaces IF EXISTS semantics.
6045    /// v7.39 (read01 round 49) — `ALTER SEQUENCE old RENAME TO new`.
6046    /// Errors when `old` is missing or `new` is taken; the SequenceDef's own
6047    /// `name` field is rewritten so it stays self-describing.
6048    pub fn rename_sequence(&mut self, old: &str, new: &str) -> Result<(), StorageError> {
6049        if !self.sequences.contains_key(old) {
6050            return Err(StorageError::Corrupt(format!(
6051                "relation {old:?} does not exist"
6052            )));
6053        }
6054        if self.sequences.contains_key(new) {
6055            return Err(StorageError::Corrupt(format!(
6056                "relation {new:?} already exists"
6057            )));
6058        }
6059        self.mark_nontable_dirty(NonTableKind::Sequence, old);
6060        self.mark_nontable_dirty(NonTableKind::Sequence, new);
6061        if let Some(mut def) = self.sequences.remove(old) {
6062            def.name = new.to_string();
6063            self.sequences.insert(new.to_string(), def);
6064        }
6065        Ok(())
6066    }
6067
6068    pub fn drop_sequence(&mut self, name: &str) -> bool {
6069        self.mark_nontable_dirty(NonTableKind::Sequence, name);
6070        self.sequences.remove(name).is_some()
6071    }
6072
6073    /// v7.17.0 — atomic nextval. Increments `last_value` per
6074    /// `increment`, returns the new value, sets `is_called`.
6075    /// Returns an error on CYCLE-less overflow.
6076    /// v7.39 (round 497) — the counter state of every sequence, for
6077    /// carrying across a commit install.
6078    ///
6079    /// A sequence's VALUE is not transactional in PG: `nextval` advances
6080    /// shared state that a rollback does not give back, because two
6081    /// sessions must never receive the same number. SPG keeps sequences in
6082    /// the catalog, and a transaction works on a catalog CLONE, so
6083    /// installing that clone at COMMIT would restore whatever the counter
6084    /// was at BEGIN. These two let the install put the live counters back.
6085    #[must_use]
6086    pub fn sequence_counters(&self) -> Vec<(String, i64, bool)> {
6087        self.sequences
6088            .iter()
6089            .map(|(k, d)| (k.clone(), d.last_value, d.is_called))
6090            .collect()
6091    }
6092
6093    /// Restore counters saved by [`Self::sequence_counters`], for the
6094    /// sequences that still exist. A sequence the transaction CREATED is
6095    /// absent from the saved set and keeps the value it was given.
6096    pub fn restore_sequence_counters(&mut self, saved: &[(String, i64, bool)]) {
6097        for (k, last, called) in saved {
6098            if let Some(d) = self.sequences.get_mut(k) {
6099                d.last_value = *last;
6100                d.is_called = *called;
6101            }
6102        }
6103    }
6104
6105    pub fn sequence_next_value(&mut self, name: &str) -> Result<i64, StorageError> {
6106        let key = self.sequence_key(name);
6107        let Some(seq) = self.sequences.get_mut(&key) else {
6108            return Err(StorageError::TableNotFound { name: name.into() });
6109        };
6110        // PG semantics: when !is_called (fresh sequence or
6111        // setval(_, false)), the next nextval returns the stored
6112        // `last_value`. When is_called, it advances by `increment`
6113        // and CYCLE-wraps on overflow.
6114        let candidate = if seq.is_called {
6115            let next = seq.last_value.checked_add(seq.increment).ok_or_else(|| {
6116                StorageError::Corrupt(format!("sequence {name:?} arithmetic overflow"))
6117            })?;
6118            if seq.increment > 0 {
6119                if next > seq.max_value {
6120                    if seq.cycle {
6121                        seq.min_value
6122                    } else {
6123                        // v7.39 (round 220) — PG's 2200H wording, not a
6124                        // Corrupt-classed error.
6125                        return Err(StorageError::SequenceExhausted {
6126                            name: name.into(),
6127                            limit: seq.max_value,
6128                            is_max: true,
6129                        });
6130                    }
6131                } else {
6132                    next
6133                }
6134            } else if next < seq.min_value {
6135                if seq.cycle {
6136                    seq.max_value
6137                } else {
6138                    return Err(StorageError::SequenceExhausted {
6139                        name: name.into(),
6140                        limit: seq.min_value,
6141                        is_max: false,
6142                    });
6143                }
6144            } else {
6145                next
6146            }
6147        } else {
6148            seq.last_value
6149        };
6150        seq.last_value = candidate;
6151        seq.is_called = true;
6152        Ok(candidate)
6153    }
6154
6155    /// v7.17.0 — currval. Errors if the session has never called
6156    /// nextval on this sequence (PG semantics). At the catalog
6157    /// level we approximate "session" with "is_called persisted";
6158    /// the engine session-tracking layer can wrap this for the
6159    /// strict per-session semantics later.
6160    pub fn sequence_current_value(&self, name: &str) -> Result<i64, StorageError> {
6161        let Some(seq) = self.sequences.get(name) else {
6162            return Err(StorageError::TableNotFound { name: name.into() });
6163        };
6164        if !seq.is_called {
6165            return Err(StorageError::Corrupt(format!(
6166                "currval of sequence {name:?} is not yet defined in this session"
6167            )));
6168        }
6169        Ok(seq.last_value)
6170    }
6171
6172    /// v7.17.0 — setval(name, value [, is_called]). PG returns
6173    /// `value` regardless. `is_called=true` means the NEXT
6174    /// nextval will return `value + increment`; `is_called=false`
6175    /// means the next nextval will return `value`.
6176    pub fn sequence_set_value(
6177        &mut self,
6178        name: &str,
6179        value: i64,
6180        is_called: bool,
6181    ) -> Result<i64, StorageError> {
6182        let key = self.sequence_key(name);
6183        let Some(seq) = self.sequences.get_mut(&key) else {
6184            return Err(StorageError::TableNotFound { name: name.into() });
6185        };
6186        // v7.39 (round 244) — PG refuses a value outside the sequence's
6187        // range (22003); SPG accepted it silently, leaving last_value out
6188        // of bounds.
6189        if value < seq.min_value || value > seq.max_value {
6190            return Err(StorageError::Unsupported(format!(
6191                "setval: value {value} is out of bounds for sequence \"{name}\" ({}..{})",
6192                seq.min_value, seq.max_value
6193            )));
6194        }
6195        seq.last_value = value;
6196        seq.is_called = is_called;
6197        Ok(value)
6198    }
6199
6200    /// v7.17.0 Phase 1.2 — read-only handle to catalogued views. Temp ones
6201    /// are in here under their mangled storage names; listing code filters
6202    /// through [`Self::listed_name`], and anything resolving ONE name by
6203    /// its logical spelling wants [`Self::view`].
6204    pub const fn views_all(&self) -> &BTreeMap<String, ViewDef> {
6205        &self.views
6206    }
6207
6208    /// v7.39 (round 469) — resolve one view by its logical name, the
6209    /// session's temporary one winning over a permanent one of the same
6210    /// name.
6211    #[must_use]
6212    pub fn view(&self, name: &str) -> Option<&ViewDef> {
6213        if let Some(mangled) = self.temp_name_for(name)
6214            && let Some(def) = self.views.get(&mangled)
6215        {
6216            return Some(def);
6217        }
6218        self.views.get(name)
6219    }
6220
6221    /// Does a view of this logical name exist for this session?
6222    #[must_use]
6223    pub fn has_view(&self, name: &str) -> bool {
6224        self.view(name).is_some()
6225    }
6226
6227    /// The storage key a view of this logical name resolves to.
6228    #[must_use]
6229    pub fn view_key(&self, name: &str) -> String {
6230        if let Some(mangled) = self.temp_name_for(name)
6231            && self.views.contains_key(&mangled)
6232        {
6233            return mangled;
6234        }
6235        name.into()
6236    }
6237
6238    /// v7.17.0 Phase 1.2 — install a VIEW. `or_replace=true`
6239    /// overwrites an existing entry; `if_not_exists=true` is a
6240    /// silent no-op when the name is taken. Errors if both flags
6241    /// are off and the name collides.
6242    pub fn create_view(
6243        &mut self,
6244        def: ViewDef,
6245        or_replace: bool,
6246        if_not_exists: bool,
6247    ) -> Result<(), StorageError> {
6248        if self.views.contains_key(&def.name) {
6249            if or_replace {
6250                self.mark_nontable_dirty(NonTableKind::View, &def.name);
6251                self.mark_nontable_dirty(NonTableKind::View, &def.name);
6252                self.views.insert(def.name.clone(), def);
6253                return Ok(());
6254            }
6255            if if_not_exists {
6256                return Ok(());
6257            }
6258            // v7.39 (read01 round 47) — a view is a relation to PG (42P07).
6259            return Err(StorageError::Corrupt(format!(
6260                "relation {:?} already exists",
6261                def.name
6262            )));
6263        }
6264        // Reject name collision with tables / sequences — same
6265        // namespace per PG.
6266        if self.by_name.contains_key(&def.name) {
6267            return Err(StorageError::Corrupt(format!(
6268                "view {:?} would shadow an existing table",
6269                def.name
6270            )));
6271        }
6272        if self.sequences.contains_key(&def.name) {
6273            return Err(StorageError::Corrupt(format!(
6274                "view {:?} would shadow an existing sequence",
6275                def.name
6276            )));
6277        }
6278        self.views.insert(def.name.clone(), def);
6279        Ok(())
6280    }
6281
6282    /// v7.17.0 Phase 1.2 — remove a view by name. Returns true if
6283    /// a view was removed.
6284    pub fn drop_view(&mut self, name: &str) -> bool {
6285        self.mark_nontable_dirty(NonTableKind::View, name);
6286        self.views.remove(name).is_some()
6287    }
6288
6289    /// v7.17.0 Phase 1.3 — read-only handle to the materialised-
6290    /// view source registry. Each entry pairs with a regular
6291    /// table of the same name that holds the cached rows.
6292    pub const fn materialized_views(&self) -> &BTreeMap<String, String> {
6293        &self.materialized_views
6294    }
6295
6296    /// v7.17.0 Phase 1.3 — register a source for a materialised
6297    /// view. Caller has already created the backing table.
6298    pub fn register_materialized_view(&mut self, name: String, body: String) {
6299        self.mark_nontable_dirty(NonTableKind::MaterializedView, &name);
6300        self.materialized_views.insert(name, body);
6301    }
6302
6303    /// v7.17.0 Phase 1.3 — drop the source registry entry. Returns
6304    /// true if a source was unregistered. Caller separately drops
6305    /// the backing table.
6306    pub fn drop_materialized_view_source(&mut self, name: &str) -> bool {
6307        self.mark_nontable_dirty(NonTableKind::MaterializedView, name);
6308        self.materialized_views.remove(name).is_some()
6309    }
6310
6311    /// v7.17.0 Phase 1.4 — read-only handle to user-defined ENUM
6312    /// catalog.
6313    pub const fn enum_types(&self) -> &BTreeMap<String, EnumDef> {
6314        &self.enum_types
6315    }
6316
6317    /// v7.17.0 Phase 1.4 — install a new ENUM type. Errors if
6318    /// `name` collides with an existing enum (no IF NOT EXISTS
6319    /// per PG semantics for CREATE TYPE).
6320    pub fn create_enum_type(&mut self, def: EnumDef) -> Result<(), StorageError> {
6321        if self.enum_types.contains_key(&def.name) {
6322            return Err(StorageError::Corrupt(format!(
6323                "type {:?} already exists",
6324                def.name
6325            )));
6326        }
6327        self.mark_nontable_dirty(NonTableKind::EnumType, &def.name);
6328        self.enum_types.insert(def.name.clone(), def);
6329        Ok(())
6330    }
6331
6332    /// v7.17.0 Phase 1.4 — drop an ENUM type by name. Returns
6333    /// true if a type was removed.
6334    /// v7.37 D.55 — `ALTER TYPE … ADD VALUE`. Appends `label` to an existing
6335    /// enum's ordered label list, or inserts it before/after an existing label.
6336    /// `if_not_exists` makes a duplicate a no-op; otherwise a duplicate errors.
6337    /// Returns `Ok(true)` if a label was added, `Ok(false)` if it already existed
6338    /// (only possible under `if_not_exists`).
6339    /// v7.39 (read01 round 49) — `ALTER TYPE t RENAME VALUE 'old' TO 'new'`.
6340    /// The parser used to swallow this form as a no-op, so the rename was
6341    /// accepted and silently ignored. Renaming in place keeps the label's
6342    /// sort position, which is what PG does (enumsortorder is untouched).
6343    pub fn rename_enum_value(
6344        &mut self,
6345        type_name: &str,
6346        old: &str,
6347        new: &str,
6348    ) -> Result<(), StorageError> {
6349        let def = self
6350            .enum_types
6351            .get_mut(type_name)
6352            .ok_or_else(|| StorageError::Corrupt(format!("type {type_name:?} does not exist")))?;
6353        if def.labels.iter().any(|l| l == new) {
6354            return Err(StorageError::Corrupt(format!(
6355                "enum label {new:?} already exists"
6356            )));
6357        }
6358        let at = def.labels.iter().position(|l| l == old).ok_or_else(|| {
6359            StorageError::Corrupt(format!("{old:?} is not an existing enum label"))
6360        })?;
6361        def.labels[at] = new.to_string();
6362        Ok(())
6363    }
6364
6365    /// v7.39 (read01 round 50) — set (or, with `None`, remove) the comment on
6366    /// an object. `key` is the canonical `"<kind>:<name>"` form.
6367    pub fn set_comment(&mut self, key: &str, text: Option<&str>) {
6368        match text {
6369            Some(t) => {
6370                self.comments.insert(key.to_string(), t.to_string());
6371            }
6372            None => {
6373                self.comments.remove(key);
6374            }
6375        }
6376    }
6377
6378    /// v7.39 (read01 round 50) — the comment on an object, if any.
6379    #[must_use]
6380    pub fn comment(&self, key: &str) -> Option<&str> {
6381        self.comments.get(key).map(String::as_str)
6382    }
6383
6384    /// v7.39 (round 547) — record a GUC default for a scope. An empty
6385    /// database or role name is PG's oid 0 ("all"). `None` value
6386    /// removes just that parameter, as PG's RESET does.
6387    pub fn set_db_role_setting(
6388        &mut self,
6389        database: &str,
6390        role: &str,
6391        param: &str,
6392        value: Option<&str>,
6393    ) {
6394        let key = (database.to_string(), role.to_string());
6395        match value {
6396            Some(v) => {
6397                self.db_role_settings
6398                    .entry(key)
6399                    .or_default()
6400                    .insert(param.to_ascii_lowercase(), v.to_string());
6401            }
6402            None => {
6403                if let Some(m) = self.db_role_settings.get_mut(&key) {
6404                    m.remove(&param.to_ascii_lowercase());
6405                    if m.is_empty() {
6406                        self.db_role_settings.remove(&key);
6407                    }
6408                }
6409            }
6410        }
6411    }
6412
6413    /// v7.39 (round 550) — create a replication slot. `Err` carries
6414    /// PG's own message for a duplicate.
6415    ///
6416    /// # Errors
6417    /// When a slot of that name already exists.
6418    pub fn create_replication_slot(
6419        &mut self,
6420        name: &str,
6421        plugin: &str,
6422        slot_type: &str,
6423    ) -> Result<(), String> {
6424        if self.replication_slots.contains_key(name) {
6425            return Err(alloc::format!("replication slot \"{name}\" already exists"));
6426        }
6427        self.replication_slots.insert(
6428            name.to_string(),
6429            (plugin.to_string(), slot_type.to_string()),
6430        );
6431        Ok(())
6432    }
6433
6434    /// # Errors
6435    /// When no slot of that name exists — PG's message, and the case
6436    /// that used to report success.
6437    pub fn drop_replication_slot(&mut self, name: &str) -> Result<(), String> {
6438        if self.replication_slots.remove(name).is_none() {
6439            return Err(alloc::format!("replication slot \"{name}\" does not exist"));
6440        }
6441        Ok(())
6442    }
6443
6444    #[must_use]
6445    /// v7.38.18 (S1) — the collation this database was created with.
6446    /// `"C"` when nothing was recorded, which is what an older catalog
6447    /// and a default `initdb`-less start both mean.
6448    pub fn db_collation(&self) -> &str {
6449        self.db_collation.as_deref().unwrap_or("C")
6450    }
6451
6452    /// Record the creation collation. Refused once one is set, because
6453    /// every index key already in this database was built under it —
6454    /// the same refusal PostgreSQL gives `ALTER DATABASE … LC_COLLATE`,
6455    /// and for the same reason.
6456    ///
6457    /// `Ok(false)` when the value asked for is the one already in force,
6458    /// so a host that passes its environment on every start is not an
6459    /// error.
6460    pub fn set_db_collation(&mut self, name: &str) -> Result<bool, StorageError> {
6461        if self.db_collation.as_deref() == Some(name) {
6462            return Ok(false);
6463        }
6464        if self.db_collation.is_none() && name.eq_ignore_ascii_case("C") {
6465            return Ok(false);
6466        }
6467        if self.db_collation.is_some() || !self.tables.is_empty() {
6468            return Err(StorageError::Corrupt(format!(
6469                "database collation is already {:?} and cannot be changed; \
6470                 PostgreSQL refuses this too, because every index key here \
6471                 was built under it",
6472                self.db_collation()
6473            )));
6474        }
6475        self.db_collation = Some(name.into());
6476        Ok(true)
6477    }
6478
6479    /// The user said so, in SQL: `CREATE DATABASE … LC_COLLATE 'x'`.
6480    ///
6481    /// Differs from [`Self::set_db_collation`] in one way, and the
6482    /// difference is the whole point: this REPLACES a collation the
6483    /// database already has, as long as no table has been created yet.
6484    /// The refusal in `set_db_collation` exists because index keys were
6485    /// built under the old collation — with no tables, none were.
6486    ///
6487    /// The case it is for: a server stamps the container's `LANG` on a
6488    /// fresh database at startup, and the customer's bootstrap script
6489    /// then says `CREATE DATABASE app LC_COLLATE 'de_DE.utf8'`. What the
6490    /// script asked for beats what the container happened to export.
6491    ///
6492    /// `Ok(false)` when a table already exists — the caller warns rather
6493    /// than failing, because PostgreSQL would have made a SEPARATE
6494    /// database here and returned success, and failing a bootstrap
6495    /// script is a customer change.
6496    pub fn declare_db_collation(&mut self, name: &str) -> bool {
6497        if self.db_collation.as_deref() == Some(name) {
6498            return true;
6499        }
6500        if !self.tables.is_empty() {
6501            return false;
6502        }
6503        self.db_collation = Some(name.into());
6504        true
6505    }
6506
6507    /// Record a name a `CREATE DATABASE` asked for; `true` when new.
6508    pub fn record_created_database(&mut self, name: &str) -> bool {
6509        self.created_databases.insert(name.to_string())
6510    }
6511
6512    /// The names `CREATE DATABASE` has been asked for.
6513    pub const fn created_databases(&self) -> &alloc::collections::BTreeSet<String> {
6514        &self.created_databases
6515    }
6516
6517    pub const fn replication_slots(&self) -> &BTreeMap<String, (String, String)> {
6518        &self.replication_slots
6519    }
6520
6521    /// PG's RESET ALL: drops this scope's whole entry, leaving the
6522    /// other scopes alone — measured on PG18, where `ALTER ROLE r RESET
6523    /// ALL` left the ALL, the database and the role-in-database rows.
6524    pub fn reset_db_role_settings(&mut self, database: &str, role: &str) {
6525        self.db_role_settings
6526            .remove(&(database.to_string(), role.to_string()));
6527    }
6528
6529    #[must_use]
6530    pub const fn db_role_settings(&self) -> &BTreeMap<(String, String), BTreeMap<String, String>> {
6531        &self.db_role_settings
6532    }
6533
6534    /// v7.39 (read01 round 50) — every `(key, text)` pair, for the
6535    /// pg_description view.
6536    #[must_use]
6537    pub const fn comments(&self) -> &BTreeMap<String, String> {
6538        &self.comments
6539    }
6540
6541    /// v7.39 (read01 round 50) — drop every comment whose key names `obj`
6542    /// (the object itself and, for a table, its columns). Called when the
6543    /// object is dropped so a later object of the same name doesn't inherit
6544    /// a stale comment.
6545    pub fn drop_comments_for(&mut self, kind: &str, name: &str) {
6546        let exact = alloc::format!("{kind}:{name}");
6547        let col_prefix = alloc::format!("column:{name}.");
6548        self.comments
6549            .retain(|k, _| *k != exact && !k.starts_with(&col_prefix));
6550    }
6551
6552    pub fn add_enum_value(
6553        &mut self,
6554        type_name: &str,
6555        label: &str,
6556        if_not_exists: bool,
6557        position: Option<(bool, String)>,
6558    ) -> Result<bool, StorageError> {
6559        self.mark_nontable_dirty(NonTableKind::EnumType, type_name);
6560        let def = self
6561            .enum_types
6562            .get_mut(type_name)
6563            .ok_or_else(|| StorageError::Corrupt(format!("type {type_name:?} does not exist")))?;
6564        if def.labels.iter().any(|l| l == label) {
6565            if if_not_exists {
6566                return Ok(false);
6567            }
6568            // v7.39 (read01 round 49) — PG wording (42710 at the wire).
6569            return Err(StorageError::Corrupt(format!(
6570                "enum label {label:?} already exists"
6571            )));
6572        }
6573        match position {
6574            None => def.labels.push(label.to_string()),
6575            Some((is_before, anchor)) => {
6576                let at = def
6577                    .labels
6578                    .iter()
6579                    .position(|l| l == &anchor)
6580                    .ok_or_else(|| {
6581                        StorageError::Corrupt(format!(
6582                            "enum label {anchor:?} does not exist in type {type_name:?}"
6583                        ))
6584                    })?;
6585                let idx = if is_before { at } else { at + 1 };
6586                def.labels.insert(idx, label.to_string());
6587            }
6588        }
6589        Ok(true)
6590    }
6591
6592    pub fn drop_enum_type(&mut self, name: &str) -> bool {
6593        self.mark_nontable_dirty(NonTableKind::EnumType, name);
6594        self.enum_types.remove(name).is_some()
6595    }
6596
6597    /// v7.17.0 Phase 1.5 — read-only handle to DOMAIN catalog.
6598    pub const fn domain_types(&self) -> &BTreeMap<String, DomainDef> {
6599        &self.domain_types
6600    }
6601
6602    /// v7.17.0 Phase 1.5 — install a DOMAIN. Errors on collision
6603    /// with an existing domain.
6604    pub fn create_domain_type(&mut self, def: DomainDef) -> Result<(), StorageError> {
6605        if self.domain_types.contains_key(&def.name) {
6606            return Err(StorageError::Corrupt(format!(
6607                "domain {:?} already exists",
6608                def.name
6609            )));
6610        }
6611        self.mark_nontable_dirty(NonTableKind::DomainType, &def.name);
6612        self.domain_types.insert(def.name.clone(), def);
6613        Ok(())
6614    }
6615
6616    /// v7.17.0 Phase 1.5 — drop a DOMAIN by name.
6617    pub fn drop_domain_type(&mut self, name: &str) -> bool {
6618        self.mark_nontable_dirty(NonTableKind::DomainType, name);
6619        self.domain_types.remove(name).is_some()
6620    }
6621
6622    /// v7.37.42-T2 ζ-B — read-only handle to user-defined COMPOSITE
6623    /// catalog. Used by the engine to resolve
6624    /// `ColumnSchema.user_composite_type` lookups + by
6625    /// information_schema-style introspection.
6626    pub const fn composite_types(&self) -> &BTreeMap<String, CompositeDef> {
6627        &self.composite_types
6628    }
6629
6630    /// v7.37.42-T2 ζ-B — install a new COMPOSITE type. Errors if
6631    /// `name` already exists in the composite registry (PG forbids
6632    /// IF NOT EXISTS on CREATE TYPE composite; the engine surfaces
6633    /// the collision with the existing name).
6634    pub fn create_composite_type(&mut self, def: CompositeDef) -> Result<(), StorageError> {
6635        if self.composite_types.contains_key(&def.name) {
6636            return Err(StorageError::Corrupt(format!(
6637                "type {:?} already exists",
6638                def.name
6639            )));
6640        }
6641        self.mark_nontable_dirty(NonTableKind::CompositeType, &def.name);
6642        self.composite_types.insert(def.name.clone(), def);
6643        Ok(())
6644    }
6645
6646    /// v7.37.42-T2 ζ-B — drop a COMPOSITE type by name. Returns
6647    /// true if a type was removed.
6648    pub fn drop_composite_type(&mut self, name: &str) -> bool {
6649        self.mark_nontable_dirty(NonTableKind::CompositeType, name);
6650        self.composite_types.remove(name).is_some()
6651    }
6652
6653    /// v7.17.0 Phase 1.6 — read-only handle to the user-created
6654    /// schema registry. Built-in schemas (`public`, `pg_catalog`,
6655    /// `information_schema`) are NOT included here; use
6656    /// [`schema_exists`](Self::schema_exists) for the full
6657    /// check.
6658    pub const fn user_schemas(&self) -> &alloc::collections::BTreeSet<String> {
6659        &self.schemas
6660    }
6661
6662    /// v7.17.0 Phase 1.6 — schema-name resolver. Returns true
6663    /// for built-in schemas + every user-CREATEd one. Used by
6664    /// CREATE SCHEMA collision checks and (future) by
6665    /// information_schema.schemata.
6666    pub fn schema_exists(&self, name: &str) -> bool {
6667        is_builtin_schema(name) || self.schemas.contains(name)
6668    }
6669
6670    /// v7.17.0 Phase 1.6 — register a new schema. Errors if the
6671    /// name already exists and `if_not_exists=false`. Built-in
6672    /// names cannot be redeclared.
6673    pub fn create_schema(&mut self, name: String, if_not_exists: bool) -> Result<(), StorageError> {
6674        if is_builtin_schema(&name) {
6675            if if_not_exists {
6676                return Ok(());
6677            }
6678            return Err(StorageError::Corrupt(format!(
6679                "schema {name:?} is built-in and cannot be redeclared"
6680            )));
6681        }
6682        if self.schemas.contains(&name) {
6683            if if_not_exists {
6684                return Ok(());
6685            }
6686            return Err(StorageError::Corrupt(format!(
6687                "schema {name:?} already exists"
6688            )));
6689        }
6690        self.schemas.insert(name);
6691        Ok(())
6692    }
6693
6694    /// v7.17.0 Phase 1.6 — drop a user-created schema. Returns
6695    /// true if a schema was removed. Built-in names always
6696    /// return false (cannot be dropped). Tables that previously
6697    /// used the schema as a prefix keep their bare name and stay
6698    /// queryable — this is the "prefix routing, not isolation"
6699    /// posture documented in v7.17 Phase 1.6.
6700    pub fn drop_schema(&mut self, name: &str) -> Result<bool, StorageError> {
6701        if is_builtin_schema(name) {
6702            return Err(StorageError::Corrupt(format!(
6703                "schema {name:?} is built-in and cannot be dropped"
6704            )));
6705        }
6706        Ok(self.schemas.remove(name))
6707    }
6708
6709    /// v7.17.0 — ALTER SEQUENCE option merge. Caller-provided
6710    /// updates overwrite the matching fields; unset fields keep
6711    /// their stored values. RESTART variants update last_value
6712    /// directly per PG: `RESTART` resets to current `start`;
6713    /// `RESTART WITH n` resets to `n`.
6714    #[allow(clippy::too_many_arguments)]
6715    pub fn alter_sequence(
6716        &mut self,
6717        name: &str,
6718        increment: Option<i64>,
6719        min_value: Option<i64>,
6720        max_value: Option<i64>,
6721        start: Option<i64>,
6722        restart: Option<Option<i64>>,
6723        cache: Option<i64>,
6724        cycle: Option<bool>,
6725        owned_by: Option<Option<(String, String)>>,
6726    ) -> Result<(), StorageError> {
6727        self.mark_nontable_dirty(NonTableKind::Sequence, name);
6728        let Some(seq) = self.sequences.get_mut(name) else {
6729            return Err(StorageError::TableNotFound { name: name.into() });
6730        };
6731        if let Some(v) = increment {
6732            seq.increment = v;
6733        }
6734        if let Some(v) = min_value {
6735            seq.min_value = v;
6736        }
6737        if let Some(v) = max_value {
6738            seq.max_value = v;
6739        }
6740        if let Some(v) = start {
6741            seq.start = v;
6742        }
6743        if let Some(restart_value) = restart {
6744            seq.last_value = restart_value.unwrap_or(seq.start);
6745            seq.is_called = false;
6746        }
6747        if let Some(v) = cache {
6748            seq.cache = v;
6749        }
6750        if let Some(v) = cycle {
6751            seq.cycle = v;
6752        }
6753        if let Some(v) = owned_by {
6754            seq.owned_by = v;
6755        }
6756        Ok(())
6757    }
6758
6759    /// v7.12.4 — read-only slice of all catalogued triggers.
6760    /// Engine row-write paths filter this by (table, event,
6761    /// timing) and fire matches in slice order.
6762    pub fn triggers(&self) -> &[TriggerDef] {
6763        &self.triggers
6764    }
6765
6766    /// v7.15.0 — mutable handle to the trigger slice for
6767    /// `ALTER TABLE … RENAME COLUMN`, which rewrites every
6768    /// `update_columns` entry that referenced the renamed
6769    /// column.
6770    pub fn triggers_mut(&mut self) -> &mut Vec<TriggerDef> {
6771        &mut self.triggers
6772    }
6773
6774    /// v7.12.4 — register a new trigger. With `or_replace = false`,
6775    /// errors when a trigger with the same name already exists on
6776    /// the same table (PG scoping rule — trigger names are
6777    /// per-table, not global). Trigger function must already
6778    /// exist in the catalog at registration time.
6779    pub fn create_trigger(
6780        &mut self,
6781        def: TriggerDef,
6782        or_replace: bool,
6783    ) -> Result<(), StorageError> {
6784        // v7.39 (round 137) — a trigger may target a base table (BEFORE / AFTER)
6785        // or a view (INSTEAD OF). The engine enforces the timing↔target rule;
6786        // storage only requires the relation to exist as one or the other.
6787        if !self.by_name.contains_key(&def.table) && !self.views.contains_key(&def.table) {
6788            return Err(StorageError::TableNotFound {
6789                name: def.table.clone(),
6790            });
6791        }
6792        // v7.39 (read01 round 62) — functions are keyed by SIGNATURE now. A
6793        // trigger names its function by NAME (a trigger function takes no
6794        // arguments), so the existence check goes through the name index.
6795        if self.functions_named(&def.function).is_empty() {
6796            // v7.39 (round 710) — PG's wording: the FUNCTION is what does
6797            // not exist (`function nosuch_fn() does not exist`), and the
6798            // old message rode `Corrupt`'s on-disk banner besides.
6799            return Err(StorageError::Corrupt(format!(
6800                "function {}() does not exist",
6801                def.function
6802            )));
6803        }
6804        let dup = self
6805            .triggers
6806            .iter()
6807            .position(|t| t.name == def.name && t.table == def.table);
6808        match (dup, or_replace) {
6809            (Some(_), false) => Err(StorageError::Corrupt(format!(
6810                "trigger {:?} already exists on table {:?}",
6811                def.name, def.table
6812            ))),
6813            (Some(i), true) => {
6814                self.triggers[i] = def;
6815                Ok(())
6816            }
6817            (None, _) => {
6818                self.triggers.push(def);
6819                Ok(())
6820            }
6821        }
6822    }
6823
6824    /// v7.12.4 — remove a trigger by `(name, table)`. Returns
6825    /// `true` if one was removed.
6826    pub fn drop_trigger(&mut self, name: &str, table: &str) -> bool {
6827        let before = self.triggers.len();
6828        self.triggers
6829            .retain(|t| !(t.name == name && t.table == table));
6830        before != self.triggers.len()
6831    }
6832
6833    /// v7.39 (round 139) — the catalogued query-rewrite RULEs.
6834    pub fn rules(&self) -> &[RuleDef] {
6835        &self.rules
6836    }
6837
6838    /// v7.39 (round 280) — the catalogued extended-statistics objects.
6839    #[must_use]
6840    pub fn statistics_ext(&self) -> &[StatisticsExtDef] {
6841        &self.statistics_ext
6842    }
6843
6844    /// v7.39 (round 287) — every large object, ascending by OID.
6845    #[must_use]
6846    pub fn large_objects(&self) -> &alloc::collections::BTreeMap<u32, Vec<u8>> {
6847        &self.large_objects
6848    }
6849
6850    /// The bytes of one large object, or `None` when no such OID exists.
6851    #[must_use]
6852    pub fn large_object(&self, oid: u32) -> Option<&[u8]> {
6853        self.large_objects.get(&oid).map(Vec::as_slice)
6854    }
6855
6856    /// Create a large object. `oid` of 0 means "pick one" — PG's
6857    /// `lo_create(0)` / `lo_creat(-1)` spelling. Errors when the
6858    /// requested OID is taken.
6859    pub fn create_large_object(&mut self, oid: u32, bytes: Vec<u8>) -> Result<u32, String> {
6860        let id = if oid == 0 {
6861            self.next_large_object_oid()
6862        } else {
6863            oid
6864        };
6865        if self.large_objects.contains_key(&id) {
6866            return Err(format!("large object {id} already exists"));
6867        }
6868        self.large_objects.insert(id, bytes);
6869        Ok(id)
6870    }
6871
6872    /// Overwrite `len` bytes at `offset` (0-based), growing the object
6873    /// with zero bytes if the write starts past the end — PG's
6874    /// `lo_put` semantics.
6875    pub fn put_large_object(&mut self, oid: u32, offset: usize, data: &[u8]) -> Result<(), String> {
6876        let Some(buf) = self.large_objects.get_mut(&oid) else {
6877            return Err(format!("large object {oid} does not exist"));
6878        };
6879        let end = offset.saturating_add(data.len());
6880        if buf.len() < end {
6881            buf.resize(end, 0);
6882        }
6883        buf[offset..end].copy_from_slice(data);
6884        Ok(())
6885    }
6886
6887    /// v7.39 (round 306) — `lo_truncate`. PG's truncate sets the object
6888    /// to exactly `len` bytes in BOTH directions: it shortens, and it
6889    /// GROWS with zero fill when `len` exceeds the current size
6890    /// (measured — `lo_truncate(fd, 8)` over a 4-byte object leaves
6891    /// eight bytes, the last four zero).
6892    pub fn truncate_large_object(&mut self, oid: u32, len: usize) -> Result<(), String> {
6893        let Some(buf) = self.large_objects.get_mut(&oid) else {
6894            return Err(format!("large object {oid} does not exist"));
6895        };
6896        buf.resize(len, 0);
6897        Ok(())
6898    }
6899
6900    /// Remove a large object. `false` when the OID was not there.
6901    pub fn unlink_large_object(&mut self, oid: u32) -> bool {
6902        self.large_objects.remove(&oid).is_some()
6903    }
6904
6905    /// The next free OID in PG's user band.
6906    /// v7.39 (round 343, V40) — large objects have their own oid band.
6907    /// It used to start at 16_384, which is where user TABLES start, so
6908    /// the first large object and the first table shared an oid — and
6909    /// `pg_largeobject_metadata.oid` is joinable against `pg_class.oid`,
6910    /// so a join across them matched a row that has nothing to do with
6911    /// it. (PG cannot collide: every oid there comes off one counter.)
6912    /// An object already stored keeps the oid it was given; only new
6913    /// ones land in the band.
6914    fn next_large_object_oid(&self) -> u32 {
6915        self.large_objects
6916            .keys()
6917            .next_back()
6918            .map_or(500_000, |m| m.saturating_add(1))
6919    }
6920
6921    /// Register one. `Err(name)` when the name is taken.
6922    pub fn create_statistics_ext(&mut self, def: StatisticsExtDef) -> Result<(), String> {
6923        if self.statistics_ext.iter().any(|s| s.name == def.name) {
6924            return Err(def.name);
6925        }
6926        self.statistics_ext.push(def);
6927        Ok(())
6928    }
6929
6930    /// Drop one by name; false when absent.
6931    pub fn drop_statistics_ext(&mut self, name: &str) -> bool {
6932        let before = self.statistics_ext.len();
6933        self.statistics_ext.retain(|s| s.name != name);
6934        before != self.statistics_ext.len()
6935    }
6936
6937    /// v7.39 (round 139) — register a RULE. Its target relation (table or view)
6938    /// must exist; `or_replace` overwrites a same-(name,table) rule.
6939    pub fn create_rule(&mut self, def: RuleDef, or_replace: bool) -> Result<(), StorageError> {
6940        if !self.by_name.contains_key(&def.table) && !self.views.contains_key(&def.table) {
6941            return Err(StorageError::TableNotFound {
6942                name: def.table.clone(),
6943            });
6944        }
6945        let dup = self
6946            .rules
6947            .iter()
6948            .position(|r| r.name == def.name && r.table == def.table);
6949        match (dup, or_replace) {
6950            (Some(_), false) => Err(StorageError::Corrupt(format!(
6951                "rule {:?} for relation {:?} already exists",
6952                def.name, def.table
6953            ))),
6954            (Some(i), true) => {
6955                self.rules[i] = def;
6956                Ok(())
6957            }
6958            (None, _) => {
6959                self.rules.push(def);
6960                Ok(())
6961            }
6962        }
6963    }
6964
6965    /// v7.39 (round 139) — drop a RULE by `(name, table)`.
6966    pub fn drop_rule(&mut self, name: &str, table: &str) -> bool {
6967        let before = self.rules.len();
6968        self.rules.retain(|r| !(r.name == name && r.table == table));
6969        before != self.rules.len()
6970    }
6971
6972    pub fn create_table(&mut self, schema: TableSchema) -> Result<(), StorageError> {
6973        if self.by_name.contains_key(&schema.name) {
6974            return Err(StorageError::DuplicateTable {
6975                name: schema.name.clone(),
6976            });
6977        }
6978        let idx = self.tables.len();
6979        let name = schema.name.clone();
6980        let mut t = Table::new(schema);
6981        // v7.38.18 (S2) — the table inherits the database's collation,
6982        // which is what its undeclared text columns compare under.
6983        t.set_db_collation(self.db_collation());
6984        self.tables.push(t);
6985        self.by_name.insert(name.clone(), idx);
6986        // v7.39 (round 496) — see `dirty_tables`.
6987        self.dirty_tables.insert(name);
6988        // v7.37.15 (Phase C.1) — stamp the new relation with a stable,
6989        // monotonic, never-reused RelId. Pre-increment so ids start at
6990        // 1 (0 = UNASSIGNED); a later DROP TABLE frees the slot but not
6991        // the id.
6992        self.next_rel_id += 1;
6993        let rid = row_header::RelId(self.next_rel_id);
6994        self.tables[idx].set_rel_id(rid);
6995        Ok(())
6996    }
6997
6998    /// v7.39 (round 436) — the session's temporary table of this name wins
6999    /// over a permanent one, as `pg_temp` does in PG's search path and as
7000    /// MySQL's TEMPORARY shadowing does. Every name → index resolution in
7001    /// this catalog goes through here.
7002    fn resolve_index(&self, name: &str) -> Option<usize> {
7003        if let Some(prefix) = &self.temp_prefix {
7004            let mut mangled = String::with_capacity(prefix.len() + name.len());
7005            mangled.push_str(prefix);
7006            mangled.push_str(name);
7007            if let Some(idx) = self.by_name.get(&mangled) {
7008                return Some(*idx);
7009            }
7010            if self.case_insensitive_names
7011                && let Some(idx) = self.index_ignoring_case(&mangled)
7012            {
7013                return Some(idx);
7014            }
7015        }
7016        if let Some(idx) = self.by_name.get(name) {
7017            return Some(*idx);
7018        }
7019        // v7.39.2 — a MySQL session finds the relation under any
7020        // spelling of its name.
7021        //
7022        // The lexer folds an unquoted identifier and leaves a backticked
7023        // one alone, so `CREATE TABLE MyTable` stored `mytable` while
7024        // ``SELECT 1 FROM `MyTable` `` looked for `MyTable` and found
7025        // nothing: the two spellings of one name were two tables.
7026        // `mysqldump` backticks every identifier, so a dump restored
7027        // here and an application that writes the name unquoted were
7028        // looking at different relations.
7029        //
7030        // This is MySQL's `lower_case_table_names = 1` — names compare
7031        // without case — which is what SPG has always half-done, and
7032        // what it now reports. Exact match first, so a catalog that
7033        // already holds two names differing only in case keeps
7034        // answering the way it did.
7035        //
7036        // PostgreSQL sessions never set this: `"MyTable"` and `mytable`
7037        // are two relations there, and the flag is off.
7038        if self.case_insensitive_names {
7039            return self.index_ignoring_case(name);
7040        }
7041        None
7042    }
7043
7044    /// The single relation whose name matches `name` without regard to
7045    /// case, or `None` when there is none — or more than one, which the
7046    /// exact lookup above has already failed to settle.
7047    fn index_ignoring_case(&self, name: &str) -> Option<usize> {
7048        let mut found = None;
7049        for (k, idx) in &self.by_name {
7050            if k.len() == name.len() && k.eq_ignore_ascii_case(name) {
7051                if found.is_some() {
7052                    return None;
7053                }
7054                found = Some(*idx);
7055            }
7056        }
7057        found
7058    }
7059
7060    /// v7.39.2 — does this session compare relation names without case?
7061    ///
7062    /// Per SESSION, and the catalog is shared, so the engine installs it
7063    /// the way it installs `temp_prefix`: on every session switch, into
7064    /// the main catalog and into every open transaction's shadow.
7065    pub fn set_case_insensitive_names(&mut self, on: bool) {
7066        self.case_insensitive_names = on;
7067    }
7068
7069    /// v7.39 (round 436) — install the calling session's temp namespace.
7070    /// `None` disables temp resolution entirely (a session that never made
7071    /// one pays a single `Option` check per lookup).
7072    pub fn set_temp_prefix(&mut self, prefix: Option<String>) {
7073        self.temp_prefix = prefix;
7074    }
7075
7076    /// The mangled storage name a temp table of `name` takes in this
7077    /// session, or `None` when the session has no temp namespace.
7078    #[must_use]
7079    pub fn temp_name_for(&self, name: &str) -> Option<String> {
7080        self.temp_prefix
7081            .as_ref()
7082            .map(|p| alloc::format!("{p}{name}"))
7083    }
7084
7085    pub fn get(&self, name: &str) -> Option<&Table> {
7086        let idx = self.resolve_index(name)?;
7087        self.tables.get(idx)
7088    }
7089
7090    pub fn get_mut(&mut self, name: &str) -> Option<&mut Table> {
7091        let idx = self.resolve_index(name)?;
7092        // v7.39 (round 496) — the choke point for changing a table, so the
7093        // record is taken here. Over-approximate on purpose: a caller that
7094        // takes the handle and writes nothing merely carries that table
7095        // through a commit, which is the old behaviour.
7096        let recorded = self.tables.get(idx).map(|t| t.schema().name.clone());
7097        if let Some(n) = recorded {
7098            self.dirty_tables.insert(n);
7099        }
7100        self.tables.get_mut(idx)
7101    }
7102
7103    /// v7.39 (round 496) — the tables changed through this handle since
7104    /// [`Self::clear_dirty_tables`]. See `dirty_tables`.
7105    #[must_use]
7106    pub fn dirty_tables(&self) -> &alloc::collections::BTreeSet<String> {
7107        &self.dirty_tables
7108    }
7109
7110    /// r1059 — mark one table dirty without taking its handle. The
7111    /// rebase/merge paths replace a tx's shadow with a fresh base
7112    /// clone and must carry the tx's OWN dirty window across (the
7113    /// base's set is an ever-growing history, never cleared).
7114    pub fn mark_table_dirty(&mut self, name: &str) {
7115        self.dirty_tables.insert(name.into());
7116    }
7117
7118    /// v7.39 (round 496) — start a fresh recording window. A transaction's
7119    /// shadow calls this at BEGIN so the set means "changed by this tx".
7120    /// 7.38.1 S3.1 — one window covers both records (tables and the
7121    /// non-table families).
7122    pub fn clear_dirty_tables(&mut self) {
7123        self.dirty_tables.clear();
7124        self.dirty_nontable.clear();
7125    }
7126
7127    /// 7.38.1 S3.1 (D4) — record a non-table object as changed by this
7128    /// window. Called from every create/alter/rename/drop of the six
7129    /// [`NonTableKind`] families; a rename records BOTH names.
7130    fn mark_nontable_dirty(&mut self, kind: NonTableKind, name: &str) {
7131        self.dirty_nontable.insert((kind, name.into()));
7132    }
7133
7134    /// 7.38.1 S3.1 (D4) — reconcile the six non-table families with
7135    /// `base` (the latest committed catalog): every entry this window
7136    /// did NOT touch is taken from base — existence, definition and
7137    /// absence alike — so a neighbour's CREATE / ALTER / DROP of a
7138    /// sequence, view, matview, enum, domain or composite type
7139    /// survives a poisoned transaction's COMMIT. Entries this window
7140    /// DID touch keep the shadow's version (the tx's own DDL wins its
7141    /// own objects, exactly like the dirty-table merge above it).
7142    pub fn merge_nontable_objects_from(&mut self, base: &Catalog) {
7143        use NonTableKind as K;
7144        fn merge_map<V: Clone>(
7145            kind: NonTableKind,
7146            dirty: &alloc::collections::BTreeSet<(NonTableKind, String)>,
7147            mine: &mut BTreeMap<String, V>,
7148            theirs: &BTreeMap<String, V>,
7149        ) {
7150            let names: alloc::vec::Vec<String> =
7151                mine.keys().chain(theirs.keys()).cloned().collect();
7152            for n in names {
7153                if dirty.contains(&(kind, n.clone())) {
7154                    continue;
7155                }
7156                match theirs.get(&n) {
7157                    Some(v) => {
7158                        mine.insert(n, v.clone());
7159                    }
7160                    None => {
7161                        mine.remove(&n);
7162                    }
7163                }
7164            }
7165        }
7166        let dirty = self.dirty_nontable.clone();
7167        merge_map(K::Sequence, &dirty, &mut self.sequences, &base.sequences);
7168        merge_map(K::View, &dirty, &mut self.views, &base.views);
7169        merge_map(
7170            K::MaterializedView,
7171            &dirty,
7172            &mut self.materialized_views,
7173            &base.materialized_views,
7174        );
7175        merge_map(K::EnumType, &dirty, &mut self.enum_types, &base.enum_types);
7176        merge_map(
7177            K::DomainType,
7178            &dirty,
7179            &mut self.domain_types,
7180            &base.domain_types,
7181        );
7182        merge_map(
7183            K::CompositeType,
7184            &dirty,
7185            &mut self.composite_types,
7186            &base.composite_types,
7187        );
7188    }
7189
7190    /// v7.39 (round 496) — put `table` in at `name`, replacing any table
7191    /// already there and keeping the rest of the catalog untouched.
7192    ///
7193    /// The commit-time table-granularity merge needs exactly this: take
7194    /// the latest committed catalog, then overwrite only the tables the
7195    /// transaction changed.
7196    pub fn install_table(&mut self, name: &str, table: Table) {
7197        match self.by_name.get(name).copied() {
7198            Some(idx) => self.tables[idx] = table,
7199            None => {
7200                let idx = self.tables.len();
7201                self.tables.push(table);
7202                self.by_name.insert(name.into(), idx);
7203            }
7204        }
7205        self.dirty_tables.insert(name.into());
7206    }
7207
7208    /// v7.37.42 (docker-fair SCALARSQ attack) — resolve a table name to
7209    /// its insertion-order index ONCE, so callers that need to fetch the
7210    /// same table many times (per-row PK probes in correlated scalar
7211    /// subqueries) can avoid the per-call `BTreeMap<String, usize>` string
7212    /// descent. The returned index is stable for the lifetime of the
7213    /// catalog snapshot the caller holds (same engine read guard).
7214    pub fn tables_position_of(&self, name: &str) -> Option<usize> {
7215        self.resolve_index(name)
7216    }
7217
7218    /// Direct positional fetch counterpart to [`tables_position_of`].
7219    /// `idx` must come from `tables_position_of` against the same catalog
7220    /// snapshot — out-of-range returns `None`.
7221    pub fn tables_at(&self, idx: usize) -> Option<&Table> {
7222        self.tables.get(idx)
7223    }
7224
7225    /// v7.34 (crash-recovery P0 #2) — replay a row-level redo log onto
7226    /// this catalog (the [`RowChange`] physical-redo apply primitive that
7227    /// row-level WAL recovery will use in place of statement re-execution).
7228    /// Applies each change in order via the same `Table` mutators the
7229    /// engine used — no uniqueness/FK/parse/plan: the original execution
7230    /// already validated, replay trusts and applies. Positions are
7231    /// physical and only valid when replayed from the matching checkpoint
7232    /// baseline in original order (see [`RowChange`] docs).
7233    ///
7234    /// A change naming an absent table, or whose position is out of range,
7235    /// is a corrupt/misaligned log and surfaces as an error rather than a
7236    /// silent skip.
7237    pub fn apply_redo(&mut self, changes: &[RowChange]) -> Result<(), StorageError> {
7238        // v7.37.5 (mailrs crash-recovery Ask 3) — true batched replay.
7239        // Pre-v7.37.5 each `RowChange::Delete` record ran a fresh
7240        // O(N) PersistentVec rebuild + O(N × indices × log N)
7241        // `rebuild_indices()` — 5000 records × 100k rows × 13 indices
7242        // ≈ 27 min on the mailrs prod-shape WAL.
7243        //
7244        // The strategy: group consecutive changes by table, and for
7245        // each run, compose all the row-level mutations through a
7246        // single "live" tracking vector + a per-table operation log,
7247        // then apply rows + indices ONCE at the end. The result:
7248        //  - DELETE blow-up: O(records × rows × indices × log rows)
7249        //    → O(rows × indices × log rows) — one rebuild per run.
7250        //  - Row-position semantics preserved: positions in a later
7251        //    `Delete` / `Update` record reference the layout produced
7252        //    by every earlier change; we walk the live-vector
7253        //    forward as each change is processed so positions
7254        //    translate correctly to the ORIGINAL row index space.
7255        //
7256        // For correctness, even with this batching `apply_redo`
7257        // remains in-order: a single per-table run only batches
7258        // a contiguous slice of changes targeting that table; a
7259        // mid-run change targeting a DIFFERENT table forces a
7260        // flush of the current run.
7261        let mut runs: alloc::vec::Vec<(String, alloc::vec::Vec<&RowChange>)> =
7262            alloc::vec::Vec::new();
7263        for change in changes {
7264            // v7.39 (flip crash-replay P0) — a replayed tombstone carries
7265            // the xmax the CRASHED process allocated, but this process's
7266            // version cursor restarted; without advancing it past every
7267            // replayed version, `Snapshot::visible`'s "deletion is in the
7268            // future" branch (xmax > snapshot.version) resurrects every
7269            // replayed delete. Same recovery contract as the snapshot
7270            // loader (`observe_persisted_version`, the pg_control-style
7271            // nextXid recovery).
7272            if let RowChange::Tombstone { xmax, .. } = change {
7273                row_header::observe_persisted_version(*xmax);
7274            }
7275            let table = match change {
7276                RowChange::Insert { table, .. }
7277                | RowChange::Update { table, .. }
7278                | RowChange::Delete { table, .. }
7279                | RowChange::Tombstone { table, .. } => table.clone(),
7280            };
7281            if runs.last().map(|(t, _)| t.as_str()) != Some(table.as_str()) {
7282                runs.push((table, alloc::vec::Vec::new()));
7283            }
7284            runs.last_mut().unwrap().1.push(change);
7285        }
7286        for (table_name, run) in runs {
7287            self.apply_redo_run_on_table(&table_name, &run)?;
7288        }
7289        Ok(())
7290    }
7291
7292    /// v7.37.5 — apply a contiguous slice of `RowChange`s all
7293    /// targeting the same `table_name`. Composes row mutations
7294    /// through a single live-tracking vector + a single tail
7295    /// for appended `Insert`s + a single in-place edit set for
7296    /// `Update`s, then writes the final row layout to
7297    /// `self.rows` and rebuilds indices ONCE.
7298    fn apply_redo_run_on_table(
7299        &mut self,
7300        table_name: &str,
7301        run: &[&RowChange],
7302    ) -> Result<(), StorageError> {
7303        // Look up the table once; the unchecked unwrap is safe
7304        // because the caller just resolved `table_name` for each
7305        // change.
7306        let table = self.get_mut(table_name).ok_or_else(|| {
7307            StorageError::Corrupt(alloc::format!("redo: unknown table {table_name:?}"))
7308        })?;
7309        // Live-tracking over both pre-existing rows and tail-
7310        // appended Insert rows. `live[i] = true` initially for
7311        // every existing row. Appended Inserts extend with `true`.
7312        // A `Delete` flips entries to `false` (using the position
7313        // mapping that walks live indices in order). An `Update`
7314        // edits in place — collected into an overlay map keyed by
7315        // ORIGINAL row position so later Updates win.
7316        let original_rows: alloc::vec::Vec<Row<'static>> = table.rows().iter().cloned().collect();
7317        let mut live: alloc::vec::Vec<bool> = alloc::vec![true; original_rows.len()];
7318        let mut tail: alloc::vec::Vec<Row<'static>> = alloc::vec::Vec::new();
7319        // Overlay: index into ORIGINAL row space (existing rows
7320        // 0..original_rows.len()) or into tail (offset
7321        // original_rows.len()). Map -> new values.
7322        let mut overlay: alloc::collections::BTreeMap<usize, alloc::vec::Vec<Value<'static>>> =
7323            alloc::collections::BTreeMap::new();
7324        // v7.37.15 (Epic W durable-tombstone slice) — extra bookkeeping
7325        // ONLY when this run actually carries an in-place `Tombstone`.
7326        // A tombstone keeps its row physically present but stamps `xmax`
7327        // on the header; the run finalizer `set_rows_and_rebuild_indices`
7328        // freezes every header (and reassigns ids), so we must re-stamp
7329        // in a post-pass keyed by RowId. When the run has no tombstone
7330        // (every default gate-off replay) this is all skipped and the
7331        // path below stays byte-for-byte the legacy one.
7332        let has_tomb = run.iter().any(|c| matches!(c, RowChange::Tombstone { .. }));
7333        // Ids of the pre-existing rows, snapshotted parallel to
7334        // `original_rows`, and ids of the tail rows filled from each
7335        // `Insert`'s carried `rowid`. Together they let a tombstone name
7336        // the exact row the writer stamped, independent of the ids the
7337        // finalizer will hand out. (When `!has_tomb`, both stay empty.)
7338        // v7.39 (flip crash-replay P0) — ids are tracked UNCONDITIONALLY
7339        // now: the finalizer preserves them so a later WAL record's
7340        // tombstone can still name rows this record produced.
7341        let orig_rowids: alloc::vec::Vec<row_header::RowId> =
7342            table.rowids().iter().copied().collect();
7343        // Headers snapshotted in lock-step: the finalizer preserves
7344        // them so earlier records' tombstone stamps survive.
7345        let orig_headers: alloc::vec::Vec<row_header::RowHeader> =
7346            table.headers().iter().copied().collect();
7347        let mut tail_rowids: alloc::vec::Vec<row_header::RowId> = alloc::vec::Vec::new();
7348        // (RowId, xmax) of every row this run tombstones.
7349        let mut tomb_targets: alloc::vec::Vec<(row_header::RowId, u64)> = alloc::vec::Vec::new();
7350        // Helper: given a "current" position (i.e. position in
7351        // the post-prior-deletes layout), translate to the
7352        // ABSOLUTE position in the unified live + tail space
7353        // by walking the live vector + tail. Returns None when
7354        // the position is out of range.
7355        fn translate(live: &[bool], tail_len: usize, current_pos: usize) -> Option<usize> {
7356            // Walk live[..] counting live entries until we hit
7357            // current_pos. Then if not yet matched, dip into tail.
7358            let mut seen = 0usize;
7359            for (i, &alive) in live.iter().enumerate() {
7360                if alive {
7361                    if seen == current_pos {
7362                        return Some(i);
7363                    }
7364                    seen += 1;
7365                }
7366            }
7367            // Position lives in tail. tail_len rows in the tail
7368            // are all live (we haven't deleted any tail rows in
7369            // this simplification; if we did, we'd extend `live`).
7370            let off = current_pos - seen;
7371            if off < tail_len {
7372                Some(live.len() + off)
7373            } else {
7374                None
7375            }
7376        }
7377        for change in run {
7378            match *change {
7379                RowChange::Insert { row, rowid, .. } => {
7380                    // Validate against schema before recording the
7381                    // change so a corrupt log surfaces as an error
7382                    // rather than silently mis-applying.
7383                    if row.len() != table.schema().columns.len() {
7384                        return Err(StorageError::ArityMismatch {
7385                            expected: table.schema().columns.len(),
7386                            actual: row.len(),
7387                        });
7388                    }
7389                    tail.push(row.clone());
7390                    // Keep the id lock-step with `tail` so a later
7391                    // tombstone (this run or a later WAL record) can
7392                    // find the row by the id the writer captured.
7393                    tail_rowids.push(*rowid);
7394                }
7395                RowChange::Update { pos, new_row, .. } => {
7396                    if new_row.len() != table.schema().columns.len() {
7397                        return Err(StorageError::ArityMismatch {
7398                            expected: table.schema().columns.len(),
7399                            actual: new_row.len(),
7400                        });
7401                    }
7402                    let abs = translate(&live, tail.len(), *pos).ok_or_else(|| {
7403                        StorageError::Corrupt(alloc::format!(
7404                            "redo: update_row position {pos} out of bounds in table {table_name:?}",
7405                        ))
7406                    })?;
7407                    // Tail edits are applied directly to `tail`
7408                    // (we own it); existing-row edits land in
7409                    // the overlay map keyed by original index.
7410                    if abs < live.len() {
7411                        overlay.insert(abs, new_row.clone());
7412                    } else {
7413                        tail[abs - live.len()] = Row::new(new_row.clone());
7414                    }
7415                }
7416                RowChange::Delete { positions, .. } => {
7417                    // De-dup + sort so the translate walk stays
7418                    // monotone (the second translate doesn't have
7419                    // to redo work the first one did, in principle;
7420                    // we keep it simple here and re-walk per
7421                    // position). Bounds-filter silently mirrors
7422                    // `Table::delete_rows`.
7423                    let mut sorted: alloc::vec::Vec<usize> = positions.clone();
7424                    sorted.sort_unstable();
7425                    sorted.dedup();
7426                    // Walk live[] once per Delete record to
7427                    // translate all positions in this record's
7428                    // post-prior-deletes layout to absolute
7429                    // indices. We MUST defer the live[] flip
7430                    // until after all positions are translated
7431                    // so two positions in the same record
7432                    // (e.g. [3, 7]) reference the same layout.
7433                    let mut to_flip_live: alloc::vec::Vec<usize> = alloc::vec::Vec::new();
7434                    let mut to_flip_tail: alloc::vec::Vec<usize> = alloc::vec::Vec::new();
7435                    // Two-pointer walk: live[i] scanned monotonically,
7436                    // sorted positions consumed in order.
7437                    let mut seen = 0usize;
7438                    let mut sp = sorted.iter().peekable();
7439                    for (i, &alive) in live.iter().enumerate() {
7440                        if !alive {
7441                            continue;
7442                        }
7443                        while let Some(&&p) = sp.peek() {
7444                            if seen == p {
7445                                to_flip_live.push(i);
7446                                sp.next();
7447                            } else {
7448                                break;
7449                            }
7450                        }
7451                        if sp.peek().is_none() {
7452                            break;
7453                        }
7454                        seen += 1;
7455                    }
7456                    // Remaining positions fall into the tail.
7457                    for &p in sp {
7458                        // p >= seen and refers to the (p - seen)-th
7459                        // entry in tail. Filter out-of-bounds.
7460                        let off = p - seen;
7461                        if off < tail.len() {
7462                            to_flip_tail.push(off);
7463                        }
7464                    }
7465                    for i in to_flip_live {
7466                        live[i] = false;
7467                        // Any pending overlay edit for this
7468                        // index is moot — the row is gone.
7469                        overlay.remove(&i);
7470                    }
7471                    // Tail deletes: remove in REVERSE order so
7472                    // shifting indices stay valid.
7473                    to_flip_tail.sort_unstable();
7474                    to_flip_tail.dedup();
7475                    for off in to_flip_tail.into_iter().rev() {
7476                        tail.remove(off);
7477                        {
7478                            // Keep the id vector lock-step with `tail`.
7479                            tail_rowids.remove(off);
7480                        }
7481                        // Re-key tail-relative overlay entries that
7482                        // were past `off` — in practice tail edits
7483                        // are applied directly so the overlay map
7484                        // only holds existing-row keys; nothing to
7485                        // do here.
7486                    }
7487                }
7488                RowChange::Tombstone { rowids, xmax, .. } => {
7489                    // An in-place tombstone leaves the row physically
7490                    // present — it does not touch `live` / `tail` /
7491                    // `overlay`. Record the (id, xmax) targets; the
7492                    // post-finalizer pass re-stamps `xmax` onto the
7493                    // matching row's (otherwise-frozen) header.
7494                    for rid in rowids {
7495                        tomb_targets.push((*rid, *xmax));
7496                    }
7497                }
7498            }
7499        }
7500        // Compose the final row layout: keep existing rows where
7501        // live[i] = true, applying overlay edits in place; then
7502        // append the surviving tail.
7503        let mut new_rows: PersistentVec<Row> = PersistentVec::new();
7504        let mut new_hot_bytes: u64 = 0;
7505        let schema_snapshot = table.schema().clone();
7506        // Parallel to `new_rows` (only built when `has_tomb`): the RowId
7507        // of each row in its FINAL slot, so the post-pass can map a
7508        // tombstone target id → the slot to re-stamp `xmax` on.
7509        let mut final_rowids: alloc::vec::Vec<row_header::RowId> = alloc::vec::Vec::new();
7510        let mut final_headers: alloc::vec::Vec<row_header::RowHeader> = alloc::vec::Vec::new();
7511        for (i, row) in original_rows.into_iter().enumerate() {
7512            if !live[i] {
7513                continue;
7514            }
7515            let final_row = if let Some(new_values) = overlay.remove(&i) {
7516                Row::new(new_values)
7517            } else {
7518                row
7519            };
7520            new_hot_bytes = new_hot_bytes
7521                .saturating_add(row_body_encoded_len(&final_row, &schema_snapshot) as u64);
7522            new_rows.push_mut(final_row);
7523            final_rowids.push(
7524                orig_rowids
7525                    .get(i)
7526                    .copied()
7527                    .unwrap_or(row_header::RowId::UNASSIGNED),
7528            );
7529            final_headers.push(
7530                orig_headers
7531                    .get(i)
7532                    .copied()
7533                    .unwrap_or_else(row_header::RowHeader::frozen),
7534            );
7535        }
7536        for (off, row) in tail.into_iter().enumerate() {
7537            new_hot_bytes =
7538                new_hot_bytes.saturating_add(row_body_encoded_len(&row, &schema_snapshot) as u64);
7539            new_rows.push_mut(row);
7540            final_rowids.push(
7541                tail_rowids
7542                    .get(off)
7543                    .copied()
7544                    .unwrap_or(row_header::RowId::UNASSIGNED),
7545            );
7546            final_headers.push(row_header::RowHeader::frozen());
7547        }
7548        // v7.39 (flip crash-replay P0) — id-preserving finalizer, so a
7549        // LATER WAL record's tombstone still resolves rows this record
7550        // produced (per-statement replay used to reassign ids between
7551        // records, orphaning every cross-record tombstone target).
7552        table.set_rows_and_rebuild_indices_with_rowids(
7553            new_rows,
7554            new_hot_bytes,
7555            &final_rowids,
7556            &final_headers,
7557        );
7558        // v7.37.15 (Epic W durable-tombstone slice) — header-preserving
7559        // re-stamp. `set_rows_and_rebuild_indices` above froze every
7560        // header, so any row this run tombstoned is currently all-
7561        // visible again. Re-apply the `xmax` stamp by matching the
7562        // tombstone's target RowId against the final-slot id map. This
7563        // is what makes a gate-on DELETE durable across replay without
7564        // changing the on-disk snapshot format (headers/ids are still
7565        // NOT serialised — that is the deferred V6 coupling; see below).
7566        if has_tomb && !tomb_targets.is_empty() {
7567            let mut id_to_slot: alloc::collections::BTreeMap<row_header::RowId, usize> =
7568                alloc::collections::BTreeMap::new();
7569            for (slot, rid) in final_rowids.iter().enumerate() {
7570                if *rid != row_header::RowId::UNASSIGNED {
7571                    id_to_slot.insert(*rid, slot);
7572                }
7573            }
7574            let table = self.get_mut(table_name).ok_or_else(|| {
7575                StorageError::Corrupt(alloc::format!("redo: unknown table {table_name:?}"))
7576            })?;
7577            for (rid, xmax) in &tomb_targets {
7578                match id_to_slot.get(rid) {
7579                    Some(&slot) => {
7580                        // First-deleter-wins + bounds handled inside.
7581                        let _ = table.mark_row_deleted(slot, *xmax);
7582                    }
7583                    None => {
7584                        // The target row was not produced by THIS redo
7585                        // run and its id was not in the run-start
7586                        // snapshot — the documented cross-checkpoint
7587                        // limitation: after a checkpoint restore the
7588                        // table's ids are reassigned (not yet persisted
7589                        // in the envelope), so a tombstone naming a
7590                        // pre-checkpoint row cannot be resolved by id.
7591                        // Skipping leaves the row visible (identical to
7592                        // the pre-Epic-W non-durable behaviour); it is
7593                        // never a correctness regression, only an
7594                        // unclosed durability gap the V6 envelope slice
7595                        // closes. Counted for observability.
7596                        UNRESOLVED_TOMBSTONES.fetch_add(1, core::sync::atomic::Ordering::Relaxed);
7597                    }
7598                }
7599            }
7600        }
7601        Ok(())
7602    }
7603
7604    fn table_for_redo(&mut self, name: &str) -> Result<&mut Table, StorageError> {
7605        self.get_mut(name)
7606            .ok_or_else(|| StorageError::Corrupt(alloc::format!("redo: unknown table {name:?}")))
7607    }
7608
7609    /// v7.34 (crash-recovery P0 #2) — enable row-level redo capture on
7610    /// every table (the engine calls this before a mutating statement
7611    /// when persistence is on; idempotent, keeps any in-flight capture).
7612    pub fn enable_redo_all(&mut self) {
7613        for t in &mut self.tables {
7614            t.enable_redo();
7615        }
7616    }
7617
7618    /// v7.34 — drain the row-level redo captured across all tables, in
7619    /// table order then per-table apply order, and stop capturing. The
7620    /// engine calls this after a successful mutating statement and writes
7621    /// the returned [`RowChange`]s to the WAL in place of the SQL text.
7622    pub fn drain_redo(&mut self) -> Vec<RowChange> {
7623        let mut all = Vec::new();
7624        for t in &mut self.tables {
7625            all.extend(t.take_redo());
7626        }
7627        all
7628    }
7629
7630    pub fn table_count(&self) -> usize {
7631        self.tables.len()
7632    }
7633
7634    /// v7.14.0 — remove a table by name. Returns `true` when the
7635    /// table existed (and is now gone), `false` when it didn't.
7636    /// Used by `DROP TABLE` from pg_dump / mysqldump preambles
7637    /// where the dump re-creates schema and starts with
7638    /// `DROP TABLE IF EXISTS`.
7639    pub fn drop_table(&mut self, name: &str) -> bool {
7640        // v7.39 (round 436) — resolve through the session's temp namespace
7641        // first, exactly as a read would: MariaDB's plain `DROP TABLE tmp`
7642        // drops the TEMPORARY one and leaves a permanent namesake standing
7643        // (measured). Removing by the raw name would have dropped the
7644        // permanent table out from under every other session.
7645        let key = match self.temp_prefix.as_ref() {
7646            Some(p) => {
7647                let mangled = alloc::format!("{p}{name}");
7648                if self.by_name.contains_key(&mangled) {
7649                    mangled
7650                } else {
7651                    name.into()
7652                }
7653            }
7654            None => name.into(),
7655        };
7656        let Some(idx) = self.by_name.remove(&key) else {
7657            return false;
7658        };
7659        // v7.39 (round 496) — see `dirty_tables`. Recorded under the
7660        // RESOLVED key, which is what a commit-time merge looks up.
7661        self.dirty_tables.insert(key.clone());
7662        // swap_remove invalidates the trailing index → rebuild
7663        // by_name for affected entries.
7664        self.tables.swap_remove(idx);
7665        // Re-stamp moved table's index slot in by_name.
7666        if idx < self.tables.len() {
7667            let moved_name = self.tables[idx].schema.name.clone();
7668            self.by_name.insert(moved_name, idx);
7669        }
7670        true
7671    }
7672
7673    /// v7.16.2 — rename a table (mailrs round-10 A.5). Updates
7674    /// the schema name, the catalog name → index map, and
7675    /// rewrites every reference dangling at the table name:
7676    ///   * every FK on every OTHER table whose `parent_table`
7677    ///     pointed at the old name now points at the new
7678    ///     name, so FK enforcement keeps working
7679    ///   * every trigger watching the table updates its `table`
7680    ///     field
7681    /// Returns `Ok` on success; `Err(StorageError::TableNotFound)`
7682    /// when the old name isn't in the catalog and
7683    /// `Err(StorageError::DuplicateTable)` when the new name is
7684    /// already taken.
7685    pub fn rename_table(&mut self, old: &str, new: &str) -> Result<(), StorageError> {
7686        if old == new {
7687            return Ok(());
7688        }
7689        if self.by_name.contains_key(new) {
7690            return Err(StorageError::Corrupt(format!(
7691                "rename_table: target name {new:?} already exists"
7692            )));
7693        }
7694        let idx = self
7695            .by_name
7696            .remove(old)
7697            .ok_or_else(|| StorageError::TableNotFound { name: old.into() })?;
7698        self.tables[idx].schema.name = new.to_string();
7699        self.by_name.insert(new.to_string(), idx);
7700        for t in &mut self.tables {
7701            for fk in &mut t.schema.foreign_keys {
7702                if fk.parent_table == old {
7703                    fk.parent_table = new.to_string();
7704                }
7705            }
7706        }
7707        for trig in &mut self.triggers {
7708            if trig.table == old {
7709                trig.table = new.to_string();
7710            }
7711        }
7712        Ok(())
7713    }
7714
7715    /// v7.16.2 — rename an index by name. Walks every table
7716    /// since the index lives on its owning table; updates the
7717    /// name in place. Errors with `IndexNotFound` when no
7718    /// index matches. mailrs round-10 A.5.
7719    pub fn rename_index(&mut self, old: &str, new: &str) -> Result<(), StorageError> {
7720        if old == new {
7721            return Ok(());
7722        }
7723        // Reject the new name if it already exists anywhere.
7724        for t in &self.tables {
7725            if t.indices.iter().any(|i| i.name == new) {
7726                return Err(StorageError::Corrupt(format!(
7727                    "rename_index: target name {new:?} already exists"
7728                )));
7729            }
7730        }
7731        for t in &mut self.tables {
7732            for i in &mut t.indices {
7733                if i.name == old {
7734                    i.name = new.to_string();
7735                    return Ok(());
7736                }
7737            }
7738        }
7739        Err(StorageError::IndexNotFound { name: old.into() })
7740    }
7741
7742    /// v7.14.0 — remove a named index across the catalog.
7743    /// Returns `true` when found + dropped.
7744    pub fn drop_named_index(&mut self, name: &str) -> bool {
7745        for t in &mut self.tables {
7746            let before = t.indices.len();
7747            t.indices.retain(|i| i.name != name);
7748            if t.indices.len() != before {
7749                return true;
7750            }
7751        }
7752        false
7753    }
7754
7755    /// Borrow-free copy of every table's name in catalog order
7756    /// (= insertion order, matching the on-disk encoding).
7757    pub fn table_names(&self) -> Vec<String> {
7758        self.tables.iter().map(|t| t.schema.name.clone()).collect()
7759    }
7760
7761    /// v7.39 (round 436) — the marker every session's temporary-table
7762    /// namespace starts with. Public so the catalog synths can tell a
7763    /// temp table from an ordinary one without knowing the session id.
7764    pub const TEMP_NAME_MARKER: &'static str = "__spg_temp_";
7765
7766    /// v7.39 (round 437) — how a stored table name should appear to the
7767    /// CALLING session in a catalog listing (SHOW TABLES, pg_class,
7768    /// information_schema, …):
7769    ///   * an ordinary table → its own name
7770    ///   * this session's temporary table → its logical name, prefix stripped
7771    ///   * another session's temporary table → `None`, i.e. not listed
7772    ///
7773    /// Measured on both oracles: MariaDB 11 and PG 18 each list the calling
7774    /// session's own temporary tables and neither lists anybody else's.
7775    /// Round 436 stored temp tables under a prefix without teaching the
7776    /// listings about it, so the mangled names leaked to every client.
7777    #[must_use]
7778    pub fn listed_name<'a>(&self, stored: &'a str) -> Option<&'a str> {
7779        if !stored.starts_with(Self::TEMP_NAME_MARKER) {
7780            return Some(stored);
7781        }
7782        let prefix = self.temp_prefix.as_ref()?;
7783        stored.strip_prefix(prefix.as_str())
7784    }
7785
7786    /// The listing names of every table this session may see, in catalog
7787    /// order. See [`Catalog::listed_name`].
7788    #[must_use]
7789    pub fn visible_table_names(&self) -> Vec<String> {
7790        self.tables
7791            .iter()
7792            .filter_map(|t| self.listed_name(&t.schema.name).map(String::from))
7793            .collect()
7794    }
7795
7796    /// v5.1: register a cold-tier segment that already lives in
7797    /// memory (caller did the file read). Returns the
7798    /// `segment_id` that `RowLocator::Cold { segment_id, .. }`
7799    /// will reference — currently this is just the index into
7800    /// `cold_segments`, but treat it as an opaque token.
7801    ///
7802    /// Storage is `no_std`, so file I/O is the caller's
7803    /// responsibility — `spg-server` reads the file and forwards
7804    /// the bytes here. The bytes stay resident in the catalog
7805    /// for the life of the `Catalog`, parsed only once.
7806    pub fn load_segment_bytes(&mut self, bytes: Vec<u8>) -> Result<u32, StorageError> {
7807        let id = u32::try_from(self.cold_segments.len()).map_err(|_| {
7808            StorageError::Corrupt("cold segment count would exceed u32::MAX".into())
7809        })?;
7810        let seg = OwnedSegment::from_bytes(bytes)
7811            .map_err(|e| StorageError::Corrupt(format!("cold segment parse failed: {e}")))?;
7812        self.cold_segments.push(Some(Arc::new(seg)));
7813        Ok(id)
7814    }
7815
7816    /// v6.7.3 — register a cold-tier segment at a specific id. Used
7817    /// by the spg-server manifest-boot path so segments whose
7818    /// neighbouring ids were retired by compaction still get back
7819    /// the same `segment_id` they had pre-restart (the
7820    /// `RowLocator::Cold { segment_id }` baked into the BTree-index
7821    /// snapshot persists across restart and must continue to
7822    /// resolve).
7823    ///
7824    /// Pads the Vec with `None` slots up to `target_id` if needed.
7825    /// Errors when the target slot is already occupied (would
7826    /// stomp another segment), the parse fails, or `target_id`
7827    /// exceeds `u32::MAX`.
7828    pub fn load_segment_bytes_at(
7829        &mut self,
7830        target_id: u32,
7831        bytes: Vec<u8>,
7832    ) -> Result<(), StorageError> {
7833        let seg = OwnedSegment::from_bytes(bytes)
7834            .map_err(|e| StorageError::Corrupt(format!("cold segment parse failed: {e}")))?;
7835        let idx = target_id as usize;
7836        while self.cold_segments.len() <= idx {
7837            self.cold_segments.push(None);
7838        }
7839        if self.cold_segments[idx].is_some() {
7840            return Err(StorageError::Corrupt(format!(
7841                "load_segment_bytes_at: segment_id {target_id} already occupied"
7842            )));
7843        }
7844        self.cold_segments[idx] = Some(Arc::new(seg));
7845        Ok(())
7846    }
7847
7848    /// v6.7.3 — retire a cold-tier segment slot (compaction-driven).
7849    /// The physical file is the caller's concern (typically kept
7850    /// on disk until the next CHECKPOINT writes a manifest that
7851    /// no longer lists it); this just flips the in-memory slot
7852    /// to `None` so later cold lookups for `segment_id` resolve
7853    /// as "unknown" instead of returning a stale row.
7854    ///
7855    /// No-op when the slot is already `None`. Errors only when
7856    /// `segment_id` is out of bounds.
7857    pub fn tombstone_segment(&mut self, segment_id: u32) -> Result<(), StorageError> {
7858        let idx = segment_id as usize;
7859        if idx >= self.cold_segments.len() {
7860            return Err(StorageError::Corrupt(format!(
7861                "tombstone_segment: segment_id {segment_id} out of bounds (len={})",
7862                self.cold_segments.len()
7863            )));
7864        }
7865        self.cold_segments[idx] = None;
7866        Ok(())
7867    }
7868
7869    /// Number of *active* (non-tombstoned) cold segments.
7870    #[must_use]
7871    pub fn cold_segment_count(&self) -> usize {
7872        self.cold_segments.iter().filter(|s| s.is_some()).count()
7873    }
7874
7875    /// v7.37.42 (docker-fair SCALARSQ attack 3) — short-circuit guard
7876    /// for scan loops that conditionally walk the cold tier. Returns
7877    /// `false` when the catalog has never loaded a cold segment (or all
7878    /// segments are tombstoned), so callers can skip the per-table cold
7879    /// PK-index walk entirely on hot-only databases. O(N segments);
7880    /// typical N is small (single-digit) so the check is sub-µs.
7881    #[must_use]
7882    pub fn has_any_cold_segments(&self) -> bool {
7883        self.cold_segments.iter().any(Option::is_some)
7884    }
7885
7886    /// Slot count including tombstones (= the next id the
7887    /// no-arg `load_segment_bytes` would allocate).
7888    #[must_use]
7889    pub fn cold_segment_slot_count(&self) -> usize {
7890        self.cold_segments.len()
7891    }
7892
7893    /// v6.2.7 — list every *active* cold-tier segment id known to
7894    /// this catalog (skips compaction tombstones since v6.7.3).
7895    /// Used by EXPLAIN ANALYZE to annotate scan nodes with the
7896    /// segments they could have walked.
7897    #[must_use]
7898    pub fn cold_segment_ids_global(&self) -> Vec<u32> {
7899        self.cold_segments
7900            .iter()
7901            .enumerate()
7902            .filter_map(|(i, s)| s.as_ref().map(|_| i as u32))
7903            .collect()
7904    }
7905
7906    /// v5.2.1: sum of `Table::hot_bytes` across every table. The v5.2
7907    /// freezer compares this against `SPG_HOT_TIER_BYTES` (parsed at
7908    /// server startup; default 4 GiB) and wakes when the budget is
7909    /// crossed. Pre-freezer (v5.2.1) this is measurement-only — the
7910    /// counter exposes whether the budget is being approached without
7911    /// triggering any demotion.
7912    #[must_use]
7913    pub fn hot_tier_bytes(&self) -> u64 {
7914        self.tables
7915            .iter()
7916            .map(Table::hot_bytes)
7917            .fold(0u64, u64::saturating_add)
7918    }
7919
7920    /// v5.2.2: freeze the **first** `max_rows` rows of `table_name`'s
7921    /// hot tier into a brand-new cold-tier segment. The named `BTree`
7922    /// index supplies the per-row PK (its column must be an integer
7923    /// type — v5.2.2 only supports `IndexKey::Int` PKs, matching the
7924    /// `index_key_as_u64` constraint used by the cold-tier lookup
7925    /// path). On success returns a [`FreezeReport`] with the
7926    /// freshly-allocated segment id, the count of rows that moved,
7927    /// the encoded segment bytes (so the caller can persist them to
7928    /// disk for later reload via `SPG_PRELOAD_COLD_SEGMENT`), and the
7929    /// hot-tier byte delta that was reclaimed.
7930    ///
7931    /// **Semantics**:
7932    /// 1. The first `max_rows` rows (by hot-tier position — same as
7933    ///    insertion order under v4.39 `PersistentVec`) are read.
7934    /// 2. Rows are sorted ascending by PK and serialised into a new
7935    ///    segment via [`encode_segment`].
7936    /// 3. The hot rows are dropped via [`Table::delete_rows`]; the
7937    ///    `rebuild_indices` it triggers regenerates `Hot` locators
7938    ///    for every remaining row (their positions shift down by
7939    ///    `max_rows`). Existing `Cold` locators in this index — from
7940    ///    a previous freeze — are also rebuilt **but with empty
7941    ///    payload** since rebuild reads only `self.rows`; this
7942    ///    routine re-registers them at the end of the call so the
7943    ///    user-visible state preserves all prior cold locators.
7944    /// 4. The new segment is loaded into `self.cold_segments` via
7945    ///    [`Catalog::load_segment_bytes`] (allocating a fresh
7946    ///    `segment_id`). New `Cold` locators are registered on the
7947    ///    named index — one per frozen row.
7948    ///
7949    /// **v5.2.2 limits** (relaxed in later sub-versions):
7950    /// - INSERT-only flow: subsequent UPDATE/DELETE on a frozen row
7951    ///   returns a stale-locator error (no promote-on-write until
7952    ///   v5.2.3).
7953    /// - Single-table scope: callers iterate tables themselves.
7954    /// - All-or-nothing: returns `Err` and leaves catalog unchanged
7955    ///   if any step fails before the atomic swap point.
7956    ///
7957    /// Errors:
7958    /// - [`StorageError::Corrupt`] for missing table/index, non-`BTree`
7959    ///   index, non-integer PK column, `max_rows == 0`, or
7960    ///   `max_rows > row_count`.
7961    /// - The encoder's [`SegmentError`] surfaces as `Corrupt` (the
7962    ///   only realistic source is "a single row is larger than the
7963    ///   page size"; SPG schemas don't hit it in practice).
7964    pub fn freeze_oldest_to_cold(
7965        &mut self,
7966        table_name: &str,
7967        index_name: &str,
7968        max_rows: usize,
7969    ) -> Result<FreezeReport, StorageError> {
7970        // --- validation phase: never mutates ---------------------
7971        if max_rows == 0 {
7972            return Err(StorageError::Corrupt(
7973                "freeze_oldest_to_cold: max_rows must be > 0".into(),
7974            ));
7975        }
7976        let table = self.get(table_name).ok_or_else(|| {
7977            StorageError::Corrupt(format!(
7978                "freeze_oldest_to_cold: table {table_name:?} not found"
7979            ))
7980        })?;
7981        if max_rows > table.rows.len() {
7982            return Err(StorageError::Corrupt(format!(
7983                "freeze_oldest_to_cold: max_rows {max_rows} > row_count {}",
7984                table.rows.len()
7985            )));
7986        }
7987        let idx = table
7988            .indices
7989            .iter()
7990            .find(|i| i.name == index_name)
7991            .ok_or_else(|| {
7992                StorageError::Corrupt(format!(
7993                    "freeze_oldest_to_cold: index {index_name:?} not found on {table_name:?}"
7994                ))
7995            })?;
7996        if !matches!(idx.kind, IndexKind::BTree(_)) {
7997            return Err(StorageError::Corrupt(format!(
7998                "freeze_oldest_to_cold: index {index_name:?} is NSW; only BTree indices may freeze"
7999            )));
8000        }
8001        let column_position = idx.column_position;
8002
8003        // --- segment build phase: reads only --------------------
8004        let schema = table.schema.clone();
8005        let mut to_freeze: Vec<(u64, Vec<u8>, IndexKey)> = Vec::with_capacity(max_rows);
8006        for row_idx in 0..max_rows {
8007            let row = table.rows.get(row_idx).expect("bounds-checked above");
8008            let key = IndexKey::from_value(&row.values[column_position]).ok_or_else(|| {
8009                StorageError::Corrupt(format!(
8010                    "freeze_oldest_to_cold: row {row_idx} has NULL / non-key value in index column"
8011                ))
8012            })?;
8013            let pk_u64 = index_key_as_u64(&key).ok_or_else(|| {
8014                StorageError::Corrupt(format!(
8015                    "freeze_oldest_to_cold: index {index_name:?} column type is non-integer; \
8016                     v5.2.2 cold tier requires IndexKey::Int (Text PK lands in v5.5+)"
8017                ))
8018            })?;
8019            to_freeze.push((pk_u64, encode_row_body_dense(row, &schema), key));
8020        }
8021        // encode_segment requires ascending u64 keys. Sort by PK
8022        // before encoding; the caller's row-position order is not
8023        // necessarily PK order (e.g. workloads that insert random
8024        // PKs).
8025        to_freeze.sort_by_key(|(k, _, _)| *k);
8026        // Reject duplicate PKs — encode_segment also rejects them
8027        // (`SegmentError::UnsortedKey`), but the resulting error
8028        // message there is misleading. Surface a clearer one.
8029        for w in to_freeze.windows(2) {
8030            if w[0].0 == w[1].0 {
8031                return Err(StorageError::Corrupt(format!(
8032                    "freeze_oldest_to_cold: duplicate PK {} in freeze batch",
8033                    w[0].0
8034                )));
8035            }
8036        }
8037        // Snapshot the (key, locator) pairs that will be registered
8038        // post-swap. Cloning the IndexKey out before the move makes
8039        // the registration loop borrow-free.
8040        let post_swap_keys: Vec<IndexKey> = to_freeze.iter().map(|(_, _, k)| k.clone()).collect();
8041        // Segment encode is now infallible w.r.t. ordering. Map the
8042        // `SegmentError` into a `StorageError::Corrupt` so the
8043        // public surface stays one error type.
8044        let seg_rows: Vec<(u64, Vec<u8>)> = to_freeze
8045            .into_iter()
8046            .map(|(k, body, _)| (k, body))
8047            .collect();
8048        let frozen_rows = seg_rows.len();
8049        let (seg_bytes, _meta) = encode_segment(seg_rows.into_iter(), 0.01, SEGMENT_PAGE_BYTES)
8050            .map_err(|e| StorageError::Corrupt(format!("freeze_oldest_to_cold: encode: {e}")))?;
8051
8052        // --- atomic swap phase: mutations only past this point ---
8053        // v5.2.3 made `Table::rebuild_indices` preserve every Cold
8054        // locator across the per-table rebuild, so `delete_rows`
8055        // below no longer wipes prior-freeze cold entries. The pre-
8056        // v5.2.3 capture-then-re-register that used to live here
8057        // was removed in v5.3.1 — keeping it would double-count
8058        // every prior-frozen key's Cold locator on each subsequent
8059        // freeze.
8060        let bytes_before = self.get(table_name).expect("just validated").hot_bytes();
8061        let positions: Vec<usize> = (0..max_rows).collect();
8062        let t_mut = self
8063            .get_mut(table_name)
8064            .expect("just validated; still present");
8065        let removed = t_mut.delete_rows(&positions);
8066        debug_assert_eq!(removed, max_rows, "delete_rows count matches request");
8067        let bytes_after = t_mut.hot_bytes();
8068        let bytes_freed = bytes_before.saturating_sub(bytes_after);
8069
8070        let segment_id = self
8071            .load_segment_bytes(seg_bytes.clone())
8072            .map_err(|e| StorageError::Corrupt(format!("freeze_oldest_to_cold: load: {e}")))?;
8073        let new_cold = post_swap_keys.into_iter().map(|k| {
8074            (
8075                k,
8076                RowLocator::Cold {
8077                    segment_id,
8078                    page_offset: 0,
8079                },
8080            )
8081        });
8082        let t_mut = self.get_mut(table_name).expect("still present");
8083        t_mut.register_cold_locators(index_name, new_cold)?;
8084        // r944 — a freeze has to say that it froze something.
8085        //
8086        // `has_cold_rows_fast()` reads the cached count, and neither
8087        // freeze path touched it, so afterwards it answered "no cold
8088        // rows" while cold rows existed. That predicate gates four join
8089        // paths, and a gate that wrongly declines the cold-aware path
8090        // drops the frozen rows from the answer.
8091        //
8092        // Marking it stale rather than adding to it: stale reads as
8093        // true, which is the safe direction, and this function cannot
8094        // know the exact total (rows may already have been cold). ANALYZE
8095        // recomputes the number.
8096        t_mut.mark_cold_row_count_stale();
8097
8098        Ok(FreezeReport {
8099            segment_id,
8100            frozen_rows,
8101            bytes_freed,
8102            segment_bytes: seg_bytes,
8103        })
8104    }
8105
8106    /// v5.1: borrow the cold segment at `segment_id`. Used by the
8107    /// spg-server preload path to enumerate (key, locator) pairs
8108    /// after loading a segment, so it can call
8109    /// [`Table::register_cold_locators`] without re-parsing the
8110    /// bytes.
8111    #[must_use]
8112    pub fn cold_segment(&self, segment_id: u32) -> Option<&OwnedSegment> {
8113        self.cold_segments
8114            .get(segment_id as usize)
8115            .and_then(|s| s.as_deref())
8116    }
8117
8118    /// v5.1: resolve a single `RowLocator::Cold` to its underlying
8119    /// `Row`. Decoupled from [`Catalog::lookup_by_pk`] so callers
8120    /// iterating a multi-locator slice (e.g. the engine's index
8121    /// seek path) can dispatch per locator instead of getting back
8122    /// only the first row for a key. Returns `None` when the
8123    /// segment isn't registered, the key isn't `u64`-coercible, or
8124    /// the segment doesn't actually carry the key (bloom or page-
8125    /// index reject).
8126    pub fn resolve_cold_locator(
8127        &self,
8128        table_name: &str,
8129        segment_id: u32,
8130        key: &IndexKey,
8131    ) -> Option<Row<'static>> {
8132        let t = self.get(table_name)?;
8133        let u64_key = index_key_as_u64(key)?;
8134        let seg = self.cold_segments.get(segment_id as usize)?.as_ref()?;
8135        let payload = seg.lookup(u64_key)?;
8136        let (row, _) = decode_row_body_dense(&payload, &t.schema, seg.codec_version()).ok()?;
8137        // v7.39 (pg_stat blks knife) — one cold-tier "block read".
8138        self.cold_read_stats
8139            .cold_reads
8140            .fetch_add(1, core::sync::atomic::Ordering::Relaxed);
8141        Some(row)
8142    }
8143
8144    /// v5.1: indexed PK lookup that dispatches per locator,
8145    /// returning the first matching row from either the hot tier
8146    /// (`Table::rows`) or a registered cold segment.
8147    ///
8148    /// The cold path requires the index column to be coercible to
8149    /// a `u64` (the segment's PK type) and the segment payload to
8150    /// be a [`encode_row_body_dense`]-encoded row body for the
8151    /// same schema. v5.1 ships this for BIGINT / INT / SMALLINT
8152    /// PKs; other types fall through to hot-only behavior.
8153    ///
8154    /// Returns `None` if (a) the table or index doesn't exist,
8155    /// (b) the key isn't in the index at all, or (c) the key was
8156    /// resolved to a stale locator (Hot index out of range, Cold
8157    /// segment id unknown, segment lookup miss). Does not surface
8158    /// segment-decode errors — those would indicate corrupted
8159    /// cold-tier files and should be caught at
8160    /// [`Catalog::load_segment_bytes`] time.
8161    pub fn lookup_by_pk(&self, table: &str, index_name: &str, key: &IndexKey) -> Option<Row<'_>> {
8162        let t = self.get(table)?;
8163        let idx = t.indices.iter().find(|i| i.name == index_name)?;
8164        let locators = idx.lookup_eq(key);
8165        let cold_u64_key = index_key_as_u64(key);
8166        for loc in locators {
8167            match *loc {
8168                RowLocator::Hot(i) => {
8169                    if let Some(row) = t.rows.get(i) {
8170                        return Some(row.clone());
8171                    }
8172                }
8173                RowLocator::Cold {
8174                    segment_id,
8175                    page_offset: _,
8176                } => {
8177                    let Some(u64_key) = cold_u64_key else {
8178                        // Key type not coercible to u64 — cold tier
8179                        // only handles BIGINT/INT/SMALLINT in v5.1.
8180                        continue;
8181                    };
8182                    let Some(seg) = self
8183                        .cold_segments
8184                        .get(segment_id as usize)
8185                        .and_then(|s| s.as_deref())
8186                    else {
8187                        // v6.7.3 — `None` slot = compaction
8188                        // retired this segment; the live locator
8189                        // on a freshly-compacted index points to
8190                        // the merged segment_id, so a Cold hit
8191                        // here against a tombstone means the BTree
8192                        // entry hasn't been swapped yet (mid-
8193                        // compaction reader race) or the caller is
8194                        // looking up a stale snapshot. Skip — the
8195                        // next locator in the list, if any, is
8196                        // typically the merged segment.
8197                        continue;
8198                    };
8199                    let Some(payload) = seg.lookup(u64_key) else {
8200                        continue;
8201                    };
8202                    let (row, _) =
8203                        decode_row_body_dense(&payload, &t.schema, seg.codec_version()).ok()?;
8204                    return Some(row);
8205                }
8206            }
8207        }
8208        None
8209    }
8210
8211    /// v5.2.3: promote a frozen row back to the hot tier so an
8212    /// UPDATE / DELETE can mutate it. Reads the cold-tier row body
8213    /// (decoded from its registered segment), pushes it into
8214    /// `table.rows` via [`Table::insert`] (which also adds a fresh
8215    /// `Hot(new_idx)` locator on `index_name`), then retires the
8216    /// shadowed `Cold` locator via
8217    /// [`Table::remove_cold_locators_for_key`]. The cold-tier row
8218    /// in the segment file becomes garbage — recoverable when a
8219    /// future cold-segment compaction job lands.
8220    ///
8221    /// Returns:
8222    /// - `Ok(Some(new_hot_idx))` when the key resolved through a
8223    ///   cold locator and the promote completed. `new_hot_idx` is
8224    ///   the position the row now occupies in `table.rows`.
8225    /// - `Ok(None)` when the key has no Cold locator on the index
8226    ///   (already hot, or wasn't present at all). Callers treat this
8227    ///   as "nothing to do here, fall back to the hot-only path".
8228    ///
8229    /// Errors when the table / index doesn't exist, the index isn't
8230    /// `BTree`, the cold segment is missing / can't decode the row,
8231    /// or the inferred row body fails `Table::insert` validation.
8232    pub fn promote_cold_row(
8233        &mut self,
8234        table_name: &str,
8235        index_name: &str,
8236        key: &IndexKey,
8237    ) -> Result<Option<usize>, StorageError> {
8238        let cold_loc = self.find_cold_locator(table_name, index_name, key)?;
8239        let Some((segment_id, _page_offset)) = cold_loc else {
8240            return Ok(None);
8241        };
8242        let u64_key = index_key_as_u64(key).ok_or_else(|| {
8243            StorageError::Corrupt(
8244                "promote_cold_row: key type not coercible to u64 (cold tier requires integer PK)"
8245                    .into(),
8246            )
8247        })?;
8248        // Read the row body from the segment. Borrow the segment +
8249        // schema short-term so we can then take `&mut self` for the
8250        // hot-side insert.
8251        let schema = self
8252            .get(table_name)
8253            .ok_or_else(|| {
8254                StorageError::Corrupt(format!("promote_cold_row: table {table_name:?} not found"))
8255            })?
8256            .schema
8257            .clone();
8258        let seg = self
8259            .cold_segments
8260            .get(segment_id as usize)
8261            .and_then(|s| s.as_ref())
8262            .ok_or_else(|| {
8263                StorageError::Corrupt(format!(
8264                    "promote_cold_row: segment {segment_id} not registered on catalog"
8265                ))
8266            })?;
8267        let payload = seg.lookup(u64_key).ok_or_else(|| {
8268            StorageError::Corrupt(format!(
8269                "promote_cold_row: key {u64_key} resolves to segment {segment_id} \
8270                 but the segment's bloom/page lookup didn't return a row"
8271            ))
8272        })?;
8273        let (row, _consumed) = decode_row_body_dense(&payload, &schema, seg.codec_version())?;
8274        // Insert the promoted row into the hot tier. `Table::insert`
8275        // appends to `self.rows`, adds a `Hot(new_idx)` locator to
8276        // every BTree index covering the row's keyed columns, and
8277        // increments `hot_bytes`.
8278        let t = self
8279            .get_mut(table_name)
8280            .expect("table existed at lookup time");
8281        t.insert(row)?;
8282        let new_hot_idx =
8283            t.rows.len().checked_sub(1).ok_or_else(|| {
8284                StorageError::Corrupt("promote_cold_row: empty after insert".into())
8285            })?;
8286        // The hot insert added Hot(new_idx) alongside the still-
8287        // present Cold locator. Drop the Cold entry so future
8288        // lookups return only the fresh hot row.
8289        t.remove_cold_locators_for_key(index_name, key)?;
8290        Ok(Some(new_hot_idx))
8291    }
8292
8293    /// v5.2.3: shadow a frozen row's index entry. Used by DELETE
8294    /// when the row to remove lives in a cold-tier segment — the
8295    /// row body stays in the segment file (becoming garbage) but
8296    /// every `Cold` locator for `key` on `index_name` is removed
8297    /// so PK lookups stop returning it.
8298    ///
8299    /// Returns the number of cold locators retired (0 when the key
8300    /// has no cold entries — the DELETE fell on a hot row or a
8301    /// key that was already absent). Errors when the table /
8302    /// index doesn't exist or the index isn't `BTree`.
8303    ///
8304    /// Cold-segment compaction (which merges shadowed-heavy
8305    /// segments and reclaims their disk footprint) lands in a
8306    /// later v5.x sub-version; until then, repeated UPDATE/DELETE
8307    /// of cold rows can amplify cold-segment disk usage by up to
8308    /// 1-2× — still well under typical LSM-tree shadowing because
8309    /// SPG segments are bulk-baked, not write-merged.
8310    pub fn shadow_cold_row(
8311        &mut self,
8312        table_name: &str,
8313        index_name: &str,
8314        key: &IndexKey,
8315    ) -> Result<usize, StorageError> {
8316        let t = self.get_mut(table_name).ok_or_else(|| {
8317            StorageError::Corrupt(format!("shadow_cold_row: table {table_name:?} not found"))
8318        })?;
8319        t.remove_cold_locators_for_key(index_name, key)
8320    }
8321
8322    /// v6.7.4 — read-only slice preparation for the parallel
8323    /// freezer. Walks rows in `row_range`, builds the
8324    /// `(pk_u64, encoded_body, IndexKey)` triples that the
8325    /// coordinator's k-way merge consumes, sorts the slice by
8326    /// `pk_u64`, and returns a [`FreezeSlice`].
8327    ///
8328    /// Caller invariants:
8329    /// - `row_range.end <= table.rows.len()` (caller's job to
8330    ///   compute the partition).
8331    /// - All slices passed to `commit_freeze_slices` must cover a
8332    ///   contiguous half-open range `[0, total_max_rows)` with no
8333    ///   gaps and no overlaps. The coordinator validates this
8334    ///   invariant before committing.
8335    ///
8336    /// `&self`-only — multiple workers can run this concurrently
8337    /// against the same `Catalog` reference under the engine's
8338    /// write lock (workers don't mutate; the coordinator does).
8339    pub fn prepare_freeze_slice(
8340        &self,
8341        table_name: &str,
8342        index_name: &str,
8343        row_range: core::ops::Range<usize>,
8344    ) -> Result<FreezeSlice, StorageError> {
8345        let table = self.get(table_name).ok_or_else(|| {
8346            StorageError::Corrupt(format!(
8347                "prepare_freeze_slice: table {table_name:?} not found"
8348            ))
8349        })?;
8350        let idx = table
8351            .indices
8352            .iter()
8353            .find(|i| i.name == index_name)
8354            .ok_or_else(|| {
8355                StorageError::Corrupt(format!(
8356                    "prepare_freeze_slice: index {index_name:?} not found on {table_name:?}"
8357                ))
8358            })?;
8359        if !matches!(idx.kind, IndexKind::BTree(_)) {
8360            return Err(StorageError::Corrupt(format!(
8361                "prepare_freeze_slice: index {index_name:?} is NSW; only BTree indices may freeze"
8362            )));
8363        }
8364        if row_range.end > table.rows.len() {
8365            return Err(StorageError::Corrupt(format!(
8366                "prepare_freeze_slice: row_range end {} > row_count {}",
8367                row_range.end,
8368                table.rows.len()
8369            )));
8370        }
8371        let column_position = idx.column_position;
8372        let schema = table.schema.clone();
8373        let mut rows: Vec<(u64, Vec<u8>, IndexKey)> = Vec::with_capacity(row_range.len());
8374        for row_idx in row_range.clone() {
8375            let row = table.rows.get(row_idx).expect("bounds-checked above");
8376            let key = IndexKey::from_value(&row.values[column_position]).ok_or_else(|| {
8377                StorageError::Corrupt(format!(
8378                    "prepare_freeze_slice: row {row_idx} has NULL / non-key value in index column"
8379                ))
8380            })?;
8381            let pk_u64 = index_key_as_u64(&key).ok_or_else(|| {
8382                StorageError::Corrupt(format!(
8383                    "prepare_freeze_slice: index {index_name:?} column type is non-integer; \
8384                     v5.2.2 cold tier requires IndexKey::Int (Text PK lands in v5.5+)"
8385                ))
8386            })?;
8387            rows.push((pk_u64, encode_row_body_dense(row, &schema), key));
8388        }
8389        rows.sort_by_key(|(k, _, _)| *k);
8390        Ok(FreezeSlice { row_range, rows })
8391    }
8392
8393    /// v6.7.4 — coordinator commit step. Merges N
8394    /// [`FreezeSlice`]s into one segment via the standard
8395    /// [`encode_segment`] path, atomically swaps the catalog
8396    /// state (delete the union row range + register Cold
8397    /// locators + load the segment).
8398    ///
8399    /// Validates that the slices cover a contiguous, gap-free,
8400    /// overlap-free half-open range starting at index 0 (the
8401    /// freezer always freezes "oldest first" — same semantics as
8402    /// the single-threaded [`Catalog::freeze_oldest_to_cold`]).
8403    ///
8404    /// Empty `slices` → no-op success (returns a zero-row report
8405    /// without mutating). Total row count = `Σ slice.rows.len()`.
8406    pub fn commit_freeze_slices(
8407        &mut self,
8408        table_name: &str,
8409        index_name: &str,
8410        slices: Vec<FreezeSlice>,
8411    ) -> Result<FreezeReport, StorageError> {
8412        // --- validation phase: never mutates ---------------------
8413        let table = self.get(table_name).ok_or_else(|| {
8414            StorageError::Corrupt(format!(
8415                "commit_freeze_slices: table {table_name:?} not found"
8416            ))
8417        })?;
8418        let idx = table
8419            .indices
8420            .iter()
8421            .find(|i| i.name == index_name)
8422            .ok_or_else(|| {
8423                StorageError::Corrupt(format!(
8424                    "commit_freeze_slices: index {index_name:?} not found on {table_name:?}"
8425                ))
8426            })?;
8427        if !matches!(idx.kind, IndexKind::BTree(_)) {
8428            return Err(StorageError::Corrupt(format!(
8429                "commit_freeze_slices: index {index_name:?} is NSW; only BTree indices may freeze"
8430            )));
8431        }
8432        // Validate slice coverage: contiguous from 0, no gaps, no
8433        // overlaps. Allow the caller to pass slices in any order —
8434        // sort by row_range.start first.
8435        let mut ordered = slices;
8436        ordered.sort_by_key(|s| s.row_range.start);
8437        // Drop fully-empty slices that fell out of an uneven
8438        // partition; they carry no data but contribute to the
8439        // contiguity check, so keep them in line.
8440        let mut expected_start = 0usize;
8441        for s in &ordered {
8442            if s.row_range.start != expected_start {
8443                return Err(StorageError::Corrupt(format!(
8444                    "commit_freeze_slices: gap/overlap at row {}; expected start {}",
8445                    s.row_range.start, expected_start
8446                )));
8447            }
8448            expected_start = s.row_range.end;
8449        }
8450        let max_rows = expected_start;
8451        if max_rows > table.rows.len() {
8452            return Err(StorageError::Corrupt(format!(
8453                "commit_freeze_slices: total row range {} exceeds row_count {}",
8454                max_rows,
8455                table.rows.len()
8456            )));
8457        }
8458        if max_rows == 0 {
8459            return Ok(FreezeReport {
8460                segment_id: u32::MAX,
8461                frozen_rows: 0,
8462                bytes_freed: 0,
8463                segment_bytes: Vec::new(),
8464            });
8465        }
8466
8467        // --- segment build phase: reads only --------------------
8468        // K-way merge of already-sorted slices. Each slice's rows
8469        // are ascending by pk_u64; we keep a per-slice cursor and
8470        // pull the next-smallest head until every cursor drains.
8471        let total_rows: usize = ordered.iter().map(|s| s.rows.len()).sum();
8472        if total_rows != max_rows {
8473            return Err(StorageError::Corrupt(format!(
8474                "commit_freeze_slices: total slice rows {total_rows} ≠ row_range coverage {max_rows}"
8475            )));
8476        }
8477        let mut cursors: Vec<usize> = alloc::vec![0; ordered.len()];
8478        let mut merged: Vec<(u64, Vec<u8>, IndexKey)> = Vec::with_capacity(total_rows);
8479        loop {
8480            // Pick the slice whose head row has the smallest key
8481            // and isn't yet exhausted.
8482            let mut pick: Option<usize> = None;
8483            for (i, c) in cursors.iter().enumerate() {
8484                let slice = &ordered[i];
8485                if *c >= slice.rows.len() {
8486                    continue;
8487                }
8488                match pick {
8489                    None => pick = Some(i),
8490                    Some(j) => {
8491                        if slice.rows[*c].0 < ordered[j].rows[cursors[j]].0 {
8492                            pick = Some(i);
8493                        }
8494                    }
8495                }
8496            }
8497            let Some(i) = pick else { break };
8498            let row = ordered[i].rows[cursors[i]].clone();
8499            cursors[i] += 1;
8500            merged.push(row);
8501        }
8502        // Reject duplicate PKs — same error as the single-threaded
8503        // path so callers get a uniform surface.
8504        for w in merged.windows(2) {
8505            if w[0].0 == w[1].0 {
8506                return Err(StorageError::Corrupt(format!(
8507                    "commit_freeze_slices: duplicate PK {} across slices",
8508                    w[0].0
8509                )));
8510            }
8511        }
8512        let post_swap_keys: Vec<IndexKey> = merged.iter().map(|(_, _, k)| k.clone()).collect();
8513        let seg_rows: Vec<(u64, Vec<u8>)> =
8514            merged.into_iter().map(|(k, body, _)| (k, body)).collect();
8515        let frozen_rows = seg_rows.len();
8516        let (seg_bytes, _meta) = encode_segment(seg_rows.into_iter(), 0.01, SEGMENT_PAGE_BYTES)
8517            .map_err(|e| StorageError::Corrupt(format!("commit_freeze_slices: encode: {e}")))?;
8518
8519        // --- atomic swap phase: mutations only past this point ---
8520        let bytes_before = self.get(table_name).expect("just validated").hot_bytes();
8521        let positions: Vec<usize> = (0..max_rows).collect();
8522        let t_mut = self
8523            .get_mut(table_name)
8524            .expect("just validated; still present");
8525        let removed = t_mut.delete_rows(&positions);
8526        debug_assert_eq!(removed, max_rows, "delete_rows count matches request");
8527        let bytes_after = t_mut.hot_bytes();
8528        let bytes_freed = bytes_before.saturating_sub(bytes_after);
8529
8530        let segment_id = self
8531            .load_segment_bytes(seg_bytes.clone())
8532            .map_err(|e| StorageError::Corrupt(format!("commit_freeze_slices: load: {e}")))?;
8533        let new_cold = post_swap_keys.into_iter().map(|k| {
8534            (
8535                k,
8536                RowLocator::Cold {
8537                    segment_id,
8538                    page_offset: 0,
8539                },
8540            )
8541        });
8542        let t_mut = self.get_mut(table_name).expect("still present");
8543        t_mut.register_cold_locators(index_name, new_cold)?;
8544        // r944 — a freeze has to say that it froze something.
8545        //
8546        // `has_cold_rows_fast()` reads the cached count, and neither
8547        // freeze path touched it, so afterwards it answered "no cold
8548        // rows" while cold rows existed. That predicate gates four join
8549        // paths, and a gate that wrongly declines the cold-aware path
8550        // drops the frozen rows from the answer.
8551        //
8552        // Marking it stale rather than adding to it: stale reads as
8553        // true, which is the safe direction, and this function cannot
8554        // know the exact total (rows may already have been cold). ANALYZE
8555        // recomputes the number.
8556        t_mut.mark_cold_row_count_stale();
8557
8558        Ok(FreezeReport {
8559            segment_id,
8560            frozen_rows,
8561            bytes_freed,
8562            segment_bytes: seg_bytes,
8563        })
8564    }
8565
8566    /// v6.7.3 — compact every cold segment on `(table, index)` whose
8567    /// `OwnedSegment::bytes().len()` is below `target_segment_bytes`
8568    /// into a single larger merged segment. Rows present in source
8569    /// segment payloads but no longer referenced by any
8570    /// `RowLocator::Cold` on the index (DELETE'd + frozen rows
8571    /// retired via [`Catalog::shadow_cold_row`]) are GC'd in the
8572    /// merge.
8573    ///
8574    /// **Semantics**:
8575    /// 1. Walk the BTree index to collect every Cold locator that
8576    ///    targets a small (< threshold) segment. Each such
8577    ///    `(key, segment_id)` becomes a row in the merged segment;
8578    ///    payload is looked up from the source segment in-place.
8579    /// 2. Encode the collected rows into one new segment via
8580    ///    [`encode_segment`]; register it via
8581    ///    [`Catalog::load_segment_bytes`] (allocating a fresh
8582    ///    `merged_segment_id` at the end of `cold_segments`).
8583    /// 3. Rewrite the BTree index in one pass: every
8584    ///    `RowLocator::Cold { segment_id ∈ sources }` becomes
8585    ///    `RowLocator::Cold { segment_id = merged_id, page_offset = 0 }`.
8586    ///    Hot locators are untouched.
8587    /// 4. Tombstone every source slot via
8588    ///    [`Catalog::tombstone_segment`]. Source segment payloads
8589    ///    are no longer reachable through the catalog; the on-disk
8590    ///    files are the caller's concern.
8591    ///
8592    /// On fewer than 2 candidate segments the catalog is **not**
8593    /// mutated and a no-op report (`merged_segment_id: None`,
8594    /// `sources: []`) is returned. This is the routine case — a
8595    /// freshly-frozen table has at most 1 small segment, no merge
8596    /// possible.
8597    ///
8598    /// Atomicity: every mutating step runs after the read-only
8599    /// gather phase, so a panic before the merge encode leaves the
8600    /// catalog unchanged. The mutation block itself (load + rewrite +
8601    /// tombstone) takes only `&mut self` — callers serialise the
8602    /// engine write lock outside this function.
8603    ///
8604    /// Errors when the table / index doesn't exist, the index isn't
8605    /// `BTree`, the index column type isn't u64-coercible (cold-tier
8606    /// pre-condition), or a source segment fails its in-place
8607    /// row-body lookup (would indicate prior catalog corruption).
8608    pub fn compact_cold_segments(
8609        &mut self,
8610        table_name: &str,
8611        index_name: &str,
8612        target_segment_bytes: u64,
8613    ) -> Result<CompactReport, StorageError> {
8614        // --- validation phase ----------------------------------
8615        let t = self.get(table_name).ok_or_else(|| {
8616            StorageError::Corrupt(format!(
8617                "compact_cold_segments: table {table_name:?} not found"
8618            ))
8619        })?;
8620        let idx = t
8621            .indices
8622            .iter()
8623            .find(|i| i.name == index_name)
8624            .ok_or_else(|| {
8625                StorageError::Corrupt(format!(
8626                    "compact_cold_segments: index {index_name:?} not found on {table_name:?}"
8627                ))
8628            })?;
8629        let map = match &idx.kind {
8630            IndexKind::BTree(m) => m,
8631            IndexKind::Nsw(_)
8632            | IndexKind::Brin { .. }
8633            | IndexKind::Gin(_)
8634            | IndexKind::GinTrgm(_)
8635            | IndexKind::GinFulltext(_)
8636            | IndexKind::GinJsonb(_)
8637            | IndexKind::BTreeMulti(_) => {
8638                return Err(StorageError::Corrupt(format!(
8639                    "compact_cold_segments: index {index_name:?} is not BTree; \
8640                     compaction applies only to BTree cold-tier indices"
8641                )));
8642            }
8643        };
8644
8645        // --- gather phase --------------------------------------
8646        // Step A: every segment_id this BTree index Cold-references.
8647        let mut referenced_ids: BTreeSet<u32> = BTreeSet::new();
8648        for (_key, locators) in map.iter() {
8649            for loc in locators {
8650                if let RowLocator::Cold { segment_id, .. } = loc {
8651                    referenced_ids.insert(*segment_id);
8652                }
8653            }
8654        }
8655        // Step B: keep only the small + still-active ones.
8656        let candidate_set: BTreeSet<u32> = referenced_ids
8657            .into_iter()
8658            .filter(|id| {
8659                self.cold_segments
8660                    .get(*id as usize)
8661                    .and_then(|s| s.as_deref())
8662                    .is_some_and(|s| (s.bytes().len() as u64) < target_segment_bytes)
8663            })
8664            .collect();
8665        if candidate_set.len() < 2 {
8666            return Ok(CompactReport {
8667                sources: Vec::new(),
8668                merged_segment_id: None,
8669                merged_segment_bytes: Vec::new(),
8670                merged_rows: 0,
8671                deleted_rows_pruned: 0,
8672                bytes_reclaimed_estimate: 0,
8673            });
8674        }
8675        // Step C: pre-count source rows for the deleted-pruned metric.
8676        let mut source_row_count: usize = 0;
8677        let mut source_byte_total: u64 = 0;
8678        for &id in &candidate_set {
8679            let seg = self.cold_segments[id as usize]
8680                .as_ref()
8681                .expect("candidate selected only when slot is Some");
8682            source_row_count = source_row_count.saturating_add(seg.meta().num_rows as usize);
8683            source_byte_total = source_byte_total.saturating_add(seg.bytes().len() as u64);
8684        }
8685        // Step D: collect (key, body) pairs from every live Cold
8686        // locator pointing at a candidate. dedupe by key — one
8687        // BTree key resolves to at most one cold payload (the
8688        // freezer + promote/shadow flow keeps Cold locators
8689        // unique per key).
8690        let mut collected: BTreeMap<u64, (Vec<u8>, IndexKey)> = BTreeMap::new();
8691        for (key, locators) in map.iter() {
8692            for loc in locators {
8693                let RowLocator::Cold { segment_id, .. } = loc else {
8694                    continue;
8695                };
8696                if !candidate_set.contains(segment_id) {
8697                    continue;
8698                }
8699                let u64_key = index_key_as_u64(key).ok_or_else(|| {
8700                    StorageError::Corrupt(format!(
8701                        "compact_cold_segments: index {index_name:?} has non-integer Cold key; \
8702                         cold tier requires IndexKey::Int (Text PK lands in v5.5+)"
8703                    ))
8704                })?;
8705                let seg = self.cold_segments[*segment_id as usize]
8706                    .as_ref()
8707                    .expect("candidate slot guaranteed Some above");
8708                let payload = seg.lookup(u64_key).ok_or_else(|| {
8709                    StorageError::Corrupt(format!(
8710                        "compact_cold_segments: BTree {index_name:?} points key={u64_key} \
8711                         at segment {segment_id} but the segment lookup missed"
8712                    ))
8713                })?;
8714                collected.insert(u64_key, (payload, key.clone()));
8715                break;
8716            }
8717        }
8718        let merged_rows = collected.len();
8719        let deleted_rows_pruned = source_row_count.saturating_sub(merged_rows);
8720
8721        // Step E: encode the merged segment. `BTreeMap<u64, _>`
8722        // iteration is ascending by key, which is what
8723        // `encode_segment` requires.
8724        let seg_rows: Vec<(u64, Vec<u8>)> = collected
8725            .iter()
8726            .map(|(k, (body, _))| (*k, body.clone()))
8727            .collect();
8728        let (seg_bytes, _meta) = encode_segment(seg_rows.into_iter(), 0.01, SEGMENT_PAGE_BYTES)
8729            .map_err(|e| StorageError::Corrupt(format!("compact_cold_segments: encode: {e}")))?;
8730        let merged_bytes_len = seg_bytes.len() as u64;
8731
8732        // --- atomic mutation phase ------------------------------
8733        let merged_segment_id = self
8734            .load_segment_bytes(seg_bytes.clone())
8735            .map_err(|e| StorageError::Corrupt(format!("compact_cold_segments: load: {e}")))?;
8736
8737        // Rewrite the BTree index: every Cold locator pointing at
8738        // a candidate source becomes a Cold locator pointing at
8739        // the merged segment. Use a flat collect-then-replace
8740        // pattern so we never hold a `&self` borrow across the
8741        // `&mut self` write.
8742        let entries: Vec<(IndexKey, crate::posting::PostingList)> = {
8743            let t = self
8744                .get(table_name)
8745                .expect("table existed at the start of this fn");
8746            let idx = t
8747                .indices
8748                .iter()
8749                .find(|i| i.name == index_name)
8750                .expect("index existed at the start of this fn");
8751            let IndexKind::BTree(map) = &idx.kind else {
8752                unreachable!("validated above");
8753            };
8754            map.iter().map(|(k, v)| (k.clone(), v.clone())).collect()
8755        };
8756        let t_mut = self
8757            .get_mut(table_name)
8758            .expect("table existed at the start of this fn");
8759        let idx_mut = t_mut
8760            .indices
8761            .iter_mut()
8762            .find(|i| i.name == index_name)
8763            .expect("index existed at the start of this fn");
8764        let IndexKind::BTree(map_mut) = &mut idx_mut.kind else {
8765            unreachable!("validated above");
8766        };
8767        for (key, locators) in entries {
8768            let mut new_locs = crate::posting::PostingList::new();
8769            let mut changed = false;
8770            for loc in &locators {
8771                match *loc {
8772                    RowLocator::Cold {
8773                        segment_id,
8774                        page_offset: _,
8775                    } if candidate_set.contains(&segment_id) => {
8776                        let replacement = RowLocator::Cold {
8777                            segment_id: merged_segment_id,
8778                            page_offset: 0,
8779                        };
8780                        if !new_locs.contains(replacement) {
8781                            new_locs.push(replacement);
8782                        }
8783                        changed = true;
8784                    }
8785                    other => new_locs.push(other),
8786                }
8787            }
8788            if changed {
8789                map_mut.insert_mut(key, new_locs);
8790            }
8791        }
8792
8793        // Tombstone every source slot. Last step — failures here
8794        // would leave the segment double-referenced in both
8795        // memory + manifest, but `tombstone_segment` only errors
8796        // on out-of-bounds, which we've already validated.
8797        for &id in &candidate_set {
8798            self.tombstone_segment(id)?;
8799        }
8800
8801        let bytes_reclaimed_estimate = source_byte_total.saturating_sub(merged_bytes_len);
8802        Ok(CompactReport {
8803            sources: candidate_set.into_iter().collect(),
8804            merged_segment_id: Some(merged_segment_id),
8805            merged_segment_bytes: seg_bytes,
8806            merged_rows,
8807            deleted_rows_pruned,
8808            bytes_reclaimed_estimate,
8809        })
8810    }
8811
8812    /// Internal helper: scan `(table, index)` for a `Cold` locator
8813    /// keyed by `key`. Returns `Ok(Some((segment_id, page_offset)))`
8814    /// when found, `Ok(None)` when the key has only hot entries
8815    /// or no entries at all, `Err` on the same input-validation
8816    /// errors as the public `promote_cold_row` / `shadow_cold_row`.
8817    fn find_cold_locator(
8818        &self,
8819        table_name: &str,
8820        index_name: &str,
8821        key: &IndexKey,
8822    ) -> Result<Option<(u32, u32)>, StorageError> {
8823        let t = self.get(table_name).ok_or_else(|| {
8824            StorageError::Corrupt(format!("find_cold_locator: table {table_name:?} not found"))
8825        })?;
8826        let idx = t
8827            .indices
8828            .iter()
8829            .find(|i| i.name == index_name)
8830            .ok_or_else(|| {
8831                StorageError::Corrupt(format!(
8832                    "find_cold_locator: index {index_name:?} not found on {table_name:?}"
8833                ))
8834            })?;
8835        if !matches!(idx.kind, IndexKind::BTree(_)) {
8836            return Err(StorageError::Corrupt(format!(
8837                "find_cold_locator: index {index_name:?} is NSW; promote-on-write only applies to BTree indices"
8838            )));
8839        }
8840        for loc in idx.lookup_eq(key) {
8841            if let RowLocator::Cold {
8842                segment_id,
8843                page_offset,
8844            } = *loc
8845            {
8846                return Ok(Some((segment_id, page_offset)));
8847            }
8848        }
8849        Ok(None)
8850    }
8851}
8852
8853/// Coerce an [`IndexKey`] to the `u64` that v5.1 cold-tier
8854/// segments use as their on-disk PK. Returns `None` for keys that
8855/// aren't representable as `u64` — Text PKs need a hash mapping
8856/// the segment writer baked in (deferred to v5.2+), Bool PKs are
8857/// almost never wide enough to be sharded into a cold tier.
8858fn index_key_as_u64(key: &IndexKey) -> Option<u64> {
8859    match key {
8860        // Reinterpret the i64 bit pattern as u64. Cold-tier segments
8861        // are sorted by this u64 view, so the chosen interpretation
8862        // only has to match between insert (bake_segment / freezer)
8863        // and lookup — using cast_unsigned keeps both sides honest
8864        // and silences clippy::cast_sign_loss.
8865        IndexKey::Int(n) => Some(n.cast_unsigned()),
8866        // Text / Bool / Uuid / Bytes / Numeric PKs aren't representable
8867        // as u64 and so can't participate in the u64-sorted cold-tier
8868        // segment PK layout. Same deferral story as Text — lookup falls
8869        // through the in-memory btree.
8870        IndexKey::Text(_)
8871        | IndexKey::Bool(_)
8872        | IndexKey::Uuid(_)
8873        | IndexKey::Bytes(_)
8874        | IndexKey::Numeric(_)
8875        | IndexKey::Null => None,
8876    }
8877}
8878
8879#[derive(Debug, Clone, PartialEq, Eq)]
8880#[non_exhaustive]
8881pub enum StorageError {
8882    DuplicateTable {
8883        name: String,
8884    },
8885    TableNotFound {
8886        name: String,
8887    },
8888    ArityMismatch {
8889        expected: usize,
8890        actual: usize,
8891    },
8892    TypeMismatch {
8893        column: String,
8894        expected: DataType,
8895        actual: DataType,
8896        position: usize,
8897    },
8898    NullInNotNull {
8899        column: String,
8900    },
8901    /// Index with this name already exists on the table.
8902    DuplicateIndex {
8903        name: String,
8904    },
8905    /// Column referenced by an index doesn't exist on the table.
8906    ColumnNotFound {
8907        column: String,
8908    },
8909    /// On-disk format failed to parse — corrupted file, wrong magic, truncated
8910    /// payload, or unknown tag bytes.
8911    Corrupt(String),
8912    /// v6.0.4 — ALTER INDEX targeted an index name that doesn't
8913    /// exist on any table in this catalog.
8914    IndexNotFound {
8915        name: String,
8916    },
8917    /// v6.0.4 — operation requested isn't supported on this index
8918    /// kind / column type (e.g. ALTER INDEX REBUILD on a `BTree`
8919    /// index, or REBUILD WITH (encoding=…) on a non-vector column).
8920    Unsupported(String),
8921    /// v7.39 (round 220) — a CYCLE-less sequence ran past its bound.
8922    /// PG's 2200H phrasing: `nextval: reached maximum value of
8923    /// sequence "s" (n)` (`is_max: false` = the MINVALUE direction).
8924    SequenceExhausted {
8925        name: String,
8926        limit: i64,
8927        is_max: bool,
8928    },
8929}
8930
8931impl fmt::Display for StorageError {
8932    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
8933        match self {
8934            // v7.39 (read01 round 47) — PG's 42P07 wording.
8935            Self::DuplicateTable { name } => write!(f, "relation \"{name}\" already exists"),
8936            // v7.39 (read01 round 47) — PG's wording for a missing relation
8937            // (42P01). DROP TABLE says "table" and raises its own error at
8938            // the engine; every other path (SELECT / ALTER / …) says
8939            // "relation", which is what this carries.
8940            Self::TableNotFound { name } => write!(f, "relation \"{name}\" does not exist"),
8941            Self::ArityMismatch { expected, actual } => write!(
8942                f,
8943                "row arity mismatch: expected {expected} columns, got {actual}"
8944            ),
8945            Self::TypeMismatch {
8946                column,
8947                expected,
8948                actual,
8949                position,
8950            } => write!(
8951                f,
8952                "type mismatch in column {column:?} (position {position}): expected {expected}, got {actual}"
8953            ),
8954            Self::NullInNotNull { column } => {
8955                // v7.39 (SQLSTATE fidelity) — PG's 23502 phrasing (the
8956                // relation-qualified long form is added by engine call
8957                // sites that know the table name).
8958                write!(
8959                    f,
8960                    "null value in column \"{column}\" violates not-null constraint"
8961                )
8962            }
8963            // v7.39 (read01 round 47) — an index is a relation to PG (42P07).
8964            Self::DuplicateIndex { name } => write!(f, "relation \"{name}\" already exists"),
8965            // v7.39 (round 701) — PG's wording, and the same fix `EvalError::
8966            // ColumnNotFound` took in read01 round 81 with the same reason:
8967            // "column not found: x" matches none of the wire layer's `does
8968            // not exist` patterns, so a missing column reached the client as
8969            // the generic error class. The eval-side variant was changed and
8970            // the storage-side one was not, so which sentence you got
8971            // depended on which layer noticed — `CREATE INDEX ix ON t(nope)`
8972            // came out of storage and kept the old spelling.
8973            Self::ColumnNotFound { column } => write!(f, "column \"{column}\" does not exist"),
8974            Self::Corrupt(detail) => write!(f, "corrupt on-disk format: {detail}"),
8975            Self::IndexNotFound { name } => write!(f, "index \"{name}\" does not exist"),
8976            Self::Unsupported(detail) => write!(f, "unsupported: {detail}"),
8977            // v7.39 (round 220) — PG's exact 2200H wording.
8978            Self::SequenceExhausted {
8979                name,
8980                limit,
8981                is_max,
8982            } => write!(
8983                f,
8984                "nextval: reached {} value of sequence \"{name}\" ({limit})",
8985                if *is_max { "maximum" } else { "minimum" }
8986            ),
8987        }
8988    }
8989}
8990
8991impl ColumnSchema {
8992    pub fn new(name: impl Into<String>, ty: DataType, nullable: bool) -> Self {
8993        Self {
8994            name: name.into(),
8995            ty,
8996            nullable,
8997            collation_name: None,
8998            default: None,
8999            runtime_default: None,
9000            auto_increment: false,
9001            user_enum_type: None,
9002            user_domain_type: None,
9003            user_composite_type: None,
9004            acl: Vec::new(),
9005            on_update_runtime: None,
9006            collation: Collation::Binary,
9007            is_unsigned: false,
9008            inline_enum_variants: None,
9009            inline_set_variants: None,
9010            generated_stored_expr: None,
9011            identity_always: false,
9012            default_text: None,
9013            auto_restart: None,
9014            scalar_row_source: false,
9015            mysql_int_width: None,
9016            mysql_fsp: None,
9017            mysql_declared_timestamp: false,
9018        }
9019    }
9020
9021    /// v7.38.14 — the SAME column, re-described.
9022    ///
9023    /// `ColumnSchema::new` is for SYNTHESISING a column: a catalog row, an
9024    /// admin view, a computed output. It sets twenty-two fields to their
9025    /// defaults, which is right when there is no source column to speak of.
9026    ///
9027    /// It is wrong, and quietly so, when there IS one -- a join's combined
9028    /// schema, an aggregate's synthetic keys, a derived table's output. Those
9029    /// sites re-describe an existing column under a new name or type, and
9030    /// have each been written as `new(..)` followed by hand-picking a few
9031    /// attributes to copy across. They all pick differently and none picks
9032    /// them all.
9033    ///
9034    /// Five fields have been lost through that shape so far -- enum identity,
9035    /// MySQL fsp, the PG collation name, `ProjectedItem::fold_exempt`, and
9036    /// the `collation` enum -- and v7.38.14 alone found four sites dropping
9037    /// the last of those. The failure is never loud: `collation` defaults to
9038    /// `Binary`, which downstream reads as "byte-wise ON PURPOSE" rather than
9039    /// as "unknown", so a dropped declaration presents as a deliberate one.
9040    ///
9041    /// This constructor copies everything by construction. A field added to
9042    /// `ColumnSchema` therefore reaches every re-describe site without anyone
9043    /// having to remember, which is the property the hand-written copy lists
9044    /// never had.
9045    ///
9046    /// The two fields a re-describe legitimately changes -- name and
9047    /// nullability -- are parameters. Callers that also retype the column
9048    /// assign `ty` afterwards.
9049    #[must_use]
9050    pub fn rederive(source: &Self, name: impl Into<String>, nullable: bool) -> Self {
9051        Self {
9052            name: name.into(),
9053            nullable,
9054            ..source.clone()
9055        }
9056    }
9057
9058    /// Builder-style helper to attach a default value to an otherwise
9059    /// plain column schema. Used by the engine when CREATE TABLE
9060    /// specifies `column TYPE DEFAULT <expr>`.
9061    #[must_use]
9062    pub fn with_default(mut self, default: Value<'static>) -> Self {
9063        self.default = Some(default);
9064        self
9065    }
9066
9067    /// v7.9.21 — builder for runtime-evaluated defaults
9068    /// (`DEFAULT now()`, `DEFAULT CURRENT_TIMESTAMP`, …).
9069    /// `expr` is the Expr's `Display` form, re-parsed by the
9070    /// engine at each INSERT.
9071    #[must_use]
9072    pub fn with_runtime_default(mut self, expr: impl Into<String>) -> Self {
9073        self.runtime_default = Some(expr.into());
9074        self
9075    }
9076
9077    /// Builder-style helper to mark a column as `AUTO_INCREMENT`.
9078    #[must_use]
9079    pub const fn with_auto_increment(mut self) -> Self {
9080        self.auto_increment = true;
9081        self
9082    }
9083}
9084
9085impl TableSchema {
9086    pub fn new(name: impl Into<String>, columns: Vec<ColumnSchema>) -> Self {
9087        Self {
9088            name: name.into(),
9089            columns,
9090            hot_tier_bytes: None,
9091            foreign_keys: Vec::new(),
9092            uniqueness_constraints: Vec::new(),
9093            exclusion_constraints: Vec::new(),
9094            checks: Vec::new(),
9095            partition_role: None,
9096            policies: Vec::new(),
9097            row_security: false,
9098            force_row_security: false,
9099            owner: None,
9100            acl: Vec::new(),
9101        }
9102    }
9103}
9104
9105// =========================================================================
9106// Persistent binary format for the catalog.
9107//
9108// Layout (little-endian throughout):
9109//
9110//   [magic "SPGDB001" 8 bytes][version u8]
9111//   [table_count u32]
9112//   for each table:
9113//       [name_len u16][name bytes]
9114//       [col_count u16]
9115//       for each col:
9116//           [name_len u16][name bytes]
9117//           [type_tag u8 + optional payload]
9118//               1=Int 2=BigInt 3=Float 4=Text 5=Bool
9119//               6=Vector(u32 dim)
9120//               7=SmallInt
9121//               8=Varchar(u32 max)
9122//               9=Char(u32 size)
9123//               10=Numeric(u8 precision, u8 scale)
9124//               11=Date
9125//               12=Timestamp
9126//           [nullable u8]   0/1
9127//           [default_tag u8] 0=none 1=value (followed by [value_tag u8] + bytes)
9128//       [row_count u32]
9129//       for each row, for each col, one [value_tag u8] + value bytes:
9130//           tag 0 (Null)     → no body
9131//           tag 1 (Int)      → i32 LE
9132//           tag 2 (BigInt)   → i64 LE
9133//           tag 3 (Float)    → f64 LE
9134//           tag 4 (Text)     → u16 LE len + UTF-8 bytes
9135//           tag 5 (Bool)     → u8 0/1
9136//           tag 6 (Vector)   → u32 LE dim + dim×f32 LE
9137//           tag 7 (SmallInt) → i16 LE
9138//           tag 8 (Numeric)  → i128 LE (16 bytes) + u8 scale
9139//           tag 9 (Date)     → i32 LE (days since Unix epoch)
9140//           tag 10 (Timestamp) → i64 LE (microseconds since Unix epoch)
9141//
9142// Bumped to version 3 when NUMERIC was added; to version 4 when
9143// AUTO_INCREMENT (per-column flag) + NSW index `kind` byte landed;
9144// to version 5 when DATE / TIMESTAMP were added; to version 6 when
9145// NSW graph topology started travelling on disk (v2.7); to version 7
9146// when the NSW topology became multi-layer HNSW (v2.13); to version 8
9147// when row encoding switched to schema-driven dense layout (v3.0.2 —
9148// per-row NULL bitmap + per-column fixed-width body, no per-cell type
9149// tag).
9150// =========================================================================
9151
9152const FILE_MAGIC: &[u8; 8] = b"SPGDB001";
9153/// Current catalog snapshot format version emitted by [`Catalog::serialize`].
9154///
9155/// v9 (v5.2) extends v8 by serialising `BTree` index entries directly — every
9156/// `(IndexKey, Vec<RowLocator>)` pair travels on disk with the v5.1
9157/// `RowLocator::write_le` tag-prefixed codec. v8 `BTree` indices stored no
9158/// entries at all (the map was rebuilt from `Table::rows` on load); v9
9159/// preserves on-disk Cold locators so freezer-produced cold-tier index
9160/// entries survive a catalog snapshot round-trip. v8 readers are accepted
9161/// by version dispatch in [`Catalog::deserialize`] — every entry decodes
9162/// as `RowLocator::Hot(_)` via `add_index` rebuild, identical to v5.1
9163/// behaviour.
9164/// v6.7.2 — bumped from 10 to 11 to append per-table
9165/// `hot_tier_bytes: Option<u64>` after the per-table indices
9166/// section. v10 catalogs (v6.7.1) load with `hot_tier_bytes =
9167/// None` for every table (the deserialiser short-circuits when
9168/// version < 11). v11 snapshots written by a pre-v6.7.2 binary
9169/// fail loudly at the version check, matching the v6.1.2 /
9170/// v6.1.4 / v6.2.0 / v6.7.1 envelope-bump upgrade fences.
9171///
9172/// v6.8.0 — bumped from 11 to 12: per-index
9173/// `included_columns: Vec<u16>` appended at the tail of each
9174/// index payload. v11 (= v6.7.2) catalogs load with
9175/// `included_columns = Vec::new()` for every index — same
9176/// "older readers, append-only extension" pattern as the v6.7.2
9177/// hot_tier_bytes byte.
9178/// v7.13.0 — bumped from 22 to 23. mailrs round-5 G3 / G10.
9179/// Per-table appendix gains two new sections:
9180///   * `checks: Vec<String>` — CHECK predicate sources (Display
9181///     form of the AST Expr); re-parsed on INSERT/UPDATE to
9182///     enforce against candidate rows. Same persistence pattern
9183///     as `Index::partial_predicate`.
9184///   * Per `UniquenessConstraint`: trailing `nulls_not_distinct:
9185///     u8` flag for PG 15+ `UNIQUE NULLS NOT DISTINCT (cols)`
9186///     semantics.
9187/// v22 catalogs deserialise with empty `checks` and every UC
9188/// at `nulls_not_distinct = false`.
9189/// v24 introduces:
9190///   * Index kind tag 4 = trigram-GIN (`gin_trgm_ops`-flavoured
9191///     `USING gin` over a TEXT/VARCHAR column). Payload shape is
9192///     identical to tag-3 GIN (String → Vec<RowLocator>); the
9193///     keys are PG-compatible 3-byte trigram shingles instead of
9194///     tsvector lexemes. v23 catalogs deserialise unchanged — no
9195///     v23 writer ever emitted tag 4.
9196/// v25 introduces:
9197///   * Per `TriggerDef`: trailing `enabled: u8` flag (mailrs
9198///     round-9 A.2.b — `ALTER TABLE … { ENABLE | DISABLE }
9199///     TRIGGER …`). v24 catalogs deserialise with every trigger
9200///     `enabled = true`, matching pre-v7.16.1 behaviour.
9201/// v26 introduces (v7.17.0 Phase 1.1):
9202///   * Trailing SEQUENCE catalog block after triggers. Encoded
9203///     as `u32 count` followed by per-sequence:
9204///     `name`, `data_type: u8` (0=SmallInt,1=Int,2=BigInt),
9205///     `start i64`, `increment i64`, `min_value i64`,
9206///     `max_value i64`, `cache i64`, `cycle u8`,
9207///     `owned_by_tag u8` (0=NONE, 1=Column → `table`,`column`),
9208///     `last_value i64`, `is_called u8`. v25-and-below catalogs
9209///     deserialise with an empty sequences map.
9210/// v27 introduces (v7.17.0 Phase 1.2):
9211///   * Trailing VIEW catalog block after sequences. Encoded as
9212///     `u32 count` followed by per-view:
9213///     `name`, `column_count u16`, then column names, then
9214///     `body` long-string. v26-and-below catalogs deserialise
9215///     with an empty views map.
9216/// v28 introduces (v7.17.0 Phase 1.3):
9217///   * Trailing MATERIALIZED VIEW source registry block after
9218///     views. Encoded as `u32 count` followed by per-entry:
9219///     `name`, `body` long-string. The materialised rows live
9220///     as a regular Table of the same name (already covered by
9221///     the pre-existing tables block). v27-and-below catalogs
9222///     deserialise with an empty map.
9223/// v29 introduces (v7.17.0 Phase 1.4):
9224///   * Per-table user_enum_type appendix (after the CHECK
9225///     appendix). Layout: `u16 count` followed by per-binding
9226///     `[u16 col_pos][str enum_name]`. Only columns whose
9227///     `user_enum_type` is Some land here; the catalog stays
9228///     compact for the common no-enum case.
9229///   * Trailing ENUM types catalog block after materialized
9230///     views. Encoded as `u32 count` followed by per-entry:
9231///     `name`, `u16 label_count`, then `label_count` short
9232///     strings. v28-and-below catalogs deserialise with an
9233///     empty enum_types map and every column's
9234///     `user_enum_type = None`.
9235/// v30 introduces (v7.17.0 Phase 1.5):
9236///   * Per-table user_domain_type appendix (after the
9237///     user_enum_type appendix). Same shape as the enum one.
9238///   * Trailing DOMAIN types catalog block after the enum
9239///     block. Encoded as `u32 count` followed by per-entry:
9240///     `name`, `data_type` byte, `nullable u8`,
9241///     `default_present u8` + optional default string,
9242///     `u16 check_count` then `check_count` Display-form
9243///     CHECK strings. v29-and-below catalogs deserialise with
9244///     an empty domain_types map and `user_domain_type = None`.
9245/// v31 introduces (v7.17.0 Phase 1.6):
9246///   * Trailing user-schemas block after the DOMAIN block.
9247///     Encoded as `u32 count` followed by `count` schema-name
9248///     short strings. Built-in schemas (`public`, `pg_catalog`,
9249///     `information_schema`) are NOT serialised — they're
9250///     hardcoded in `is_builtin_schema`. v30-and-below catalogs
9251///     deserialise with an empty user-schemas set.
9252/// v32 introduces (v7.17.0 Phase 2.1):
9253///   * Per-table on_update_runtime appendix (after the
9254///     user_domain_type appendix). Layout: `u16 count` followed
9255///     by per-binding `[u16 col_pos][str expr_src]`. Only
9256///     columns whose `on_update_runtime` is Some land here;
9257///     the catalog stays compact when no MySQL-shaped table
9258///     uses the attribute. v31-and-below catalogs deserialise
9259///     with every column's `on_update_runtime = None`.
9260/// v33 introduces (v7.17.0 Phase 2.2):
9261///   * Index kind tag 5 = fulltext-GIN (MySQL `FULLTEXT KEY`
9262///     surface over a TEXT / VARCHAR column). Payload shape is
9263///     identical to tag-3 / tag-4 GIN (`String → Vec<RowLocator>`);
9264///     the keys are lower-cased word lexemes (same rule as
9265///     `to_tsvector('simple', text)`). v32 catalogs deserialise
9266///     unchanged — no v32 writer ever emitted tag 5, and FULLTEXT
9267///     KEY was silently dropped pre-v7.17 so no rebuild shim is
9268///     needed for round-tripped catalogs.
9269/// v34 introduces (v7.17.0 Phase 2.5):
9270///   * Per-table collation appendix (after the on_update_runtime
9271///     appendix). Sparse layout: only columns whose `collation`
9272///     is non-Binary land here. `u16 count` then per-binding
9273///     `[u16 col_pos][u8 collation_tag]` where the tag matches
9274///     `Collation::TAG_*`. Snapshots written by v33-and-below
9275///     readers deserialise every column with `collation =
9276///     Binary`, preserving the prior byte-wise compare
9277///     semantics. Unknown tags read back as Binary too — keeps
9278///     a forward-compat path if a future v35 adds variants
9279///     and someone rolls back to a v34 reader.
9280/// v35 introduces (v7.17.0 Phase 4.4):
9281///   * Per-table is_unsigned appendix (after the collation
9282///     appendix). Sparse layout: only `is_unsigned = true`
9283///     columns land. `u16 count` then per-binding `[u16 col_pos]`.
9284///     v34-and-below catalogs deserialise every column as
9285///     `is_unsigned = false`, preserving the prior silent-
9286///     accept behaviour for negative inserts on UNSIGNED columns.
9287/// v46 introduces (v7.23, mailrs round-14):
9288///   * Escaped short-string codec — `write_str` lengths >= 0xFFFF
9289///     emit `[u16 0xFFFF][u32 real_len]` so TEXT cells (mail bodies,
9290///     document text) above 64 KiB encode instead of panicking.
9291///     One-way upgrade: v45-and-below readers reject v46 catalogs
9292///     loudly via the version gate; v46 readers decode v45 catalogs
9293///     with the plain-u16 rules (0xFFFF is a legitimate length
9294///     there).
9295/// v47 introduces (v7.27, mailrs round-21):
9296///   * Escaped lengths for the REMAINING u16-length cell payloads —
9297///     BYTEA cells, TEXT[] elements, tsvector lexemes and tsquery
9298///     terms — the same `[u16 0xFFFF][u32 real_len]` escape v46
9299///     gave short strings. Round-14 fixed TEXT and missed these;
9300///     round-21 fired the BYTEA twin during a production migration.
9301///     One-way upgrade, same posture as v46.
9302/// v48 introduces (v7.37.5 β-P2, sentori cutover window):
9303///   * `INTERVAL` becomes a real column type. Catalog tag 34 in
9304///     `write_data_type`; per-row body is a fixed 16 bytes
9305///     (i64 micros + i32 days + i32 months, LE, PG-byte-equal
9306///     field order). The runtime-only days collapse is gone —
9307///     `'1 day'` and `'24 hours'` are stored distinctly. One-way
9308///     upgrade: v47 catalogs without INTERVAL columns deserialise
9309///     identically; v47 readers fed a v48 catalog that contains
9310///     INTERVAL hit the explicit "unknown data type tag: 34"
9311///     fence in `read_data_type`.
9312/// v49 introduces (v7.37.6-B, sentori Epic 2 P0):
9313///   * Per-table partition role appendix(declarative
9314///     `PARTITION BY RANGE` parent / range child / DEFAULT
9315///     child)。Layout, written **after** the inline_set_variants
9316///     appendix and **before** the per-table block close:
9317///       `[u8 role_tag]`
9318///         0 = `None`(普通表,后向兼容默认)
9319///         1 = `Parent`:  `[u8 kind_tag (0=Range)]`
9320///                        `[u16 key_col_count]` `(× u16 col_pos)`
9321///                        `[u16 tmpl_count]` `(× str source)`
9322///         2 = `Range`:   `[str parent_name]` `[Bound]` `[Bound]`
9323///         3 = `Default`: `[str parent_name]`
9324///     `PartitionBound` codec:
9325///       `[u8 bound_tag]` 0=MinValue 1=MaxValue 2=TimestampTz(`[i64 LE micros]`)
9326///     v48-and-below readers stop after the inline_set_variants
9327///     block — they don't see this appendix and deserialise every
9328///     table with `partition_role = None`. v49 writers always emit
9329///     `[0]` for plain tables, so the encoding stays one-byte-cheap.
9330/// v50 introduces (v7.37.7, sentori Epic 3 P1):
9331///   * Per-table `generated_stored_expr` appendix(stored generated
9332///     columns — `GENERATED ALWAYS AS (<expr>) STORED`)。Layout,
9333///     written **after** the partition_role appendix and before
9334///     the per-table block close:
9335///       `[u16 binding_count]`
9336///       `binding_count × { [u16 col_pos][str expr_source] }`
9337///     Sparse — only generated columns land here, so plain-shape
9338///     catalogs stay byte-for-byte identical save for the new
9339///     u16 zero count. v49-and-below readers stop after the
9340///     partition_role appendix; v50 readers default every column
9341///     to `generated_stored_expr = None` when this block is absent.
9342/// v51 introduces (v7.37.8, sentori Epic 5 P2):
9343///   * Per-index tag byte 6 = `GinJsonb`(real posting-list GIN
9344///     over a JSONB column). Payload shape mirrors tag-3 / 4 / 5:
9345///     `[u32 posting_list_count]` then `(str token, u32 locator_count,
9346///     locators …)` per posting list. Same `write_str` /
9347///     `RowLocator::write_le` codec as the rest of the GIN family.
9348///     v50 catalogs never wrote tag 6(the same DDL loaded as a
9349///     BTree fallback); v51 readers see tag 6 explicitly and dispatch
9350///     into `IndexKind::GinJsonb`.
9351/// v52 introduces (v7.37.42-T2 ζ-B composite + domain metasystem):
9352///   * Trailing COMPOSITE-types catalog block after the
9353///     user-schemas block. Encoded as `u32 count` followed by
9354///     per-entry: `name`, `u16 field_count`, then `field_count`
9355///     `[str field_name][data_type]` pairs (`write_data_type` is
9356///     reused). v51-and-below catalogs deserialise with an empty
9357///     composite_types map; v52 readers tolerate v51 catalogs by
9358///     stopping at the schema block (no composite block present
9359///     ⇒ empty map). Composite types are referenced by columns
9360///     via `ColumnSchema.user_composite_type`, mirroring the
9361///     `user_enum_type` / `user_domain_type` pattern. The block
9362///     lands here (not as a per-table appendix) so dropping the
9363///     composite type registers globally and DROP TYPE can find it
9364///     without a table scan.
9365/// v53 introduces (v7.37.16 Epic W — cross-checkpoint tombstone
9366///   durability):
9367///   * Trailing per-table MVCC appendix carrying, for every row,
9368///     its `RowHeader` (`xmin:u64`, `xmax:u64`, `flags:u8`) and its
9369///     stable `RowId` (`u64`), followed by the relation's
9370///     `next_rowid:u64`. Layout per table (after the v50
9371///     generated_stored_expr block, before the table loop closes):
9372///       `[u32 row_count]` (== `Table::rows().len()`, cross-check)
9373///       per row in physical order:
9374///         `[u64 xmin][u64 xmax][u8 flags][u64 rowid]`
9375///       `[u64 next_rowid]`
9376///     v52-and-below catalogs never wrote this block; their reader
9377///     stops after the last per-table appendix and
9378///     `deserialize_rows` leaves every row `RowHeader::frozen()`
9379///     with dense 1..=N ids — the exact pre-v53 contract. A v53
9380///     reader instead reconstructs headers + ids VERBATIM, so a
9381///     tombstone-redo naming a row inserted before the last
9382///     checkpoint resolves by `RowId` across the base-snapshot
9383///     boundary (closing the coupling the Epic W WAL slices deferred
9384///     to this format bump). Because the reader routes on `version`,
9385///     the block is strictly backward-compatible: old images load
9386///     byte-for-byte as before. `SPG_MVCC_INPLACE` is unaffected —
9387///     a gate-off database's rows are all frozen/alive, so
9388///     persisting + restoring their headers is observationally a
9389///     no-op.
9390/// v7.38 (read01 P5.05) — v54 appends a CRC32C over the whole preceding
9391/// image so a corrupted `base.spg` is caught on load instead of silently
9392/// deserialising garbage. Older images (v8..=53) carry no trailer and load
9393/// unchanged.
9394/// v7.39 (round 210) — v72 appends a per-table EXCLUDE-constraint appendix
9395/// (sparse: only tables carrying an EXCLUDE write it) at the very end of the
9396/// per-table block, after the column-ACL appendix. A v71 reader stops before
9397/// it and its tables read back with no exclusion constraints, which is what
9398/// they were.
9399/// v7.39 (round 220) — v73 appends a per-table identity-RESTART appendix
9400/// (sparse: [u16 count] then per entry [u16 col_pos][i64 LE floor]) after
9401/// the EXCLUDE appendix. A v72 reader stops before it; its columns read
9402/// back with no RESTART floor, losing only an un-consumed
9403/// `ALTER … RESTART WITH` across a restart.
9404/// r1039 — v90 adds index-key tags 4 (bytea) and 5 (the canonical
9405/// numeric key), so BYTEA and NUMERIC columns carry a real B-tree
9406/// instead of falling back to a scan. A v89 reader meeting either tag
9407/// reports a corrupt catalog rather than mis-reading it, which is the
9408/// same forward-compatibility story tag 3 (uuid) had at v36.
9409const FILE_VERSION: u8 = 93;
9410
9411/// v7.37 (round 833) — the codec version to decode a row that
9412/// [`encode_row_body_dense`] has just produced.
9413///
9414/// That encoder always writes the newest form, and every decoder gate is
9415/// a `codec_version >= N` feature test, so a freshly encoded row must be
9416/// read at the current version. Cold segments carry their own version in
9417/// their header and keep passing that; this is for in-process round
9418/// trips — sort runs on temp storage — where the bytes never outlive the
9419/// build that wrote them.
9420pub const CURRENT_ROW_CODEC_VERSION: u8 = FILE_VERSION;
9421/// First version that appends the trailing CRC32C integrity trailer.
9422const FILE_VERSION_CRC_TRAILER: u8 = 54;
9423/// Oldest format version [`Catalog::deserialize`] still accepts. v8 is the
9424/// v3.0.2 dense-row layout; pre-v8 catalogs require an offline migration.
9425const MIN_SUPPORTED_FILE_VERSION: u8 = 8;
9426
9427// IndexKey wire format (v9):
9428//   tag 0 = Int  → [i64 LE]
9429//   tag 1 = Text → [u16 LE len + UTF-8 bytes] (via write_str / read_str)
9430//   tag 2 = Bool → [u8 0/1]
9431const INDEX_KEY_TAG_INT: u8 = 0;
9432const INDEX_KEY_TAG_TEXT: u8 = 1;
9433const INDEX_KEY_TAG_BOOL: u8 = 2;
9434/// v7.17.0 — `IndexKey::Uuid([u8; 16])`. Body = raw 16 bytes
9435/// (RFC 4122 byte order). Persisted only in FILE_VERSION 36+
9436/// catalogs.
9437const INDEX_KEY_TAG_UUID: u8 = 3;
9438/// r1039 — `IndexKey::Bytes`. Body = [u32 LE len][raw bytes].
9439/// Persisted only in FILE_VERSION 90+ catalogs.
9440const INDEX_KEY_TAG_BYTES: u8 = 4;
9441/// r1039 — `IndexKey::Numeric`. Body = [u8 class][u8 neg][i32 LE exp]
9442/// [u32 LE digit count][one byte per decimal digit, 0..=9, MSD first].
9443/// Persisted only in FILE_VERSION 90+ catalogs.
9444const INDEX_KEY_TAG_NUMERIC: u8 = 5;
9445/// v7.38.1 (L12) — `IndexKey::Null`, a NULL component inside a
9446/// composite key. No body. Persisted only inside tag-7 multi-index
9447/// payloads, FILE_VERSION 91+.
9448const INDEX_KEY_TAG_NULL: u8 = 6;
9449
9450impl Catalog {
9451    /// Serialize the whole catalog (schema + every row) into a self-contained
9452    /// byte buffer. Format is documented above the impl block.
9453    pub fn serialize(&self) -> Vec<u8> {
9454        let mut out = Vec::with_capacity(64);
9455        out.extend_from_slice(FILE_MAGIC);
9456        out.push(FILE_VERSION);
9457        write_u32(
9458            &mut out,
9459            u32::try_from(self.tables.len()).expect("≤ 4G tables"),
9460        );
9461        for t in &self.tables {
9462            write_str(&mut out, &t.schema.name);
9463            write_u16(
9464                &mut out,
9465                u16::try_from(t.schema.columns.len()).expect("≤ 65k columns/table"),
9466            );
9467            for c in &t.schema.columns {
9468                write_str(&mut out, &c.name);
9469                write_data_type(&mut out, c.ty);
9470                out.push(u8::from(c.nullable));
9471                match &c.default {
9472                    None => out.push(0),
9473                    Some(v) => {
9474                        out.push(1);
9475                        write_value(&mut out, v);
9476                    }
9477                }
9478                out.push(u8::from(c.auto_increment));
9479            }
9480            write_u32(
9481                &mut out,
9482                u32::try_from(t.rows.len()).expect("≤ 4G rows/table"),
9483            );
9484            // v3.0.2 dense row encoding (FILE_VERSION 8): per-row NULL
9485            // bitmap, then tightly-packed bodies. Identical wire format
9486            // as before — extracted into `encode_row_body_dense` so cold-
9487            // tier segments (v5.1+) can share the encoding.
9488            for row in &t.rows {
9489                out.extend_from_slice(&encode_row_body_dense(row, &t.schema));
9490            }
9491            // Index definitions. Per-index payload:
9492            //   [name][col_pos u16][kind u8]
9493            //     kind 0 = B-tree           (no params — rebuilt on load)
9494            //     kind 1 = NSW graph        (u16 M + serialized graph)
9495            // For NSW the graph topology travels on disk so startup
9496            // doesn't re-run the O(n²M) rebuild — see v2.7 notes.
9497            write_u16(
9498                &mut out,
9499                u16::try_from(t.indices.len()).expect("≤ 65k indices/table"),
9500            );
9501            for idx in &t.indices {
9502                write_str(&mut out, &idx.name);
9503                write_u16(
9504                    &mut out,
9505                    u16::try_from(idx.column_position).expect("≤ 65k columns/table"),
9506                );
9507                match &idx.kind {
9508                    IndexKind::BTree(map) => {
9509                        out.push(0);
9510                        // v9: serialise the full PB map. Each entry's
9511                        // RowLocator list travels with the tag-prefixed
9512                        // codec from `row_locator::write_le`, so freezer-
9513                        // produced Cold locators survive a snapshot
9514                        // round-trip. v8 BTree wrote nothing here and
9515                        // rebuilt from rows — v9 readers tolerate v8 by
9516                        // version dispatch in `Catalog::deserialize`.
9517                        write_u32(
9518                            &mut out,
9519                            u32::try_from(map.len()).expect("≤ 4G index entries/index"),
9520                        );
9521                        for (key, locators) in map {
9522                            write_index_key(&mut out, key);
9523                            write_u32(
9524                                &mut out,
9525                                u32::try_from(locators.len()).expect("≤ 4G locators/key"),
9526                            );
9527                            for loc in locators {
9528                                loc.write_le(&mut out);
9529                            }
9530                        }
9531                    }
9532                    // v7.38.1 (L12) — tag byte 7 = BTreeMulti. Payload
9533                    // mirrors the tag-0 BTree encoding, with each key
9534                    // written as `[u16 arity]` followed by that many
9535                    // `write_index_key` components. FILE_VERSION 91+;
9536                    // older catalogs never carried a multi index, so no
9537                    // migration shim is needed.
9538                    IndexKind::BTreeMulti(map) => {
9539                        out.push(7);
9540                        write_u32(
9541                            &mut out,
9542                            u32::try_from(map.len()).expect("≤ 4G index entries/index"),
9543                        );
9544                        for (key, locators) in map {
9545                            write_u16(
9546                                &mut out,
9547                                u16::try_from(key.len()).expect("≤ 65k key components"),
9548                            );
9549                            for component in key.iter() {
9550                                write_index_key(&mut out, component);
9551                            }
9552                            write_u32(
9553                                &mut out,
9554                                u32::try_from(locators.len()).expect("≤ 4G locators/key"),
9555                            );
9556                            for loc in locators {
9557                                loc.write_le(&mut out);
9558                            }
9559                        }
9560                    }
9561                    IndexKind::Nsw(g) => {
9562                        out.push(1);
9563                        write_u16(&mut out, u16::try_from(g.m).expect("≤ 65k NSW neighbours"));
9564                        write_nsw_graph(&mut out, g);
9565                    }
9566                    IndexKind::Brin { column_type, .. } => {
9567                        // v6.7.1 — tag byte 2 = BRIN. Payload is the
9568                        // column type code (1 byte mapping to the
9569                        // shared DataType numeric encoding); no
9570                        // further data — BRIN summaries live in
9571                        // cold segments, not the catalog.
9572                        out.push(2);
9573                        write_data_type(&mut out, *column_type);
9574                    }
9575                    IndexKind::Gin(map) => {
9576                        // v7.12.3 — tag byte 3 = GIN. Payload mirrors
9577                        // the BTree encoding but with String (lexeme
9578                        // word) keys instead of IndexKey. Tag-prefixed
9579                        // RowLocator codec so freezer-produced Cold
9580                        // locators survive snapshot round-trip.
9581                        // FILE_VERSION 21+; v20 catalogs never wrote a
9582                        // GIN index (the AM degraded to BTree fallback
9583                        // pre-v7.12.3), so no migration shim is needed.
9584                        out.push(3);
9585                        write_u32(
9586                            &mut out,
9587                            u32::try_from(map.len()).expect("≤ 4G GIN posting lists"),
9588                        );
9589                        for (word, locators) in map {
9590                            write_str(&mut out, word);
9591                            write_u32(
9592                                &mut out,
9593                                u32::try_from(locators.len()).expect("≤ 4G locators/posting list"),
9594                            );
9595                            for loc in locators {
9596                                loc.write_le(&mut out);
9597                            }
9598                        }
9599                    }
9600                    IndexKind::GinTrgm(map) => {
9601                        // v7.15.0 — tag byte 4 = GinTrgm
9602                        // (`gin_trgm_ops` GIN over a TEXT column).
9603                        // Payload shape is identical to tag-3 GIN —
9604                        // `String → Vec<RowLocator>` posting lists.
9605                        // The String keys are 3-byte trigrams instead
9606                        // of tsvector lexemes; the deserializer
9607                        // dispatches on the tag, not the key shape.
9608                        // FILE_VERSION 24+; v23 catalogs never wrote
9609                        // a trigram-GIN.
9610                        out.push(4);
9611                        write_u32(
9612                            &mut out,
9613                            u32::try_from(map.len()).expect("≤ 4G trigram-GIN posting lists"),
9614                        );
9615                        for (tri, locators) in map {
9616                            write_str(&mut out, tri);
9617                            write_u32(
9618                                &mut out,
9619                                u32::try_from(locators.len()).expect("≤ 4G locators/posting list"),
9620                            );
9621                            for loc in locators {
9622                                loc.write_le(&mut out);
9623                            }
9624                        }
9625                    }
9626                    IndexKind::GinFulltext(map) => {
9627                        // v7.17.0 Phase 2.2 — tag byte 5 =
9628                        // GinFulltext (MySQL `FULLTEXT KEY` GIN
9629                        // over a TEXT/VARCHAR column). Payload
9630                        // shape mirrors tag-3 / tag-4 GIN —
9631                        // `String → Vec<RowLocator>` posting
9632                        // lists keyed by lower-cased word
9633                        // lexemes. FILE_VERSION 33+; v32 catalogs
9634                        // never wrote a fulltext-GIN (FULLTEXT
9635                        // KEY was silently dropped pre-v7.17).
9636                        out.push(5);
9637                        write_u32(
9638                            &mut out,
9639                            u32::try_from(map.len()).expect("≤ 4G fulltext-GIN posting lists"),
9640                        );
9641                        for (lex, locators) in map {
9642                            write_str(&mut out, lex);
9643                            write_u32(
9644                                &mut out,
9645                                u32::try_from(locators.len()).expect("≤ 4G locators/posting list"),
9646                            );
9647                            for loc in locators {
9648                                loc.write_le(&mut out);
9649                            }
9650                        }
9651                    }
9652                    IndexKind::GinJsonb(map) => {
9653                        // v7.37.8 — tag byte 6 = GinJsonb
9654                        // (real posting-list GIN over a JSONB
9655                        // column; sentori Epic 5 P2). Payload
9656                        // shape mirrors tag-3 / 4 / 5 — keys are
9657                        // the canonical `(path, leaf)` tokens
9658                        // from `jsonb_gin::extract_tokens`.
9659                        // FILE_VERSION 51+; v50 catalogs never
9660                        // wrote a JSONB-GIN (the same DDL loaded
9661                        // as a BTree fallback).
9662                        out.push(6);
9663                        write_u32(
9664                            &mut out,
9665                            u32::try_from(map.len()).expect("≤ 4G JSONB-GIN posting lists"),
9666                        );
9667                        for (token, locators) in map {
9668                            write_str(&mut out, token);
9669                            write_u32(
9670                                &mut out,
9671                                u32::try_from(locators.len()).expect("≤ 4G locators/posting list"),
9672                            );
9673                            for loc in locators {
9674                                loc.write_le(&mut out);
9675                            }
9676                        }
9677                    }
9678                }
9679                // v6.8.0 — included_columns appendix per index.
9680                // Layout: [u16 num_included][num × u16 column_position].
9681                // v11 readers stop before this u16 (deserialise loop
9682                // gated on version >= 12); v12+ readers always
9683                // consume it. Empty Vec serialises as a bare 0u16.
9684                write_u16(
9685                    &mut out,
9686                    u16::try_from(idx.included_columns.len()).expect("≤ 65k INCLUDE columns/index"),
9687                );
9688                for col_pos in &idx.included_columns {
9689                    write_u16(
9690                        &mut out,
9691                        u16::try_from(*col_pos).expect("≤ 65k columns/table"),
9692                    );
9693                }
9694                // v6.8.1 — partial_predicate appendix per index.
9695                // Layout: [u8 has_pred][u16 LE len][bytes (if has_pred)].
9696                // Same v12 gate as included_columns.
9697                match &idx.partial_predicate {
9698                    None => out.push(0),
9699                    Some(pred) => {
9700                        out.push(1);
9701                        write_str(&mut out, pred);
9702                    }
9703                }
9704                // v6.8.2 — expression appendix. Same shape as
9705                // partial_predicate.
9706                match &idx.expression {
9707                    None => out.push(0),
9708                    Some(expr) => {
9709                        out.push(1);
9710                        write_str(&mut out, expr);
9711                    }
9712                }
9713                // v7.9.29 — is_unique appendix (FILE_VERSION 16+).
9714                // Single byte 0/1. v15-and-below readers stop before
9715                // this byte; v16 readers always consume it. mailrs K1.
9716                out.push(u8::from(idx.is_unique));
9717                // v7.9.29 — extra_column_positions appendix.
9718                // Layout: [u16 count][count × u16 column_position].
9719                write_u16(
9720                    &mut out,
9721                    u16::try_from(idx.extra_column_positions.len())
9722                        .expect("≤ 65k extra cols / index"),
9723                );
9724                for cp in &idx.extra_column_positions {
9725                    write_u16(&mut out, u16::try_from(*cp).expect("≤ 65k columns/table"));
9726                }
9727                // v7.39 (read01 round 52) — nulls_not_distinct (FILE_VERSION
9728                // 62+). Appended at the end of the per-index block so the v16
9729                // layout above is untouched; v61-and-below readers stop before
9730                // this byte and default the flag to false (NULLS DISTINCT).
9731                out.push(u8::from(idx.nulls_not_distinct));
9732                // v7.39 (round 537) — the key column's ordering clause
9733                // (FILE_VERSION 83+).
9734                out.push(u8::from(idx.descending));
9735                out.push(match idx.nulls_first {
9736                    None => 0,
9737                    Some(true) => 1,
9738                    Some(false) => 2,
9739                });
9740                // v7.39 (round 538) — the key's explicit collation
9741                // (FILE_VERSION 84+).
9742                match &idx.collation {
9743                    Some(c) => {
9744                        out.push(1);
9745                        write_str(&mut out, c);
9746                    }
9747                    None => out.push(0),
9748                }
9749            }
9750            // v6.7.2 — per-table hot_tier_bytes Option<u64>.
9751            // Layout: [u8 has_value][u64 LE value (if has_value)].
9752            // v10 readers stop before this byte (deserialise loop
9753            // gated on version >= 11); v11+ readers always
9754            // consume it.
9755            match t.schema.hot_tier_bytes {
9756                None => out.push(0),
9757                Some(n) => {
9758                    out.push(1);
9759                    out.extend_from_slice(&n.to_le_bytes());
9760                }
9761            }
9762            // v7.6.1 — FOREIGN KEY appendix (catalog FILE_VERSION 13+).
9763            // Layout: [u16 LE fk_count]
9764            //   per fk:
9765            //     [u8 has_name] [str name (if has_name)]
9766            //     [u16 LE local_arity] [u16 LE local_pos]*arity
9767            //     [str parent_table]
9768            //     [u16 LE parent_arity] [u16 LE parent_pos]*arity
9769            //     [u8 on_delete_tag] [u8 on_update_tag]
9770            // Older catalogs (v12 and below) skip this block entirely;
9771            // their reader stops before this byte.
9772            write_u16(
9773                &mut out,
9774                u16::try_from(t.schema.foreign_keys.len()).expect("≤ 65k FKs/table"),
9775            );
9776            for fk in &t.schema.foreign_keys {
9777                match &fk.name {
9778                    None => out.push(0),
9779                    Some(n) => {
9780                        out.push(1);
9781                        write_str(&mut out, n);
9782                    }
9783                }
9784                write_u16(
9785                    &mut out,
9786                    u16::try_from(fk.local_columns.len()).expect("≤ 65k FK columns"),
9787                );
9788                for &p in &fk.local_columns {
9789                    write_u16(&mut out, u16::try_from(p).expect("≤ 65k columns/table"));
9790                }
9791                write_str(&mut out, &fk.parent_table);
9792                write_u16(
9793                    &mut out,
9794                    u16::try_from(fk.parent_columns.len()).expect("≤ 65k FK parent columns"),
9795                );
9796                for &p in &fk.parent_columns {
9797                    write_u16(&mut out, u16::try_from(p).expect("≤ 65k columns/table"));
9798                }
9799                out.push(fk.on_delete.tag());
9800                out.push(fk.on_update.tag());
9801                // v7.38 (read01, T29) — MATCH type tag (FILE_VERSION 55+).
9802                out.push(fk.match_type.tag());
9803                // v7.39 (round 288) — constraint timing (FILE_VERSION 79+).
9804                // One byte, bit 0 = DEFERRABLE, bit 1 = INITIALLY DEFERRED.
9805                out.push(u8::from(fk.deferrable) | (u8::from(fk.initially_deferred) << 1));
9806            }
9807            // v7.9.19 — UniquenessConstraint appendix (catalog
9808            // FILE_VERSION 15+). Layout per table after the FK
9809            // block:
9810            //   [u16 count]
9811            //     per constraint:
9812            //       [u8 is_primary_key]
9813            //       [u16 arity][u16 col_pos]*arity
9814            // Older catalogs (v14 and below) skip this block.
9815            write_u16(
9816                &mut out,
9817                u16::try_from(t.schema.uniqueness_constraints.len())
9818                    .expect("≤ 65k uniqueness constraints/table"),
9819            );
9820            for uc in &t.schema.uniqueness_constraints {
9821                out.push(u8::from(uc.is_primary_key));
9822                write_u16(
9823                    &mut out,
9824                    u16::try_from(uc.columns.len()).expect("≤ 65k cols in uniqueness constraint"),
9825                );
9826                for &p in &uc.columns {
9827                    write_u16(&mut out, u16::try_from(p).expect("≤ 65k columns/table"));
9828                }
9829                // v7.13.0 — `nulls_not_distinct` flag
9830                // (FILE_VERSION 23+). Always written by writers at
9831                // version 23+; deserialise gates on `version >= 23`
9832                // so v22-and-below catalogs round-trip cleanly.
9833                out.push(u8::from(uc.nulls_not_distinct));
9834            }
9835            // v7.9.21 — runtime_default appendix per table.
9836            // Layout: [u16 count] then for each:
9837            //   [u16 col_pos][str expr]
9838            // Only columns whose runtime_default is Some land here;
9839            // catalog stays compact for the common literal-default
9840            // case.
9841            let mut rt_defaults: Vec<(usize, &str)> = Vec::new();
9842            for (i, c) in t.schema.columns.iter().enumerate() {
9843                if let Some(e) = &c.runtime_default {
9844                    rt_defaults.push((i, e.as_str()));
9845                }
9846            }
9847            write_u16(
9848                &mut out,
9849                u16::try_from(rt_defaults.len()).expect("≤ 65k runtime defaults/table"),
9850            );
9851            for (pos, expr) in rt_defaults {
9852                write_u16(&mut out, u16::try_from(pos).expect("≤ 65k columns/table"));
9853                write_str(&mut out, expr);
9854            }
9855            // v7.13.0 — CHECK constraint appendix per table.
9856            // Layout: [u16 count] then `count` Display-form
9857            // expression strings. Re-parsed on every INSERT/UPDATE
9858            // by the engine. FILE_VERSION 23+ only; v22 readers
9859            // never reach this block because the writer also moves
9860            // to v23 in lock-step.
9861            write_u16(
9862                &mut out,
9863                u16::try_from(t.schema.checks.len()).expect("≤ 65k CHECK constraints/table"),
9864            );
9865            for c in &t.schema.checks {
9866                // v7.39 (read01 round 48) — the expr stays in this v23
9867                // appendix (byte layout unchanged for old readers); the
9868                // name rides the v60 constraint-name appendix at the tail.
9869                write_str(&mut out, c.expr.as_str());
9870            }
9871            // v7.17.0 Phase 1.4 — per-table user_enum_type
9872            // appendix. Layout: [u16 count] then
9873            // [u16 col_pos][str enum_name] per binding. Only
9874            // columns whose user_enum_type is Some land here.
9875            let mut enum_bindings: Vec<(usize, &str)> = Vec::new();
9876            for (i, c) in t.schema.columns.iter().enumerate() {
9877                if let Some(e) = &c.user_enum_type {
9878                    enum_bindings.push((i, e.as_str()));
9879                }
9880            }
9881            write_u16(
9882                &mut out,
9883                u16::try_from(enum_bindings.len()).expect("≤ 65k enum-typed columns/table"),
9884            );
9885            for (pos, ename) in enum_bindings {
9886                write_u16(&mut out, u16::try_from(pos).expect("≤ 65k columns/table"));
9887                write_str(&mut out, ename);
9888            }
9889            // v7.17.0 Phase 1.5 — per-table user_domain_type
9890            // appendix. Same layout as the enum one. v29-and-
9891            // below readers stop after the enum appendix.
9892            let mut domain_bindings: Vec<(usize, &str)> = Vec::new();
9893            for (i, c) in t.schema.columns.iter().enumerate() {
9894                if let Some(d) = &c.user_domain_type {
9895                    domain_bindings.push((i, d.as_str()));
9896                }
9897            }
9898            write_u16(
9899                &mut out,
9900                u16::try_from(domain_bindings.len()).expect("≤ 65k domain-typed columns/table"),
9901            );
9902            for (pos, dname) in domain_bindings {
9903                write_u16(&mut out, u16::try_from(pos).expect("≤ 65k columns/table"));
9904                write_str(&mut out, dname);
9905            }
9906            // v7.17.0 Phase 2.1 — per-table on_update_runtime
9907            // appendix. Sparse: only ON UPDATE-bound columns.
9908            let mut on_update_bindings: Vec<(usize, &str)> = Vec::new();
9909            for (i, c) in t.schema.columns.iter().enumerate() {
9910                if let Some(e) = &c.on_update_runtime {
9911                    on_update_bindings.push((i, e.as_str()));
9912                }
9913            }
9914            write_u16(
9915                &mut out,
9916                u16::try_from(on_update_bindings.len()).expect("≤ 65k ON UPDATE columns/table"),
9917            );
9918            for (pos, expr_src) in on_update_bindings {
9919                write_u16(&mut out, u16::try_from(pos).expect("≤ 65k columns/table"));
9920                write_str(&mut out, expr_src);
9921            }
9922            // v7.17.0 Phase 2.5 — per-table collation appendix.
9923            // Sparse: only non-Binary columns land. Layout:
9924            // `[u16 count][u16 col_pos][u8 tag] × count`.
9925            let mut coll_bindings: Vec<(usize, u8)> = Vec::new();
9926            for (i, c) in t.schema.columns.iter().enumerate() {
9927                let tag = match c.collation {
9928                    Collation::Binary => continue,
9929                    Collation::CaseInsensitive => Collation::TAG_CASE_INSENSITIVE,
9930                };
9931                coll_bindings.push((i, tag));
9932            }
9933            write_u16(
9934                &mut out,
9935                u16::try_from(coll_bindings.len()).expect("≤ 65k collation bindings/table"),
9936            );
9937            for (pos, tag) in coll_bindings {
9938                write_u16(&mut out, u16::try_from(pos).expect("≤ 65k columns/table"));
9939                out.push(tag);
9940            }
9941            // v7.17.0 Phase 4.4 — per-table is_unsigned appendix.
9942            // Sparse: only UNSIGNED columns land. Layout:
9943            // `[u16 count][u16 col_pos] × count`.
9944            let mut unsigned_bindings: Vec<usize> = Vec::new();
9945            for (i, c) in t.schema.columns.iter().enumerate() {
9946                if c.is_unsigned {
9947                    unsigned_bindings.push(i);
9948                }
9949            }
9950            write_u16(
9951                &mut out,
9952                u16::try_from(unsigned_bindings.len()).expect("≤ 65k UNSIGNED columns/table"),
9953            );
9954            for pos in unsigned_bindings {
9955                write_u16(&mut out, u16::try_from(pos).expect("≤ 65k columns/table"));
9956            }
9957            // v7.17.0 Phase 3.P0-36 — per-table inline_enum_variants
9958            // appendix. Sparse: only ENUM columns land. Layout:
9959            // `[u16 count] then per binding [u16 col_pos]
9960            // [u16 variant_count] then variant strings`.
9961            // FILE_VERSION 41+; v40 readers never reach this block.
9962            let mut enum_inline_bindings: Vec<(usize, &[String])> = Vec::new();
9963            for (i, c) in t.schema.columns.iter().enumerate() {
9964                if let Some(vs) = &c.inline_enum_variants {
9965                    enum_inline_bindings.push((i, vs.as_slice()));
9966                }
9967            }
9968            write_u16(
9969                &mut out,
9970                u16::try_from(enum_inline_bindings.len()).expect("≤ 65k inline-ENUM columns/table"),
9971            );
9972            for (pos, variants) in enum_inline_bindings {
9973                write_u16(&mut out, u16::try_from(pos).expect("≤ 65k columns/table"));
9974                write_u16(
9975                    &mut out,
9976                    u16::try_from(variants.len()).expect("≤ 65k variants/ENUM"),
9977                );
9978                for v in variants {
9979                    write_str(&mut out, v.as_str());
9980                }
9981            }
9982            // v7.17.0 Phase 3.P0-37 — per-table inline_set_variants
9983            // appendix. Same layout as the inline ENUM block.
9984            // FILE_VERSION 42+; v41 readers never reach this block.
9985            let mut set_inline_bindings: Vec<(usize, &[String])> = Vec::new();
9986            for (i, c) in t.schema.columns.iter().enumerate() {
9987                if let Some(vs) = &c.inline_set_variants {
9988                    set_inline_bindings.push((i, vs.as_slice()));
9989                }
9990            }
9991            write_u16(
9992                &mut out,
9993                u16::try_from(set_inline_bindings.len()).expect("≤ 65k inline-SET columns/table"),
9994            );
9995            for (pos, variants) in set_inline_bindings {
9996                write_u16(&mut out, u16::try_from(pos).expect("≤ 65k columns/table"));
9997                write_u16(
9998                    &mut out,
9999                    u16::try_from(variants.len()).expect("≤ 65k variants/SET"),
10000                );
10001                for v in variants {
10002                    write_str(&mut out, v.as_str());
10003                }
10004            }
10005            // v7.37.6-B — partition role appendix(FILE_VERSION 49+)。
10006            // Layout 详见 FILE_VERSION 49 docstring。普通表 = 单字节 0。
10007            write_partition_role(&mut out, t.schema.partition_role.as_ref());
10008            // v7.37.7 — per-table generated_stored_expr appendix
10009            // (FILE_VERSION 50+). Sparse: only columns whose
10010            // generated_stored_expr is Some land here.
10011            let mut gen_bindings: Vec<(usize, &str)> = Vec::new();
10012            for (i, c) in t.schema.columns.iter().enumerate() {
10013                if let Some(src) = &c.generated_stored_expr {
10014                    gen_bindings.push((i, src.as_str()));
10015                }
10016            }
10017            write_u16(
10018                &mut out,
10019                u16::try_from(gen_bindings.len()).expect("≤ 65k GENERATED STORED columns/table"),
10020            );
10021            for (pos, src) in gen_bindings {
10022                write_u16(&mut out, u16::try_from(pos).expect("≤ 65k columns/table"));
10023                write_str(&mut out, src);
10024            }
10025            // v7.38 (read01) — per-table default_text appendix
10026            // (FILE_VERSION 58+). Sparse: only columns whose default_text
10027            // is Some land here. Mirrors the generated_stored_expr shape.
10028            let mut default_texts: Vec<(usize, &str)> = Vec::new();
10029            for (i, c) in t.schema.columns.iter().enumerate() {
10030                if let Some(src) = &c.default_text {
10031                    default_texts.push((i, src.as_str()));
10032                }
10033            }
10034            write_u16(
10035                &mut out,
10036                u16::try_from(default_texts.len()).expect("≤ 65k defaulted columns/table"),
10037            );
10038            for (pos, src) in default_texts {
10039                write_u16(&mut out, u16::try_from(pos).expect("≤ 65k columns/table"));
10040                write_str(&mut out, src);
10041            }
10042            // v7.39 (RLS) — per-table policy appendix + the two RLS flags
10043            // (FILE_VERSION 59+). Written after the default_text block and
10044            // before the MVCC row appendix, so a v58 reader stops before it.
10045            // Layout: [u8 row_security][u8 force] [u16 policy_count] then per
10046            // policy: [str name][u8 cmd][u8 permissive][u16 role_count]
10047            // (role_count × str) [u8 has_using](+str)[u8 has_check](+str).
10048            out.push(u8::from(t.schema.row_security));
10049            out.push(u8::from(t.schema.force_row_security));
10050            write_u16(
10051                &mut out,
10052                u16::try_from(t.schema.policies.len()).expect("≤ 65k policies/table"),
10053            );
10054            for p in &t.schema.policies {
10055                write_str(&mut out, &p.name);
10056                out.push(p.cmd.to_wire_byte());
10057                out.push(u8::from(p.permissive));
10058                write_u16(
10059                    &mut out,
10060                    u16::try_from(p.roles.len()).expect("≤ 65k roles/policy"),
10061                );
10062                for r in &p.roles {
10063                    write_str(&mut out, r);
10064                }
10065                match &p.using_expr {
10066                    Some(s) => {
10067                        out.push(1);
10068                        write_str(&mut out, s);
10069                    }
10070                    None => out.push(0),
10071                }
10072                match &p.with_check_expr {
10073                    Some(s) => {
10074                        out.push(1);
10075                        write_str(&mut out, s);
10076                    }
10077                    None => out.push(0),
10078                }
10079            }
10080            // v7.37.16 (Epic W) — per-row MVCC header + stable RowId
10081            // appendix (FILE_VERSION 53+). Persists xmin/xmax/flags +
10082            // RowId for every row so a tombstone naming a pre-checkpoint
10083            // row survives a serialize→deserialize base restore
10084            // (cross-checkpoint tombstone durability). `headers` /
10085            // `rowids` are lock-step parallel to `rows` (invariant held
10086            // at every mutation boundary), so the count is `rows.len()`
10087            // and the zipped walk visits them in physical row order —
10088            // the same order the rows block above was written in. v52
10089            // readers never reach this block (the writer also moves to
10090            // v53 in lock-step); a v53 reader restores headers + ids
10091            // verbatim instead of freezing + dense-assigning.
10092            debug_assert_eq!(
10093                t.rows.len(),
10094                t.headers.len(),
10095                "headers must be lock-step with rows at serialize"
10096            );
10097            debug_assert_eq!(
10098                t.rows.len(),
10099                t.rowids.len(),
10100                "rowids must be lock-step with rows at serialize"
10101            );
10102            write_u32(
10103                &mut out,
10104                u32::try_from(t.rows.len()).expect("≤ 4G rows/table"),
10105            );
10106            for (h, rid) in t.headers.iter().zip(t.rowids.iter()) {
10107                out.extend_from_slice(&h.xmin.to_le_bytes());
10108                out.extend_from_slice(&h.xmax.to_le_bytes());
10109                out.push(h.flags);
10110                out.extend_from_slice(&rid.0.to_le_bytes());
10111            }
10112            out.extend_from_slice(
10113                &t.next_rowid
10114                    .load(core::sync::atomic::Ordering::Relaxed)
10115                    .to_le_bytes(),
10116            );
10117            // v7.39 (read01 round 48) — constraint-name appendix
10118            // (FILE_VERSION 60+). Index-aligned to the CHECK and
10119            // uniqueness-constraint appendices written above, so the
10120            // existing byte layouts stay untouched and a v59 catalog still
10121            // decodes (its constraints just come back unnamed).
10122            // Layout: [u16 check_count] then per check
10123            //         [u8 has_name] ([str name] when has_name)
10124            //         [u16 uc_count] then per uc the same pair.
10125            write_u16(
10126                &mut out,
10127                u16::try_from(t.schema.checks.len()).expect("≤ 65k CHECK constraints/table"),
10128            );
10129            for c in &t.schema.checks {
10130                match &c.name {
10131                    Some(n) => {
10132                        out.push(1);
10133                        write_str(&mut out, n);
10134                    }
10135                    None => out.push(0),
10136                }
10137            }
10138            write_u16(
10139                &mut out,
10140                u16::try_from(t.schema.uniqueness_constraints.len())
10141                    .expect("≤ 65k uniqueness constraints/table"),
10142            );
10143            for uc in &t.schema.uniqueness_constraints {
10144                match &uc.name {
10145                    Some(n) => {
10146                        out.push(1);
10147                        write_str(&mut out, n);
10148                    }
10149                    None => out.push(0),
10150                }
10151            }
10152            // v7.39 (read01 round 56) — user_composite_type appendix
10153            // (FILE_VERSION 63+). Sparse, at the very end of the per-table
10154            // block: only composite-typed columns land here, so a v62 reader
10155            // stops before it and its composite columns stay plain JSON.
10156            let mut comp_bindings: Vec<(usize, &str)> = Vec::new();
10157            for (i, c) in t.schema.columns.iter().enumerate() {
10158                if let Some(n) = &c.user_composite_type {
10159                    comp_bindings.push((i, n.as_str()));
10160                }
10161            }
10162            write_u16(
10163                &mut out,
10164                u16::try_from(comp_bindings.len()).expect("≤ 65k composite-typed columns/table"),
10165            );
10166            for (pos, n) in comp_bindings {
10167                write_u16(&mut out, u16::try_from(pos).expect("≤ 65k columns/table"));
10168                write_str(&mut out, n);
10169            }
10170            // v7.39 (read01 round 57) — owner + ACL appendix (FILE_VERSION
10171            // 64+), at the very end of the per-table block so a v63 reader
10172            // stops before it (its tables then read back owner-less, i.e.
10173            // owned by the login role, with no grants — which is exactly what
10174            // they were).
10175            match &t.schema.owner {
10176                Some(o) => {
10177                    out.push(1);
10178                    write_str(&mut out, o);
10179                }
10180                None => out.push(0),
10181            }
10182            write_u16(
10183                &mut out,
10184                u16::try_from(t.schema.acl.len()).expect("≤ 65k aclitems/table"),
10185            );
10186            for a in &t.schema.acl {
10187                write_str(&mut out, &a.grantee);
10188                write_u16(&mut out, a.privs);
10189                write_u16(&mut out, a.grantable);
10190                write_str(&mut out, &a.grantor);
10191            }
10192            // v7.39 (read01 round 59) — COLUMN acl appendix (FILE_VERSION 65+),
10193            // sparse: only columns that carry a grant land here, so a v64 reader
10194            // stops before it and its columns read back un-granted, which is
10195            // what they were.
10196            let granted: Vec<(usize, &ColumnSchema)> = t
10197                .schema
10198                .columns
10199                .iter()
10200                .enumerate()
10201                .filter(|(_, c)| !c.acl.is_empty())
10202                .collect();
10203            write_u16(
10204                &mut out,
10205                u16::try_from(granted.len()).expect("≤ 65k granted columns/table"),
10206            );
10207            for (pos, c) in granted {
10208                write_u16(&mut out, u16::try_from(pos).expect("≤ 65k columns/table"));
10209                write_u16(
10210                    &mut out,
10211                    u16::try_from(c.acl.len()).expect("≤ 65k aclitems/column"),
10212                );
10213                for a in &c.acl {
10214                    write_str(&mut out, &a.grantee);
10215                    write_u16(&mut out, a.privs);
10216                    write_u16(&mut out, a.grantable);
10217                    write_str(&mut out, &a.grantor);
10218                }
10219            }
10220            // v7.39 (round 210) — EXCLUDE-constraint appendix (FILE_VERSION
10221            // 72+), at the very end of the per-table block so a v71 reader
10222            // stops before it and its tables read back with no exclusion
10223            // constraints. Layout: [u16 excl_count] then per constraint
10224            // [str name] [u8 has_method](+str) [u16 elem_count] then per
10225            // element [u16 col_pos][str op].
10226            write_u16(
10227                &mut out,
10228                u16::try_from(t.schema.exclusion_constraints.len())
10229                    .expect("≤ 65k exclusion constraints/table"),
10230            );
10231            for ex in &t.schema.exclusion_constraints {
10232                write_str(&mut out, &ex.name);
10233                match &ex.method {
10234                    Some(m) => {
10235                        out.push(1);
10236                        write_str(&mut out, m);
10237                    }
10238                    None => out.push(0),
10239                }
10240                write_u16(
10241                    &mut out,
10242                    u16::try_from(ex.elements.len()).expect("≤ 65k elements/exclusion"),
10243                );
10244                for (pos, op) in &ex.elements {
10245                    write_u16(&mut out, u16::try_from(*pos).expect("≤ 65k columns/table"));
10246                    write_str(&mut out, op);
10247                }
10248            }
10249            // v7.39 (round 220) — identity-RESTART appendix (FILE_VERSION
10250            // 73+), sparse: only columns carrying a RESTART floor land here.
10251            let restarts: Vec<(usize, i64)> = t
10252                .schema
10253                .columns
10254                .iter()
10255                .enumerate()
10256                .filter_map(|(i, c)| c.auto_restart.map(|n| (i, n)))
10257                .collect();
10258            write_u16(
10259                &mut out,
10260                u16::try_from(restarts.len()).expect("≤ 65k restart columns/table"),
10261            );
10262            for (pos, n) in restarts {
10263                write_u16(&mut out, u16::try_from(pos).expect("≤ 65k columns/table"));
10264                out.extend_from_slice(&n.to_le_bytes());
10265            }
10266            // v7.39 (round 386, type-fidelity epic P1) — per-table
10267            // mysql_int_width appendix (FILE_VERSION 81+). Sparse: only
10268            // TINYINT / MEDIUMINT columns land. Layout:
10269            // `[u16 count]([u16 col_pos][u8 width_tag]) × count`
10270            // (tag 0 = Tiny, 1 = Medium). v80-and-below readers stop after
10271            // the identity-RESTART appendix, leaving every column at None.
10272            let int_widths: Vec<(usize, u8)> = t
10273                .schema
10274                .columns
10275                .iter()
10276                .enumerate()
10277                .filter_map(|(i, c)| {
10278                    c.mysql_int_width.map(|w| {
10279                        let tag = match w {
10280                            MysqlIntWidth::Tiny => 0u8,
10281                            MysqlIntWidth::Medium => 1u8,
10282                            MysqlIntWidth::Small => 2u8,
10283                            MysqlIntWidth::Int => 3u8,
10284                            MysqlIntWidth::Big => 4u8,
10285                        };
10286                        (i, tag)
10287                    })
10288                })
10289                .collect();
10290            write_u16(
10291                &mut out,
10292                u16::try_from(int_widths.len()).expect("≤ 65k narrow-int columns/table"),
10293            );
10294            for (pos, tag) in int_widths {
10295                write_u16(&mut out, u16::try_from(pos).expect("≤ 65k columns/table"));
10296                out.push(tag);
10297            }
10298            // v7.39 (round 424, type-fidelity epic) — per-table mysql_fsp
10299            // appendix (FILE_VERSION 82+). Sparse: only MySQL-declared
10300            // temporal columns land. Layout:
10301            // `[u16 count]([u16 col_pos][u8 fsp]) × count`, fsp in 0..=6.
10302            // v81-and-below readers stop after the int-width appendix,
10303            // leaving every column at None (PG microsecond behaviour).
10304            let fsps: Vec<(usize, u8)> = t
10305                .schema
10306                .columns
10307                .iter()
10308                .enumerate()
10309                .filter_map(|(i, c)| c.mysql_fsp.map(|p| (i, p)))
10310                .collect();
10311            write_u16(
10312                &mut out,
10313                u16::try_from(fsps.len()).expect("≤ 65k temporal columns/table"),
10314            );
10315            for (pos, fsp) in fsps {
10316                write_u16(&mut out, u16::try_from(pos).expect("≤ 65k columns/table"));
10317                out.push(fsp);
10318            }
10319            // v7.39.2 — the declared-TIMESTAMP appendix (FILE_VERSION
10320            // 93+). Sparse: only the columns written as `TIMESTAMP` in a
10321            // MySQL session. Layout: `[u16 count]([u16 col_pos]) × count`.
10322            // v92-and-below readers stop after the CHECK appendix below,
10323            // leaving every column at `false` — which is what they meant.
10324            let declared_ts: Vec<usize> = t
10325                .schema
10326                .columns
10327                .iter()
10328                .enumerate()
10329                .filter_map(|(i, c)| c.mysql_declared_timestamp.then_some(i))
10330                .collect();
10331            write_u16(
10332                &mut out,
10333                u16::try_from(declared_ts.len()).expect("≤ 65k timestamp columns/table"),
10334            );
10335            for pos in declared_ts {
10336                write_u16(&mut out, u16::try_from(pos).expect("≤ 65k columns/table"));
10337            }
10338            // v7.39 (round 652) — CHECK-validated appendix (FILE_VERSION
10339            // 87+). Sparse the other way round from the ones above: the
10340            // common case is every constraint validated, so only the
10341            // NOT VALID ones are written, by their index into the CHECK
10342            // appendix. Layout: `[u16 count]([u16 check_idx]) × count`.
10343            let unvalidated: Vec<usize> = t
10344                .schema
10345                .checks
10346                .iter()
10347                .enumerate()
10348                .filter_map(|(i, c)| (!c.validated).then_some(i))
10349                .collect();
10350            write_u16(
10351                &mut out,
10352                u16::try_from(unvalidated.len()).expect("≤ 65k CHECK constraints/table"),
10353            );
10354            for idx in unvalidated {
10355                write_u16(&mut out, u16::try_from(idx).expect("≤ 65k CHECK/table"));
10356            }
10357            // v7.39 (round 677) — per-column collation names (FILE_VERSION
10358            // 88+). Sparse: only the columns that were written with an
10359            // explicit `COLLATE` appear, so a table that declares none pays
10360            // two bytes. Layout: `[u16 count]([u16 col_idx][str]) × count`.
10361            //
10362            // Without this the declaration survives CREATE TABLE and dies
10363            // at the next restart — measured: a column declared
10364            // `COLLATE "C"` reported attcollation 950 in the session that
10365            // created it and 100 after a reload.
10366            let collated: Vec<(usize, &str)> = t
10367                .schema
10368                .columns
10369                .iter()
10370                .enumerate()
10371                .filter_map(|(i, c)| c.collation_name.as_deref().map(|n| (i, n)))
10372                .collect();
10373            write_u16(
10374                &mut out,
10375                u16::try_from(collated.len()).expect("≤ 65k columns/table"),
10376            );
10377            for (idx, name) in collated {
10378                write_u16(&mut out, u16::try_from(idx).expect("≤ 65k columns/table"));
10379                write_str(&mut out, name);
10380            }
10381            // v7.39 (round 711) — PK/UNIQUE constraint timing (FILE_VERSION
10382            // 89+). Dense, one byte per uniqueness constraint in
10383            // declaration order, the same bit layout the FK block has
10384            // carried since round 288: bit 0 = DEFERRABLE, bit 1 =
10385            // INITIALLY DEFERRED. A v88 reader stops before it.
10386            write_u16(
10387                &mut out,
10388                u16::try_from(t.schema.uniqueness_constraints.len())
10389                    .expect("≤ 65k uniqueness constraints/table"),
10390            );
10391            for uc in &t.schema.uniqueness_constraints {
10392                out.push(u8::from(uc.deferrable) | (u8::from(uc.initially_deferred) << 1));
10393            }
10394        }
10395        // v7.12.4 — catalog-wide appendix: user-defined functions
10396        // then triggers. FILE_VERSION 22+ only. v21 and earlier
10397        // readers stop after the last table; v22 readers always
10398        // consume two `u32` counts (possibly zero).
10399        //
10400        // Function entry layout:
10401        //   [str name] [str args_repr] [str returns]
10402        //   [str language] [str body]
10403        // Trigger entry layout:
10404        //   [str name] [str table] [str timing]
10405        //   [u16 event_count] (event_count × str)
10406        //   [str for_each] [str function]
10407        write_u32(
10408            &mut out,
10409            u32::try_from(self.functions.len()).expect("≤ 4G functions"),
10410        );
10411        for fd in self.functions.values() {
10412            write_str(&mut out, &fd.name);
10413            write_str(&mut out, &fd.args_repr);
10414            write_str(&mut out, &fd.returns);
10415            write_str(&mut out, &fd.language);
10416            write_str_long(&mut out, &fd.body);
10417        }
10418        write_u32(
10419            &mut out,
10420            u32::try_from(self.triggers.len()).expect("≤ 4G triggers"),
10421        );
10422        for td in &self.triggers {
10423            write_str(&mut out, &td.name);
10424            write_str(&mut out, &td.table);
10425            write_str(&mut out, &td.timing);
10426            write_u16(
10427                &mut out,
10428                u16::try_from(td.events.len()).expect("≤ 65k events / trigger"),
10429            );
10430            for ev in &td.events {
10431                write_str(&mut out, ev);
10432            }
10433            write_str(&mut out, &td.for_each);
10434            write_str(&mut out, &td.function);
10435            // v7.13.0 — `UPDATE OF cols` filter
10436            // (FILE_VERSION 23+). v22 readers omit; v23 writers
10437            // always emit (possibly zero).
10438            write_u16(
10439                &mut out,
10440                u16::try_from(td.update_columns.len()).expect("≤ 65k cols / trigger"),
10441            );
10442            for c in &td.update_columns {
10443                write_str(&mut out, c);
10444            }
10445            // v7.16.1 — TriggerDef.enabled (FILE_VERSION 25+).
10446            out.push(u8::from(td.enabled));
10447            // v7.39 (round 138) — WHEN condition text (FILE_VERSION 70+).
10448            write_str(&mut out, &td.when_condition);
10449        }
10450        // v7.17.0 Phase 1.1 — SEQUENCE catalog block (FILE_VERSION 26+).
10451        write_u32(
10452            &mut out,
10453            u32::try_from(self.sequences.len()).expect("≤ 4G sequences"),
10454        );
10455        for seq in self.sequences.values() {
10456            write_str(&mut out, &seq.name);
10457            out.push(match seq.data_type {
10458                SequenceDataType::SmallInt => 0,
10459                SequenceDataType::Int => 1,
10460                SequenceDataType::BigInt => 2,
10461            });
10462            out.extend_from_slice(&seq.start.to_le_bytes());
10463            out.extend_from_slice(&seq.increment.to_le_bytes());
10464            out.extend_from_slice(&seq.min_value.to_le_bytes());
10465            out.extend_from_slice(&seq.max_value.to_le_bytes());
10466            out.extend_from_slice(&seq.cache.to_le_bytes());
10467            out.push(u8::from(seq.cycle));
10468            match &seq.owned_by {
10469                None => out.push(0),
10470                Some((table, column)) => {
10471                    out.push(1);
10472                    write_str(&mut out, table);
10473                    write_str(&mut out, column);
10474                }
10475            }
10476            out.extend_from_slice(&seq.last_value.to_le_bytes());
10477            out.push(u8::from(seq.is_called));
10478        }
10479        // v7.17.0 Phase 1.2 — VIEW catalog block (FILE_VERSION 27+).
10480        write_u32(
10481            &mut out,
10482            u32::try_from(self.views.len()).expect("≤ 4G views"),
10483        );
10484        for view in self.views.values() {
10485            write_str(&mut out, &view.name);
10486            write_u16(
10487                &mut out,
10488                u16::try_from(view.columns.len()).expect("≤ 65k cols / view"),
10489            );
10490            for c in &view.columns {
10491                write_str(&mut out, c);
10492            }
10493            write_str_long(&mut out, &view.body);
10494            // v7.39 (round 132, FILE_VERSION 69+) — WITH CHECK OPTION marker.
10495            out.push(view.check_option);
10496        }
10497        // v7.17.0 Phase 1.3 — MATERIALIZED VIEW source registry
10498        // (FILE_VERSION 28+). The backing rows live as a regular
10499        // table of the same name already in the tables block.
10500        write_u32(
10501            &mut out,
10502            u32::try_from(self.materialized_views.len()).expect("≤ 4G materialized views"),
10503        );
10504        for (name, body) in &self.materialized_views {
10505            write_str(&mut out, name);
10506            write_str_long(&mut out, body);
10507        }
10508        // v7.17.0 Phase 1.4 — ENUM types catalog block
10509        // (FILE_VERSION 29+).
10510        write_u32(
10511            &mut out,
10512            u32::try_from(self.enum_types.len()).expect("≤ 4G enum types"),
10513        );
10514        for e in self.enum_types.values() {
10515            write_str(&mut out, &e.name);
10516            write_u16(
10517                &mut out,
10518                u16::try_from(e.labels.len()).expect("≤ 65k labels / enum"),
10519            );
10520            for l in &e.labels {
10521                write_str(&mut out, l);
10522            }
10523        }
10524        // v7.17.0 Phase 1.5 — DOMAIN types catalog block
10525        // (FILE_VERSION 30+).
10526        write_u32(
10527            &mut out,
10528            u32::try_from(self.domain_types.len()).expect("≤ 4G domain types"),
10529        );
10530        for d in self.domain_types.values() {
10531            write_str(&mut out, &d.name);
10532            write_data_type(&mut out, d.base_type);
10533            out.push(u8::from(d.nullable));
10534            match &d.default {
10535                None => out.push(0),
10536                Some(s) => {
10537                    out.push(1);
10538                    write_str(&mut out, s);
10539                }
10540            }
10541            write_u16(
10542                &mut out,
10543                u16::try_from(d.checks.len()).expect("≤ 65k CHECKs / domain"),
10544            );
10545            for c in &d.checks {
10546                write_str(&mut out, &c.expr);
10547                // v7.39 (round 260) — the constraint name (FILE_VERSION 75+).
10548                write_str(&mut out, &c.name);
10549            }
10550            // v7.39 (round 259) — the parent domain (FILE_VERSION 74+).
10551            match &d.base_domain {
10552                None => out.push(0),
10553                Some(s) => {
10554                    out.push(1);
10555                    write_str(&mut out, s);
10556                }
10557            }
10558        }
10559        // v7.17.0 Phase 1.6 — user-schemas registry
10560        // (FILE_VERSION 31+). Built-ins are hardcoded in
10561        // `is_builtin_schema` and not persisted.
10562        write_u32(
10563            &mut out,
10564            u32::try_from(self.schemas.len()).expect("≤ 4G schemas"),
10565        );
10566        for name in &self.schemas {
10567            write_str(&mut out, name);
10568        }
10569        // v7.37.42-T2 ζ-B — COMPOSITE types catalog block
10570        // (FILE_VERSION 52+). Each entry: name, u16 field_count,
10571        // then field_count `[str field_name][data_type]` pairs.
10572        write_u32(
10573            &mut out,
10574            u32::try_from(self.composite_types.len()).expect("≤ 4G composite types"),
10575        );
10576        for c in self.composite_types.values() {
10577            write_str(&mut out, &c.name);
10578            write_u16(
10579                &mut out,
10580                u16::try_from(c.fields.len()).expect("≤ 65k fields / composite"),
10581            );
10582            for (i, (fname, fty)) in c.fields.iter().enumerate() {
10583                write_str(&mut out, fname);
10584                write_data_type(&mut out, *fty);
10585                // v7.39 (round 264) — the field's user type (v76+).
10586                match c.field_user_types.get(i).and_then(Option::as_ref) {
10587                    None => out.push(0),
10588                    Some(n) => {
10589                        out.push(1);
10590                        write_str(&mut out, n);
10591                    }
10592                }
10593            }
10594        }
10595        // v7.39 (read01 round 50) — COMMENT store (FILE_VERSION 61+).
10596        // Catalog-wide, written last (before the CRC trailer) so every older
10597        // reader stops before it. Layout: [u32 count] then [str key][str text].
10598        write_u32(
10599            &mut out,
10600            u32::try_from(self.comments.len()).expect("≤ 4G comments"),
10601        );
10602        for (k, v) in &self.comments {
10603            write_str(&mut out, k);
10604            write_str_long(&mut out, v);
10605        }
10606        // v7.39 (read01 round 60) — non-table ACLs (FILE_VERSION 66+), catalog-
10607        // wide and written last so a v65 reader stops before them. The sequence
10608        // block itself sits mid-image and cannot grow without breaking older
10609        // readers, so a sequence's owner + ACL rides here, keyed by name.
10610        let acl_out = |out: &mut Vec<u8>, acl: &[AclItem]| {
10611            write_u16(out, u16::try_from(acl.len()).expect("≤ 65k aclitems"));
10612            for a in acl {
10613                write_str(out, &a.grantee);
10614                write_u16(out, a.privs);
10615                write_u16(out, a.grantable);
10616                write_str(out, &a.grantor);
10617            }
10618        };
10619        let owned: Vec<&SequenceDef> = self
10620            .sequences
10621            .values()
10622            .filter(|s| s.owner.is_some() || !s.acl.is_empty())
10623            .collect();
10624        write_u32(
10625            &mut out,
10626            u32::try_from(owned.len()).expect("≤ 4G sequences"),
10627        );
10628        for seq in owned {
10629            write_str(&mut out, &seq.name);
10630            match &seq.owner {
10631                Some(o) => {
10632                    out.push(1);
10633                    write_str(&mut out, o);
10634                }
10635                None => out.push(0),
10636            }
10637            acl_out(&mut out, &seq.acl);
10638        }
10639        acl_out(&mut out, &self.schema_acl);
10640        acl_out(&mut out, &self.database_acl);
10641        // v7.39 (read01 round 61) — FUNCTION owner + ACL (FILE_VERSION 67+).
10642        // The function block sits mid-image like the sequence one, so this
10643        // rides the catalog-wide tail too, keyed by name.
10644        let fns: Vec<&FunctionDef> = self
10645            .functions
10646            .values()
10647            .filter(|f| f.owner.is_some() || !f.acl.is_empty())
10648            .collect();
10649        write_u32(&mut out, u32::try_from(fns.len()).expect("≤ 4G functions"));
10650        for f in fns {
10651            // v7.39 (read01 round 62) — keyed by SIGNATURE now: two overloads
10652            // have two ACLs.
10653            write_str(&mut out, &function_signature_key(&f.name, &f.args_repr));
10654            match &f.owner {
10655                Some(o) => {
10656                    out.push(1);
10657                    write_str(&mut out, o);
10658                }
10659                None => out.push(0),
10660            }
10661            acl_out(&mut out, &f.acl);
10662        }
10663        // v7.39 (round 139) — RULE catalog block (FILE_VERSION 71+), catalog-
10664        // wide and written last (right before the CRC trailer) so every older
10665        // reader stops cleanly before it. Layout: [u32 count] then per rule
10666        // [str name][str table][str event][u8 instead][str when]
10667        // [u16 cmd_count]([str cmd] × cmd_count).
10668        write_u32(
10669            &mut out,
10670            u32::try_from(self.rules.len()).expect("≤ 4G rules"),
10671        );
10672        for r in &self.rules {
10673            write_str(&mut out, &r.name);
10674            write_str(&mut out, &r.table);
10675            write_str(&mut out, &r.event);
10676            out.push(u8::from(r.instead));
10677            write_str(&mut out, &r.when_condition);
10678            write_u16(
10679                &mut out,
10680                u16::try_from(r.commands.len()).expect("≤ 65k commands / rule"),
10681            );
10682            for c in &r.commands {
10683                write_str(&mut out, c);
10684            }
10685        }
10686        // v7.39 (round 280) — extended-statistics block (FILE_VERSION
10687        // 77+), appended after the RULE block for the same reason: an
10688        // older reader stops cleanly before it. Layout: [u32 count]
10689        // then per object [str name][str table][u16 n]([str kind] × n)
10690        // [u16 m]([str column] × m).
10691        write_u32(
10692            &mut out,
10693            u32::try_from(self.statistics_ext.len()).expect("≤ 4G statistics objects"),
10694        );
10695        for st in &self.statistics_ext {
10696            write_str(&mut out, &st.name);
10697            write_str(&mut out, &st.table);
10698            write_u16(
10699                &mut out,
10700                u16::try_from(st.kinds.len()).expect("≤ 65k kinds"),
10701            );
10702            for k in &st.kinds {
10703                write_str(&mut out, k);
10704            }
10705            write_u16(
10706                &mut out,
10707                u16::try_from(st.columns.len()).expect("≤ 65k columns"),
10708            );
10709            for c in &st.columns {
10710                write_str(&mut out, c);
10711            }
10712        }
10713        // v7.39 (round 287) — large-object block (FILE_VERSION 78+),
10714        // appended after the statistics block for the same reason: an
10715        // older reader stops cleanly before it. Layout: [u32 count]
10716        // then per object [u32 oid][u32 len][len bytes].
10717        write_u32(
10718            &mut out,
10719            u32::try_from(self.large_objects.len()).expect("≤ 4G large objects"),
10720        );
10721        for (oid, bytes) in &self.large_objects {
10722            write_u32(&mut out, *oid);
10723            write_u32(
10724                &mut out,
10725                u32::try_from(bytes.len()).expect("≤ 4G per object"),
10726            );
10727            out.extend_from_slice(bytes);
10728        }
10729        // v7.39 (round 322, V46) — function-attribute block (FILE_VERSION
10730        // 80+), appended last for the same reason as every block before
10731        // it: an older reader stops cleanly ahead of it and simply sees
10732        // functions with PG's default attributes. Only functions that
10733        // declared something non-default are written. Layout: [u32 count]
10734        // then per function [str signature_key][u8 volatility][u8 flags]
10735        // [u8 parallel][f64 cost or NaN][f64 rows or NaN], where flags bit
10736        // 0 = strict, 1 = security definer, 2 = leakproof.
10737        let attr_fns: Vec<(&String, &FunctionDef)> = self
10738            .functions
10739            .iter()
10740            .filter(|(_, f)| {
10741                f.volatility != FN_VOLATILE
10742                    || f.strict
10743                    || f.security_definer
10744                    || f.leakproof
10745                    || f.parallel != FN_PARALLEL_UNSAFE
10746                    || f.cost.is_some()
10747                    || f.rows.is_some()
10748            })
10749            .collect();
10750        write_u32(
10751            &mut out,
10752            u32::try_from(attr_fns.len()).expect("≤ 4G functions"),
10753        );
10754        for (key, f) in attr_fns {
10755            write_str(&mut out, key);
10756            out.push(f.volatility);
10757            let flags = u8::from(f.strict)
10758                | (u8::from(f.security_definer) << 1)
10759                | (u8::from(f.leakproof) << 2);
10760            out.push(flags);
10761            out.push(f.parallel);
10762            out.extend_from_slice(&f.cost.unwrap_or(f64::NAN).to_le_bytes());
10763            out.extend_from_slice(&f.rows.unwrap_or(f64::NAN).to_le_bytes());
10764        }
10765        // v7.38 (read01 P5.05) — CRC32C trailer over the whole image so a
10766        // corrupted snapshot is rejected on load. FILE_VERSION is >= the
10767        // trailer version, so this always runs for freshly-written images.
10768        // v7.39 (round 547) — pg_db_role_setting (FILE_VERSION 85+),
10769        // catalog-wide and written LAST so a v84 reader stops before it.
10770        // Layout: [u32 scopes] then [str database][str role][u32 params]
10771        // then [str name][str value] per param.
10772        write_u32(
10773            &mut out,
10774            u32::try_from(self.db_role_settings.len()).expect("≤ 4G scopes"),
10775        );
10776        for ((db, role), params) in &self.db_role_settings {
10777            write_str(&mut out, db);
10778            write_str(&mut out, role);
10779            write_u32(&mut out, u32::try_from(params.len()).expect("≤ 4G params"));
10780            for (name, value) in params {
10781                write_str(&mut out, name);
10782                write_str(&mut out, value);
10783            }
10784        }
10785        // v7.39 (round 550) — replication slots (FILE_VERSION 86+),
10786        // written LAST so a v85 reader stops before them.
10787        write_u32(
10788            &mut out,
10789            u32::try_from(self.replication_slots.len()).expect("≤ 4G slots"),
10790        );
10791        for (name, (plugin, slot_type)) in &self.replication_slots {
10792            write_str(&mut out, name);
10793            write_str(&mut out, plugin);
10794            write_str(&mut out, slot_type);
10795        }
10796        // v7.38.18 (S1) — the database collation (FILE_VERSION 92+).
10797        // Absent on an older image, which reads back as `C`.
10798        match &self.db_collation {
10799            None => out.push(0),
10800            Some(c) => {
10801                out.push(1);
10802                write_str(&mut out, c);
10803            }
10804        }
10805        let crc = spg_crypto::crc32c::crc32c(&out);
10806        write_u32(&mut out, crc);
10807        out
10808    }
10809
10810    /// Deserialize a previously-serialized catalog. Rejects bad magic, version
10811    /// mismatch, unknown tags, truncation, and trailing bytes.
10812    pub fn deserialize(buf: &[u8]) -> Result<Self, StorageError> {
10813        let mut cur = Cursor::new(buf);
10814        let magic = cur.take(8)?;
10815        if magic != FILE_MAGIC {
10816            return Err(StorageError::Corrupt(format!(
10817                "bad magic: expected SPGDB001, got {magic:?}"
10818            )));
10819        }
10820        let version = cur.read_u8()?;
10821        if !(MIN_SUPPORTED_FILE_VERSION..=FILE_VERSION).contains(&version) {
10822            return Err(StorageError::Corrupt(format!(
10823                "unsupported file version: {version} (supported: {MIN_SUPPORTED_FILE_VERSION}..={FILE_VERSION})"
10824            )));
10825        }
10826        // v7.23/v7.27 — escape decoding is version-gated (see
10827        // STR_LEN_ESCAPE / Cursor::codec_version).
10828        cur.codec_version = version;
10829        let table_count = cur.read_u32()? as usize;
10830        let mut cat = Self::new();
10831        for _ in 0..table_count {
10832            deserialize_table(&mut cur, &mut cat, version)?;
10833        }
10834        // v7.37.15 (Phase C.1) — stamp dense stable RelIds on load.
10835        // Pre-V6 envelopes carry no ids; a dense 1..=N assignment is
10836        // sufficient while RelId is process-local bookkeeping (the V6
10837        // envelope, Phase C.6, will round-trip real ids). Sets the
10838        // allocator above the loaded ids so a post-load CREATE TABLE
10839        // never collides.
10840        for (i, t) in cat.tables.iter_mut().enumerate() {
10841            t.set_rel_id(row_header::RelId((i as u64) + 1));
10842        }
10843        cat.next_rel_id = cat.tables.len() as u64;
10844        // v7.12.4 — catalog-wide function + trigger appendix.
10845        // FILE_VERSION 22+ only; v21 and earlier catalogs stop
10846        // after the last table.
10847        if version >= 22 {
10848            let fn_count = cur.read_u32()? as usize;
10849            for _ in 0..fn_count {
10850                let name = cur.read_str()?;
10851                let args_repr = cur.read_str()?;
10852                let returns = cur.read_str()?;
10853                let language = cur.read_str()?;
10854                let body = cur.read_str_long()?;
10855                let key = function_signature_key(&name, &args_repr);
10856                cat.functions.insert(
10857                    key,
10858                    FunctionDef {
10859                        name,
10860                        args_repr,
10861                        returns,
10862                        language,
10863                        body,
10864                        owner: None,
10865                        acl: Vec::new(),
10866                        volatility: FN_VOLATILE,
10867                        strict: false,
10868                        security_definer: false,
10869                        leakproof: false,
10870                        parallel: FN_PARALLEL_UNSAFE,
10871                        cost: None,
10872                        rows: None,
10873                    },
10874                );
10875            }
10876            let trg_count = cur.read_u32()? as usize;
10877            for _ in 0..trg_count {
10878                let name = cur.read_str()?;
10879                let table = cur.read_str()?;
10880                let timing = cur.read_str()?;
10881                let ev_count = cur.read_u16()? as usize;
10882                let mut events = Vec::with_capacity(ev_count);
10883                for _ in 0..ev_count {
10884                    events.push(cur.read_str()?);
10885                }
10886                let for_each = cur.read_str()?;
10887                let function = cur.read_str()?;
10888                // v7.13.0 — trailing `UPDATE OF cols` filter
10889                // (FILE_VERSION 23+ only; v22 catalogs omit and
10890                // deserialise with an empty vec).
10891                let update_columns = if version >= 23 {
10892                    let n = cur.read_u16()? as usize;
10893                    let mut cols = Vec::with_capacity(n);
10894                    for _ in 0..n {
10895                        cols.push(cur.read_str()?);
10896                    }
10897                    cols
10898                } else {
10899                    Vec::new()
10900                };
10901                // v7.16.1 — TriggerDef.enabled (FILE_VERSION 25+).
10902                // v24-and-below catalogs deserialise with `true`
10903                // — pre-v7.16.1 every trigger always fired.
10904                let enabled = if version >= 25 {
10905                    cur.read_u8()? != 0
10906                } else {
10907                    true
10908                };
10909                // v7.39 (round 138) — WHEN condition text added at FILE_VERSION
10910                // 70; older catalogs read back empty (no WHEN filter).
10911                let when_condition = if version >= 70 {
10912                    cur.read_str()?
10913                } else {
10914                    String::new()
10915                };
10916                cat.triggers.push(TriggerDef {
10917                    name,
10918                    table,
10919                    timing,
10920                    events,
10921                    for_each,
10922                    function,
10923                    update_columns,
10924                    enabled,
10925                    when_condition,
10926                });
10927            }
10928        }
10929        // v7.17.0 Phase 1.1 — SEQUENCE block (FILE_VERSION 26+).
10930        // v25-and-below catalogs omit; we leave the map empty.
10931        if version >= 26 {
10932            let seq_count = cur.read_u32()? as usize;
10933            for _ in 0..seq_count {
10934                let name = cur.read_str()?;
10935                let data_type = match cur.read_u8()? {
10936                    0 => SequenceDataType::SmallInt,
10937                    1 => SequenceDataType::Int,
10938                    2 => SequenceDataType::BigInt,
10939                    other => {
10940                        return Err(StorageError::Corrupt(format!(
10941                            "unknown SEQUENCE data-type tag {other}"
10942                        )));
10943                    }
10944                };
10945                let start = cur.read_i64()?;
10946                let increment = cur.read_i64()?;
10947                let min_value = cur.read_i64()?;
10948                let max_value = cur.read_i64()?;
10949                let cache = cur.read_i64()?;
10950                let cycle = cur.read_u8()? != 0;
10951                let owned_by = match cur.read_u8()? {
10952                    0 => None,
10953                    1 => {
10954                        let t = cur.read_str()?;
10955                        let c = cur.read_str()?;
10956                        Some((t, c))
10957                    }
10958                    other => {
10959                        return Err(StorageError::Corrupt(format!(
10960                            "unknown SEQUENCE owned-by tag {other}"
10961                        )));
10962                    }
10963                };
10964                let last_value = cur.read_i64()?;
10965                let is_called = cur.read_u8()? != 0;
10966                cat.sequences.insert(
10967                    name.clone(),
10968                    SequenceDef {
10969                        name,
10970                        data_type,
10971                        start,
10972                        increment,
10973                        min_value,
10974                        max_value,
10975                        cache,
10976                        cycle,
10977                        owned_by,
10978                        last_value,
10979                        is_called,
10980                        owner: None,
10981                        acl: Vec::new(),
10982                    },
10983                );
10984            }
10985        }
10986        // v7.17.0 Phase 1.2 — VIEW block (FILE_VERSION 27+).
10987        // v26-and-below catalogs omit; we leave the map empty.
10988        if version >= 27 {
10989            let view_count = cur.read_u32()? as usize;
10990            for _ in 0..view_count {
10991                let name = cur.read_str()?;
10992                let col_count = cur.read_u16()? as usize;
10993                let mut columns = Vec::with_capacity(col_count);
10994                for _ in 0..col_count {
10995                    columns.push(cur.read_str()?);
10996                }
10997                let body = cur.read_str_long()?;
10998                // v7.39 (round 132) — check-option marker added at FILE_VERSION
10999                // 69; older catalogs default to 0 (no check option).
11000                let check_option = if version >= 69 { cur.read_u8()? } else { 0 };
11001                cat.views.insert(
11002                    name.clone(),
11003                    ViewDef {
11004                        name,
11005                        columns,
11006                        body,
11007                        check_option,
11008                    },
11009                );
11010            }
11011        }
11012        // v7.17.0 Phase 1.3 — MATERIALIZED VIEW source registry
11013        // (FILE_VERSION 28+). v27-and-below catalogs omit.
11014        if version >= 28 {
11015            let mv_count = cur.read_u32()? as usize;
11016            for _ in 0..mv_count {
11017                let name = cur.read_str()?;
11018                let body = cur.read_str_long()?;
11019                cat.materialized_views.insert(name, body);
11020            }
11021        }
11022        // v7.17.0 Phase 1.4 — ENUM types catalog block
11023        // (FILE_VERSION 29+).
11024        if version >= 29 {
11025            let etype_count = cur.read_u32()? as usize;
11026            for _ in 0..etype_count {
11027                let name = cur.read_str()?;
11028                let label_count = cur.read_u16()? as usize;
11029                let mut labels = Vec::with_capacity(label_count);
11030                for _ in 0..label_count {
11031                    labels.push(cur.read_str()?);
11032                }
11033                cat.enum_types
11034                    .insert(name.clone(), EnumDef { name, labels });
11035            }
11036        }
11037        // v7.17.0 Phase 1.5 — DOMAIN types catalog block
11038        // (FILE_VERSION 30+).
11039        if version >= 30 {
11040            let dtype_count = cur.read_u32()? as usize;
11041            for _ in 0..dtype_count {
11042                let name = cur.read_str()?;
11043                let base_type = cur.read_data_type()?;
11044                let nullable = cur.read_u8()? != 0;
11045                let default = match cur.read_u8()? {
11046                    0 => None,
11047                    1 => Some(cur.read_str()?),
11048                    other => {
11049                        return Err(StorageError::Corrupt(format!(
11050                            "unknown DOMAIN default tag {other}"
11051                        )));
11052                    }
11053                };
11054                let check_count = cur.read_u16()? as usize;
11055                let mut checks: Vec<DomainCheck> = Vec::with_capacity(check_count);
11056                for i in 0..check_count {
11057                    let expr = cur.read_str()?;
11058                    // v7.39 (round 260) — names arrived in FILE_VERSION 75.
11059                    // An older catalog gets PG's auto-naming applied to the
11060                    // checks it stored, which is what they would have been.
11061                    let cname = if version >= 75 {
11062                        cur.read_str()?
11063                    } else if i == 0 {
11064                        alloc::format!("{name}_check")
11065                    } else {
11066                        alloc::format!("{name}_check{i}")
11067                    };
11068                    checks.push(DomainCheck { name: cname, expr });
11069                }
11070                // v7.39 (round 259) — the parent domain. Absent before
11071                // FILE_VERSION 74; an older catalog reads as a domain over
11072                // a scalar, which is what it was.
11073                let base_domain = if version >= 74 {
11074                    match cur.read_u8()? {
11075                        0 => None,
11076                        1 => Some(cur.read_str()?),
11077                        other => {
11078                            return Err(StorageError::Corrupt(alloc::format!(
11079                                "domain base_domain tag {other}"
11080                            )));
11081                        }
11082                    }
11083                } else {
11084                    None
11085                };
11086                cat.domain_types.insert(
11087                    name.clone(),
11088                    DomainDef {
11089                        name,
11090                        base_type,
11091                        nullable,
11092                        default,
11093                        checks,
11094                        base_domain,
11095                    },
11096                );
11097            }
11098        }
11099        // v7.17.0 Phase 1.6 — user-schemas registry
11100        // (FILE_VERSION 31+).
11101        if version >= 31 {
11102            let sch_count = cur.read_u32()? as usize;
11103            for _ in 0..sch_count {
11104                let name = cur.read_str()?;
11105                cat.schemas.insert(name);
11106            }
11107        }
11108        // v7.37.42-T2 ζ-B — COMPOSITE types catalog block
11109        // (FILE_VERSION 52+). v51-and-below readers stop at the
11110        // user-schemas block; v52 readers fed a v51 catalog see no
11111        // composite block and default to an empty map.
11112        if version >= 52 {
11113            let ctype_count = cur.read_u32()? as usize;
11114            for _ in 0..ctype_count {
11115                let name = cur.read_str()?;
11116                let field_count = cur.read_u16()? as usize;
11117                let mut fields = Vec::with_capacity(field_count);
11118                let mut field_user_types: Vec<Option<String>> = Vec::with_capacity(field_count);
11119                for _ in 0..field_count {
11120                    let fname = cur.read_str()?;
11121                    let fty = cur.read_data_type()?;
11122                    // v7.39 (round 264) — present from FILE_VERSION 76.
11123                    let ut = if version >= 76 {
11124                        match cur.read_u8()? {
11125                            0 => None,
11126                            1 => Some(cur.read_str()?),
11127                            other => {
11128                                return Err(StorageError::Corrupt(alloc::format!(
11129                                    "composite field user-type tag {other}"
11130                                )));
11131                            }
11132                        }
11133                    } else {
11134                        None
11135                    };
11136                    fields.push((fname, fty));
11137                    field_user_types.push(ut);
11138                }
11139                cat.composite_types.insert(
11140                    name.clone(),
11141                    CompositeDef {
11142                        name,
11143                        fields,
11144                        field_user_types,
11145                    },
11146                );
11147            }
11148        }
11149        // v7.39 (read01 round 50) — COMMENT store (FILE_VERSION 61+).
11150        if version >= 61 {
11151            let comment_count = cur.read_u32()? as usize;
11152            for _ in 0..comment_count {
11153                let key = cur.read_str()?;
11154                let text = cur.read_str_long()?;
11155                cat.comments.insert(key, text);
11156            }
11157        }
11158        // v7.39 (read01 round 60) — non-table ACLs (FILE_VERSION 66+).
11159        if version >= 66 {
11160            let read_acl = |cur: &mut Cursor| -> Result<Vec<AclItem>, StorageError> {
11161                let n = cur.read_u16()? as usize;
11162                let mut acl = Vec::with_capacity(n);
11163                for _ in 0..n {
11164                    let grantee = cur.read_str()?;
11165                    let privs = cur.read_u16()?;
11166                    let grantable = cur.read_u16()?;
11167                    let grantor = cur.read_str()?;
11168                    acl.push(AclItem {
11169                        grantee,
11170                        privs,
11171                        grantable,
11172                        grantor,
11173                    });
11174                }
11175                Ok(acl)
11176            };
11177            let seq_count = cur.read_u32()? as usize;
11178            for _ in 0..seq_count {
11179                let name = cur.read_str()?;
11180                let owner = if cur.read_u8()? == 1 {
11181                    Some(cur.read_str()?)
11182                } else {
11183                    None
11184                };
11185                let acl = read_acl(&mut cur)?;
11186                if let Some(seq) = cat.sequences.get_mut(&name) {
11187                    seq.owner = owner;
11188                    seq.acl = acl;
11189                }
11190            }
11191            cat.schema_acl = read_acl(&mut cur)?;
11192            cat.database_acl = read_acl(&mut cur)?;
11193            // v7.39 (read01 round 61) — FUNCTION owner + ACL (v67+; keyed by
11194            // signature from v68, when overloads became possible).
11195            if version >= 67 {
11196                let fn_count = cur.read_u32()? as usize;
11197                for _ in 0..fn_count {
11198                    let name = cur.read_str()?;
11199                    let owner = if cur.read_u8()? == 1 {
11200                        Some(cur.read_str()?)
11201                    } else {
11202                        None
11203                    };
11204                    let acl = read_acl(&mut cur)?;
11205                    // v7.39 (round 315, V19) — the stored key was computed
11206                    // by whichever formula was current when the image was
11207                    // written. A miss is not "no such function": before the
11208                    // multi-word fix, `f(double precision)` keyed as
11209                    // `f(precision)`, so an older image's grants would land
11210                    // nowhere and vanish silently. Fall back to matching by
11211                    // the old formula, which re-attaches them.
11212                    let target = resolve_stored_function_key(&cat.functions, &name);
11213                    if let Some(k) = target
11214                        && let Some(f) = cat.functions.get_mut(&k)
11215                    {
11216                        f.owner = owner;
11217                        f.acl = acl;
11218                    }
11219                }
11220            }
11221        }
11222        // v7.39 (round 139) — RULE catalog block (FILE_VERSION 71+), read from
11223        // the tail right before the CRC trailer. Pre-71 images stop before it.
11224        if version >= 71 {
11225            let rule_count = cur.read_u32()? as usize;
11226            for _ in 0..rule_count {
11227                let name = cur.read_str()?;
11228                let table = cur.read_str()?;
11229                let event = cur.read_str()?;
11230                let instead = cur.read_u8()? != 0;
11231                let when_condition = cur.read_str()?;
11232                let cmd_count = cur.read_u16()? as usize;
11233                let mut commands = Vec::with_capacity(cmd_count);
11234                for _ in 0..cmd_count {
11235                    commands.push(cur.read_str()?);
11236                }
11237                cat.rules.push(RuleDef {
11238                    name,
11239                    table,
11240                    event,
11241                    instead,
11242                    when_condition,
11243                    commands,
11244                });
11245            }
11246        }
11247        // v7.39 (round 280) — extended-statistics block (FILE_VERSION
11248        // 77+). Pre-77 images stop before it.
11249        if version >= 77 {
11250            let count = cur.read_u32()? as usize;
11251            for _ in 0..count {
11252                let name = cur.read_str()?;
11253                let table = cur.read_str()?;
11254                let nk = cur.read_u16()? as usize;
11255                let mut kinds = Vec::with_capacity(nk);
11256                for _ in 0..nk {
11257                    kinds.push(cur.read_str()?);
11258                }
11259                let nc = cur.read_u16()? as usize;
11260                let mut columns = Vec::with_capacity(nc);
11261                for _ in 0..nc {
11262                    columns.push(cur.read_str()?);
11263                }
11264                cat.statistics_ext.push(StatisticsExtDef {
11265                    name,
11266                    table,
11267                    kinds,
11268                    columns,
11269                });
11270            }
11271        }
11272        // v7.39 (round 287) — large-object block (FILE_VERSION 78+).
11273        // Pre-78 images stop before it.
11274        if version >= 78 {
11275            let count = cur.read_u32()? as usize;
11276            for _ in 0..count {
11277                let oid = cur.read_u32()?;
11278                let len = cur.read_u32()? as usize;
11279                let bytes = cur.read_bytes(len)?;
11280                cat.large_objects.insert(oid, bytes);
11281            }
11282        }
11283        // v7.39 (round 322, V46) — function-attribute block (FILE_VERSION
11284        // 80+). Pre-80 images stop before it and keep PG's defaults.
11285        if version >= 80 {
11286            let count = cur.read_u32()? as usize;
11287            for _ in 0..count {
11288                let key = cur.read_str()?;
11289                let volatility = cur.read_u8()?;
11290                let flags = cur.read_u8()?;
11291                let parallel = cur.read_u8()?;
11292                let cost = f64::from_le_bytes(cur.read_bytes(8)?.try_into().unwrap_or([0; 8]));
11293                let rows = f64::from_le_bytes(cur.read_bytes(8)?.try_into().unwrap_or([0; 8]));
11294                if let Some(f) = cat.functions.get_mut(&key) {
11295                    f.volatility = volatility;
11296                    f.strict = flags & 1 != 0;
11297                    f.security_definer = flags & 2 != 0;
11298                    f.leakproof = flags & 4 != 0;
11299                    f.parallel = parallel;
11300                    f.cost = (!cost.is_nan()).then_some(cost);
11301                    f.rows = (!rows.is_nan()).then_some(rows);
11302                }
11303            }
11304        }
11305        // v7.39 (round 547) — pg_db_role_setting (FILE_VERSION 85+).
11306        // Pre-85 images stop before it and carry no GUC defaults.
11307        if version >= 85 {
11308            let scopes = cur.read_u32()? as usize;
11309            for _ in 0..scopes {
11310                let db = cur.read_str()?;
11311                let role = cur.read_str()?;
11312                let params = cur.read_u32()? as usize;
11313                let mut m: BTreeMap<String, String> = BTreeMap::new();
11314                for _ in 0..params {
11315                    let name = cur.read_str()?;
11316                    let value = cur.read_str()?;
11317                    m.insert(name, value);
11318                }
11319                if !m.is_empty() {
11320                    cat.db_role_settings.insert((db, role), m);
11321                }
11322            }
11323        }
11324        // v7.39 (round 550) — replication slots (FILE_VERSION 86+).
11325        if version >= 86 {
11326            let count = cur.read_u32()? as usize;
11327            for _ in 0..count {
11328                let name = cur.read_str()?;
11329                let plugin = cur.read_str()?;
11330                let slot_type = cur.read_str()?;
11331                cat.replication_slots.insert(name, (plugin, slot_type));
11332            }
11333        }
11334        // v7.38.18 (S1) — the database collation (FILE_VERSION 92+).
11335        if version >= 92 {
11336            match cur.read_u8()? {
11337                0 => {}
11338                1 => cat.db_collation = Some(cur.read_str()?),
11339                other => {
11340                    return Err(StorageError::Corrupt(format!(
11341                        "db_collation tag: unknown byte {other}"
11342                    )));
11343                }
11344            }
11345        }
11346        // v7.38.18 (S3) — a database created under a collation this
11347        // build cannot perform does not open.
11348        //
11349        // Falling back to bytes would answer with a different comparator
11350        // than every index key in it was built under, which is the one
11351        // failure this whole layer exists to prevent — and it would do
11352        // it silently, since a byte-ordered answer looks exactly like a
11353        // correct one. The check is a NAME classification here; the
11354        // engine, which owns the collator, verifies it can actually
11355        // perform the name before recording it.
11356        if let Some(c) = &cat.db_collation
11357            && c.trim().is_empty()
11358        {
11359            return Err(StorageError::Corrupt(format!(
11360                "database collation is recorded as {c:?}, which names nothing"
11361            )));
11362        }
11363        // v7.38.18 (S2) — and every table read back learns it, because a
11364        // table decides for itself which of its indexes key under a
11365        // collation. Done here rather than per-table in the loop above
11366        // because the byte that says so is written after the tables.
11367        let db_coll = cat.db_collation().to_string();
11368        for t in &mut cat.tables {
11369            t.set_db_collation(&db_coll);
11370        }
11371        // v7.38 (read01 P5.05) — v54+ images end with a CRC32C over every
11372        // preceding byte; verify it before accepting the snapshot. Older
11373        // images have no trailer and fall through to the trailing-byte check.
11374        if version >= FILE_VERSION_CRC_TRAILER {
11375            let crc_start = cur.pos;
11376            let stored = cur.read_u32()?;
11377            let computed = spg_crypto::crc32c::crc32c(&buf[..crc_start]);
11378            if computed != stored {
11379                return Err(StorageError::Corrupt(format!(
11380                    "base snapshot CRC mismatch: computed {computed:#010x}, stored {stored:#010x}"
11381                )));
11382            }
11383        }
11384        if cur.pos < buf.len() {
11385            return Err(StorageError::Corrupt(format!(
11386                "trailing bytes: {} unread",
11387                buf.len() - cur.pos
11388            )));
11389        }
11390        Ok(cat)
11391    }
11392}
11393
11394#[cfg(test)]
11395mod tests;