Skip to main content

core_api/
db.rs

1use crate::ingest::{IngestOptions, IngestReport};
2use crate::roles::{PropPredicate, RoleDef, RolesFile, WriteScope};
3use crate::subscription::{
4    event_matches, DbEvent, SubEntry, SubFilter, SubInner, Subscription, DEFAULT_SUB_CAPACITY,
5};
6use core_query::cypher::ast::{ret_val_label, ArithOp};
7use core_query::cypher::{
8    execute, execute_union, is_subscribable, is_write_tokens, lex, parse, parse_read, parse_write,
9    plan, Expr, MatchDeleteNodeStmt, NodePat, Operand, Params, Pattern, PlanOp, Query, RetItem,
10    RetVal, WriteStatement,
11};
12use core_query::{eval_cmp, eval_filter, expand, neighborhood, Dir, Filter, GraphView, ResultSet};
13use core_rules::{
14    decode_rule_def, ef_max, evaluate, BuildProgress, EngineEdgeDelta, GraphMut, NodeView,
15    Predicate, RuleDef, RuleEngine, ViewDef, ViewStore,
16};
17use core_storage::fs::{FileId, Fs, FsIntrospect, RealFs};
18use core_storage::fulltext::FulltextIndex;
19use core_storage::property_index::PropertyIndex;
20use core_storage::v8::encode::{
21    archived_hnsw_to_owned, archived_rules_meta_to_owned, archived_to_idmap, archived_to_interner,
22    archived_views_to_owned, decode_last_change_bytes, decode_meta, encode_v8, V8Meta,
23};
24use core_storage::v8::seam::TopologyView;
25use core_storage::wal::{decode_all, encode_record, WalRecord};
26use core_storage::EdgePropsView;
27use core_storage::{
28    namespace_of_value, ColumnStore, Direction, EdgeProps, GraphError, IdMap, Interner, Result,
29    Topology, Value,
30};
31pub use core_storage::{valid_namespace, NS_DEFAULT, NS_MAX_LEN, NS_PROP};
32
33/// Index of [`NS_DEFAULT`] in `GraphDb::ns_names` — always zero, so the
34/// open-time pass over a store with no `ns` column fills `node_ns` with one
35/// constant and allocates no names.
36const NS_DEFAULT_IDX: u32 = 0;
37
38/// The reserved edge property holding a pair's insert count (§5.13).
39///
40/// Absent means 1 — the count is written only from the second insert of a
41/// triple onward, and only on a store that called
42/// [`GraphDb::enable_multiplicity`]. The engine owns the name: Cypher `SET` on
43/// it is refused, as the other reserved names are.
44pub const EDGE_COUNT_PROP: &str = "count";
45use serde::{Deserialize, Serialize};
46use std::collections::{BTreeMap, BTreeSet, HashMap, HashSet};
47use std::sync::Arc;
48
49/// Print a timing checkpoint when MUSHROOMDB_TRACE_OPEN is set.
50/// Zero-cost when the env var is absent (the var check is O(1) after first call).
51macro_rules! trace_open {
52    ($phase:literal, $t:expr) => {
53        if std::env::var("MUSHROOMDB_TRACE_OPEN").is_ok() {
54            eprintln!(
55                "[MUSHROOMDB_TRACE_OPEN] {:40} {:>9.3?}",
56                $phase,
57                $t.elapsed()
58            );
59        }
60    };
61}
62
63/// Print a migration phase checkpoint when MUSHROOMDB_TRACE_MIGRATE is set.
64/// Zero-cost when the env var is absent (the var check is O(1) after first call).
65macro_rules! trace_migrate {
66    ($phase:literal, $t:expr) => {
67        if std::env::var("MUSHROOMDB_TRACE_MIGRATE").is_ok() {
68            eprintln!(
69                "[MUSHROOMDB_TRACE_MIGRATE] {:40} {:>9.3?}",
70                $phase,
71                $t.elapsed()
72            );
73        }
74    };
75}
76
77// Test-only: counts how many times `pending_deltas_since().to_vec()` actually
78// executes (i.e., at least one view is defined). Used to verify the fast-path
79// guard skips the allocation when `view_store.is_empty()`.
80#[cfg(test)]
81thread_local! {
82    static DELTA_COPY_COUNT: std::cell::Cell<usize> = const { std::cell::Cell::new(0) };
83}
84
85// Per-thread count of query-subscription `execute` calls in `distribute_events`.
86//
87// Incremented each time a query subscription actually runs its plan (i.e.,
88// the label-skip fast-path did not fire). Because `distribute_events` is
89// called synchronously on the writer thread, this thread-local correctly
90// isolates each test thread's count even when integration tests run in
91// parallel. Read via [`query_sub_exec_count`].
92thread_local! {
93    static QUERY_SUB_EXECS_TL: std::cell::Cell<usize> = const { std::cell::Cell::new(0) };
94}
95
96/// Return the number of query-subscription re-executions logged on this
97/// thread since the process started (or since last reset via
98/// [`reset_query_sub_exec_count`]).
99///
100/// Primarily for integration tests that verify the label-skip fast-path.
101#[doc(hidden)]
102pub fn query_sub_exec_count() -> usize {
103    QUERY_SUB_EXECS_TL.with(|c| c.get())
104}
105
106/// Reset the per-thread query-subscription execution counter to zero.
107#[doc(hidden)]
108pub fn reset_query_sub_exec_count() {
109    QUERY_SUB_EXECS_TL.with(|c| c.set(0));
110}
111
112// Exact-versus-approximate warnings emitted on this thread. Thread-local for
113// the same reason [`QUERY_SUB_EXECS_TL`] is: integration tests run in parallel
114// and each gets its own thread, so a neighbour's masked search cannot be
115// mistaken for this test's.
116thread_local! {
117    static AMBIGUOUS_EXACTNESS_WARNS: std::cell::Cell<u64> = const { std::cell::Cell::new(0) };
118    static AMBIGUOUS_EXACTNESS_LAST: std::cell::RefCell<Option<String>> =
119        const { std::cell::RefCell::new(None) };
120}
121
122/// How many times a masked, non-exact vector search has explained itself on
123/// this thread since the last [`ambiguous_exactness_warns_reset`].
124///
125/// The line itself is the product; this counter exists so a test can assert it
126/// is printed **once per index** rather than once per call.
127///
128/// **Single-threaded assertions only.** The suppression set this counts is a
129/// `Mutex<HashSet<_>>` on the `GraphDb` — shared by every thread — while the
130/// counter is thread-local. Under a concurrent caller (`serve`, which is the
131/// deployment the warning exists for) the thread that prints the line is not
132/// necessarily the thread that asked, so a zero here does not mean the line was
133/// not printed and a one does not mean it was printed once. It answers
134/// "once per index" only in a test that owns the store.
135#[doc(hidden)]
136pub fn ambiguous_exactness_warns() -> u64 {
137    AMBIGUOUS_EXACTNESS_WARNS.with(|c| c.get())
138}
139
140/// The most recent exactness warning printed on this thread, verbatim.
141///
142/// The advice a caller reads has to be advice that caller can act on, which the
143/// counter alone cannot witness — see
144/// `the_hybrid_path_advises_a_call_the_hybrid_caller_can_make`. Carries the
145/// same single-threaded caveat as [`ambiguous_exactness_warns`].
146#[doc(hidden)]
147pub fn ambiguous_exactness_last_warning() -> Option<String> {
148    AMBIGUOUS_EXACTNESS_LAST.with(|c| c.borrow().clone())
149}
150
151/// Reset this thread's exactness-warning counter and recorded line.
152#[doc(hidden)]
153pub fn ambiguous_exactness_warns_reset() {
154    AMBIGUOUS_EXACTNESS_WARNS.with(|c| c.set(0));
155    AMBIGUOUS_EXACTNESS_LAST.with(|c| *c.borrow_mut() = None);
156}
157
158/// Which signature reached the approximate masked vector leg.
159///
160/// One code path, two entry points, and the advice cannot be the same: a
161/// warning that names an argument the caller's function does not take sends
162/// them looking for a parameter that is not there. `search_hybrid` and
163/// `search_hybrid_scoped` take `(text_field, query_text, vector_field,
164/// query_vec, label, k[, mask])` — no `exact`, no `where`.
165///
166/// The warning is not suppressed on the hybrid path. The approximation is the
167/// same one, and a caller who read `mask=` as a promise of exhaustiveness is
168/// the reader it was written for whichever door they came in by; only the
169/// remedy differs, so only the remedy changes.
170#[derive(Clone, Copy, PartialEq, Eq, Hash, Debug)]
171enum ExactnessCaller {
172    /// `find_similar` / `find_similar_vector_*` — `exact` and `where` are its
173    /// own parameters.
174    Vector,
175    /// `search_hybrid` / `search_hybrid_scoped` — neither argument exists, and
176    /// the leg is one half of a fusion.
177    Hybrid,
178}
179
180impl ExactnessCaller {
181    fn subject(self) -> &'static str {
182        match self {
183            Self::Vector => "a masked vector search",
184            Self::Hybrid => "the vector leg of a masked hybrid search",
185        }
186    }
187
188    fn advice(self) -> &'static str {
189        match self {
190            Self::Vector => "pass exact=True or a where= predicate.",
191            // Names the call that does take the argument, because this one
192            // does not: the caller's own next step, not a parameter hunt.
193            Self::Hybrid => {
194                "run the vector leg on its own with find_similar(field, vector, mask=…, \
195                 exact=True) and fuse it with search() yourself — search_hybrid itself \
196                 takes no exactness argument."
197            }
198        }
199    }
200}
201
202/// Internal state for a single `subscribe_query` subscription.
203///
204/// On every commit, `distribute_events` re-executes `ops` against the current
205/// graph state, diffs the result against `prev_rows`, and pushes
206/// `DbEvent::QueryRowAdded` / `QueryRowRemoved` events to `inner`.
207///
208/// **Full re-run per commit; use LIMIT to bound execution cost.**
209/// (Differential evaluation is roadmap / Phase 5.)
210pub(crate) struct QuerySubEntry {
211    /// Compiled plan for the subscribed Cypher query.
212    ops: Vec<PlanOp>,
213    /// Column names from the first execution (fixed for the subscription lifetime).
214    columns: Vec<String>,
215    /// Serialized (JSON) row key → row data, representing the result set at
216    /// the end of the last commit. Used to diff against the new result.
217    prev_row_map: std::collections::HashMap<String, Vec<Option<Value>>>,
218    /// Weak pointer to the subscriber queue; dead Weak → subscription dropped.
219    inner: std::sync::Weak<SubInner>,
220    /// Interned label sym captured at subscribe time from the plan's leading scan
221    /// (`ScanLabel`, `IndexScan`, or `IndexIntersect` with a concrete label).
222    ///
223    /// `None` means the plan has an `Expand` op (or no recognizable leading scan
224    /// with a concrete label), and this subscription must re-execute on every
225    /// commit without skipping. This is the conservative v0.4.3 boundary: Expand
226    /// queries are never skipped because edges can alter join results regardless
227    /// of which node labels were written.
228    scan_label: Option<u32>,
229}
230
231/// A post-commit mutation notification.
232///
233/// Emitted from `log_then_apply` after the WAL append, fsync, and
234/// in-memory `apply` all succeed. Never emitted for rejected operations
235/// (validation errors, [`GraphError::RuleOwned`], duplicate keys, no-op
236/// deletes/removes). Event payloads carry user keys and rule names, never
237/// internal ids.
238///
239/// **Replay:** [`GraphDb::open`] / [`GraphDb::open_with`] replay the WAL via
240/// `apply` only. Emission lives exclusively in `log_then_apply`, so
241/// recovery is silent even if a sink were installed (it cannot be: the
242/// sink is in-memory and set after open).
243///
244/// **Ordering:** a `Batch` WAL frame emits one event per inner record, then
245/// [`MutationEvent::BatchApplied`]. An ingest commit emits those same inner
246/// events, then [`MutationEvent::Ingested`] (not `BatchApplied`). An empty
247/// or all-noop batch writes no WAL and emits nothing (including no summary).
248///
249/// **Derived edges:** rule-created or retracted edges are not individually
250/// evented — they are recoverable from the triggering mutation plus the live
251/// rule set. Only the triggering record is emitted.
252///
253/// **Wire form:** externally tagged snake_case JSON
254/// (`{"node_inserted":{"label":"A","key":"k"}}`).
255#[derive(Debug, Clone, PartialEq, Eq, Serialize)]
256#[serde(rename_all = "snake_case")]
257pub enum MutationEvent {
258    NodeInserted {
259        label: String,
260        key: String,
261    },
262    PropSet {
263        key: String,
264        field: String,
265    },
266    PropRemoved {
267        key: String,
268        field: String,
269    },
270    EdgeInserted {
271        edge_type: String,
272        src: String,
273        dst: String,
274    },
275    EdgeDeleted {
276        edge_type: String,
277        src: String,
278        dst: String,
279    },
280    NodeDeleted {
281        key: String,
282    },
283    RuleCreated {
284        name: String,
285    },
286    RuleDeleted {
287        name: String,
288    },
289    RuleRebuilt {
290        name: String,
291    },
292    BatchApplied {
293        ops: usize,
294    },
295    Ingested {
296        label: String,
297        inserted: usize,
298    },
299}
300
301/// Whether `rec` is the frame [`GraphDb::create_rule`] logs: a `CreateRule`,
302/// alone or behind the `Intern` record for its edge type.
303///
304/// `create_rule` goes through the dense rewrite like every other write, so its
305/// frame is `Batch([Intern { edge_type }, CreateRule])` rather than a bare
306/// `CreateRule`. Anything that asks "was this commit a rule creation" must
307/// accept both shapes, or it silently stops recognising the one the
308/// standalone call writes. A batch carrying anything else is a user batch and
309/// is not this frame.
310fn is_create_rule_frame(rec: &WalRecord) -> bool {
311    match rec {
312        WalRecord::CreateRule { .. } => true,
313        WalRecord::Batch(inner) => {
314            inner
315                .iter()
316                .any(|r| matches!(r, WalRecord::CreateRule { .. }))
317                && inner
318                    .iter()
319                    .all(|r| matches!(r, WalRecord::CreateRule { .. } | WalRecord::Intern { .. }))
320        }
321        _ => false,
322    }
323}
324
325fn event_from_record(rec: &WalRecord, intern: &Interner, ids: &IdMap) -> Option<MutationEvent> {
326    match rec {
327        WalRecord::InsertNode { label, key, .. } => Some(MutationEvent::NodeInserted {
328            label: label.clone(),
329            key: key.clone(),
330        }),
331        WalRecord::InsertNodeId { label, key, .. } => Some(MutationEvent::NodeInserted {
332            label: intern.resolve(*label)?.to_string(),
333            key: key.clone(),
334        }),
335        WalRecord::SetProp { key, field, .. } => Some(MutationEvent::PropSet {
336            key: key.clone(),
337            field: field.clone(),
338        }),
339        WalRecord::SetPropId { id, field, .. } => Some(MutationEvent::PropSet {
340            key: ids.key_of(*id)?.to_string(),
341            field: intern.resolve(*field)?.to_string(),
342        }),
343        WalRecord::RemoveProp { key, field } => Some(MutationEvent::PropRemoved {
344            key: key.clone(),
345            field: field.clone(),
346        }),
347        WalRecord::InsertEdge {
348            edge_type,
349            src_key,
350            dst_key,
351        } => Some(MutationEvent::EdgeInserted {
352            edge_type: edge_type.clone(),
353            src: src_key.clone(),
354            dst: dst_key.clone(),
355        }),
356        WalRecord::InsertEdgeId { etype, src, dst } => Some(MutationEvent::EdgeInserted {
357            edge_type: intern.resolve(*etype)?.to_string(),
358            src: ids.key_of(*src)?.to_string(),
359            dst: ids.key_of(*dst)?.to_string(),
360        }),
361        WalRecord::DeleteEdge {
362            edge_type,
363            src_key,
364            dst_key,
365        } => Some(MutationEvent::EdgeDeleted {
366            edge_type: edge_type.clone(),
367            src: src_key.clone(),
368            dst: dst_key.clone(),
369        }),
370        WalRecord::DeleteNode { key } => Some(MutationEvent::NodeDeleted { key: key.clone() }),
371        WalRecord::CreateRule { def_bytes } => {
372            let def: RuleDef = decode_rule_def(def_bytes).ok()?;
373            Some(MutationEvent::RuleCreated { name: def.name })
374        }
375        WalRecord::DeleteRule { name } => Some(MutationEvent::RuleDeleted { name: name.clone() }),
376        WalRecord::RebuildRule { name } => Some(MutationEvent::RuleRebuilt { name: name.clone() }),
377        WalRecord::Batch(_)
378        | WalRecord::CreateView { .. }
379        | WalRecord::DeleteView { .. }
380        | WalRecord::EnableFulltext { .. }
381        | WalRecord::DisableFulltext { .. }
382        | WalRecord::EnableIndex { .. }
383        | WalRecord::DisableIndex { .. }
384        | WalRecord::Intern { .. }
385        // History markers are no-ops for mutation events — they carry no new
386        // state and rules re-derive deterministically on replay.
387        | WalRecord::DerivedEdgeAdded { .. }
388        | WalRecord::DerivedEdgeRetracted { .. }
389        // A count changes neither the node nor the edge population: the pair it
390        // counts was already there, which is why it is written at all.
391        | WalRecord::SetEdgeCount { .. }
392        // RenameNode carries no node/edge count change; no special event.
393        | WalRecord::RenameNode { .. } => None,
394    }
395}
396
397/// Database-wide counters plus per-rule budget/fire stats.
398#[derive(Debug, Clone, PartialEq, Eq, Serialize)]
399pub struct Stats {
400    pub nodes_live: usize,
401    pub nodes_tombstoned: usize,
402    pub edges: u64,
403    pub rules: Vec<RuleStats>,
404    /// How many writes hit the rule-chaining depth cap with work still pending,
405    /// since this handle was opened. Non-zero means some derived edges beyond
406    /// the cap are stale and no single later write will repair them: split the
407    /// rule chain or shorten it. Never persisted, so it resets on reopen.
408    #[serde(default)]
409    pub chain_truncations: u64,
410    /// The oldest commit index history still reaches (the WAL horizon floor).
411    /// `0` means nothing has been pruned and history is complete; a non-zero
412    /// value means events before that commit were pruned and are gone.
413    #[serde(default)]
414    pub history_floor: u64,
415    /// Live node counts per namespace, in name order. Always carries
416    /// `default` — a store is at least its default namespace — so a
417    /// single-tenant store reads `[{"name":"default", …}]` and a reader can
418    /// tell "no namespaces in use" from one entry.
419    #[serde(default)]
420    pub namespaces: Vec<NamespaceStats>,
421}
422
423/// Live node count for one namespace; one entry of [`Stats::namespaces`].
424#[derive(Debug, Clone, PartialEq, Eq, Serialize)]
425pub struct NamespaceStats {
426    pub name: String,
427    pub nodes_live: usize,
428}
429
430/// One rule's provenance size, trip latch, and fire counter.
431///
432/// `tripped` is a one-way latch: once set, the engine adds no new edges for
433/// that rule until [`GraphDb::rebuild_rule`] (and only if the full desired
434/// set then fits). `fires` counts `on_node_changed` evaluations plus
435/// backfill/rebuild participant ticks (rebuild counts even when it is a
436/// provenance no-op).
437#[derive(Debug, Clone, PartialEq, Eq, Serialize)]
438pub struct RuleStats {
439    pub name: String,
440    pub edges: u64,
441    pub tripped: bool,
442    pub fires: u64,
443    /// Whether this rule uses the approximate IVF-Flat candidate path.
444    pub approximate: bool,
445    /// `Some` while this rule's vector index is still being built.
446    ///
447    /// The rule derives **no** edges until it is `None`: the backfill is one
448    /// commit that runs after the index is whole, so a caller never sees a
449    /// partial edge set. Absent from the JSON when the rule is not building,
450    /// which is every rule created over a corpus at or below
451    /// [`core_rules::HNSW_BUILD_BATCH`] vectors.
452    #[serde(default, skip_serializing_if = "Option::is_none")]
453    pub building: Option<BuildProgress>,
454}
455
456/// One entry in the slow-query ring buffer.
457#[derive(Debug, Clone, Serialize)]
458pub struct SlowQueryEntry {
459    /// Execution time in whole milliseconds.
460    pub ms: u64,
461    /// The Cypher query string that was slow.
462    pub query: String,
463    /// The commit sequence number at the time the query ran.
464    pub at_commit: u64,
465}
466
467/// Snapshot of the slow-query log returned by [`GraphDb::slow_query_snapshot`].
468#[derive(Debug, Clone, Serialize)]
469pub struct SlowQuerySnapshot {
470    /// Current threshold in milliseconds (0 = disabled).
471    pub threshold_ms: u64,
472    /// Total number of slow queries ever recorded (not capped by ring size).
473    pub count: u64,
474    /// Most-recent slow queries (up to 16), oldest first.
475    pub last: Vec<SlowQueryEntry>,
476}
477
478/// Internal ring-buffer state protected by a `Mutex` so `query(&self)` can
479/// write to it without a mutable borrow.
480struct SlowQueryLog {
481    entries: std::collections::VecDeque<SlowQueryEntry>,
482    total: u64,
483}
484
485/// Maximum number of entries kept in the slow-query ring buffer.
486const SLOW_QUERY_RING_CAP: usize = 16;
487
488/// Wire summary of a [`Predicate`]. JSON only — `Explanation` is never
489/// bincode-persisted (WAL/snapshots store `RuleDef` bytes, not this type).
490#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
491pub struct PredicateSummary {
492    pub kind: String,
493    pub fields: Vec<String>,
494    pub min: Option<f64>,
495    pub tolerance: Option<f64>,
496    pub km: Option<f64>,
497    pub parts: Option<Vec<PredicateSummary>>,
498    /// True when the owning rule has `approximate=true` (IVF-Flat candidate path).
499    /// Always false for predicates reported without rule context (sub-predicates in `parts`).
500    #[serde(default)]
501    pub approximate: bool,
502}
503
504impl From<&Predicate> for PredicateSummary {
505    fn from(p: &Predicate) -> Self {
506        match p {
507            Predicate::KeyMatch { field } => PredicateSummary {
508                kind: "key_match".into(),
509                fields: vec![field.clone()],
510                min: None,
511                tolerance: None,
512                km: None,
513                parts: None,
514                approximate: false,
515            },
516            Predicate::FieldEqual { field } => PredicateSummary {
517                kind: "field_equal".into(),
518                fields: vec![field.clone()],
519                min: None,
520                tolerance: None,
521                km: None,
522                parts: None,
523                approximate: false,
524            },
525            Predicate::Overlap { field, min } => PredicateSummary {
526                kind: "overlap".into(),
527                fields: vec![field.clone()],
528                min: Some(*min),
529                tolerance: None,
530                km: None,
531                parts: None,
532                approximate: false,
533            },
534            Predicate::NumericWithin { field, tolerance } => PredicateSummary {
535                kind: "numeric_within".into(),
536                fields: vec![field.clone()],
537                min: None,
538                tolerance: Some(*tolerance),
539                km: None,
540                parts: None,
541                approximate: false,
542            },
543            Predicate::GeoRadius { field, km } => PredicateSummary {
544                kind: "geo_radius".into(),
545                fields: vec![field.clone()],
546                min: None,
547                tolerance: None,
548                km: Some(*km),
549                parts: None,
550                approximate: false,
551            },
552            Predicate::VectorSimilar { field, min } => PredicateSummary {
553                kind: "vector_similar".into(),
554                fields: vec![field.clone()],
555                min: Some(*min),
556                tolerance: None,
557                km: None,
558                parts: None,
559                approximate: false,
560            },
561            Predicate::All(inner) => {
562                let parts: Vec<PredicateSummary> = inner.iter().map(Self::from).collect();
563                let mut fields = Vec::new();
564                for part in &parts {
565                    for f in &part.fields {
566                        if !fields.contains(f) {
567                            fields.push(f.clone());
568                        }
569                    }
570                }
571                PredicateSummary {
572                    kind: "all".into(),
573                    fields,
574                    min: None,
575                    tolerance: None,
576                    km: None,
577                    parts: Some(parts),
578                    approximate: false,
579                }
580            }
581            Predicate::Any(inner) => {
582                let parts: Vec<PredicateSummary> = inner.iter().map(Self::from).collect();
583                let mut fields = Vec::new();
584                for part in &parts {
585                    for f in &part.fields {
586                        if !fields.contains(f) {
587                            fields.push(f.clone());
588                        }
589                    }
590                }
591                PredicateSummary {
592                    kind: "any".into(),
593                    fields,
594                    min: None,
595                    tolerance: None,
596                    km: None,
597                    parts: Some(parts),
598                    approximate: false,
599                }
600            }
601        }
602    }
603}
604
605/// Snapshot of a live node's key, label, and columnar properties.
606///
607/// `props` is a [`BTreeMap`] so field order is deterministic (sorted by name)
608/// regardless of insert order or the columnar store's `HashMap` iteration.
609///
610/// Deliberately does not derive `Serialize`: `Value`'s serde form is
611/// internally tagged. Wire JSON is built by `value_to_json` in the server.
612#[derive(Debug, Clone, PartialEq)]
613pub struct NodeInfo {
614    pub key: String,
615    pub label: String,
616    pub props: BTreeMap<String, Value>,
617}
618
619/// Counts returned by [`GraphDb::delete_node`].
620#[derive(Debug, Clone, PartialEq, Eq, Default)]
621pub struct DeleteReport {
622    /// Number of manual (user-inserted) edges removed.
623    pub manual_edges: u64,
624    /// Number of derived (rule-owned) edges retracted.
625    pub derived_edges: u64,
626}
627
628/// One directed edge incident on a node, with provenance membership.
629///
630/// `derived` is true iff `(edge_type, src, dst)` is in the rule engine's
631/// Plan-8 `by_node` provenance index.
632#[derive(Debug, Clone, PartialEq, Eq, Serialize)]
633pub struct EdgeInfo {
634    pub edge_type: String,
635    pub src_key: String,
636    pub dst_key: String,
637    pub derived: bool,
638}
639
640/// One directed edge incident on a node at a point in WAL history, with the
641/// rule that derived it when it is rule-owned.
642///
643/// Returned by [`GraphDb::edges_at`] (sorted by `(edge_type, src_key, dst_key)`)
644/// and by [`GraphDb::what_if_set_prop`].
645#[derive(Debug, Clone, PartialEq, Eq, PartialOrd, Ord, Serialize)]
646pub struct EdgeAt {
647    pub edge_type: String,
648    pub src_key: String,
649    pub dst_key: String,
650    /// `true` when a rule wrote the edge (`DerivedEdgeAdded` in the WAL, or a
651    /// live provenance entry).
652    pub derived: bool,
653    /// The rule that derived the edge. `None` for a manual edge.
654    pub rule: Option<String>,
655}
656
657/// The derived edges a hypothetical property change would retract and derive.
658///
659/// Returned by [`GraphDb::what_if_set_prop`]. Both lists are sorted by
660/// `(edge_type, src_key, dst_key)` and every entry is rule-derived.
661#[derive(Debug, Clone, PartialEq, Eq, Serialize)]
662pub struct WhatIf {
663    /// Derived edges that exist now and would be retracted.
664    pub lost: Vec<EdgeAt>,
665    /// Derived edges that do not exist now and would be derived.
666    pub gained: Vec<EdgeAt>,
667}
668
669/// An edge with mask-aware endpoint visibility.
670///
671/// Returned by [`GraphDb::node_edges_masked`] in [`crate::mask::MaskMode::Stub`]
672/// mode — hidden endpoints carry `*_restricted: true`.
673#[derive(Debug, Clone, PartialEq, Eq)]
674pub struct MaskedEdge {
675    pub edge_type: String,
676    pub src_key: String,
677    /// `true` when `src_key` is in the DB but hidden from the mask.
678    pub src_restricted: bool,
679    pub dst_key: String,
680    /// `true` when `dst_key` is in the DB but hidden from the mask.
681    pub dst_restricted: bool,
682    pub derived: bool,
683}
684
685/// Result of a mask-aware node lookup via [`GraphDb::node_info_masked`].
686///
687/// `None` from that method means the key does not exist (→ 404).
688/// `Some(Restricted)` is only produced when `mask.mode() == MaskMode::Stub`.
689#[derive(Debug, PartialEq)]
690pub enum MaskedNodeResult {
691    Visible(NodeInfo),
692    /// Node exists in the DB but is hidden from this mask.
693    Restricted,
694}
695
696/// One rule-owned edge between two nodes, with the rule name, edge type,
697/// direction (src_key → dst_key), and weight if the rule stores one.
698#[derive(Debug, Clone, PartialEq, Serialize)]
699pub struct Explanation {
700    pub rule: String,
701    pub edge_type: String,
702    pub src_key: String,
703    pub dst_key: String,
704    pub weight: Option<f64>,
705    pub predicate: PredicateSummary,
706    /// For a via-hop rule, the edge type the rule hops over to reach its
707    /// candidates. `None` for a plain two-node rule. A via-hop rule whose
708    /// `via_edge` is itself rule-derived is the chaining case: the hop edge
709    /// was written by another rule in the same commit.
710    #[serde(default)]
711    pub via_edge: Option<String>,
712}
713
714/// Report returned by [`GraphDb::backup_to`].
715#[derive(Debug, Clone)]
716pub struct BackupReport {
717    /// Filenames copied into the destination directory (sorted ascending).
718    pub files: Vec<String>,
719    /// Total bytes written across all copied files.
720    pub bytes: u64,
721    /// `true` when the destination opened cleanly and passed post-copy checks.
722    ///
723    /// For stores that have a `snapshot.bin` this means: all V8 section CRCs
724    /// matched **and** the destination opened without error.
725    ///
726    /// For WAL-only stores (no `snapshot.bin`) there is no snapshot to
727    /// CRC-check; `verified` is `true` when the destination opened and
728    /// replayed the WAL without error (record-level checksums in the WAL
729    /// provide the integrity signal, not section CRCs).
730    pub verified: bool,
731}
732
733/// One directed edge in export form, with optional rule attribution for derived edges.
734///
735/// Returned by [`GraphDb::all_edges_for_export`].
736///
737/// Does not derive `Eq`/`Ord`: `weight` is an `f64` and NaN breaks a total
738/// order. Callers that need a stable edge ordering already sort by
739/// `(edge_type, src, dst)` explicitly (see `all_edges_for_export`).
740#[derive(Debug, Clone, PartialEq, PartialOrd)]
741pub struct ExportEdge {
742    pub edge_type: String,
743    pub src: String,
744    pub dst: String,
745    pub derived: bool,
746    /// Rule name that created this edge, if derived. `None` for manual edges.
747    pub rule: Option<String>,
748    /// The creating rule's declared `weight_prop`, read off this edge, when
749    /// derived and numeric (`Int`/`Float`). `None` for manual edges, derived
750    /// edges whose rule declares no `weight_prop`, or a non-numeric value.
751    pub weight: Option<f64>,
752}
753
754/// One edge type's shape, as [`GraphDb::edge_type_census`] counts it.
755///
756/// Deliberately per *type* and not per edge: everything here is a summary a
757/// caller can print in one line, and none of it costs a record per edge.
758#[derive(Debug, Clone, PartialEq, Eq)]
759pub struct EdgeTypeCensus {
760    pub edge_type: String,
761    /// Directed edges of this type. Counted the way
762    /// [`GraphDb::edge_count`] counts: each edge once, from its source.
763    pub edges: u64,
764    /// Every label seen on a source of this type, sorted.
765    pub src_labels: Vec<String>,
766    /// Every label seen on a destination of this type, sorted.
767    pub dst_labels: Vec<String>,
768    /// The rules that declare this `edge_type`, sorted. Empty for a type
769    /// written by hand.
770    pub rules: Vec<String>,
771    /// `(src key, dst key)` of the first edge of this type in the store's own
772    /// id order — a real pair to quote in an example.
773    pub sample: Option<(String, String)>,
774}
775
776/// Construct the standard write-query result set (columns: created, properties_set, deleted).
777fn write_result_set() -> ResultSet {
778    ResultSet::new(vec![
779        "created".into(),
780        "properties_set".into(),
781        "deleted".into(),
782    ])
783}
784
785fn resolve_merge_set_value(op: &Operand, params: &BTreeMap<String, Value>) -> Result<Value> {
786    match op {
787        Operand::Lit(v) => Ok(v.clone()),
788        Operand::Param(name) => params
789            .get(name)
790            .cloned()
791            .ok_or_else(|| GraphError::QueryError {
792                detail: format!("missing parameter `{name}`"),
793            }),
794        _ => Err(GraphError::QueryError {
795            detail: "ON CREATE/ON MATCH SET value must be a literal or $parameter".into(),
796        }),
797    }
798}
799
800fn operand_node_vars(op: &Operand, out: &mut Vec<String>) {
801    match op {
802        Operand::Prop { var, .. } | Operand::Var(var) => {
803            if !out.contains(var) {
804                out.push(var.clone());
805            }
806        }
807        Operand::FuncCall { args, .. } => {
808            for arg in args {
809                operand_node_vars(arg, out);
810            }
811        }
812        Operand::BinArith { left, right, .. } => {
813            operand_node_vars(left, out);
814            operand_node_vars(right, out);
815        }
816        Operand::Case { branches, default } => {
817            // Branch conditions reference vars already bound (and mask-filtered)
818            // by the MATCH phase, so collecting from the value operands + ELSE
819            // is sufficient for RETURN-projection var discovery.
820            for (_, value) in branches {
821                operand_node_vars(value, out);
822            }
823            if let Some(d) = default {
824                operand_node_vars(d, out);
825            }
826        }
827        Operand::Index { base, index } => {
828            operand_node_vars(base, out);
829            operand_node_vars(index, out);
830        }
831        Operand::Lit(_) | Operand::Param(_) => {}
832    }
833}
834
835fn ret_node_vars(items: &[RetItem]) -> Vec<String> {
836    let mut out = Vec::new();
837    for item in items {
838        match &item.value {
839            RetVal::Var(v) | RetVal::Prop { var: v, .. } => {
840                if !out.contains(v) {
841                    out.push(v.clone());
842                }
843            }
844            RetVal::FuncCall { args, .. } => {
845                for arg in args {
846                    operand_node_vars(arg, &mut out);
847                }
848            }
849            RetVal::ScalarExpr(op) => operand_node_vars(op, &mut out),
850            RetVal::Agg { .. } => {}
851        }
852    }
853    out
854}
855
856fn add_var(out: &mut Vec<String>, v: &str) {
857    if !out.iter().any(|x| x == v) {
858        out.push(v.to_string());
859    }
860}
861
862fn pattern_node_vars(pats: &[Pattern]) -> Vec<String> {
863    let mut out = Vec::new();
864    for p in pats {
865        if let Some(v) = &p.start.var {
866            add_var(&mut out, v);
867        }
868        for (_, dest) in &p.chain {
869            if let Some(v) = &dest.var {
870                add_var(&mut out, v);
871            }
872        }
873    }
874    out
875}
876
877fn pattern_rel_vars(pats: &[Pattern]) -> Vec<String> {
878    let mut out = Vec::new();
879    for p in pats {
880        for (rel, _) in &p.chain {
881            if rel.hops.is_none() {
882                if let Some(v) = &rel.var {
883                    add_var(&mut out, v);
884                }
885            }
886        }
887    }
888    out
889}
890
891fn rel_type_alias(var: &str) -> String {
892    format!("__rt_{var}")
893}
894
895fn ret_column_name(item: &RetItem) -> String {
896    if let Some(alias) = &item.alias {
897        return alias.clone();
898    }
899    // The same naming rule the planner and the executor use, so a
900    // write-statement RETURN names its columns exactly as a read query does.
901    // An aggregate is not legal in a write-statement RETURN; it keeps the
902    // placeholder it always had.
903    ret_val_label(&item.value).unwrap_or_else(|| "<agg>".to_string())
904}
905
906fn eval_set_return_operand<F: Fs>(
907    db: &GraphDb<F>,
908    match_rs: &ResultSet,
909    row: usize,
910    rel_vars: &[String],
911    op: &Operand,
912    params: &BTreeMap<String, Value>,
913) -> Result<Option<Value>> {
914    match op {
915        Operand::Lit(v) => Ok(Some(v.clone())),
916        Operand::Param(name) => params.get(name).cloned().ok_or_else(|| GraphError::QueryError {
917            detail: format!("missing parameter `{name}`"),
918        }).map(Some),
919        Operand::Var(name) if rel_vars.iter().any(|r| r == name) => Err(GraphError::QueryError {
920            detail: format!(
921                "cannot return relationship variable '{name}' bare; return its properties ({name}.field) instead"
922            ),
923        }),
924        Operand::Var(name) => Ok(match_rs.get(row, name).cloned()),
925        Operand::Prop { var, field } => {
926            if rel_vars.iter().any(|r| r == var) {
927                return Ok(None);
928            }
929            let Some(Value::Str(key)) = match_rs.get(row, var) else {
930                return Ok(None);
931            };
932            if let Some(v) = db.get_prop(key, field) {
933                return Ok(Some(v));
934            }
935            // Same stored-wins identity fallback as the read path:
936            // n.key / n.id / n.label, not only get_prop.
937            Ok(match field.as_str() {
938                "key" | "id" => Some(Value::Str(key.clone())),
939                "label" => db
940                    .node_ref(key)
941                    .map(|n| Value::Str(n.label().to_owned())),
942                _ => None,
943            })
944        }
945        Operand::FuncCall { name, args } => {
946            eval_set_return_func(db, match_rs, row, rel_vars, name, args, params)
947        }
948        Operand::BinArith { op, left, right } => {
949            let lv = eval_set_return_operand(db, match_rs, row, rel_vars, left, params)?;
950            let rv = eval_set_return_operand(db, match_rs, row, rel_vars, right, params)?;
951            eval_set_return_arith(op, lv, rv)
952        }
953        Operand::Case { branches, default } => {
954            for (cond, value) in branches {
955                if eval_set_return_expr(db, match_rs, row, rel_vars, cond, params, 0)? {
956                    return eval_set_return_operand(db, match_rs, row, rel_vars, value, params);
957                }
958            }
959            match default {
960                Some(d) => eval_set_return_operand(db, match_rs, row, rel_vars, d, params),
961                None => Ok(None),
962            }
963        }
964        Operand::Index { base, index } => {
965            let base_val = eval_set_return_operand(db, match_rs, row, rel_vars, base, params)?;
966            let idx_val = eval_set_return_operand(db, match_rs, row, rel_vars, index, params)?;
967            Ok(core_query::value_ops::index_list(base_val, idx_val))
968        }
969    }
970}
971
972fn eval_set_return_expr<F: Fs>(
973    db: &GraphDb<F>,
974    match_rs: &ResultSet,
975    row: usize,
976    rel_vars: &[String],
977    expr: &Expr,
978    params: &BTreeMap<String, Value>,
979    depth: u32,
980) -> Result<bool> {
981    if depth > 256 {
982        return Err(GraphError::QueryError {
983            detail: "expression nesting too deep".into(),
984        });
985    }
986    match expr {
987        Expr::And(lhs, rhs) => {
988            let l = eval_set_return_expr(db, match_rs, row, rel_vars, lhs, params, depth + 1)?;
989            let r = eval_set_return_expr(db, match_rs, row, rel_vars, rhs, params, depth + 1)?;
990            Ok(l && r)
991        }
992        Expr::Or(lhs, rhs) => {
993            let l = eval_set_return_expr(db, match_rs, row, rel_vars, lhs, params, depth + 1)?;
994            let r = eval_set_return_expr(db, match_rs, row, rel_vars, rhs, params, depth + 1)?;
995            Ok(l || r)
996        }
997        Expr::Not(inner) => Ok(!eval_set_return_expr(
998            db,
999            match_rs,
1000            row,
1001            rel_vars,
1002            inner,
1003            params,
1004            depth + 1,
1005        )?),
1006        Expr::Cmp { lhs, op, rhs } => {
1007            let l = eval_set_return_operand(db, match_rs, row, rel_vars, lhs, params)?;
1008            let r = eval_set_return_operand(db, match_rs, row, rel_vars, rhs, params)?;
1009            match (l, r) {
1010                (Some(a), Some(b)) => Ok(eval_cmp(op, &a, &b)),
1011                _ => Ok(false),
1012            }
1013        }
1014        Expr::Truthy(op) => {
1015            let val = eval_set_return_operand(db, match_rs, row, rel_vars, op, params)?;
1016            Ok(match val {
1017                None => false,
1018                Some(Value::Bool(b)) => b,
1019                Some(Value::Int(n)) => n != 0,
1020                Some(Value::Float(f)) => f != 0.0,
1021                Some(Value::Str(s)) => !s.is_empty(),
1022                Some(Value::List(v)) => !v.is_empty(),
1023                Some(Value::Map(m)) => !m.is_empty(),
1024            })
1025        }
1026        Expr::IsNull(op) => {
1027            let val = eval_set_return_operand(db, match_rs, row, rel_vars, op, params)?;
1028            Ok(val.is_none())
1029        }
1030        Expr::IsNotNull(op) => {
1031            let val = eval_set_return_operand(db, match_rs, row, rel_vars, op, params)?;
1032            Ok(val.is_some())
1033        }
1034        Expr::In { expr, list } => {
1035            let Some(needle) = eval_set_return_operand(db, match_rs, row, rel_vars, expr, params)?
1036            else {
1037                return Ok(false);
1038            };
1039            for item_op in list {
1040                match eval_set_return_operand(db, match_rs, row, rel_vars, item_op, params)? {
1041                    None => {}
1042                    Some(Value::List(items)) => {
1043                        for item in items {
1044                            if eval_cmp(&core_query::CmpOp::Eq, &needle, &item) {
1045                                return Ok(true);
1046                            }
1047                        }
1048                    }
1049                    Some(item) if eval_cmp(&core_query::CmpOp::Eq, &needle, &item) => {
1050                        return Ok(true);
1051                    }
1052                    Some(_) => {}
1053                }
1054            }
1055            Ok(false)
1056        }
1057    }
1058}
1059
1060fn eval_set_return_arith(
1061    op: &ArithOp,
1062    lv: Option<Value>,
1063    rv: Option<Value>,
1064) -> Result<Option<Value>> {
1065    match (lv, rv) {
1066        (None, _) | (_, None) => Ok(None),
1067        (Some(Value::Int(a)), Some(Value::Int(b))) => {
1068            let result = match op {
1069                ArithOp::Sub => a.saturating_sub(b),
1070                ArithOp::Mul => a.saturating_mul(b),
1071                ArithOp::Add => a.saturating_add(b),
1072                ArithOp::Div => {
1073                    if b == 0 {
1074                        return Err(GraphError::QueryError {
1075                            detail: "division by zero".into(),
1076                        });
1077                    }
1078                    a.checked_div(b).unwrap_or(i64::MAX)
1079                }
1080            };
1081            Ok(Some(Value::Int(result)))
1082        }
1083        (Some(lv), Some(rv)) => {
1084            let a = match &lv {
1085                Value::Float(f) => *f,
1086                Value::Int(i) => *i as f64,
1087                _ => {
1088                    return Err(GraphError::QueryError {
1089                        detail: format!("arithmetic operand must be numeric, got {lv:?}"),
1090                    })
1091                }
1092            };
1093            let b = match &rv {
1094                Value::Float(f) => *f,
1095                Value::Int(i) => *i as f64,
1096                _ => {
1097                    return Err(GraphError::QueryError {
1098                        detail: format!("arithmetic operand must be numeric, got {rv:?}"),
1099                    })
1100                }
1101            };
1102            let result = match op {
1103                ArithOp::Sub => a - b,
1104                ArithOp::Mul => a * b,
1105                ArithOp::Add => a + b,
1106                ArithOp::Div => {
1107                    if b == 0.0 {
1108                        return Err(GraphError::QueryError {
1109                            detail: "division by zero".into(),
1110                        });
1111                    }
1112                    a / b
1113                }
1114            };
1115            Ok(Some(Value::Float(result)))
1116        }
1117    }
1118}
1119
1120fn eval_set_return_func<F: Fs>(
1121    db: &GraphDb<F>,
1122    match_rs: &ResultSet,
1123    row: usize,
1124    rel_vars: &[String],
1125    name: &str,
1126    args: &[Operand],
1127    params: &BTreeMap<String, Value>,
1128) -> Result<Option<Value>> {
1129    let norm = name.to_ascii_lowercase();
1130    if norm == "type" {
1131        if args.len() != 1 {
1132            return Err(GraphError::QueryError {
1133                detail: format!("type() requires exactly 1 argument, got {}", args.len()),
1134            });
1135        }
1136        let Operand::Var(rel) = &args[0] else {
1137            return Err(GraphError::QueryError {
1138                detail: "type() argument must be a relationship variable (e.g. type(r))".into(),
1139            });
1140        };
1141        return Ok(match_rs.get(row, &rel_type_alias(rel)).cloned());
1142    }
1143    if norm == "key" || norm == "id" {
1144        let fname = if norm == "id" { "id" } else { "key" };
1145        if args.len() != 1 {
1146            return Err(GraphError::QueryError {
1147                detail: format!("{fname}() requires exactly 1 argument, got {}", args.len()),
1148            });
1149        }
1150        let Operand::Var(var) = &args[0] else {
1151            return Err(GraphError::QueryError {
1152                detail: format!("{fname}() argument must be a node variable (e.g. {fname}(n))"),
1153            });
1154        };
1155        if rel_vars.iter().any(|r| r == var) {
1156            return Err(GraphError::QueryError {
1157                detail: format!("{fname}() argument `{var}` is a relationship, not a node"),
1158            });
1159        }
1160        // MATCH rows bind node variables to their key string, so the column
1161        // value *is* the key. `id()` aliases `key()`.
1162        return Ok(match_rs.get(row, var).cloned());
1163    }
1164    let mut vals = Vec::with_capacity(args.len());
1165    for arg in args {
1166        vals.push(eval_set_return_operand(
1167            db, match_rs, row, rel_vars, arg, params,
1168        )?);
1169    }
1170    match norm.as_str() {
1171        "tolower" => {
1172            if vals.len() != 1 {
1173                return Err(GraphError::QueryError {
1174                    detail: format!("toLower() requires exactly 1 argument, got {}", vals.len()),
1175                });
1176            }
1177            Ok(vals[0].clone().map(|val| match val {
1178                Value::Str(s) => Value::Str(s.to_ascii_lowercase()),
1179                other => other,
1180            }))
1181        }
1182        "toupper" => {
1183            if vals.len() != 1 {
1184                return Err(GraphError::QueryError {
1185                    detail: format!("toUpper() requires exactly 1 argument, got {}", vals.len()),
1186                });
1187            }
1188            Ok(vals[0].clone().map(|val| match val {
1189                Value::Str(s) => Value::Str(s.to_ascii_uppercase()),
1190                other => other,
1191            }))
1192        }
1193        "size" => match vals.first().cloned().flatten() {
1194            None => Ok(None),
1195            Some(Value::Str(s)) => Ok(Some(Value::Int(s.len() as i64))),
1196            Some(Value::List(items)) => Ok(Some(Value::Int(items.len() as i64))),
1197            Some(_) => Ok(None),
1198        },
1199        "coalesce" => Ok(vals.into_iter().flatten().next()),
1200        "abs" => match vals.first().cloned().flatten() {
1201            None => Ok(None),
1202            Some(Value::Int(n)) => Ok(Some(Value::Int(n.saturating_abs()))),
1203            Some(Value::Float(f)) => Ok(Some(Value::Float(f.abs()))),
1204            Some(_) => Ok(None),
1205        },
1206        "round" => match vals.first().cloned().flatten() {
1207            None => Ok(None),
1208            Some(Value::Float(f)) => Ok(Some(Value::Float(f.round()))),
1209            Some(Value::Int(n)) => Ok(Some(Value::Int(n))),
1210            Some(_) => Ok(None),
1211        },
1212        "decay" => {
1213            if vals.len() != 3 {
1214                return Err(GraphError::QueryError {
1215                    detail: format!("decay() requires exactly 3 arguments, got {}", vals.len()),
1216                });
1217            }
1218            match (vals[0].clone(), vals[1].clone(), vals[2].clone()) {
1219                (None, _, _) | (_, None, _) | (_, _, None) => Ok(None),
1220                (Some(b), Some(a), Some(h)) => {
1221                    let numeric = |v: Value| -> Result<f64> {
1222                        match v {
1223                            Value::Int(n) => Ok(n as f64),
1224                            Value::Float(f) => Ok(f),
1225                            other => Err(GraphError::QueryError {
1226                                detail: format!(
1227                                    "decay() requires numeric arguments, got {other:?}"
1228                                ),
1229                            }),
1230                        }
1231                    };
1232                    let b = numeric(b)?;
1233                    let a = numeric(a)?;
1234                    let h = numeric(h)?;
1235                    if h <= 0.0 {
1236                        return Err(GraphError::QueryError {
1237                            detail: "decay() requires halflife > 0".into(),
1238                        });
1239                    }
1240                    Ok(Some(Value::Float(b * 0.5f64.powf(a / h))))
1241                }
1242            }
1243        }
1244        _ => Err(GraphError::QueryError {
1245            detail: format!(
1246                "unknown function `{name}`; supported: toLower, toUpper, size, coalesce, type, abs, round, decay, key, id"
1247            ),
1248        }),
1249    }
1250}
1251
1252fn eval_set_return_item<F: Fs>(
1253    db: &GraphDb<F>,
1254    match_rs: &ResultSet,
1255    row: usize,
1256    rel_vars: &[String],
1257    item: &RetItem,
1258    params: &BTreeMap<String, Value>,
1259) -> Result<Option<Value>> {
1260    match &item.value {
1261        RetVal::Var(v) => eval_set_return_operand(
1262            db,
1263            match_rs,
1264            row,
1265            rel_vars,
1266            &Operand::Var(v.clone()),
1267            params,
1268        ),
1269        RetVal::Prop { var, field } => eval_set_return_operand(
1270            db,
1271            match_rs,
1272            row,
1273            rel_vars,
1274            &Operand::Prop {
1275                var: var.clone(),
1276                field: field.clone(),
1277            },
1278            params,
1279        ),
1280        RetVal::FuncCall { name, args } => {
1281            eval_set_return_func(db, match_rs, row, rel_vars, name, args, params)
1282        }
1283        RetVal::ScalarExpr(op) => eval_set_return_operand(db, match_rs, row, rel_vars, op, params),
1284        RetVal::Agg { .. } => Err(GraphError::QueryError {
1285            detail: "aggregates are not supported in MATCH … SET … RETURN".into(),
1286        }),
1287    }
1288}
1289
1290/// Project user RETURN from original MATCH rows after SET. No rematch.
1291fn project_set_return_rows<F: Fs>(
1292    db: &GraphDb<F>,
1293    rel_vars: &[String],
1294    match_rs: &ResultSet,
1295    returns: &[RetItem],
1296    params: &BTreeMap<String, Value>,
1297) -> Result<ResultSet> {
1298    let columns: Vec<String> = returns.iter().map(ret_column_name).collect();
1299    let mut out = ResultSet::new(columns);
1300    for row in 0..match_rs.len() {
1301        let mut cells = Vec::with_capacity(returns.len());
1302        for item in returns {
1303            cells.push(eval_set_return_item(
1304                db, match_rs, row, rel_vars, item, params,
1305            )?);
1306        }
1307        out.push_row(cells);
1308    }
1309    Ok(out)
1310}
1311
1312/// Single construction point for a `GraphMut` view over the split-borrowed graph fields.
1313/// Callers use `std::mem::take` on the engine before calling this, then restore it after.
1314/// Extract a `Vec<f64>` from a `Value::List` whose items are all numeric.
1315/// Returns `None` for non-list values or lists with non-numeric elements.
1316/// Extra candidates pulled from an approximate index before re-scoring, over and
1317/// above the `k` asked for.
1318///
1319/// The index orders candidates by `f32` distances, which agree with the exact
1320/// `f64` cosine to about 1e-6. Re-scoring can therefore only reshuffle
1321/// candidates inside a band that narrow — it cannot move a hit past one that is
1322/// further away by more than 1e-6 — so the only way a true top-`k` member can be
1323/// lost is if the index ranked it just outside `k` on the `f32` order. Fetching
1324/// `k + 16` covers any such band up to 16 members wide, which at 1e-6 means 16
1325/// vectors within a millionth of each other in cosine: a duplicate cluster, and
1326/// then the members are interchangeable anyway. `min` is applied to the exact
1327/// score, never to the index's, so a hit sitting on the threshold is decided
1328/// exactly.
1329const VECTOR_RESCORE_MARGIN: usize = 16;
1330
1331/// Cosine similarity between an already-unit query and node `id`'s `field`
1332/// vector, read from the **`f64`** properties. `None` when the node has no
1333/// numeric-list vector there, or its norm is zero.
1334///
1335/// The single definition of the score this API reports. Both the brute-force
1336/// scan and the re-scoring step that follows an index lookup go through it, so
1337/// the two paths cannot disagree — which is the property
1338/// `index_and_brute_force_agree_on_scores` pins.
1339fn exact_vector_similarity(
1340    view: &GraphView<'_>,
1341    id: u32,
1342    field: &str,
1343    q_unit: &[f64],
1344) -> Option<f64> {
1345    let v = view.prop(id, field)?;
1346    let xs = value_as_float_list(&v.into_value())?;
1347    let v_norm: f64 = xs.iter().map(|x| x * x).sum::<f64>().sqrt();
1348    if v_norm == 0.0 {
1349        return None;
1350    }
1351    Some(
1352        q_unit
1353            .iter()
1354            .zip(xs.iter())
1355            .map(|(a, b)| a * (b / v_norm))
1356            .sum(),
1357    )
1358}
1359
1360fn value_as_float_list(v: &Value) -> Option<Vec<f64>> {
1361    match v {
1362        Value::List(items) => items
1363            .iter()
1364            .map(|item| match item {
1365                Value::Float(f) => Some(*f),
1366                Value::Int(i) => Some(*i as f64),
1367                _ => None,
1368            })
1369            .collect(),
1370        _ => None,
1371    }
1372}
1373
1374fn make_graph_mut<'a>(
1375    ids: &'a IdMap,
1376    syms: &'a mut Interner,
1377    labels: &'a [u32],
1378    props: core_storage::v8::seam::ColumnsView<'a>,
1379    topo: &'a mut Topology,
1380    base: &'a Option<std::sync::Arc<core_storage::v8::MappedBase>>,
1381    edge_props: &'a mut EdgeProps,
1382) -> GraphMut<'a> {
1383    GraphMut {
1384        ids,
1385        syms,
1386        labels,
1387        props,
1388        topo,
1389        base_topo: base_csr(base),
1390        edge_props,
1391    }
1392}
1393
1394/// The archived CSR of an open V8 snapshot, for the rule engine's graph reads.
1395///
1396/// A store opened from a snapshot keeps its edges in the mapping and its
1397/// overlay empty, so a rule that reads the graph's shape has to see both.
1398fn base_csr(
1399    base: &Option<std::sync::Arc<core_storage::v8::MappedBase>>,
1400) -> Option<&core_storage::v8::layout::ArchivedCsr> {
1401    base.as_ref().map(|b| {
1402        b.topology()
1403            .expect("base topology section bounds validated at open")
1404    })
1405}
1406
1407/// Build a `ColumnsView` from the disjoint `props` overlay and optional V8 base.
1408///
1409/// Takes explicit field references rather than `&self` so the caller can hold
1410/// simultaneous mutable borrows of other fields (e.g. `syms`, `topo`).
1411fn build_props_view<'a>(
1412    props: &'a ColumnStore,
1413    base: &'a Option<std::sync::Arc<core_storage::v8::MappedBase>>,
1414) -> core_storage::v8::seam::ColumnsView<'a> {
1415    match base {
1416        None => core_storage::v8::seam::ColumnsView::owned(props),
1417        Some(b) => {
1418            let archived = b
1419                .columns()
1420                .expect("base columns section bounds validated at open");
1421            core_storage::v8::seam::ColumnsView::with_base_cached(props, archived, b.mixed_cache())
1422                .with_shared_strings(base_string_table(b))
1423        }
1424    }
1425}
1426
1427/// The base columns section paired with the string table that resolves its
1428/// string ids — what `ViewStore` needs to read a neighbour's string property
1429/// out of a V9 snapshot.
1430fn base_columns(
1431    base: &Option<std::sync::Arc<core_storage::v8::MappedBase>>,
1432) -> Option<core_storage::v8::seam::BaseColumns<'_>> {
1433    base.as_ref().map(|b| core_storage::v8::seam::BaseColumns {
1434        cols: b
1435            .columns()
1436            .expect("base columns section bounds validated at open"),
1437        strings: base_string_table(b),
1438    })
1439}
1440
1441/// The shared string table of a V9 base, or `None` for a pre-V9 one.
1442///
1443/// Every `ColumnsView` built over a base must carry it: without it a V9
1444/// snapshot's string columns, whose own tables are empty, read back as absent.
1445fn base_string_table(
1446    base: &core_storage::v8::MappedBase,
1447) -> Option<&core_storage::v8::layout::ArchivedStringTable> {
1448    base.string_table()
1449        .transpose()
1450        .expect("base strings section bounds validated at open")
1451}
1452
1453fn build_topo_view<'a>(
1454    overlay: &'a Topology,
1455    base: &'a Option<std::sync::Arc<core_storage::v8::MappedBase>>,
1456) -> core_storage::v8::seam::TopologyView<'a> {
1457    match base {
1458        None => core_storage::v8::seam::TopologyView::owned(overlay),
1459        Some(b) => {
1460            let archived_csr = b
1461                .topology()
1462                .expect("base topology section bounds validated at open");
1463            core_storage::v8::seam::TopologyView::with_base(overlay, archived_csr)
1464        }
1465    }
1466}
1467
1468/// When [`GraphDb`] calls `Fs::sync` after a WAL append.
1469///
1470/// Default is [`Strict`](FsyncPolicy::Strict): every `log_then_apply_with`
1471/// fsyncs (single `insert_node` / `set_prop`). Ingest and `write_batch`
1472/// emit one `WalRecord::Batch` and fsync once at that frame (Batched).
1473/// [`Relaxed`](FsyncPolicy::Relaxed) skips WAL sync; [`GraphDb::snapshot`]
1474/// is still durable via `write_atomic`. Crash-recovery DST stays Strict.
1475#[derive(Clone, Copy, Debug, PartialEq, Eq, Default)]
1476pub enum FsyncPolicy {
1477    /// Every WAL commit calls `fs.sync` (today's behavior).
1478    #[default]
1479    Strict,
1480    /// Sync only at a `Batch` frame end. Single-op path stays Strict unless
1481    /// this policy is set on the database.
1482    Batched,
1483    /// Never call `fs.sync`. [`GraphDb::snapshot`] still syncs via `write_atomic`.
1484    Relaxed,
1485}
1486
1487/// A precondition for a compare-and-set batch write.
1488///
1489/// All preconditions in a [`GraphDb::write_batch_cas`] or
1490/// [`crate::SharedDb::submit_batch_cas`] call are checked atomically before
1491/// any operation in the batch is applied.  If any precondition fails, the
1492/// entire batch is rejected with [`GraphError::CasConflict`] and no WAL frame
1493/// is written.
1494///
1495/// # Touch definition
1496///
1497/// A node's last-change commit (`last_changed`) is updated when any of the
1498/// following state-changing WAL records touch it:
1499///
1500/// - `InsertNode` / `InsertNodeId` — the newly-inserted node.
1501/// - `SetProp` / `SetPropId` / `RemoveProp` — the property-bearing node.
1502/// - `InsertEdge` / `InsertEdgeId` / `DeleteEdge` — **both** src and dst
1503///   endpoints (an edge change touches both sides).
1504/// - `DeleteNode` — the node is tombstoned; `last_changed` returns `None`
1505///   for deleted keys so the pre-deletion entry is never observed.
1506///
1507/// History markers (`DerivedEdgeAdded` / `DerivedEdgeRetracted`) are
1508/// state no-ops.  The underlying mutation that triggered rule firing already
1509/// updated the relevant nodes' last-change entries.  Rule-management records
1510/// (`CreateRule`, `DeleteRule`, `RebuildRule`) and view/full-text declarations
1511/// do not touch any node's last-change.
1512#[derive(Debug, Clone, PartialEq, Eq)]
1513pub enum Precondition {
1514    /// The node's last-change commit must equal `expected`.
1515    ///
1516    /// Fails with [`GraphError::CasConflict`] when:
1517    /// - The node does not exist (`last_changed` returns `None`), or
1518    /// - The recorded commit seq does not match `expected`.
1519    NodeUnchangedSince { key: String, expected: u64 },
1520    /// The node must not exist (not inserted, or already deleted).
1521    ///
1522    /// Fails with [`GraphError::CasConflict`] (expected=`u64::MAX`,
1523    /// actual=`last_changed(key).unwrap_or(0)`) when the node is live.
1524    NodeAbsent { key: String },
1525}
1526
1527pub struct GraphDb<F: Fs> {
1528    fs: F,
1529    ids: Arc<IdMap>,
1530    syms: Arc<Interner>,
1531    topo: Arc<Topology>,
1532    props: Arc<ColumnStore>,
1533    labels: Arc<Vec<u32>>, // node id -> label symbol
1534    /// Namespace names by index; index [`NS_DEFAULT_IDX`] is always
1535    /// [`NS_DEFAULT`]. Derived beside [`Self::node_ns`], never persisted.
1536    ///
1537    /// A private table rather than the shared [`Interner`]: interning
1538    /// `"default"` at open would add a symbol to the store's symbol table and
1539    /// change the bytes of the next snapshot of a store that has no namespaces
1540    /// at all.
1541    ns_names: Vec<String>,
1542    /// Namespace index per dense node id, into [`Self::ns_names`];
1543    /// [`NS_DEFAULT_IDX`] for a node with no `ns` property.
1544    ///
1545    /// Derived: built by one pass over the `ns` column at open (which reads
1546    /// nothing when the column does not exist) and maintained at every node
1547    /// insert. Never written to a snapshot or the WAL, because the property it
1548    /// mirrors already is. A namespace cannot change, so no other record shape
1549    /// can move a node between namespaces.
1550    node_ns: Vec<u32>,
1551    edge_props: Arc<EdgeProps>,
1552    engine: RuleEngine,
1553    view_store: ViewStore,
1554    /// Incremental inverted index for full-text-lite search.
1555    /// Rebuild-on-open: populated from WAL replay + rebuild_all at open end.
1556    fulltext: Arc<FulltextIndex>,
1557    /// Opt-in equality index over scalar node properties.
1558    /// Rebuild-on-open: declarations replay from the WAL, postings rebuild at
1559    /// open end (mirrors `fulltext`).
1560    prop_index: PropertyIndex,
1561    /// Whether this store records insert-count multiplicity (§5.13).
1562    ///
1563    /// Declared like `prop_index`'s enabled pairs — a WAL record replayed at
1564    /// open, re-emitted into the baseline by a truncating snapshot — but it
1565    /// gates a *format* step rather than an index: `WalRecord::SetEdgeCount`
1566    /// (discriminant 23) is written only when this is `true`, so a store that
1567    /// never opts in stays readable by a binary that predates the record.
1568    multiplicity: bool,
1569    event_sink: Option<Box<dyn Fn(MutationEvent) + Send + Sync>>,
1570    /// WAL fsync cadence. Default [`FsyncPolicy::Strict`].
1571    fsync: FsyncPolicy,
1572    /// Monotonically increasing per-commit counter.  A single `log_then_apply_with`
1573    /// call increments this once; all events emitted from that call share the same
1574    /// `commit_seq` value.
1575    commit_seq: u64,
1576    /// Commit → wall-clock map, loaded from the `commit_times.bin` sidecar at
1577    /// open and appended to by `log_then_apply_with` — the one place a commit
1578    /// is born. Replay does **not** stamp: `apply_frames` re-applies commits
1579    /// that already happened, and `SystemTime::now()` there would record replay
1580    /// time as commit time. Empty on a store written before v0.6.11, which
1581    /// makes every date query answer `NoRecordedTime` rather than guess.
1582    commit_times: core_storage::commit_times::CommitTimes,
1583    /// When set, subsequent commits are recorded at this instant instead of the
1584    /// system clock.
1585    ///
1586    /// Sticky on purpose. A backfill replays history that happened over months,
1587    /// and a day's worth of rows genuinely share one instant — a one-shot flag
1588    /// would mean setting it before every row of a bulk load, and forgetting one
1589    /// would stamp that row "now" in the middle of 2026-06. Sticky makes the
1590    /// failure visible instead: forget to move it and every commit carries the
1591    /// same timestamp, which a date query answers oddly and an inspection shows
1592    /// at once.
1593    commit_time_override: Option<i64>,
1594    /// `true` only while the open path is replaying, where `load_from_disk`
1595    /// calls `fulltext.rebuild_all` unconditionally afterwards.
1596    ///
1597    /// Replaying an `EnableFulltext` record backfills its pair with a full
1598    /// `0..ids.len()` scan, and every snapshot re-emits one such record per
1599    /// enabled pair — so on a snapshotted store the open does that scan once per
1600    /// pair and then `rebuild_all` clears every posting and does it all again.
1601    /// The backfill is pure waste *when a rebuild follows*, which is true of the
1602    /// open path and **false** of `refresh()`: refresh applies peer frames and
1603    /// then only folds, so its backfill is the only thing that indexes them.
1604    fulltext_rebuild_follows: bool,
1605    /// `true` when `commit_times.bin` was present but would not decode.
1606    ///
1607    /// Mirrors `roles: None`: a damaged map must not read as "this store
1608    /// records no times", because that is also what an honest pre-v0.6.11 store
1609    /// says. Date queries answer `Corrupt` instead, and nothing is appended to
1610    /// a file already known to be damaged.
1611    commit_times_poisoned: bool,
1612    /// RBAC role definitions loaded from `roles.json` at open.
1613    ///
1614    /// `Some(roles)` — loaded successfully (may be empty when no roles are defined).
1615    /// `None` — `roles.json` was present but corrupt; `mask_for_role` returns
1616    /// `Err` for any request (fail-loud, never silently grant empty visibility).
1617    roles: Option<Vec<RoleDef>>,
1618    /// Memo for [`mask_for_role`](GraphDb::mask_for_role), keyed by
1619    /// `(role, commit_seq)` — a scoped reader between two writes resolves once.
1620    ///
1621    /// Shared by `Arc` with every [`ReaderSnapshot`](crate::reader::ReaderSnapshot)
1622    /// taken from this handle. Replaced (not cleared) whenever the role
1623    /// definitions change or the store is reloaded, which `commit_seq` does not
1624    /// record; see [`RoleMaskCache`](crate::mask::RoleMaskCache).
1625    role_masks: Arc<crate::mask::RoleMaskCache>,
1626    /// Which loaded store this handle is, for memos that outlive it.
1627    ///
1628    /// `role_masks` needs no such thing — the handle owns it and replaces it —
1629    /// but a [`Scope`](crate::mask::Scope) is the caller's, so its resolved key
1630    /// leg is stamped with this alongside `commit_seq`. Minted fresh here and
1631    /// again in [`reset_for_reload`](GraphDb::reset_for_reload), at exactly the
1632    /// two points a fresh `RoleMaskCache` is installed; see
1633    /// [`StoreStamp`](crate::mask::StoreStamp) for the invariant.
1634    store_id: crate::mask::StoreId,
1635    /// Live subscriptions.  Entries with a dead `Weak` are pruned on the next
1636    /// distribute_events call.
1637    subscriptions: Vec<SubEntry>,
1638    /// Live query subscriptions. Re-executed on every commit when non-empty.
1639    /// Dead `Weak` entries are pruned inside `distribute_events`.
1640    query_subscriptions: Vec<QuerySubEntry>,
1641    /// Queue capacity for new subscriptions created by this db.  Default is
1642    /// [`DEFAULT_SUB_CAPACITY`]; can be overridden via [`set_sub_capacity`]
1643    /// to test Lagged behaviour with small queues.
1644    sub_capacity: usize,
1645    /// True for as-of instances opened via [`GraphDb::open_at`].
1646    /// Every mutation method and `snapshot()` returns [`GraphError::ReadOnly`]
1647    /// when this flag is set.
1648    read_only: bool,
1649    /// Total WAL commit count at the time [`open_at`] was called.
1650    /// 0 for normal (non-as-of) instances.
1651    total_wal_commits: u64,
1652    /// Immutable mmap-backed base snapshot (V8).  When `Some`, `self.topo` is
1653    /// the WAL-replay overlay (empty at open time, populated by apply()) and
1654    /// reads go through a merged `TopologyView`.  `self.props` is always
1655    /// fully materialized (base + WAL replay) for HNSW/IVF and view compat.
1656    base: Option<Arc<core_storage::v8::MappedBase>>,
1657    // ── MVCC epoch reader state ───────────────────────────────────────────────
1658    /// Most-recent full overlay clone.  Initialized at end of `open_with` /
1659    /// `open_at_with`; refreshed every `FOLD_EVERY_K` commits.
1660    /// `None` only between struct creation and the first fold.
1661    fold_overlay: Option<Arc<crate::reader::FrozenOverlay>>,
1662    /// Per-commit deltas accumulated since the last fold.
1663    delta_tail: Vec<Arc<crate::reader::CommitDelta>>,
1664    /// How many commits have occurred since the last fold.
1665    commits_since_fold: usize,
1666    /// When true, `log_then_apply_with` buffers event notifications instead of
1667    /// firing them immediately.  Used by the group-commit drain thread to defer
1668    /// events until after the group fsync (R2: durability before notification).
1669    /// Cleared to false once the drain thread flushes or discards the buffer.
1670    defer_events: bool,
1671    /// Buffered events accumulated while `defer_events` is true.
1672    deferred_events: Vec<DeferredEvent>,
1673    /// Set to true by the group-commit drain thread when a group fsync fails
1674    /// after WAL truncation.  All subsequent mutation attempts return an IO
1675    /// error until the database is reopened.
1676    degraded: bool,
1677    /// Set to `true` after `ensure_v8_base_sections_loaded` has read provenance,
1678    /// HNSW, and IVF sections from the mmap base into the engine's retained
1679    /// fields.  `false` on all opens until first use; always `true` for non-V8
1680    /// opens (base is None, fast-path sets flag immediately).
1681    v8_sections_loaded: std::sync::atomic::AtomicBool,
1682    /// Serializes the one-time section population in `ensure_v8_base_sections_loaded`.
1683    v8_sections_mutex: std::sync::Mutex<()>,
1684    /// Per-node last-change commit sequence.  `last_change[node_id] = seq` means
1685    /// the node was last modified by commit `seq`.
1686    ///
1687    /// Loaded from V8 section 11 at open; updated on every state-changing commit
1688    /// and WAL replay frame.  V5-V7 stores start with an empty map; pre-WAL-horizon
1689    /// nodes return `None` from `last_changed` until they are next mutated.
1690    ///
1691    /// See [`Precondition`] for the full touch definition.
1692    last_change: HashMap<u32, u64>,
1693    /// WAL archive retention policy set by [`set_wal_archive_retention`].
1694    /// `None` = unlimited (keep all archives); `Some(N)` = keep N newest archives,
1695    /// pruning older ones at snapshot time.  0 is treated as unlimited.
1696    wal_archive_retention: Option<u32>,
1697    /// Global frame index of the first commit that is still reachable through
1698    /// surviving archives.  Persisted to `wal.floor` sidecar when pruning occurs.
1699    /// Default 0 = all history reachable.
1700    wal_horizon_floor: u64,
1701    /// True when the surviving archive chain forms a continuous WAL history
1702    /// starting from the store's first commit (the genesis chain).
1703    ///
1704    /// `open_at` may replay archive-resident commits from empty state only when
1705    /// this flag is true AND `wal_horizon_floor == 0`.  Cleared whenever:
1706    ///   - a WAL-truncating snapshot (`keep_wal=false`) is taken after archives
1707    ///     already exist (breaks the chain for subsequent archives), or
1708    ///   - any archive is pruned (floor advances past zero).
1709    ///
1710    /// Persisted via the `wal.genesis` marker file; loaded from it at open.
1711    archive_genesis_chain: bool,
1712    /// True when this handle can *prove* the live WAL has never been truncated:
1713    /// there was no `snapshot.bin` when it opened the store, and it has taken no
1714    /// truncating snapshot since.
1715    ///
1716    /// The archive path's genesis check asks "did a snapshot exist before this
1717    /// one?" as a proxy for "was the WAL ever truncated". The proxy is sound
1718    /// across sessions — this binary cannot tell a history-preserving snapshot
1719    /// from a truncating one once the handle that took it is gone — but inside
1720    /// one session it is not, and `enable_multiplicity` made that visible: its
1721    /// forced `keep_wal` snapshot left the WAL entirely intact and yet
1722    /// permanently disqualified the store from ever receiving a genesis marker
1723    /// (defect #23). This flag is what the proxy defers to when the answer is
1724    /// actually known.
1725    snapshot_preserved_history: bool,
1726    /// Transient write-authz context set by `write_batch_authz` /
1727    /// `query_write_authz` for the duration of ONE mutation call.
1728    /// Always `None` at rest.  Never serialized, never WAL-replayed.
1729    pending_write_authz: Option<WriteAuthz>,
1730    /// Slow-query threshold in milliseconds.  0 = disabled.
1731    /// Seeded from `MUSHROOMDB_SLOW_QUERY_MS` at open; override via
1732    /// [`GraphDb::set_slow_query_threshold_ms`] (tests must use the setter
1733    /// — env vars are process-global and race parallel test threads).
1734    slow_query_threshold_ms: u64,
1735    /// Ring buffer of recent slow queries (interior-mutable so `query(&self)`
1736    /// can record entries without requiring `&mut self`).
1737    slow_queries: std::sync::Mutex<SlowQueryLog>,
1738    /// `(field, label, caller)` triples whose exact-versus-approximate
1739    /// ambiguity this handle has already explained once. See
1740    /// [`note_ambiguous_exactness`](GraphDb::note_ambiguous_exactness).
1741    /// The caller shape is part of the key because the two shapes give
1742    /// different advice — silencing one with the other would leave a caller
1743    /// reading advice meant for a signature it does not have.
1744    /// Advice bookkeeping, not graph state: a reload keeps it, as the
1745    /// slow-query log does.
1746    warned_ambiguous_exactness: std::sync::Mutex<HashSet<(String, String, ExactnessCaller)>>,
1747    /// Instant at which the database was opened (used by `/metrics` uptime).
1748    started_at: std::time::Instant,
1749    // ── Multi-process state (cross-process lock + WAL tailing) ────────────────
1750    /// Byte offset of the WAL prefix already applied to in-memory state.
1751    ///
1752    /// Advanced by exactly the encoded length of every frame this handle
1753    /// appends, and by the decoded byte count of every tail
1754    /// [`refresh`](GraphDb::refresh) absorbs. Rewound by
1755    /// [`set_wal_consumed`](GraphDb::set_wal_consumed) when the group-commit
1756    /// drain thread truncates a failed group. Compared against the WAL's
1757    /// on-disk length to decide staleness.
1758    wal_consumed: u64,
1759    /// The **global 0-based frame index the next appended WAL frame will
1760    /// occupy** — `wal_horizon_floor` plus every frame currently reachable
1761    /// through archives and the live WAL.
1762    ///
1763    /// This is the space every history surface addresses: `edges_at`,
1764    /// `was_linked`, both history readouts and `open_at` all index the sequence
1765    /// [`all_frames`](GraphDb::all_frames) returns, and
1766    /// [`wal_total_commits`](GraphDb::wal_total_commits) counts it.
1767    ///
1768    /// It exists because **a commit is not a frame**. `commit_seq` counts
1769    /// commits; a commit whose rules fire appends a *second* frame — the
1770    /// derived-edge history marker — that no counter of commits ever sees. The
1771    /// two diverge by one frame per rule-firing commit, cumulatively, so
1772    /// deriving a frame index from `commit_seq` under-reports by more and more
1773    /// as history grows and resolves every date to an earlier graph. Silently:
1774    /// an older graph is a plausible answer, not an error.
1775    ///
1776    /// Maintained in lockstep with [`wal_consumed`](GraphDb::wal_consumed) —
1777    /// the same appends advance both, one in frames and one in bytes — so the
1778    /// two are seeded and rewound at exactly the same places. Keep it that way.
1779    wal_frames_written: u64,
1780    /// Identity of the snapshot this handle's base state came from, as
1781    /// `(len, mtime_nanos)`. A different value means another process replaced
1782    /// the snapshot and the WAL no longer continues our state: refresh reloads.
1783    snapshot_ident: Option<(u64, u64)>,
1784    /// The options this handle was opened with. Replayed verbatim when
1785    /// `refresh` has to rebuild from disk.
1786    open_opts: OpenOptions,
1787    /// True when this handle holds the cross-process write lock for its whole
1788    /// lifetime (a plain read-write open). Per-write lock acquisition is a
1789    /// no-op on such a handle, and never releases the lock.
1790    holds_lifetime_lock: bool,
1791    /// True between a failed lock acquisition and the end of the write scope
1792    /// that failed. Makes every WAL-appending mutation in that scope return
1793    /// [`GraphError::Busy`] instead of writing.
1794    lock_denied: bool,
1795    /// True for an as-of view opened via [`GraphDb::open_at`]. Such a view is
1796    /// pinned to one commit, so it is never stale and never refreshes — later
1797    /// commits by any process are deliberately invisible to it.
1798    pinned: bool,
1799}
1800
1801/// One group of deferred event notifications, held until the group fsync
1802/// completes.  Replayed by [`GraphDb::flush_deferred_events`].
1803struct DeferredEvent {
1804    rec: core_storage::WalRecord,
1805    engine_deltas: Vec<EngineEdgeDelta>,
1806    seq: u64,
1807    ingest: Option<(String, usize)>,
1808}
1809
1810/// Options for [`GraphDb::open_with_options`].
1811#[derive(Clone, Copy, Debug)]
1812pub struct OpenOptions {
1813    /// Rewrite an old-format snapshot to the current VERSION after a
1814    /// successful load (default `true`). The old snapshot is kept as
1815    /// `snapshot.bin.bak` until the next clean open at the current version,
1816    /// at which point the `.bak` is deleted.
1817    ///
1818    /// Set to `false` to open a store without touching any on-disk files
1819    /// (useful for read-only inspection of a store at an older format).
1820    pub auto_migrate: bool,
1821
1822    /// Write the valid WAL prefix back over a torn tail on open (default
1823    /// `true`). Truncating a genuinely torn tail is correct crash recovery.
1824    ///
1825    /// Set to `false` for an unattended reader. The valid prefix is still
1826    /// decoded and replayed in memory, but nothing is written: a reader that
1827    /// opens while another process is mid-append would otherwise discard a
1828    /// frame that writer believes durable. `mushroomdb recall`, which runs on
1829    /// every prompt, passes `false` for exactly this reason.
1830    pub repair_wal: bool,
1831
1832    /// Open without ever writing to the store (default `false`).
1833    ///
1834    /// A read-only handle:
1835    /// - returns [`GraphError::ReadOnly`] from every mutation and from
1836    ///   `snapshot()`;
1837    /// - performs no disk write at open — no WAL repair write-back and no
1838    ///   auto-migration rewrite, whatever the other two flags say;
1839    /// - never takes the cross-process write lock, so it opens immediately even
1840    ///   while another process is writing, and never makes a writer wait.
1841    ///
1842    /// [`refresh`](GraphDb::refresh) and [`is_stale`](GraphDb::is_stale) work
1843    /// normally, so a read-only handle can follow another process's commits.
1844    pub read_only: bool,
1845}
1846
1847impl Default for OpenOptions {
1848    fn default() -> Self {
1849        Self {
1850            auto_migrate: true,
1851            repair_wal: true,
1852            read_only: false,
1853        }
1854    }
1855}
1856
1857/// How long a writer polls for the cross-process write lock before giving up
1858/// with [`GraphError::Busy`].
1859///
1860/// Long enough to ride out another process's commit (a batch apply plus one
1861/// fsync), short enough that a stuck peer surfaces as an error rather than a
1862/// hang.
1863pub const WRITE_LOCK_WAIT: std::time::Duration = std::time::Duration::from_secs(2);
1864
1865/// Refusal when a `MERGE` create cannot choose a namespace.
1866///
1867/// A role bound to two or more namespaces cannot have its create arm land in
1868/// `default`, and the statement did not name `ns`. The role must name one.
1869pub const MERGE_CREATE_NEEDS_ONE_NAMESPACE: &str =
1870    "role-bound token: MERGE create requires the role to name one namespace";
1871
1872/// Interval between poll attempts while waiting for the cross-process lock.
1873pub(crate) const LOCK_POLL_INTERVAL: std::time::Duration = std::time::Duration::from_millis(10);
1874
1875/// Why `load_from_disk` is running, which decides whether it may repair.
1876#[derive(Clone, Copy, Debug, PartialEq, Eq)]
1877enum LoadOrigin {
1878    /// A fresh open. Crash recovery is this handle's job: a torn WAL tail is
1879    /// the signature of a crash and truncating it is correct, and archives
1880    /// orphaned by an interrupted prune can be swept.
1881    Open,
1882    /// A reload driven by [`GraphDb::refresh`], because another process
1883    /// replaced the snapshot. Nothing here is crash recovery — the store is
1884    /// live and someone else is writing it — so this origin writes nothing.
1885    Reload,
1886}
1887
1888/// Authorization context carried by `write_batch_authz` / `query_write_authz`.
1889///
1890/// `None` at the call site = full authority (today's zero-cost behavior).
1891/// `Some(WriteAuthz)` = role-scoped: the decision table (plan §"authz decision
1892/// table") is evaluated per-op inside `commit_logged_batch` BEFORE any WAL
1893/// record is built.  A denial returns an error with no WAL frame written.
1894///
1895/// The mask is ALWAYS `Omit`-mode: role-token paths must never acknowledge
1896/// hidden-node existence to callers.
1897#[derive(Clone, Debug)]
1898pub struct WriteAuthz {
1899    pub role: String,
1900    pub scope: WriteScope,
1901    /// Resolved by `mask_for_role` under the same write guard as the mutation.
1902    /// Always `Omit`-mode — never `Stub`.
1903    pub mask: crate::mask::NodeMask,
1904}
1905
1906/// The error every role surface gives when `roles.json` did not parse at open.
1907///
1908/// One text, so `mask_for_role` and [`GraphDb::roles_checked`] cannot drift
1909/// apart on the same cause.
1910fn roles_poisoned() -> GraphError {
1911    GraphError::Corrupt {
1912        detail: "roles.json was corrupt at open; fix the file and re-open to restore role access"
1913            .into(),
1914    }
1915}
1916
1917/// Write `bytes` to `snapshot.bin.bak` atomically with full fsync.
1918///
1919/// Uses [`RealFs::write_atomic`] which applies `F_FULLFSYNC` on macOS and
1920/// `sync_all` on other platforms, then renames the `.tmp` file into place and
1921/// syncs the directory entry. This is the only correct path for writing the
1922/// `.bak` — plain `std::fs::write + sync_all` misses both `F_FULLFSYNC` and
1923/// the directory sync.
1924pub fn write_snapshot_bak(dir: &std::path::Path, bytes: &[u8]) -> crate::Result<()> {
1925    use core_storage::fs::{FileId, Fs as _};
1926    RealFs::new(dir)
1927        .map_err(core_storage::GraphError::Io)?
1928        .write_atomic(FileId::SnapshotBak, bytes)
1929        .map_err(core_storage::GraphError::Io)
1930}
1931
1932/// Return the on-disk snapshot format version without decoding the full snapshot.
1933///
1934/// Reads only the 6-byte header (magic + version LE). Returns `None` when no
1935/// snapshot file exists (WAL-only store). Returns an error if the header is
1936/// malformed.
1937pub fn snapshot_version_at(dir: &std::path::Path) -> crate::Result<Option<u16>> {
1938    use std::io::Read as _;
1939    let path = dir.join("snapshot.bin");
1940    let mut header = [0u8; 6];
1941    let n = match std::fs::File::open(&path) {
1942        Ok(mut f) => f.read(&mut header).map_err(core_storage::GraphError::Io)?,
1943        Err(e) if e.kind() == std::io::ErrorKind::NotFound => return Ok(None),
1944        Err(e) => return Err(core_storage::GraphError::Io(e)),
1945    };
1946    core_storage::snapshot::peek_version(&header[..n])
1947}
1948
1949/// Options for [`GraphDb::snapshot_with`].
1950#[derive(Debug, Clone, Default)]
1951pub struct SnapshotOptions {
1952    /// When `true`, the WAL is preserved after the snapshot write.
1953    /// Pre-snapshot commits remain reachable via [`GraphDb::open_at`].
1954    /// When `false` (the default), the WAL is truncated to a minimal
1955    /// baseline so cold-start replay stays fast.
1956    pub keep_wal: bool,
1957    /// When `true`, the current WAL is renamed to `wal.<commit_seq>.archive`
1958    /// before a fresh WAL baseline is written (history-preserving snapshot).
1959    ///
1960    /// This is the feature opt-in: `false` (the default) leaves the existing
1961    /// truncation / keep-wal behaviour byte-identical.  `archive_wal` takes
1962    /// precedence over `keep_wal` when both are set.
1963    ///
1964    /// Archives can be scanned by [`GraphDb::node_history`],
1965    /// [`GraphDb::edge_history`], [`GraphDb::was_linked`], and
1966    /// [`GraphDb::open_at`], extending the reachable history horizon across
1967    /// snapshot boundaries.
1968    pub archive_wal: bool,
1969}
1970
1971/// Derive the scan-label sym for the commit-skip fast-path.
1972///
1973/// Walks `ops` to find the plan's leading scan op (`ScanLabel`, `IndexScan`,
1974/// or `IndexIntersect`) with a concrete label string, then interns it.
1975///
1976/// Returns `None` in all cases where skipping is unsafe:
1977/// - Any `Expand` op is present (edge traversal; edges change results regardless
1978///   of node labels).
1979/// - The leading scan has no label (`ScanLabel { label: None }` — full scan).
1980/// - No recognizable leading scan op is found.
1981///
1982/// This is the conservative v0.4.3 boundary. The caller stores the result in
1983/// [`QuerySubEntry::scan_label`] at subscribe time; `None` means always execute.
1984fn extract_scan_label(ops: &[PlanOp], syms: &mut Interner) -> Option<u32> {
1985    // Any Expand → must always re-execute (edges can change join results).
1986    if ops.iter().any(|op| matches!(op, PlanOp::Expand { .. })) {
1987        return None;
1988    }
1989    for op in ops {
1990        match op {
1991            PlanOp::ScanLabel {
1992                label: Some(label), ..
1993            } => return Some(syms.intern(label)),
1994            PlanOp::IndexScan {
1995                label: Some(label), ..
1996            } => return Some(syms.intern(label)),
1997            PlanOp::IndexIntersect {
1998                label: Some(label), ..
1999            } => return Some(syms.intern(label)),
2000            _ => {}
2001        }
2002    }
2003    None
2004}
2005
2006/// How an as-of read is restricted — the argument to
2007/// [`GraphDb::query_at_scoped`].
2008///
2009/// Every variant is resolved against the graph **as it was at the requested
2010/// commit**, not against the current graph.
2011#[derive(Debug, Clone, Copy)]
2012pub enum AsOfScope<'a> {
2013    /// Everything the named role may see. The role *definition* is the current
2014    /// one — `roles.json` is a sidecar and has no past version — but its
2015    /// `keys` and `labels` are resolved against the as-of graph.
2016    Role(&'a str),
2017    /// An explicit node-key allow-list. Keys that did not exist at that commit
2018    /// resolve to nothing.
2019    Keys(&'a [String]),
2020    /// A role intersected with a client-supplied allow-list. The intersection
2021    /// is the never-widen rule: a client mask can only narrow a role.
2022    RoleAndKeys(&'a str, &'a [String]),
2023    /// Every live node in one namespace, as the graph was at that commit.
2024    ///
2025    /// A namespace cannot change — it is set at insert and immutable — so the
2026    /// answer is simply "the nodes that existed then and are in this
2027    /// namespace". A name no node uses resolves to nothing, never to
2028    /// everything.
2029    Namespace(&'a str),
2030}
2031
2032impl GraphDb<RealFs> {
2033    /// Open the database at `dir` with default options.
2034    ///
2035    /// Equivalent to `open_with_options(dir, OpenOptions::default())`.
2036    /// Old-format snapshots (V5, V6) are automatically migrated to the
2037    /// current version on a successful load (see [`OpenOptions::auto_migrate`]).
2038    pub fn open(dir: &std::path::Path) -> Result<Self> {
2039        Self::open_with_options(dir, OpenOptions::default())
2040    }
2041
2042    /// Open the database at `dir` with explicit options.
2043    ///
2044    /// When `opts.auto_migrate` is `true` (the default) and the on-disk
2045    /// snapshot is an older format version, this function:
2046    ///   1. Copies the current `snapshot.bin` to `snapshot.bin.bak` (atomic
2047    ///      + fsynced) before any modification.
2048    ///   2. Rewrites `snapshot.bin` at the current format version via
2049    ///      [`GraphDb::snapshot_with`] with `keep_wal: true` (WAL preserved).
2050    ///
2051    /// If migration fails the error is returned and the original files are
2052    /// intact (the `.bak` was written before the new snapshot was attempted).
2053    ///
2054    /// A clean open that finds the snapshot already at the current version
2055    /// deletes any leftover `.bak` file.
2056    ///
2057    /// WAL-only stores (no snapshot) are never auto-migrated on open.
2058    ///
2059    /// `opts.repair_wal` controls the other write this function can make; see
2060    /// [`OpenOptions::repair_wal`]. With both flags `false` the open touches
2061    /// no file on disk.
2062    pub fn open_with_options(dir: &std::path::Path, opts: OpenOptions) -> Result<Self> {
2063        Self::open_dir(dir, opts, true)
2064    }
2065
2066    /// Open without taking the cross-process write lock for the handle's
2067    /// lifetime.
2068    ///
2069    /// Only [`SharedDb`](crate::SharedDb) uses this: a long-lived server holds
2070    /// its handle open indefinitely, so it takes the lock per write instead of
2071    /// keeping every other process out of the store for as long as it runs.
2072    pub(crate) fn open_unlocked(dir: &std::path::Path) -> Result<Self> {
2073        Self::open_dir(dir, OpenOptions::default(), false)
2074    }
2075
2076    fn open_dir(dir: &std::path::Path, opts: OpenOptions, hold_lock: bool) -> Result<Self> {
2077        // Header-only peek — 6 bytes, no full decode.
2078        let snap_version = snapshot_version_at(dir)?;
2079
2080        // Full load: decode snapshot + replay WAL + rebuild indexes.
2081        let mut db = Self::open_generic(RealFs::new(dir)?, opts, hold_lock)?;
2082
2083        // A read-only handle writes nothing at open, so it never migrates —
2084        // the old-format snapshot is loaded and left exactly as it is.
2085        if opts.auto_migrate && !opts.read_only {
2086            match snap_version {
2087                Some(ver) if ver < core_storage::snapshot::VERSION => {
2088                    let _tm = std::time::Instant::now();
2089                    // Copy the original snapshot to .bak at OS level — no in-memory
2090                    // buffer required for a 2+ GiB file.
2091                    //
2092                    // Crash-safety: snapshot.bin remains intact (write_atomic inside
2093                    // snapshot_with uses a .tmp+rename) until the V8 write succeeds.
2094                    // A torn .bak on crash is acceptable because the original
2095                    // snapshot.bin is the authoritative source until after the rename.
2096                    std::fs::copy(dir.join("snapshot.bin"), dir.join("snapshot.bin.bak"))
2097                        .map_err(core_storage::GraphError::Io)?;
2098                    trace_migrate!("bak copy done", _tm);
2099                    // Rewrite snapshot at current version; keep WAL intact.
2100                    db.snapshot_with(SnapshotOptions {
2101                        keep_wal: true,
2102                        ..SnapshotOptions::default()
2103                    })?;
2104                    trace_migrate!("snapshot_with done", _tm);
2105                }
2106                Some(_) => {
2107                    // Already current version: remove any leftover .bak.
2108                    let bak = dir.join("snapshot.bin.bak");
2109                    if bak.exists() {
2110                        std::fs::remove_file(&bak).map_err(core_storage::GraphError::Io)?;
2111                    }
2112                }
2113                None => {
2114                    // WAL-only store — nothing to migrate on open.
2115                }
2116            }
2117        }
2118
2119        Ok(db)
2120    }
2121
2122    /// Open a read-only view of the database as it existed after `commit`.
2123    ///
2124    /// Commit indices are 0-based over the current WAL: commit 0 is the state
2125    /// after the first WAL frame, commit N-1 is the state after the N-th (most
2126    /// recent) frame.  Call [`GraphDb::open`] to read the full current state.
2127    ///
2128    /// **Replay base.** [`GraphDb::snapshot`] truncates the WAL when it runs,
2129    /// so as-of can only reach commits recorded in the current WAL (those
2130    /// written after the most recent snapshot, or all commits if no snapshot
2131    /// was ever taken).  Commit 0 in `open_at` always refers to the first
2132    /// frame in the WAL that exists on disk, not the first ever write to the
2133    /// database.  When the on-disk snapshot recorded that it truncated the
2134    /// WAL (V7, default `keep_wal: false`), it is loaded as the base state
2135    /// before frame replay, so the as-of view includes all pre-snapshot data.
2136    /// Snapshots written with `keep_wal: true` (and legacy V5/V6 snapshots)
2137    /// are ignored and replay is WAL-only, as before.
2138    ///
2139    /// **Read-only.** Every mutation method and `snapshot()` on the returned
2140    /// instance returns [`GraphError::ReadOnly`].  Queries, `explain()`, and
2141    /// `stats()` work normally.
2142    ///
2143    /// # Errors
2144    /// - [`GraphError::CommitOutOfRange`] if `commit >= wal_commit_count` (including
2145    ///   when the WAL is empty after a snapshot).
2146    pub fn open_at(dir: &std::path::Path, commit: u64) -> Result<Self> {
2147        Self::open_at_with(RealFs::new(dir)?, commit)
2148    }
2149
2150    /// Run a **read-only** Cypher query against the graph as it existed at
2151    /// `commit` — the "time-travel" / agent-replay query. Opens a temporal view
2152    /// of this store's directory at that commit and executes the read there.
2153    ///
2154    /// The current instance is unaffected. Write statements are rejected (the
2155    /// temporal view is read-only). `commit` is a 0-based WAL commit index;
2156    /// `commit == wal_commit_count` (or `open_at`'s range) yields the newest
2157    /// state. Prefer this over holding many historical instances open.
2158    ///
2159    /// # Errors
2160    /// - [`GraphError::CommitOutOfRange`] if `commit` is past the WAL horizon.
2161    /// - A query error for a malformed or write query.
2162    pub fn query_at(
2163        &self,
2164        commit: u64,
2165        cypher: &str,
2166        params: &std::collections::BTreeMap<String, Value>,
2167    ) -> Result<ResultSet> {
2168        let temporal = self.open_at_for_read(commit, cypher)?;
2169        temporal.query(cypher, params)
2170    }
2171
2172    /// Run a **read-only** Cypher query at `commit`, restricted by `scope`.
2173    ///
2174    /// The **graph** is as of `commit`; the **role definition** is as it is
2175    /// now, because `roles.json` is a sidecar and is never a WAL record — it
2176    /// has no past version to read. A role's `keys` and `labels` are resolved
2177    /// against the commit-`commit` graph, so a role that may see a label sees
2178    /// exactly the nodes that carried it then, and an explicit key that did
2179    /// not exist yet resolves to nothing.
2180    ///
2181    /// [`AsOfScope::RoleAndKeys`] intersects the two: a client allow-list can
2182    /// only narrow what a role may see, never widen it.
2183    ///
2184    /// Write statements are rejected, exactly as [`GraphDb::query_at`] rejects
2185    /// them.
2186    ///
2187    /// # Errors
2188    /// - [`GraphError::CommitOutOfRange`] if `commit` is outside the retained
2189    ///   range; the error carries that range.
2190    /// - [`GraphError::KeyNotFound`] with a `role:` prefix for an unknown role,
2191    ///   or [`GraphError::Corrupt`] when `roles.json` was corrupt at open.
2192    /// - A query error for a malformed or write query.
2193    pub fn query_at_scoped(
2194        &self,
2195        commit: u64,
2196        cypher: &str,
2197        params: &std::collections::BTreeMap<String, Value>,
2198        scope: AsOfScope<'_>,
2199    ) -> Result<ResultSet> {
2200        let temporal = self.open_at_for_read(commit, cypher)?;
2201        let mask = temporal.mask_at_scope(scope)?;
2202        temporal.query_masked(cypher, params, &mask)
2203    }
2204
2205    /// As [`GraphDb::query_at_scoped`], with `namespace` intersected into
2206    /// whatever `scope` resolves to.
2207    ///
2208    /// This is what a surface needs when a caller passes `namespace` beside a
2209    /// `role` or a client mask on a time-travel read: [`AsOfScope`] names one
2210    /// restriction, and the namespace is a second one that composes with it
2211    /// rather than replacing it. The intersection is the never-widen rule — a
2212    /// namespace can only narrow what the scope already allows — and both legs
2213    /// are resolved against the graph as it was at `commit`.
2214    ///
2215    /// `AsOfScope::Namespace(ns)` is still the way to ask for a namespace alone.
2216    pub fn query_at_scoped_in_namespace(
2217        &self,
2218        commit: u64,
2219        cypher: &str,
2220        params: &std::collections::BTreeMap<String, Value>,
2221        scope: AsOfScope<'_>,
2222        namespace: &str,
2223    ) -> Result<ResultSet> {
2224        let temporal = self.open_at_for_read(commit, cypher)?;
2225        let mask = temporal
2226            .mask_at_scope(scope)?
2227            .intersect(&temporal.mask_for_namespace(namespace));
2228        temporal.query_masked(cypher, params, &mask)
2229    }
2230
2231    /// Run a **read-only** Cypher query at `commit`, restricted by a
2232    /// [`Scope`](crate::mask::Scope).
2233    ///
2234    /// [`AsOfScope`] names *one* restriction — a role, a key list, a namespace,
2235    /// or a role-and-keys pair. A `Scope` is the general shape a handle carries,
2236    /// and nesting can give it several role or namespace legs at once, so it
2237    /// cannot be spelled as an `AsOfScope`. This is the entry point a scoped
2238    /// handle uses for time travel; `query_at_scoped` stays the way to ask for
2239    /// one named restriction.
2240    ///
2241    /// Both the graph and the scope's key and namespace legs are resolved
2242    /// against `commit`; a role's *definition* is the current one, because
2243    /// `roles.json` is a sidecar with no past version — the same split
2244    /// [`GraphDb::query_at_scoped`] documents.
2245    ///
2246    /// The scope resolves **cold** here: a temporal handle is its own store, so
2247    /// its ids could never be served to a live read, but filling the scope's
2248    /// one-entry key memo from a handle thrown away at the end of this call
2249    /// would evict the live entry for nothing. See
2250    /// [`Scope::resolve_uncached`](crate::mask::Scope::resolve_uncached).
2251    ///
2252    /// # Errors
2253    /// - [`GraphError::CommitOutOfRange`] if `commit` is outside the retained
2254    ///   range; the error carries that range.
2255    /// - [`GraphError::KeyNotFound`] with a `role:` prefix for an unknown role,
2256    ///   or [`GraphError::Corrupt`] when `roles.json` was corrupt at open.
2257    /// - A query error for a malformed or write query.
2258    pub fn query_at_with_scope(
2259        &self,
2260        commit: u64,
2261        cypher: &str,
2262        params: &std::collections::BTreeMap<String, Value>,
2263        scope: &crate::mask::Scope,
2264    ) -> Result<ResultSet> {
2265        let temporal = self.open_at_for_read(commit, cypher)?;
2266        let mask = scope.resolve_uncached(&temporal)?;
2267        temporal.query_masked(cypher, params, &mask)
2268    }
2269
2270    /// Open the temporal view for a time-travel read and refuse write Cypher.
2271    ///
2272    /// Shared by [`GraphDb::query_at`] and [`GraphDb::query_at_scoped`] so both
2273    /// resolve the commit and reject writes identically.
2274    fn open_at_for_read(&self, commit: u64, cypher: &str) -> Result<Self> {
2275        let dir = self.fs.dir().to_path_buf();
2276        let temporal = Self::open_at(&dir, commit)?;
2277        if is_write_tokens(&lex(cypher).map_err(|e| GraphError::QueryError {
2278            detail: format!("lex: {e}"),
2279        })?) {
2280            return Err(GraphError::QueryError {
2281                detail: "query_at is read-only: write statements are not permitted in a \
2282                         time-travel query"
2283                    .into(),
2284            });
2285        }
2286        Ok(temporal)
2287    }
2288}
2289
2290impl<F: Fs> GraphDb<F> {
2291    /// Open over an arbitrary [`Fs`], repairing a torn WAL tail as usual.
2292    pub fn open_with(fs: F) -> Result<Self> {
2293        Self::open_with_repair(fs, true)
2294    }
2295
2296    /// As [`GraphDb::open_with`], but `repair_wal: false` decodes the valid WAL
2297    /// prefix without writing the truncation back. See
2298    /// [`OpenOptions::repair_wal`].
2299    pub fn open_with_repair(fs: F, repair_wal: bool) -> Result<Self> {
2300        Self::open_generic(
2301            fs,
2302            OpenOptions {
2303                repair_wal,
2304                ..OpenOptions::default()
2305            },
2306            true,
2307        )
2308    }
2309
2310    /// Shared open path.
2311    ///
2312    /// `hold_lock` requests the cross-process write lock for the whole handle
2313    /// lifetime — the right behaviour for a plain read-write `GraphDb`, whose
2314    /// owner writes through it directly. [`SharedDb`](crate::SharedDb) passes
2315    /// `false` and takes the lock per write instead, so that a long-lived
2316    /// server does not keep every other process out of the store.
2317    ///
2318    /// A read-only open never takes the lock regardless of `hold_lock`.
2319    fn open_generic(fs: F, opts: OpenOptions, hold_lock: bool) -> Result<Self> {
2320        let mut db = Self::new_empty(fs, opts);
2321        db.read_only = opts.read_only;
2322        if hold_lock && !opts.read_only {
2323            if !db.poll_lock(WRITE_LOCK_WAIT)? {
2324                return Err(GraphError::Busy { holder: None });
2325            }
2326            db.holds_lifetime_lock = true;
2327        }
2328        db.load_from_disk(LoadOrigin::Open)?;
2329        Ok(db)
2330    }
2331
2332    /// A handle with no state loaded: every field at its empty value, the
2333    /// filesystem and options in place. Only [`load_from_disk`] makes it
2334    /// usable.
2335    fn new_empty(fs: F, opts: OpenOptions) -> Self {
2336        Self {
2337            fs,
2338            ids: Arc::new(IdMap::new()),
2339            syms: Arc::new(Interner::new()),
2340            topo: Arc::new(Topology::new()),
2341            props: Arc::new(ColumnStore::new()),
2342            labels: Arc::new(Vec::new()),
2343            ns_names: vec![NS_DEFAULT.to_string()],
2344            node_ns: Vec::new(),
2345            edge_props: Arc::new(EdgeProps::new()),
2346            engine: RuleEngine::new(),
2347            view_store: ViewStore::new(),
2348            fulltext: Arc::new(FulltextIndex::new()),
2349            prop_index: PropertyIndex::new(),
2350            multiplicity: false,
2351            event_sink: None,
2352            fsync: FsyncPolicy::Strict,
2353            commit_seq: 0,
2354            commit_times: core_storage::commit_times::CommitTimes::default(),
2355            commit_times_poisoned: false,
2356            fulltext_rebuild_follows: false,
2357            commit_time_override: None,
2358            roles: Some(vec![]),
2359            role_masks: Arc::new(crate::mask::RoleMaskCache::new()),
2360            store_id: crate::mask::StoreId::next(),
2361            subscriptions: Vec::new(),
2362            query_subscriptions: Vec::new(),
2363            sub_capacity: DEFAULT_SUB_CAPACITY,
2364            read_only: false,
2365            total_wal_commits: 0,
2366            base: None,
2367            fold_overlay: None,
2368            delta_tail: Vec::new(),
2369            commits_since_fold: 0,
2370            defer_events: false,
2371            deferred_events: Vec::new(),
2372            degraded: false,
2373            v8_sections_loaded: std::sync::atomic::AtomicBool::new(false),
2374            v8_sections_mutex: std::sync::Mutex::new(()),
2375            last_change: HashMap::new(),
2376            wal_archive_retention: None,
2377            wal_horizon_floor: 0,
2378            archive_genesis_chain: false,
2379            // Nothing is proven until `load_from_disk` has looked at the store.
2380            snapshot_preserved_history: false,
2381            pending_write_authz: None,
2382            slow_query_threshold_ms: std::env::var("MUSHROOMDB_SLOW_QUERY_MS")
2383                .ok()
2384                .and_then(|v| v.parse().ok())
2385                .unwrap_or(100),
2386            slow_queries: std::sync::Mutex::new(SlowQueryLog {
2387                entries: std::collections::VecDeque::new(),
2388                total: 0,
2389            }),
2390            warned_ambiguous_exactness: std::sync::Mutex::new(HashSet::new()),
2391            started_at: std::time::Instant::now(),
2392            wal_consumed: 0,
2393            wal_frames_written: 0,
2394            snapshot_ident: None,
2395            open_opts: opts,
2396            holds_lifetime_lock: false,
2397            lock_denied: false,
2398            pinned: false,
2399        }
2400    }
2401
2402    /// Return every field describing stored graph state to its empty value,
2403    /// leaving this handle's own identity alone.
2404    ///
2405    /// Preserved on purpose: the filesystem, open options, lock ownership, the
2406    /// event sink and subscriptions, fsync policy, degraded flag, and the
2407    /// slow-query configuration and log. A caller that registered a sink or a
2408    /// subscription keeps it across a reload.
2409    fn reset_for_reload(&mut self) {
2410        self.ids = Arc::new(IdMap::new());
2411        self.syms = Arc::new(Interner::new());
2412        self.topo = Arc::new(Topology::new());
2413        self.props = Arc::new(ColumnStore::new());
2414        self.labels = Arc::new(Vec::new());
2415        self.ns_names = vec![NS_DEFAULT.to_string()];
2416        self.node_ns = Vec::new();
2417        self.edge_props = Arc::new(EdgeProps::new());
2418        self.engine = RuleEngine::new();
2419        self.view_store = ViewStore::new();
2420        self.fulltext = Arc::new(FulltextIndex::new());
2421        self.prop_index = PropertyIndex::new();
2422        // Cleared like every other declaration: a reload replays the store's own
2423        // WAL, and the opt-in comes back from it or not at all.
2424        self.multiplicity = false;
2425        self.commit_seq = 0;
2426        self.commit_times = core_storage::commit_times::CommitTimes::default();
2427        self.commit_times_poisoned = false;
2428        self.fulltext_rebuild_follows = false;
2429        self.commit_time_override = None;
2430        self.roles = Some(vec![]);
2431        // A fresh cache, not a cleared one: any reader snapshot still holding
2432        // the old `Arc` keeps it to itself, so nothing it memoised against the
2433        // pre-reload store can be read back through this handle.
2434        self.role_masks = Arc::new(crate::mask::RoleMaskCache::new());
2435        // The same move for memos this handle does not own. `commit_seq` is
2436        // zeroed just above and reseeded from `max(last_change)`, which a
2437        // delete-only commit leaves where it was — so a reload can land back on
2438        // a sequence a caller's `Scope` already cached a mask at. A new id is
2439        // what makes that entry stop matching.
2440        self.store_id = crate::mask::StoreId::next();
2441        self.total_wal_commits = 0;
2442        self.base = None;
2443        self.fold_overlay = None;
2444        self.delta_tail = Vec::new();
2445        self.commits_since_fold = 0;
2446        self.deferred_events = Vec::new();
2447        self.v8_sections_loaded
2448            .store(false, std::sync::atomic::Ordering::Release);
2449        self.last_change = HashMap::new();
2450        self.wal_horizon_floor = 0;
2451        self.archive_genesis_chain = false;
2452        // Re-derived by `load_from_disk` from the store it is about to read.
2453        self.snapshot_preserved_history = false;
2454        self.pending_write_authz = None;
2455        self.wal_consumed = 0;
2456        self.wal_frames_written = 0;
2457        self.snapshot_ident = None;
2458    }
2459
2460    /// Load the snapshot base and replay the WAL into an empty handle — the
2461    /// whole of what opening a store does after the struct exists.
2462    ///
2463    /// Split out of the open path so that [`refresh`](GraphDb::refresh) can
2464    /// rebuild a handle in place, without ownership of `F`, when another
2465    /// process replaces the snapshot underneath it.
2466    ///
2467    /// `origin` decides whether the two repair writes this function can make
2468    /// are appropriate; see [`LoadOrigin`].
2469    fn load_from_disk(&mut self, origin: LoadOrigin) -> Result<usize> {
2470        // Both writes below are crash recovery, and only an open is entitled to
2471        // perform them. A read-only handle promises to touch nothing, and a
2472        // reload driven by `refresh` is looking at a store another process is
2473        // actively writing: what looks like a torn tail there is a peer
2474        // mid-append, and what looks like an orphaned archive may be one that
2475        // peer is about to reference.
2476        let may_repair = origin == LoadOrigin::Open && !self.open_opts.read_only;
2477        let repair_wal = self.open_opts.repair_wal && may_repair;
2478        let db = self;
2479        db.wal_horizon_floor = db.fs.read_horizon_floor()?;
2480        db.archive_genesis_chain = db.fs.has_genesis_marker();
2481        // Opening cleanup: remove orphaned archives — archives whose frames all
2482        // fall below the horizon floor.  Orphans arise when a crash interrupted
2483        // the retention-prune sequence after the floor was written but before
2484        // all surplus archives were deleted.  Safe to delete: floor already
2485        // accounts for their frames.
2486        if may_repair {
2487            db.cleanup_orphaned_archives()?;
2488        }
2489        let _t0 = std::time::Instant::now();
2490        // Peek 6 bytes to determine snapshot version without reading the full
2491        // file. For RealFs this is a true partial read (O(1)); for SimFs the
2492        // default impl reads all bytes and truncates (still correct).
2493        let snap_header = db.fs.read_prefix(FileId::Snapshot, 6)?;
2494        // V8, V9 and V10 share the mmap-able container; V9 only adds section 12
2495        // and V10 adds nothing but its version stamp. A version outside that set
2496        // falls through to the full-read path below, where `snapshot::decode`
2497        // either handles it (V5–V7) or refuses it by name — which is what stops
2498        // an older binary before it reaches the WAL.
2499        let is_v8 = snap_header.len() >= 6
2500            && &snap_header[0..4] == b"GDB1"
2501            && core_storage::snapshot::is_mmap_container(u16::from_le_bytes([
2502                snap_header[4],
2503                snap_header[5],
2504            ]));
2505        // The version this store is stamped with, or `None` when it has never
2506        // been snapshotted. Read from the same six bytes, with no second read.
2507        let snapshot_version = if snap_header.len() >= 6 && &snap_header[0..4] == b"GDB1" {
2508            Some(u16::from_le_bytes([snap_header[4], snap_header[5]]))
2509        } else {
2510            None
2511        };
2512        // No snapshot means no snapshot has ever truncated the WAL, so this
2513        // handle can prove the history is whole. Once a snapshot exists that
2514        // this handle did not take, it cannot: see `snapshot_preserved_history`.
2515        db.snapshot_preserved_history = snap_header.is_empty();
2516        if is_v8 {
2517            // V8: map the file zero-copy (RealFs) or read full bytes (SimFs).
2518            // No 2.4GB heap Vec is allocated on RealFs.
2519            let mapped = Arc::new(
2520                if let Some(snap_path) = db.fs.snapshot_path() {
2521                    core_storage::v8::MappedBase::map(&snap_path)
2522                } else {
2523                    let snap_bytes = db.fs.read(FileId::Snapshot)?;
2524                    core_storage::v8::MappedBase::from_bytes(snap_bytes)
2525                }
2526                .map_err(|e| GraphError::Corrupt {
2527                    detail: format!("v8: mmap open: {e:?}"),
2528                })?,
2529            );
2530            db.restore_v8_base(Arc::clone(&mapped))?;
2531            trace_open!("restore_v8_base", _t0);
2532            db.base = Some(mapped);
2533            trace_open!("base assigned", _t0);
2534        } else if !snap_header.is_empty() {
2535            // Legacy V5-V7: full read required for decode.
2536            let snap_bytes = db.fs.read(FileId::Snapshot)?;
2537            if let Some(state) = core_storage::snapshot::decode(&snap_bytes)? {
2538                db.restore_snapshot_state(state)?;
2539            }
2540        }
2541        // else: snap_header is empty = no snapshot file, fresh store.
2542        //
2543        // Seed commit_seq from the highest seq persisted in last_change so that
2544        // WAL-replay frames (which start at commit_seq+1) always exceed any seq
2545        // already stored in the snapshot.  Without this, a db with one snapshot
2546        // commit would save last_change["a"]=1, then on reopen the first WAL
2547        // frame would replay at seq=1 again — colliding and making WAL-tail
2548        // mutations indistinguishable from the snapshot baseline.
2549        //
2550        // Safety invariant (seq-recycling):
2551        //   Recycled seqs (those below the seeded baseline) were NEVER stored in
2552        //   last_change because they belonged to a previous db lifetime — a new
2553        //   db starts at commit_seq=0 with an empty last_change.  Therefore no
2554        //   CAS precondition can carry a recycled seq as its `expected` value
2555        //   and accidentally match a live node's last_change entry.
2556        //
2557        // `expected:0` on a deleted-then-reinserted node:
2558        //   After deletion, last_changed() returns None; callers that call
2559        //   last_changed() and then use NodeUnchangedSince get None.unwrap_or(0)
2560        //   = 0.  The reinserted node gets seq > 0, so a subsequent CAS with
2561        //   expected=0 correctly conflicts.  The only way to observe actual=0 in
2562        //   a CasConflict would be a caller that invented expected=0 without ever
2563        //   calling last_changed() — unreachable via the documented API contract.
2564        if let Some(&max_seq) = db.last_change.values().max() {
2565            db.commit_seq = db.commit_seq.max(max_seq);
2566        }
2567        let bytes = db.fs.read(FileId::Wal)?;
2568        let (records, valid_len) = decode_all(&bytes);
2569        // The valid prefix is replayed either way; `repair_wal` only decides
2570        // whether the truncation is written back. A reader that races a live
2571        // appender must not persist a truncation the writer never asked for.
2572        if valid_len < bytes.len() && repair_wal {
2573            db.fs.write_atomic(FileId::Wal, &bytes[..valid_len])?;
2574        }
2575        // WAL-present path: build indexes eagerly BEFORE replay so that the
2576        // first replayed record does not trigger the lazy-init guard (which
2577        // would call reindex_all_load_state on an empty graph, defeating the
2578        // point of restoring IVF/HNSW blobs from the snapshot).
2579        if !records.is_empty() {
2580            db.ensure_v8_base_sections_loaded();
2581            trace_open!("lazy sections loaded (WAL path)", _t0);
2582        }
2583        // Scoped to this call: `refresh()` also replays frames and is *not*
2584        // followed by a rebuild, so its backfill must still run.
2585        db.fulltext_rebuild_follows = true;
2586        // Every decoded frame occupies an index, replayed or not: markers
2587        // are state no-ops but they are not index no-ops.
2588        let decoded_frames = records.len() as u64;
2589        let replayed = db.apply_frames(records);
2590        db.fulltext_rebuild_follows = false;
2591        let replayed = replayed?;
2592        // ── The multiplicity declaration, recovered from the stamp ───────────
2593        //
2594        // The opt-in is re-emitted into every baseline WAL a snapshot writes, so
2595        // ordinarily the replay above has already found it. But
2596        // `snapshot_with(archive_wal)` renames the live WAL away and writes its
2597        // replacement afterwards, and between those two points the store holds
2598        // no live declaration at all. A crash there — or a single `Err` from any
2599        // call in between — used to opt the store back out on the next open
2600        // (defect #22): it would stop counting and write a **V9** snapshot while
2601        // the archives still carried discriminant 23, which is the exact state
2602        // the V10 stamp exists to prevent.
2603        //
2604        // The V10 stamp is what carries the conclusion. The archive clause is a
2605        // scope restriction, not a second proof — an earlier version of this
2606        // comment, and defect #22, claimed otherwise, and defect #33 corrects
2607        // it. Taking the two in order:
2608        //
2609        // **The stamp.** `snapshot_with` stamps the snapshot from
2610        // `self.multiplicity` *before* it touches the WAL, and nothing rewrites
2611        // a V10 snapshot at V9 while the store believes it is opted in. So a
2612        // V10 stamp says this store reached `enable_multiplicity` far enough to
2613        // write the snapshot — and, decisively, that every older binary already
2614        // refuses this store by name. Opting in here can cost such a reader
2615        // nothing it was not already being told.
2616        //
2617        // **What the archive clause does not prove.** It is *not* evidence that
2618        // the archive was taken while the store was opted in. A store can
2619        // archive at V9 and opt in afterwards, leaving a V10 snapshot standing
2620        // beside an archive whose WAL carries no declaration at all — see
2621        // `a_failed_opt_in_beside_an_archive_comes_back_opted_in`. The inference
2622        // held in the success case by coincidence, not by construction.
2623        //
2624        // **What it does buy: scope.** Without it the recovery would also fire
2625        // on a store that reached the V10 snapshot write and then failed with no
2626        // archive in sight. That store must stay opted out, and can: no WAL was
2627        // renamed away, nothing carries discriminant 23, and its next snapshot
2628        // rewrites at V9, which puts it back within reach of every older reader.
2629        // An archive is the marker for the one state that is not recoverable
2630        // that way — a WAL renamed away that may hold the only copy of the
2631        // declaration. `no_crash_leaves_discriminant_23_unguarded` pins that
2632        // line: it sweeps a workload with no archives at all and refuses a
2633        // V10-implies-enabled rule.
2634        //
2635        // **The invariant, whichever way the clause goes:** the recovery never
2636        // opts in a store whose snapshot is not V10. A V9 store has made no
2637        // promise to an older reader, so opting it in would start writing
2638        // discriminant 23 behind a stamp that does not guard it. Pinned by
2639        // `the_recovery_never_opts_in_a_store_whose_snapshot_is_not_v10` and
2640        // `the_recovery_does_not_opt_a_store_in_by_itself`.
2641        //
2642        // What this recovery cannot do is make the opt-in atomic; it is not,
2643        // and `enable_multiplicity` says so. See defects #32-#34.
2644        if !db.multiplicity
2645            && snapshot_version == Some(core_storage::snapshot::VERSION_10)
2646            && !db.fs.list_archives()?.is_empty()
2647        {
2648            db.multiplicity = true;
2649        }
2650        // The cursor sits at the end of the valid prefix, not the end of the
2651        // file: a torn or still-being-written tail is unconsumed by definition
2652        // and stays visible to `is_stale` until it decodes.
2653        db.wal_consumed = valid_len as u64;
2654        // The frame cursor counts the same sequence `all_frames` returns:
2655        // surviving archives first, then the live WAL, offset by the floor.
2656        // Counting the archives separately rather than calling
2657        // `wal_total_commits` keeps the live WAL from being decoded twice on
2658        // every open, and costs nothing on a store that has never archived.
2659        db.wal_frames_written = db.wal_horizon_floor + db.archive_frame_count()? + decoded_frames;
2660        db.snapshot_ident = db.fs.snapshot_ident().map_err(GraphError::Io)?;
2661        trace_open!("wal replay done", _t0);
2662        // Rebuild view values after WAL replay only when there is no V8 base.
2663        // With a V8 base, view values are correct in the snapshot and are updated
2664        // incrementally during WAL replay (on_edge_changed / on_prop_changed).
2665        // A full rebuild would read overlay-only props (empty after restore_v8_base)
2666        // and overwrite correct base values with wrong results (e.g. NeighborAgg
2667        // Sum reads no "score" in overlay → writes 0.0, shadowing the correct
2668        // base value).
2669        if db.base.is_none() {
2670            let topo_view = TopologyView::owned(&db.topo);
2671            db.view_store.rebuild_all(
2672                Arc::make_mut(&mut db.props),
2673                &topo_view,
2674                &db.ids,
2675                &db.syms,
2676                &db.labels,
2677            );
2678        }
2679        // Rebuild full-text index after WAL replay.  Corrects drift from
2680        // per-record incremental apply during replay.
2681        Arc::make_mut(&mut db.fulltext).rebuild_all(
2682            &db.ids,
2683            &db.labels,
2684            &db.syms,
2685            build_props_view(&db.props, &db.base),
2686        );
2687        db.prop_index.rebuild_all(
2688            &db.ids,
2689            &db.labels,
2690            &db.syms,
2691            build_props_view(&db.props, &db.base),
2692        );
2693        // Namespaces: one pass over the `ns` column, after the snapshot is
2694        // restored and the WAL replayed. Replay maintains `node_ns` record by
2695        // record as well; this pass is what makes a snapshot-only open right,
2696        // and it reads nothing on a store with no `ns` column.
2697        db.rebuild_node_ns();
2698        // A mid-build snapshot's HNSW blob carries `complete == false`.
2699        // Register it so `serve`'s ticker sees work without waiting for a write.
2700        db.register_outstanding_index_builds();
2701        // Load roles sidecar. Missing file = no roles (Some(vec![])).
2702        // Corrupt/unparseable = poisoned (None); mask_for_role will fail-loud.
2703        db.roles = Self::load_roles_from_fs(&db.fs)?;
2704        // The time sidecar. Absent is the normal case for any store written
2705        // before v0.6.11 and is not an error; unreadable is recorded so date
2706        // queries can say "damaged" rather than "none recorded".
2707        db.load_commit_times_from_fs();
2708        // Capture the initial MVCC fold so reader() is ready immediately.
2709        db.fold_now();
2710        trace_open!("open_with complete", _t0);
2711        Ok(replayed)
2712    }
2713
2714    /// Apply decoded WAL frames to in-memory state, exactly as the open-path
2715    /// replay does — same `apply` calls, same per-frame delta drain, same
2716    /// commit-seq and last-change bookkeeping. Rules therefore fire and derived
2717    /// edges appear identically whether a frame arrives at open, from a local
2718    /// commit, or from another process by way of [`refresh`](GraphDb::refresh).
2719    ///
2720    /// Returns the number of frames applied.
2721    ///
2722    /// Deltas are drained and discarded per frame: replayed frames are already
2723    /// reflected on disk, so they are not news to a subscriber, and draining
2724    /// inside the loop keeps `pending_deltas` O(1) over a large WAL (I-2).
2725    fn apply_frames(&mut self, records: Vec<WalRecord>) -> Result<usize> {
2726        if records.is_empty() {
2727            return Ok(0);
2728        }
2729        // Materialize any state retained in the mmap base before the first
2730        // frame lands, so a replayed record cannot trip the lazy-init guard and
2731        // rebuild indexes from an empty graph. Both calls are idempotent.
2732        self.ensure_v8_base_sections_loaded();
2733        self.engine.consume_retained_state_eager(
2734            &self.ids,
2735            &self.syms,
2736            &self.labels,
2737            build_props_view(&self.props, &self.base),
2738        );
2739        let applied = records.len();
2740        for rec in records {
2741            self.apply(&rec)?;
2742            let _ = self.engine.drain_deltas();
2743            // Track commit_seq during replay so last_change entries are
2744            // consistent with the seqs assigned by log_then_apply_with on
2745            // subsequent live commits.  After N replayed frames, commit_seq=N;
2746            // live commits begin at N+1.
2747            self.commit_seq += 1;
2748            let replay_seq = self.commit_seq;
2749            self.update_last_change_from_rec(&rec, replay_seq);
2750        }
2751        // Enforce I-2: if the per-frame drain above is ever removed or skipped,
2752        // this assert catches the regression in debug builds immediately.
2753        debug_assert_eq!(
2754            self.engine.pending_delta_count(),
2755            0,
2756            "pending_deltas non-empty after replay — \
2757             per-frame drain must run inside the loop to keep memory O(1)"
2758        );
2759        // T2 note: the per-frame drain IS the suppression seam for replay.
2760        // Any future as-of replay path (Plan-15 T2) must drain here to feed
2761        // replaying subscribers; the mechanism is already in place.
2762        let _ = self.engine.drain_deltas(); // belt-and-braces no-op after loop drain
2763        Ok(applied)
2764    }
2765
2766    // ── Multi-process safety: cross-process write lock + WAL tailing ──────────
2767    //
2768    // mushroomdb is many-readers / one-writer across processes. Writers take an
2769    // advisory exclusive lock on the store's `LOCK` file; readers never do.
2770    // Every handle tracks how much of the WAL it has consumed, so it can pick
2771    // up another process's commits by decoding only the new tail rather than
2772    // reopening. See `docs/site/concurrency.md`.
2773
2774    /// Whether the store on disk has moved ahead of (or out from under) this
2775    /// handle's in-memory state.
2776    ///
2777    /// True when the WAL's length differs from this handle's cursor — another
2778    /// process committed, or is mid-append — or when the snapshot file's
2779    /// identity changed. Costs two metadata lookups and reads no file contents,
2780    /// so it is cheap enough for a read path to call.
2781    ///
2782    /// Always false for an as-of view from [`GraphDb::open_at`]: such a view is
2783    /// pinned to one commit and later commits are deliberately invisible to it.
2784    pub fn is_stale(&self) -> Result<bool> {
2785        if self.pinned {
2786            return Ok(false);
2787        }
2788        if self.fs.wal_len().map_err(GraphError::Io)? != self.wal_consumed {
2789            return Ok(true);
2790        }
2791        Ok(self.fs.snapshot_ident().map_err(GraphError::Io)? != self.snapshot_ident)
2792    }
2793
2794    /// Bring this handle up to date with every commit other processes have made,
2795    /// and return how many frames were applied.
2796    ///
2797    /// The WAL tail is decoded from this handle's cursor and applied through the
2798    /// same path the open replay uses, so rules fire and derived edges appear
2799    /// exactly as they would on a fresh open. Interners, id maps and indexes
2800    /// stay valid for the same reason.
2801    ///
2802    /// A frame another process is still writing is left alone: a trailing
2803    /// partial frame is a wait, not a corruption, and the handle stays stale
2804    /// until that frame is complete. Nothing is written to disk, so a read-only
2805    /// handle can refresh freely.
2806    ///
2807    /// When the snapshot file's identity changed, or the WAL is shorter than
2808    /// this handle's cursor, the WAL no longer continues our state — another
2809    /// process snapshotted or archived. The handle is then rebuilt from disk
2810    /// with the options it was opened with, and the return value is the number
2811    /// of frames in the new WAL.
2812    ///
2813    /// Returns 0 for an as-of view, which never follows later commits.
2814    ///
2815    /// # Errors
2816    ///
2817    /// An error here leaves the handle **degraded**: it got partway through
2818    /// applying the tail, or partway through a reload, so its in-memory state
2819    /// no longer matches any point on disk. Further mutations are refused and
2820    /// the handle must be reopened. Nothing on disk was damaged — the store
2821    /// itself is fine, and a fresh open recovers it.
2822    pub fn refresh(&mut self) -> Result<u64> {
2823        if self.pinned {
2824            return Ok(0);
2825        }
2826        let disk_ident = self.fs.snapshot_ident().map_err(GraphError::Io)?;
2827        let wal_len = self.fs.wal_len().map_err(GraphError::Io)?;
2828        if disk_ident != self.snapshot_ident || wal_len < self.wal_consumed {
2829            // The WAL no longer continues our state: rebuild from disk. State
2830            // is cleared first, so a failed load leaves an empty handle — mark
2831            // it degraded rather than let a caller read an empty graph as if
2832            // it were the store's contents.
2833            self.reset_for_reload();
2834            return match self.load_from_disk(LoadOrigin::Reload) {
2835                Ok(frames) => Ok(frames as u64),
2836                Err(e) => {
2837                    self.degraded = true;
2838                    Err(e)
2839                }
2840            };
2841        }
2842        if wal_len == self.wal_consumed {
2843            return Ok(0);
2844        }
2845        let tail = self
2846            .fs
2847            .read_range(FileId::Wal, self.wal_consumed)
2848            .map_err(GraphError::Io)?;
2849        let (records, valid_len) = decode_all(&tail);
2850        let decoded_frames = records.len() as u64;
2851        let applied = match self.apply_frames(records) {
2852            Ok(n) => n,
2853            Err(e) => {
2854                // Some frames landed and some did not, and the cursor cannot
2855                // say how many. Advancing it would skip the rest; leaving it
2856                // would replay what already applied. Neither is recoverable in
2857                // place, so refuse further writes and require a reopen.
2858                self.degraded = true;
2859                return Err(e);
2860            }
2861        };
2862        // Advance by the bytes actually decoded, never by the file length: an
2863        // incomplete trailing frame stays unconsumed for the next refresh.
2864        self.wal_consumed += valid_len as u64;
2865        self.wal_frames_written += decoded_frames;
2866        // The peer that wrote those frames also stamped them. Absorbing the
2867        // frames without the stamps leaves this handle resolving dates from a
2868        // prefix of the store's history, and — while our own map is still
2869        // empty — one commit away from rewriting the peer's file out of
2870        // existence (`first` below decides on the map, and the map is what we
2871        // just brought up to date).
2872        self.load_commit_times_from_fs();
2873        if applied > 0 {
2874            // Peer commits must reach `reader()` snapshots taken from here on.
2875            // A full fold is what open does; refresh does not build per-commit
2876            // deltas, so there is nothing cheaper that stays correct.
2877            self.fold_now();
2878        }
2879        Ok(applied as u64)
2880    }
2881
2882    /// Byte offset of the WAL prefix this handle has applied.
2883    ///
2884    /// Exposed for tests that assert the cursor tracks appended bytes exactly.
2885    #[doc(hidden)]
2886    pub fn wal_consumed(&self) -> u64 {
2887        self.wal_consumed
2888    }
2889
2890    /// Rewind the WAL cursor after the group-commit drain thread truncated a
2891    /// failed group off the tail, so the cursor still describes the file.
2892    pub(crate) fn set_wal_consumed(&mut self, len: u64) {
2893        self.wal_consumed = len;
2894    }
2895
2896    /// One non-blocking attempt at the cross-process write lock.
2897    ///
2898    /// Takes `&self` so a caller can poll for the lock *before* it acquires the
2899    /// in-process write guard. That ordering is what keeps a busy peer in
2900    /// another process from stalling this process's readers.
2901    ///
2902    /// A handle that owns the lock for its lifetime always succeeds.
2903    pub(crate) fn try_cross_process_lock(&self) -> Result<bool> {
2904        if self.holds_lifetime_lock {
2905            return Ok(true);
2906        }
2907        self.fs.try_lock_exclusive().map_err(GraphError::Io)
2908    }
2909
2910    /// Poll for the cross-process write lock until `wait` elapses.
2911    ///
2912    /// One attempt is always made, so a zero wait is a single try. Returns
2913    /// `false` when the lock is still held elsewhere at the deadline; nothing
2914    /// has been written and retrying later is safe.
2915    ///
2916    /// Only the plain-`GraphDb` open path uses this, where the caller owns the
2917    /// handle outright. [`SharedDb`](crate::SharedDb) polls
2918    /// [`try_cross_process_lock`](GraphDb::try_cross_process_lock) itself so
2919    /// that it holds no in-process guard while it waits.
2920    fn poll_lock(&self, wait: std::time::Duration) -> Result<bool> {
2921        let deadline = std::time::Instant::now() + wait;
2922        loop {
2923            if self.try_cross_process_lock()? {
2924                return Ok(true);
2925            }
2926            let now = std::time::Instant::now();
2927            if now >= deadline {
2928                return Ok(false);
2929            }
2930            std::thread::sleep(LOCK_POLL_INTERVAL.min(deadline.saturating_duration_since(now)));
2931        }
2932    }
2933
2934    /// Open a cross-process write scope, given the outcome of an already-made
2935    /// lock attempt.
2936    ///
2937    /// The caller polls for the lock first — outside any in-process guard — and
2938    /// passes what it got. On success this refreshes, so the writes about to
2939    /// happen land on top of every other process's commits. On failure the
2940    /// handle refuses WAL-appending mutations and `snapshot()` with
2941    /// [`GraphError::Busy`] until [`end_write_lock`](GraphDb::end_write_lock)
2942    /// closes the scope, so a caller holding a guard cannot write behind
2943    /// another process's back.
2944    ///
2945    /// A handle that already owns the lock for its lifetime skips the refresh:
2946    /// no other process can have written, so there is nothing to pick up.
2947    pub(crate) fn enter_write_scope(&mut self, acquired: bool) -> Result<()> {
2948        self.lock_denied = !acquired;
2949        if !acquired || self.holds_lifetime_lock {
2950            return Ok(());
2951        }
2952        if let Err(e) = self.refresh() {
2953            // Do not hold a lock we cannot use: release it and let the caller
2954            // see the underlying failure.
2955            let _ = self.fs.unlock();
2956            self.lock_denied = true;
2957            return Err(e);
2958        }
2959        Ok(())
2960    }
2961
2962    /// Close a cross-process write scope opened by
2963    /// [`enter_write_scope`](GraphDb::enter_write_scope): release the lock and
2964    /// clear the Busy latch. Safe to call when the lock was never taken.
2965    pub(crate) fn end_write_lock(&mut self) {
2966        self.lock_denied = false;
2967        if !self.holds_lifetime_lock {
2968            // Releasing a lock we do not hold is a no-op; a failure to release
2969            // is reported by the OS closing the descriptor at handle drop.
2970            let _ = self.fs.unlock();
2971        }
2972    }
2973
2974    /// As-of replay for [`GraphDb::open_at`]: snapshot base (only when the
2975    /// snapshot truncated the WAL) plus the first `commit + 1` WAL frames;
2976    /// see [`GraphDb::open_at`] for the semantics.  The per-frame drain
2977    /// mirrors `open_with` exactly so pending_delta_count is 0 on exit.
2978    /// Restore all persisted state from a decoded snapshot. Shared by
2979    /// `open_with` and (when the snapshot truncated the WAL) `open_at_with`.
2980    fn restore_snapshot_state(
2981        &mut self,
2982        state: core_storage::snapshot::SnapshotState,
2983    ) -> Result<()> {
2984        self.ids = Arc::new(state.ids);
2985        self.syms = Arc::new(state.syms);
2986        self.topo = Arc::new(state.topo);
2987        self.props = Arc::new(state.props);
2988        self.labels = Arc::new(state.labels);
2989        self.edge_props = Arc::new(state.edge_props);
2990        // Cross-section label integrity for V5/V7 snapshots: same invariants as
2991        // restore_v8_base.  A crafted bincode snapshot with a short `labels` vec,
2992        // out-of-range sym ids, or a sentinel label on a live node would otherwise
2993        // open successfully and panic later in `NodeRef::label()` or
2994        // `neighborhood_masked()`.  Catching it here turns those into typed
2995        // `GraphError::Corrupt` at open time.
2996        {
2997            let ids_len = self.ids.len();
2998            if self.labels.len() != ids_len {
2999                return Err(GraphError::Corrupt {
3000                    detail: format!(
3001                        "snapshot: labels vec has {} entries but id table has {} total slots",
3002                        self.labels.len(),
3003                        ids_len,
3004                    ),
3005                });
3006            }
3007            let syms_len = self.syms.len() as u32;
3008            for (i, &sym) in self.labels.iter().enumerate() {
3009                let is_tombstoned = self.ids.is_tombstoned(i as u32);
3010                if sym == u32::MAX {
3011                    if !is_tombstoned {
3012                        return Err(GraphError::Corrupt {
3013                            detail: format!(
3014                                "snapshot: live node at id slot {i} has sentinel label (u32::MAX)"
3015                            ),
3016                        });
3017                    }
3018                } else if sym >= syms_len {
3019                    return Err(GraphError::Corrupt {
3020                        detail: format!(
3021                            "snapshot: label at id slot {i} references sym {sym} \
3022                             which is out of interner range ({syms_len})"
3023                        ),
3024                    });
3025                }
3026            }
3027        }
3028        let defs: Vec<RuleDef> = state
3029            .rule_defs
3030            .iter()
3031            .map(|b| {
3032                decode_rule_def(b).map_err(|e| GraphError::Corrupt {
3033                    detail: format!("snapshot rule_def deserialize: {e}"),
3034                })
3035            })
3036            .collect::<Result<Vec<_>>>()?;
3037        self.engine =
3038            RuleEngine::from_persist(defs, state.provenance, state.rule_tripped, state.rule_fires);
3039        // Candidate indexes are rebuilt lazily on the first mutation (see
3040        // RuleEngine::on_node_changed).  HNSW blobs and IVF centroids from the
3041        // snapshot are retained without deserializing so that:
3042        //   - clean-open (empty WAL): indexes stay empty; blobs load on first
3043        //     ANN query via ensure_hnsw_loaded, or on first mutation via the
3044        //     lazy-init guard which calls reindex_all_load_state (the scan
3045        //     skips the HNSW build for every side the blob supplies).
3046        //   - WAL-present: open_with calls consume_retained_state_eager before
3047        //     replay so HNSW/IVF are live before any record fires the hooks.
3048        let ivf_bytes = if state.ivf_state.is_empty() {
3049            Vec::new()
3050        } else {
3051            bincode::serialize(&state.ivf_state).expect("IVF state serialize cannot fail")
3052        };
3053        // Store blobs without eagerly deserializing them.
3054        // `self.ids` is the snapshot's id table at this point — WAL replay has
3055        // not run — so its length is the line an interrupted build is detected
3056        // against.
3057        let snapshot_ids = self.ids.len() as u32;
3058        self.engine
3059            .store_snapshot_state(state.hnsw_state, ivf_bytes, snapshot_ids);
3060        // Restore view defs from snapshot (V5).
3061        // The ColumnStore already contains view values from the snapshot;
3062        // use restore_view (no collision check, no backfill) so the store
3063        // is aware of the definitions.  rebuild_all runs after WAL replay.
3064        for def_bytes in &state.view_defs {
3065            let def: ViewDef =
3066                bincode::deserialize(def_bytes).map_err(|e| GraphError::Corrupt {
3067                    detail: format!("snapshot view_def deserialize: {e}"),
3068                })?;
3069            self.view_store
3070                .restore_view(def)
3071                .map_err(|e| GraphError::Corrupt {
3072                    detail: format!("snapshot view restore: {e}"),
3073                })?;
3074        }
3075        Ok(())
3076    }
3077
3078    /// Restore all persisted state from a V8 `MappedBase` snapshot, **except**
3079    /// topology (`self.topo` stays empty and serves as the WAL-replay overlay).
3080    ///
3081    /// `self.props` IS fully materialised from the base so that HNSW/IVF blob
3082    /// deserialization and view rebuild have access to all column data.
3083    fn restore_v8_base(&mut self, mapped: Arc<core_storage::v8::MappedBase>) -> Result<()> {
3084        self.ids = Arc::new(archived_to_idmap(mapped.ids().map_err(|e| {
3085            GraphError::Corrupt {
3086                detail: format!("v8: ids section: {e:?}"),
3087            }
3088        })?));
3089        self.syms = Arc::new(archived_to_interner(mapped.syms().map_err(|e| {
3090            GraphError::Corrupt {
3091                detail: format!("v8: syms section: {e:?}"),
3092            }
3093        })?));
3094
3095        // C1: self.props is left as an empty overlay. Column reads go through
3096        // props_view() (ColumnsView::with_base), which consults the archived base
3097        // section zero-copy. This avoids the O(columns) heap copy at every open.
3098
3099        // self.topo deliberately left as Topology::new() — overlay path.
3100
3101        let meta = decode_meta(mapped.meta_bytes().map_err(|e| GraphError::Corrupt {
3102            detail: format!("v8: meta section: {e:?}"),
3103        })?)
3104        .map_err(|e| GraphError::Corrupt {
3105            detail: format!("v8: meta decode: {e:?}"),
3106        })?;
3107        self.labels = Arc::new(meta.labels);
3108        // Cross-section label integrity: labels must cover every id slot (live
3109        // and tombstoned), every non-sentinel sym must be within the interner's
3110        // bound, and no live (non-tombstoned) node may carry the u32::MAX
3111        // sentinel label.  Without this check, a crafted snapshot where the META
3112        // section (small, CRC-validated) holds a short `labels` vec, out-of-range
3113        // sym ids, or a sentinel label on a live node, would open successfully
3114        // and then panic in `NodeRef::label()`, `neighborhood_masked()`, and
3115        // related read paths.  Catching the inconsistency here converts those
3116        // panics into typed `GraphError::Corrupt` at open time.
3117        {
3118            let ids_len = self.ids.len();
3119            if self.labels.len() != ids_len {
3120                return Err(GraphError::Corrupt {
3121                    detail: format!(
3122                        "v8: labels section has {} entries but id table has {} total slots",
3123                        self.labels.len(),
3124                        ids_len,
3125                    ),
3126                });
3127            }
3128            let syms_len = self.syms.len() as u32;
3129            for (i, &sym) in self.labels.iter().enumerate() {
3130                let is_tombstoned = self.ids.is_tombstoned(i as u32);
3131                if sym == u32::MAX {
3132                    // Sentinel is only valid for tombstoned slots.
3133                    if !is_tombstoned {
3134                        return Err(GraphError::Corrupt {
3135                            detail: format!(
3136                                "v8: live node at id slot {i} has sentinel label (u32::MAX)"
3137                            ),
3138                        });
3139                    }
3140                } else if sym >= syms_len {
3141                    return Err(GraphError::Corrupt {
3142                        detail: format!(
3143                            "v8: label at id slot {i} references sym {sym} \
3144                             which is out of interner range ({syms_len})"
3145                        ),
3146                    });
3147                }
3148            }
3149        }
3150        // C3: self.edge_props stays as an empty overlay.  Reads go through
3151        // edge_props_view() which consults the mmap'd base section zero-copy
3152        // via EdgePropsView::with_base.  No heap decode at open time.
3153
3154        // Restore rule engine.
3155        let (rule_def_bytes, rule_tripped, rule_fires) =
3156            archived_rules_meta_to_owned(mapped.rules_meta_section().map_err(|e| {
3157                GraphError::Corrupt {
3158                    detail: format!("v8: rules_meta section: {e:?}"),
3159                }
3160            })?);
3161        let defs: Vec<RuleDef> = rule_def_bytes
3162            .iter()
3163            .map(|b| {
3164                decode_rule_def(b).map_err(|e| GraphError::Corrupt {
3165                    detail: format!("v8: rule_def deserialize: {e}"),
3166                })
3167            })
3168            .collect::<Result<Vec<_>>>()?;
3169        self.engine = RuleEngine::from_persist(defs, BTreeMap::new(), rule_tripped, rule_fires);
3170        // C4+C5: provenance, HNSW, and IVF sections are NOT read here.
3171        // `ensure_v8_base_sections_loaded` reads them on first use from
3172        // `self.base` (set by the caller immediately after this returns).
3173        // A clean open touches only: header + IDS + SYMS + META + RULES_META.
3174
3175        // Restore view definitions.
3176        let view_defs =
3177            archived_views_to_owned(mapped.views_section().map_err(|e| GraphError::Corrupt {
3178                detail: format!("v8: views section: {e:?}"),
3179            })?);
3180        for def_bytes in &view_defs {
3181            let def: ViewDef =
3182                bincode::deserialize(def_bytes).map_err(|e| GraphError::Corrupt {
3183                    detail: format!("v8: view_def deserialize: {e}"),
3184                })?;
3185            self.view_store
3186                .restore_view(def)
3187                .map_err(|e| GraphError::Corrupt {
3188                    detail: format!("v8: view restore: {e}"),
3189                })?;
3190        }
3191        // Load the last-change map from section 11 (small section; load eagerly).
3192        // Pre-Task-3 snapshots lack this section; `last_change_bytes` returns &[]
3193        // in that case and `decode_last_change_bytes` returns an empty map.
3194        let last_change_raw = mapped
3195            .last_change_bytes()
3196            .map_err(|e| GraphError::Corrupt {
3197                detail: format!("v8: last_change section: {e:?}"),
3198            })?;
3199        self.last_change = decode_last_change_bytes(last_change_raw);
3200
3201        // Validate that all deferred sections (provenance, HNSW, IVF) fit within
3202        // the file.  Pure bounds check — no bytes read, no page faults triggered.
3203        // Catches truncated snapshots at open time before the lazy deferred reads.
3204        mapped.validate_section_bounds().map_err(|e| match e {
3205            GraphError::Corrupt { detail } => GraphError::Corrupt {
3206                detail: format!("v8: section bounds: {detail}"),
3207            },
3208            other => other,
3209        })?;
3210        Ok(())
3211    }
3212
3213    /// Read provenance, HNSW, and IVF sections from the mmap base into the
3214    /// engine's retained fields on first call.  Subsequent calls are a no-op
3215    /// (AtomicBool fast-path).
3216    ///
3217    /// Must be called before any code path that reads or mutates engine
3218    /// provenance, HNSW, or IVF state:
3219    /// - WAL replay (before `consume_retained_state_eager`)
3220    /// - First mutation (`log_then_apply_with`)
3221    /// - Read-only paths (`stats`, `explain`, `node_edges`)
3222    /// - Snapshot (`snapshot_with`)
3223    ///
3224    /// No-op for fresh stores and V5-V7 opens (`self.base` is `None`).
3225    fn ensure_v8_base_sections_loaded(&self) {
3226        use std::sync::atomic::Ordering;
3227        if self.v8_sections_loaded.load(Ordering::Acquire) {
3228            return;
3229        }
3230        let _guard = self
3231            .v8_sections_mutex
3232            .lock()
3233            .expect("v8 sections mutex poisoned");
3234        if self.v8_sections_loaded.load(Ordering::Acquire) {
3235            return; // another caller populated while we waited
3236        }
3237        let _t = std::time::Instant::now();
3238        if let Some(base) = &self.base {
3239            // Provenance: raw rkyv bytes; CRC validated inside section_bytes.
3240            // Bounds are already validated at open time (restore_v8_base →
3241            // validate_section_bounds) — unreachable post-validate_section_bounds;
3242            // unwrap_or_default is a safety belt against impossible errors.
3243            let prov_bytes = base
3244                .provenance_raw_bytes()
3245                .map(|b| b.to_vec())
3246                .unwrap_or_default();
3247            self.engine.store_provenance_bytes(prov_bytes);
3248            // HNSW: decode rkyv blobs into owned map.
3249            let hnsw_state = base
3250                .hnsw_section()
3251                .map(archived_hnsw_to_owned)
3252                .unwrap_or_default();
3253            // IVF: raw bincode bytes; deserialized on first mutation/query.
3254            let ivf_bytes = base.ivf_bytes().map(|b| b.to_vec()).unwrap_or_default();
3255            // Called before WAL replay on a WAL-present open (`open_with`) and
3256            // before any write on a clean one, so this is the snapshot's count.
3257            let snapshot_ids = self.ids.len() as u32;
3258            self.engine
3259                .store_snapshot_state(hnsw_state, ivf_bytes, snapshot_ids);
3260        }
3261        self.v8_sections_loaded.store(true, Ordering::Release);
3262        if std::env::var("MUSHROOMDB_TRACE_OPEN").is_ok() {
3263            eprintln!(
3264                "[MUSHROOMDB_TRACE_OPEN] ensure_v8_base_sections_loaded: {:>9.3?}",
3265                _t.elapsed()
3266            );
3267        }
3268    }
3269
3270    /// Return a `TopologyView` that merges the mmap'd base (when present) with
3271    /// the in-memory WAL overlay.  Used by all read paths in db.rs that need
3272    /// the full merged topology without going through `self.view()`.
3273    fn topo_view(&self) -> TopologyView<'_> {
3274        match self.base {
3275            None => TopologyView::owned(&self.topo),
3276            Some(ref base) => {
3277                // SAFETY: base lives as long as self; section bounds validated at open.
3278                // topology() uses access_unchecked; all field reads are bounds-checked in seam.rs.
3279                let archived = base
3280                    .topology()
3281                    .expect("base topology section bounds validated at open");
3282                TopologyView::with_base(&self.topo, archived)
3283            }
3284        }
3285    }
3286
3287    /// Return a `ColumnsView` that merges the mmap'd base columns (when a V8
3288    /// snapshot is open) with the in-memory WAL overlay.  Reads consult the
3289    /// overlay first, then fall through to the archived base section zero-copy.
3290    fn props_view(&self) -> core_storage::v8::seam::ColumnsView<'_> {
3291        match self.base {
3292            None => core_storage::v8::seam::ColumnsView::owned(&self.props),
3293            Some(ref base) => {
3294                // columns() uses access_unchecked; field reads are bounds-checked in seam.rs.
3295                let archived = base
3296                    .columns()
3297                    .expect("base columns section bounds validated at open");
3298                core_storage::v8::seam::ColumnsView::with_base_cached(
3299                    &self.props,
3300                    archived,
3301                    base.mixed_cache(),
3302                )
3303                .with_shared_strings(base_string_table(base))
3304            }
3305        }
3306    }
3307
3308    /// Return an `EdgePropsView` that merges the mmap'd base edge-props section
3309    /// (when a V8 snapshot is open) with the in-memory WAL overlay.
3310    ///
3311    /// Reads consult the overlay first (for post-snapshot mutations), then fall
3312    /// through to the archived base section zero-copy.  Tombstones in the
3313    /// overlay mask deleted-from-base entries.
3314    fn edge_props_view(&self) -> EdgePropsView<'_> {
3315        match self.base {
3316            None => EdgePropsView::owned(&self.edge_props),
3317            Some(ref base) => {
3318                // edge_props_section() uses access_unchecked; field reads bounds-checked in seam.rs.
3319                let archived = base
3320                    .edge_props_section()
3321                    .expect("base edge_props section bounds validated at open");
3322                EdgePropsView::with_base(&self.edge_props, archived)
3323            }
3324        }
3325    }
3326
3327    fn open_at_with(fs: F, commit: u64) -> Result<Self> {
3328        // An as-of view never writes and is pinned to one commit: it takes no
3329        // cross-process lock and does not follow later commits.
3330        let mut db = Self::new_empty(
3331            fs,
3332            OpenOptions {
3333                repair_wal: false,
3334                auto_migrate: false,
3335                read_only: true,
3336            },
3337        );
3338        db.pinned = true; // read_only is set after replay, but pinning is immediate
3339        db.wal_horizon_floor = db.fs.read_horizon_floor()?;
3340        db.archive_genesis_chain = db.fs.has_genesis_marker();
3341        // Same orphaned-archive cleanup as open_with: floor was written first
3342        // during pruning, so a crash may have left stale archives below floor.
3343        db.cleanup_orphaned_archives()?;
3344        // Collect archive frames (oldest-first) and live WAL frames.
3345        // Archives represent pre-snapshot history; the snapshot captures the
3346        // cumulative state at the time of archiving.  Crash-window guarantee:
3347        //   A: crash before rename → WAL intact, no archive. Reopen: normal.
3348        //   B: crash after rename, before new WAL → archive present, WAL
3349        //      absent. Reopen: snapshot loaded (full state), no WAL replay.
3350        //   C: crash after new baseline WAL written → normal post-archive.
3351        let archive_ns = db.fs.list_archives()?;
3352        let mut archive_frames_all: Vec<WalRecord> = Vec::new();
3353        for n in &archive_ns {
3354            let arc_bytes = db.fs.read_archive(*n)?;
3355            let (arc_frames, _) = decode_all(&arc_bytes);
3356            archive_frames_all.extend(arc_frames);
3357        }
3358        let total_archive_frames = archive_frames_all.len() as u64;
3359
3360        let live_bytes = db.fs.read(FileId::Wal)?;
3361        let (live_records, _valid_len) = decode_all(&live_bytes);
3362        let total_surviving = total_archive_frames + live_records.len() as u64;
3363        // Global total including any pruned history below the horizon floor.
3364        let total = db.wal_horizon_floor + total_surviving;
3365
3366        // Horizon and range check.
3367        if commit < db.wal_horizon_floor {
3368            return Err(GraphError::CommitOutOfRange {
3369                commit,
3370                total,
3371                floor: db.wal_horizon_floor,
3372            });
3373        }
3374        if commit >= total {
3375            return Err(GraphError::CommitOutOfRange {
3376                commit,
3377                total,
3378                floor: db.wal_horizon_floor,
3379            });
3380        }
3381
3382        // Local index into surviving frames (0 = first frame of oldest archive).
3383        let local = commit - db.wal_horizon_floor;
3384
3385        if local < total_archive_frames {
3386            // Target commit is in an archive.  Correct replay from empty state
3387            // is only possible when the archive chain is an uninterrupted
3388            // genesis chain (first archive taken from a fresh store, no prior
3389            // WAL truncation) and no archives have been pruned (floor == 0).
3390            //
3391            // If either condition is violated the prefix needed to reconstruct
3392            // the requested state is gone; refuse rather than return wrong data.
3393            if db.wal_horizon_floor > 0 || !db.archive_genesis_chain {
3394                return Err(GraphError::CommitOutOfRange {
3395                    commit,
3396                    total,
3397                    floor: db.wal_horizon_floor,
3398                });
3399            }
3400            // Replay all archive frames up to and including the target commit
3401            // from an empty database state.  Archives must be replayed in order
3402            // so that dense-id intern tables are built up correctly.
3403            for rec in archive_frames_all.into_iter().take((local + 1) as usize) {
3404                db.apply(&rec)?;
3405                let _ = db.engine.drain_deltas();
3406            }
3407        } else {
3408            // Target commit is in the live WAL: load snapshot as base, then
3409            // replay the needed live WAL prefix.
3410            //
3411            // Base state: a truncating snapshot (wal_truncated=true) compacts
3412            // all pre-truncation / pre-archive commits.  Dense-id records in
3413            // the live WAL reference ids/interns that the snapshot provides.
3414            // Peek 6 bytes (same pattern as open_with).
3415            let snap_header = db.fs.read_prefix(FileId::Snapshot, 6)?;
3416            let is_v8 = snap_header.len() >= 6
3417                && &snap_header[0..4] == b"GDB1"
3418                && core_storage::snapshot::is_mmap_container(u16::from_le_bytes([
3419                    snap_header[4],
3420                    snap_header[5],
3421                ]));
3422            if is_v8 {
3423                let state = if let Some(snap_path) = db.fs.snapshot_path() {
3424                    let mapped = core_storage::v8::MappedBase::map(&snap_path).map_err(|e| {
3425                        GraphError::Corrupt {
3426                            detail: format!("v8: open_at mmap: {e:?}"),
3427                        }
3428                    })?;
3429                    core_storage::snapshot::decode_v8_from_mapped(&mapped)?
3430                } else {
3431                    let snap_bytes = db.fs.read(FileId::Snapshot)?;
3432                    core_storage::snapshot::decode(&snap_bytes)?
3433                };
3434                if let Some(state) = state {
3435                    if state.wal_truncated {
3436                        db.restore_snapshot_state(state)?;
3437                    }
3438                }
3439            } else if !snap_header.is_empty() {
3440                let snap_bytes = db.fs.read(FileId::Snapshot)?;
3441                if let Some(state) = core_storage::snapshot::decode(&snap_bytes)? {
3442                    if state.wal_truncated {
3443                        db.restore_snapshot_state(state)?;
3444                    }
3445                }
3446            }
3447            // else: snap_header empty = no snapshot file.
3448            let live_local = local - total_archive_frames;
3449            for rec in live_records.into_iter().take((live_local + 1) as usize) {
3450                db.apply(&rec)?;
3451                let _ = db.engine.drain_deltas();
3452            }
3453        }
3454        // Pin: pending_delta_count must be 0 after as-of replay, mirroring T1's
3455        // post-loop assert in open_with.
3456        debug_assert_eq!(
3457            db.engine.pending_delta_count(),
3458            0,
3459            "pending_deltas non-empty after open_at replay — \
3460             per-frame drain must run inside the loop to keep memory O(1)"
3461        );
3462        let _ = db.engine.drain_deltas(); // belt-and-braces no-op
3463                                          // Rebuild view values after WAL replay so derived-edge-driven views
3464                                          // reflect the as-of state.  open_at always uses the legacy path (no V8
3465                                          // base), so topo_view is always owned.
3466        {
3467            let topo_view = TopologyView::owned(&db.topo);
3468            db.view_store.rebuild_all(
3469                Arc::make_mut(&mut db.props),
3470                &topo_view,
3471                &db.ids,
3472                &db.syms,
3473                &db.labels,
3474            );
3475        }
3476        // Rebuild full-text index for as-of view (mirrors open_with pattern).
3477        Arc::make_mut(&mut db.fulltext).rebuild_all(
3478            &db.ids,
3479            &db.labels,
3480            &db.syms,
3481            build_props_view(&db.props, &db.base),
3482        );
3483        db.prop_index.rebuild_all(
3484            &db.ids,
3485            &db.labels,
3486            &db.syms,
3487            build_props_view(&db.props, &db.base),
3488        );
3489        // Namespaces on the temporal handle, built by the same pass the live
3490        // open uses, so an as-of mask narrows by the namespaces of that commit.
3491        db.rebuild_node_ns();
3492        // Load roles sidecar (current roles, not point-in-time).
3493        db.roles = Self::load_roles_from_fs(&db.fs)?;
3494        db.read_only = true;
3495        db.total_wal_commits = total;
3496        // Capture initial fold so reader() is immediately usable.
3497        db.fold_now();
3498        Ok(db)
3499    }
3500
3501    /// Whether this instance is a read-only as-of view.
3502    pub fn is_read_only(&self) -> bool {
3503        self.read_only
3504    }
3505
3506    // ── MVCC epoch reader ─────────────────────────────────────────────────────
3507
3508    /// Clone the current overlay state into a new `FrozenOverlay` and reset
3509    /// the delta tail. Called automatically every `FOLD_EVERY_K` commits and at
3510    /// the end of `open_with` / `open_at_with` to prime the reader.
3511    fn fold_now(&mut self) {
3512        // Eight `Arc::clone`s — refcount bumps, O(1). This used to deep-copy the
3513        // whole overlay: `IdMap` alone is a `HashMap<String, u32>` plus a
3514        // `Vec<String>`, so every node key was copied twice, on every open,
3515        // after every snapshot, every FOLD_EVERY_K commits on the write path,
3516        // and — with no commit threshold — on every `refresh()` that applied a
3517        // peer commit. A refreshing reader now pays nothing for a fold.
3518        //
3519        // `props` and `topo` were already cheap for a different reason: on a
3520        // snapshotted store `restore_v8_base` leaves them as empty overlays over
3521        // the zero-copy mmap. `ids`, `syms` and `fulltext` were not, and that
3522        // inconsistency was the defect.
3523        let frozen = crate::reader::FrozenOverlay {
3524            ids: std::sync::Arc::clone(&self.ids),
3525            syms: std::sync::Arc::clone(&self.syms),
3526            topo: std::sync::Arc::clone(&self.topo),
3527            props: std::sync::Arc::clone(&self.props),
3528            labels: std::sync::Arc::clone(&self.labels),
3529            edge_props: std::sync::Arc::clone(&self.edge_props),
3530            roles: self.roles.clone().map(std::sync::Arc::new),
3531            fulltext: std::sync::Arc::clone(&self.fulltext),
3532        };
3533        self.fold_overlay = Some(Arc::new(frozen));
3534        self.delta_tail.clear();
3535        self.commits_since_fold = 0;
3536    }
3537
3538    /// Capture a lock-free reader snapshot of the current db state.
3539    ///
3540    /// The read lock is held only for the duration of this call (to clone a
3541    /// handful of `Arc` handles). Subsequent query operations run without any
3542    /// lock.
3543    pub fn reader(&self) -> crate::reader::ReaderSnapshot {
3544        crate::reader::ReaderSnapshot::new(
3545            self.fold_overlay
3546                .clone()
3547                .expect("fold_overlay is always Some after open_with; call reader() after open"),
3548            self.base.clone(),
3549            self.delta_tail.clone(),
3550            // The snapshot's effective state is exactly this handle's state at
3551            // this commit, so it shares the memo and its version key.
3552            self.commit_seq,
3553            Arc::clone(&self.role_masks),
3554        )
3555    }
3556
3557    /// Append a delta the reader cannot apply, so that a corrupt overlay is
3558    /// reachable from a test.
3559    ///
3560    /// Compiled only under `test-hooks`, which the server's dev-dependency on
3561    /// this crate turns on. One call permanently corrupts every
3562    /// [`ReaderSnapshot`](crate::reader::ReaderSnapshot) taken from the handle,
3563    /// so it must not be in the published surface: `#[doc(hidden)]` hides it
3564    /// from rustdoc and from nothing else. The feature gate — not
3565    /// `#[cfg(test)]` — because its only callers are in `crates/server/tests`,
3566    /// a different crate, exactly as `core_rules`'s index counters are.
3567    ///
3568    /// [`ReaderSnapshot::effective`](crate::reader::ReaderSnapshot) folds the
3569    /// delta tail into a clone of the frozen overlay and answers
3570    /// [`GraphError::Corrupt`] when a record will not apply. Nothing a caller
3571    /// can do produces that state — `apply_one`'s failures are disagreements
3572    /// between the tail and the fold it is applied to, which the write path
3573    /// cannot create — so the `Corrupt` arm of every scoped reader method was
3574    /// reachable only by inspection until this hook existed. An `Intern` record
3575    /// claiming an id the frozen interner will not hand back is the smallest
3576    /// such disagreement.
3577    ///
3578    /// Only the tail is touched. This handle's own state is untouched and
3579    /// `commit_seq` does not move, so a role mask already memoised at this
3580    /// version stays memoised — which is exactly the state in which the HTTP
3581    /// role branches reach a scoped read with a corrupt overlay under them.
3582    #[cfg(any(test, feature = "test-hooks"))]
3583    #[doc(hidden)]
3584    pub fn push_unapplyable_delta_for_test(&mut self) {
3585        self.delta_tail.push(Arc::new(crate::reader::CommitDelta {
3586            records: vec![WalRecord::Intern {
3587                id: u32::MAX,
3588                text: "delta-tail-corruption".into(),
3589            }],
3590            derived_inserts: Vec::new(),
3591            derived_deletes: Vec::new(),
3592        }));
3593    }
3594
3595    /// Total number of WAL commits at the time [`open_at`] was called.
3596    /// Returns 0 for normal (non-as-of) instances.
3597    pub fn total_wal_commits(&self) -> u64 {
3598        self.total_wal_commits
3599    }
3600
3601    /// Apply a record to in-memory state. Used by both live writes and replay,
3602    /// so replay is definitionally identical to the original execution.
3603    fn apply(&mut self, rec: &WalRecord) -> Result<()> {
3604        // Before the record mutates anything: a store restored from a snapshot
3605        // defers building its candidate indexes until the first write, and that
3606        // build is a full node scan. Left where it used to fire — inside the
3607        // engine hook, after `props.set` and the label assignment — the scan
3608        // read the half-applied record and took the in-flight node's vector for
3609        // one the snapshot should have carried, which read as an interrupted
3610        // vector-index build and cost a full `RebuildRule` on the first
3611        // embedded write after every reopen. Hoisted here the scan sees exactly
3612        // the persisted state; the record's own hook then files its vector
3613        // through the ordinary insert path a line later.
3614        self.populate_indexes_before_write();
3615        match rec {
3616            WalRecord::InsertNode { label, key, props } => {
3617                let id = Arc::make_mut(&mut self.ids).try_insert(key)?;
3618                let sym = Arc::make_mut(&mut self.syms).intern(label);
3619                if self.labels.len() <= id as usize {
3620                    // gap slots are sentinels, never valid label symbols
3621                    Arc::make_mut(&mut self.labels).resize(id as usize + 1, u32::MAX);
3622                }
3623                Arc::make_mut(&mut self.labels)[id as usize] = sym;
3624                let mut ns_name = NS_DEFAULT.to_string();
3625                for (field, value) in props {
3626                    if field == NS_PROP {
3627                        ns_name = namespace_of_value(Some(value)).to_string();
3628                    }
3629                    Arc::make_mut(&mut self.props).set(id, field, value.clone());
3630                }
3631                self.set_node_ns(id, &ns_name);
3632                // Initialize view values for the new node before the engine runs so
3633                // delta-based increments start from a known zero baseline.
3634                self.view_store.init_node_views(
3635                    id,
3636                    Arc::make_mut(&mut self.props),
3637                    &self.syms,
3638                    &self.labels,
3639                );
3640                // Fire rules for the newly inserted node.
3641                let cursor = self.engine.pending_delta_count();
3642                let mut eng = std::mem::take(&mut self.engine);
3643                {
3644                    let mut gm = make_graph_mut(
3645                        &self.ids,
3646                        Arc::make_mut(&mut self.syms),
3647                        &self.labels,
3648                        build_props_view(&self.props, &self.base),
3649                        Arc::make_mut(&mut self.topo),
3650                        &self.base,
3651                        Arc::make_mut(&mut self.edge_props),
3652                    );
3653                    eng.on_node_changed(id, None, &mut gm);
3654                }
3655                self.engine = eng;
3656                // Process derived-edge deltas for view maintenance.
3657                // Fast path: skip the O(delta_count) allocation when no views exist.
3658                if !self.view_store.is_empty() {
3659                    #[cfg(test)]
3660                    DELTA_COPY_COUNT.with(|c| c.set(c.get() + 1));
3661                    let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
3662                    for d in &new_deltas {
3663                        self.view_store.on_edge_changed(
3664                            d.etype_sym,
3665                            d.src_id,
3666                            d.dst_id,
3667                            d.fired,
3668                            Arc::make_mut(&mut self.props),
3669                            &build_topo_view(&self.topo, &self.base),
3670                            &self.ids,
3671                            &self.syms,
3672                            &self.labels,
3673                            base_columns(&self.base),
3674                        );
3675                    }
3676                }
3677                // Full-text index maintenance: index enabled fields for this label.
3678                if self.fulltext.has_label(label) {
3679                    for (field, value) in props {
3680                        if self.fulltext.is_enabled(label, field) {
3681                            Arc::make_mut(&mut self.fulltext).add_tokens(id, field, value);
3682                        }
3683                    }
3684                }
3685                // Property (equality) index maintenance.
3686                if self.prop_index.has_label(label) {
3687                    for (field, value) in props {
3688                        self.prop_index.set(label, field, id, value);
3689                    }
3690                }
3691            }
3692            WalRecord::InsertEdge {
3693                edge_type,
3694                src_key,
3695                dst_key,
3696            } => {
3697                let src = self.ids.get(src_key).ok_or_else(|| GraphError::Corrupt {
3698                    detail: format!("wal replay references unknown key {src_key}"),
3699                })?;
3700                let dst = self.ids.get(dst_key).ok_or_else(|| GraphError::Corrupt {
3701                    detail: format!("wal replay references unknown key {dst_key}"),
3702                })?;
3703                let etype = Arc::make_mut(&mut self.syms).intern(edge_type);
3704                // Skip if the edge is already visible in the merged base+overlay
3705                // view.  This keeps WAL replay idempotent when the WAL contains
3706                // pre-snapshot records that are already encoded in a V8 base
3707                // (keep_wal=true opens and crash-before-truncation scenarios).
3708                if self.base.is_some()
3709                    && self
3710                        .topo_view()
3711                        .neighbors(etype, Direction::Out, src)
3712                        .contains(&dst)
3713                {
3714                    return Ok(());
3715                }
3716                Arc::make_mut(&mut self.topo).add_edge(etype, src, dst);
3717                // View maintenance for manual edge insert.
3718                self.view_store.on_edge_changed(
3719                    etype,
3720                    src,
3721                    dst,
3722                    true,
3723                    Arc::make_mut(&mut self.props),
3724                    &build_topo_view(&self.topo, &self.base),
3725                    &self.ids,
3726                    &self.syms,
3727                    &self.labels,
3728                    base_columns(&self.base),
3729                );
3730                // Rule engine: via-hop rules must update when user edges change.
3731                let cursor = self.engine.pending_delta_count();
3732                let mut eng = std::mem::take(&mut self.engine);
3733                {
3734                    let mut gm = make_graph_mut(
3735                        &self.ids,
3736                        Arc::make_mut(&mut self.syms),
3737                        &self.labels,
3738                        build_props_view(&self.props, &self.base),
3739                        Arc::make_mut(&mut self.topo),
3740                        &self.base,
3741                        Arc::make_mut(&mut self.edge_props),
3742                    );
3743                    eng.on_edge_changed(edge_type, src, dst, &mut gm);
3744                }
3745                self.engine = eng;
3746                if !self.view_store.is_empty() {
3747                    let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
3748                    for d in &new_deltas {
3749                        self.view_store.on_edge_changed(
3750                            d.etype_sym,
3751                            d.src_id,
3752                            d.dst_id,
3753                            d.fired,
3754                            Arc::make_mut(&mut self.props),
3755                            &build_topo_view(&self.topo, &self.base),
3756                            &self.ids,
3757                            &self.syms,
3758                            &self.labels,
3759                            base_columns(&self.base),
3760                        );
3761                    }
3762                }
3763            }
3764            WalRecord::SetProp { key, field, value } => {
3765                let id = self.ids.get(key).ok_or_else(|| GraphError::Corrupt {
3766                    detail: format!("wal replay references unknown key {key}"),
3767                })?;
3768                let old_value = build_props_view(&self.props, &self.base)
3769                    .get(id, field)
3770                    .map(|vr| vr.into_value());
3771                Arc::make_mut(&mut self.props).set(id, field, value.clone());
3772                // Fire rules for the changed field.
3773                let cursor = self.engine.pending_delta_count();
3774                let mut eng = std::mem::take(&mut self.engine);
3775                {
3776                    let mut gm = make_graph_mut(
3777                        &self.ids,
3778                        Arc::make_mut(&mut self.syms),
3779                        &self.labels,
3780                        build_props_view(&self.props, &self.base),
3781                        Arc::make_mut(&mut self.topo),
3782                        &self.base,
3783                        Arc::make_mut(&mut self.edge_props),
3784                    );
3785                    eng.on_node_changed(id, Some((field, old_value)), &mut gm);
3786                }
3787                self.engine = eng;
3788                // Derived-edge deltas → view updates.
3789                if !self.view_store.is_empty() {
3790                    #[cfg(test)]
3791                    DELTA_COPY_COUNT.with(|c| c.set(c.get() + 1));
3792                    let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
3793                    for d in &new_deltas {
3794                        self.view_store.on_edge_changed(
3795                            d.etype_sym,
3796                            d.src_id,
3797                            d.dst_id,
3798                            d.fired,
3799                            Arc::make_mut(&mut self.props),
3800                            &build_topo_view(&self.topo, &self.base),
3801                            &self.ids,
3802                            &self.syms,
3803                            &self.labels,
3804                            base_columns(&self.base),
3805                        );
3806                    }
3807                }
3808                // Neighbor-aggregate views that read `field` must also update.
3809                self.view_store.on_prop_changed(
3810                    id,
3811                    field,
3812                    Arc::make_mut(&mut self.props),
3813                    &build_topo_view(&self.topo, &self.base),
3814                    &self.ids,
3815                    &self.syms,
3816                    &self.labels,
3817                    base_columns(&self.base),
3818                );
3819                // Full-text index maintenance: update tokens for this field if indexed.
3820                if self.fulltext.field_indexed(field) {
3821                    let label_opt = self.labels.get(id as usize).and_then(|&sym| {
3822                        if sym == u32::MAX {
3823                            None
3824                        } else {
3825                            self.syms.resolve(sym)
3826                        }
3827                    });
3828                    if let Some(label) = label_opt {
3829                        if self.fulltext.is_enabled(label, field) {
3830                            Arc::make_mut(&mut self.fulltext).remove_node_field(id, field);
3831                            Arc::make_mut(&mut self.fulltext).add_tokens(id, field, value);
3832                        }
3833                    }
3834                }
3835                // Property (equality) index maintenance: re-key this node's value.
3836                if self.prop_index.field_indexed(field) {
3837                    let label_opt = self.labels.get(id as usize).and_then(|&sym| {
3838                        if sym == u32::MAX {
3839                            None
3840                        } else {
3841                            self.syms.resolve(sym)
3842                        }
3843                    });
3844                    if let Some(label) = label_opt {
3845                        self.prop_index.set(label, field, id, value);
3846                    }
3847                }
3848            }
3849            WalRecord::Intern { id, text } => {
3850                if let Some(existing) = self.syms.get(text) {
3851                    if existing != *id {
3852                        return Err(GraphError::Corrupt {
3853                            detail: format!(
3854                                "wal intern mismatch for {text:?}: have {existing}, record {id}"
3855                            ),
3856                        });
3857                    }
3858                } else {
3859                    let got = Arc::make_mut(&mut self.syms).intern(text);
3860                    if got != *id {
3861                        return Err(GraphError::Corrupt {
3862                            detail: format!(
3863                                "wal intern assigned {got} for {text:?}, record wanted {id}"
3864                            ),
3865                        });
3866                    }
3867                }
3868            }
3869            WalRecord::InsertNodeId { label, key, props } => {
3870                let id = Arc::make_mut(&mut self.ids).try_insert(key)?;
3871                if self.labels.len() <= id as usize {
3872                    Arc::make_mut(&mut self.labels).resize(id as usize + 1, u32::MAX);
3873                }
3874                Arc::make_mut(&mut self.labels)[id as usize] = *label;
3875                let label_str = self
3876                    .syms
3877                    .resolve(*label)
3878                    .ok_or_else(|| GraphError::Corrupt {
3879                        detail: format!("wal InsertNodeId unknown label intern {label}"),
3880                    })?
3881                    .to_string();
3882                let mut ns_name = NS_DEFAULT.to_string();
3883                for (field_sym, value) in props {
3884                    let field =
3885                        self.syms
3886                            .resolve(*field_sym)
3887                            .ok_or_else(|| GraphError::Corrupt {
3888                                detail: format!(
3889                                    "wal InsertNodeId unknown field intern {field_sym}"
3890                                ),
3891                            })?;
3892                    if field == NS_PROP {
3893                        ns_name = namespace_of_value(Some(value)).to_string();
3894                    }
3895                    Arc::make_mut(&mut self.props).set(id, field, value.clone());
3896                }
3897                self.set_node_ns(id, &ns_name);
3898                self.view_store.init_node_views(
3899                    id,
3900                    Arc::make_mut(&mut self.props),
3901                    &self.syms,
3902                    &self.labels,
3903                );
3904                let cursor = self.engine.pending_delta_count();
3905                let mut eng = std::mem::take(&mut self.engine);
3906                {
3907                    let mut gm = make_graph_mut(
3908                        &self.ids,
3909                        Arc::make_mut(&mut self.syms),
3910                        &self.labels,
3911                        build_props_view(&self.props, &self.base),
3912                        Arc::make_mut(&mut self.topo),
3913                        &self.base,
3914                        Arc::make_mut(&mut self.edge_props),
3915                    );
3916                    eng.on_node_changed(id, None, &mut gm);
3917                }
3918                self.engine = eng;
3919                if !self.view_store.is_empty() {
3920                    #[cfg(test)]
3921                    DELTA_COPY_COUNT.with(|c| c.set(c.get() + 1));
3922                    let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
3923                    for d in &new_deltas {
3924                        self.view_store.on_edge_changed(
3925                            d.etype_sym,
3926                            d.src_id,
3927                            d.dst_id,
3928                            d.fired,
3929                            Arc::make_mut(&mut self.props),
3930                            &build_topo_view(&self.topo, &self.base),
3931                            &self.ids,
3932                            &self.syms,
3933                            &self.labels,
3934                            base_columns(&self.base),
3935                        );
3936                    }
3937                }
3938                if self.fulltext.has_label(&label_str) {
3939                    for (field_sym, value) in props {
3940                        let Some(field) = self.syms.resolve(*field_sym) else {
3941                            continue;
3942                        };
3943                        if self.fulltext.is_enabled(&label_str, field) {
3944                            Arc::make_mut(&mut self.fulltext).add_tokens(id, field, value);
3945                        }
3946                    }
3947                }
3948                if self.prop_index.has_label(&label_str) {
3949                    for (field_sym, value) in props {
3950                        let Some(field) = self.syms.resolve(*field_sym) else {
3951                            continue;
3952                        };
3953                        self.prop_index.set(&label_str, field, id, value);
3954                    }
3955                }
3956            }
3957            WalRecord::InsertEdgeId { etype, src, dst } => {
3958                // Replay-over-snapshot: dense ids in the pre-snapshot WAL may
3959                // already be tombstoned. Skip rather than attaching edges to
3960                // dead ids (DeleteNode keys the live re-insert, not the old id).
3961                if self.ids.is_tombstoned(*src)
3962                    || self.ids.is_tombstoned(*dst)
3963                    || self.ids.key_of(*src).is_none()
3964                    || self.ids.key_of(*dst).is_none()
3965                {
3966                    return Ok(());
3967                }
3968                // Skip if already visible in the merged view (same idempotency
3969                // guard as InsertEdge above: prevents double-counting when
3970                // pre-snapshot WAL records are replayed over a V8 base).
3971                if self.base.is_some()
3972                    && self
3973                        .topo_view()
3974                        .neighbors(*etype, Direction::Out, *src)
3975                        .contains(dst)
3976                {
3977                    return Ok(());
3978                }
3979                Arc::make_mut(&mut self.topo).add_edge(*etype, *src, *dst);
3980                self.view_store.on_edge_changed(
3981                    *etype,
3982                    *src,
3983                    *dst,
3984                    true,
3985                    Arc::make_mut(&mut self.props),
3986                    &build_topo_view(&self.topo, &self.base),
3987                    &self.ids,
3988                    &self.syms,
3989                    &self.labels,
3990                    base_columns(&self.base),
3991                );
3992                // Rule engine: via-hop rules fire when user via-edges are inserted.
3993                // Resolve etype back to string so on_edge_changed can match rules by name.
3994                if let Some(etype_str) = self.syms.resolve(*etype).map(|s| s.to_string()) {
3995                    let cursor = self.engine.pending_delta_count();
3996                    let mut eng = std::mem::take(&mut self.engine);
3997                    {
3998                        let mut gm = make_graph_mut(
3999                            &self.ids,
4000                            Arc::make_mut(&mut self.syms),
4001                            &self.labels,
4002                            build_props_view(&self.props, &self.base),
4003                            Arc::make_mut(&mut self.topo),
4004                            &self.base,
4005                            Arc::make_mut(&mut self.edge_props),
4006                        );
4007                        eng.on_edge_changed(&etype_str, *src, *dst, &mut gm);
4008                    }
4009                    self.engine = eng;
4010                    if !self.view_store.is_empty() {
4011                        let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
4012                        for d in &new_deltas {
4013                            self.view_store.on_edge_changed(
4014                                d.etype_sym,
4015                                d.src_id,
4016                                d.dst_id,
4017                                d.fired,
4018                                Arc::make_mut(&mut self.props),
4019                                &build_topo_view(&self.topo, &self.base),
4020                                &self.ids,
4021                                &self.syms,
4022                                &self.labels,
4023                                base_columns(&self.base),
4024                            );
4025                        }
4026                    }
4027                }
4028            }
4029            WalRecord::SetPropId { id, field, value } => {
4030                if self.ids.is_tombstoned(*id) || self.ids.key_of(*id).is_none() {
4031                    return Ok(());
4032                }
4033                let field_str = self
4034                    .syms
4035                    .resolve(*field)
4036                    .ok_or_else(|| GraphError::Corrupt {
4037                        detail: format!("wal SetPropId unknown field intern {field}"),
4038                    })?
4039                    .to_string();
4040                let old_value = build_props_view(&self.props, &self.base)
4041                    .get(*id, &field_str)
4042                    .map(|vr| vr.into_value());
4043                Arc::make_mut(&mut self.props).set(*id, &field_str, value.clone());
4044                let cursor = self.engine.pending_delta_count();
4045                let mut eng = std::mem::take(&mut self.engine);
4046                {
4047                    let mut gm = make_graph_mut(
4048                        &self.ids,
4049                        Arc::make_mut(&mut self.syms),
4050                        &self.labels,
4051                        build_props_view(&self.props, &self.base),
4052                        Arc::make_mut(&mut self.topo),
4053                        &self.base,
4054                        Arc::make_mut(&mut self.edge_props),
4055                    );
4056                    eng.on_node_changed(*id, Some((field_str.as_str(), old_value)), &mut gm);
4057                }
4058                self.engine = eng;
4059                if !self.view_store.is_empty() {
4060                    #[cfg(test)]
4061                    DELTA_COPY_COUNT.with(|c| c.set(c.get() + 1));
4062                    let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
4063                    for d in &new_deltas {
4064                        self.view_store.on_edge_changed(
4065                            d.etype_sym,
4066                            d.src_id,
4067                            d.dst_id,
4068                            d.fired,
4069                            Arc::make_mut(&mut self.props),
4070                            &build_topo_view(&self.topo, &self.base),
4071                            &self.ids,
4072                            &self.syms,
4073                            &self.labels,
4074                            base_columns(&self.base),
4075                        );
4076                    }
4077                }
4078                self.view_store.on_prop_changed(
4079                    *id,
4080                    &field_str,
4081                    Arc::make_mut(&mut self.props),
4082                    &build_topo_view(&self.topo, &self.base),
4083                    &self.ids,
4084                    &self.syms,
4085                    &self.labels,
4086                    base_columns(&self.base),
4087                );
4088                if self.fulltext.field_indexed(&field_str) {
4089                    let label_opt = self.labels.get(*id as usize).and_then(|&sym| {
4090                        if sym == u32::MAX {
4091                            None
4092                        } else {
4093                            self.syms.resolve(sym)
4094                        }
4095                    });
4096                    if let Some(label) = label_opt {
4097                        if self.fulltext.is_enabled(label, &field_str) {
4098                            Arc::make_mut(&mut self.fulltext).remove_node_field(*id, &field_str);
4099                            Arc::make_mut(&mut self.fulltext).add_tokens(*id, &field_str, value);
4100                        }
4101                    }
4102                }
4103                if self.prop_index.field_indexed(&field_str) {
4104                    let label_opt = self.labels.get(*id as usize).and_then(|&sym| {
4105                        if sym == u32::MAX {
4106                            None
4107                        } else {
4108                            self.syms.resolve(sym)
4109                        }
4110                    });
4111                    if let Some(label) = label_opt {
4112                        self.prop_index.set(label, &field_str, *id, value);
4113                    }
4114                }
4115            }
4116            WalRecord::CreateRule { def_bytes } => {
4117                let def: RuleDef = decode_rule_def(def_bytes).map_err(|e| GraphError::Corrupt {
4118                    detail: format!("CreateRule def_bytes deserialize failed: {e}"),
4119                })?;
4120                // Replay-over-snapshot idempotency: the rule was captured in the snapshot
4121                // so the engine already has it; silently skip to avoid a spurious
4122                // RuleInvalid error in the crash window between snapshot write and WAL
4123                // truncation.
4124                if self.engine.rules().any(|r| r.name == def.name) {
4125                    return Ok(());
4126                }
4127                let cursor = self.engine.pending_delta_count();
4128                let mut eng = std::mem::take(&mut self.engine);
4129                let result = {
4130                    let mut gm = make_graph_mut(
4131                        &self.ids,
4132                        Arc::make_mut(&mut self.syms),
4133                        &self.labels,
4134                        build_props_view(&self.props, &self.base),
4135                        Arc::make_mut(&mut self.topo),
4136                        &self.base,
4137                        Arc::make_mut(&mut self.edge_props),
4138                    );
4139                    eng.create_rule(def, &mut gm)
4140                };
4141                self.engine = eng;
4142                result.map_err(|e| GraphError::RuleInvalid { detail: e })?;
4143                // Derived-edge fires from backfill → view updates.
4144                // Fast path: skip O(edge_count) allocation when no views exist.
4145                if !self.view_store.is_empty() {
4146                    #[cfg(test)]
4147                    DELTA_COPY_COUNT.with(|c| c.set(c.get() + 1));
4148                    let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
4149                    for d in &new_deltas {
4150                        self.view_store.on_edge_changed(
4151                            d.etype_sym,
4152                            d.src_id,
4153                            d.dst_id,
4154                            d.fired,
4155                            Arc::make_mut(&mut self.props),
4156                            &build_topo_view(&self.topo, &self.base),
4157                            &self.ids,
4158                            &self.syms,
4159                            &self.labels,
4160                            base_columns(&self.base),
4161                        );
4162                    }
4163                }
4164            }
4165            WalRecord::DeleteRule { name } => {
4166                // Replay-over-snapshot idempotency: the snapshot already captured the
4167                // post-delete state so the rule is absent; silently skip to avoid a
4168                // spurious RuleNotFound error in the crash window between snapshot write
4169                // and WAL truncation.
4170                if !self.engine.rules().any(|r| r.name == *name) {
4171                    return Ok(());
4172                }
4173                let cursor = self.engine.pending_delta_count();
4174                let mut eng = std::mem::take(&mut self.engine);
4175                let result = {
4176                    let mut gm = make_graph_mut(
4177                        &self.ids,
4178                        Arc::make_mut(&mut self.syms),
4179                        &self.labels,
4180                        build_props_view(&self.props, &self.base),
4181                        Arc::make_mut(&mut self.topo),
4182                        &self.base,
4183                        Arc::make_mut(&mut self.edge_props),
4184                    );
4185                    eng.delete_rule(name, &mut gm)
4186                };
4187                self.engine = eng;
4188                result.map_err(|_| GraphError::RuleNotFound { name: name.clone() })?;
4189                // Derived-edge retractions → view updates.
4190                if !self.view_store.is_empty() {
4191                    #[cfg(test)]
4192                    DELTA_COPY_COUNT.with(|c| c.set(c.get() + 1));
4193                    let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
4194                    for d in &new_deltas {
4195                        self.view_store.on_edge_changed(
4196                            d.etype_sym,
4197                            d.src_id,
4198                            d.dst_id,
4199                            d.fired,
4200                            Arc::make_mut(&mut self.props),
4201                            &build_topo_view(&self.topo, &self.base),
4202                            &self.ids,
4203                            &self.syms,
4204                            &self.labels,
4205                            base_columns(&self.base),
4206                        );
4207                    }
4208                }
4209            }
4210            WalRecord::RemoveProp { key, field } => {
4211                // Recovery-safe: unknown key or already-absent field is a
4212                // clean no-op. Crash-window replay over a snapshot that
4213                // already applied this record must not Err.
4214                let Some(id) = self.ids.get(key) else {
4215                    return Ok(());
4216                };
4217                // Read old value through the seam for rule retraction.
4218                let old = build_props_view(&self.props, &self.base)
4219                    .get(id, field)
4220                    .map(|vr| vr.into_value());
4221                Arc::make_mut(&mut self.props).remove(id, field);
4222                // If the base still supplies the value after the overlay removal,
4223                // record a tombstone so ColumnsView::get does not resurrect it.
4224                // This covers both the base-only case AND the both-resident case:
4225                //   base-only (in_overlay=false): old prop was only in base, remove
4226                //     is a no-op on overlay, base still visible → tombstone needed.
4227                //   both-resident (in_overlay=true): overlay had v2, base has v1;
4228                //     removing overlay uncovers v1 → tombstone needed.
4229                // Idempotent on double-replay: second pass sees the tombstone →
4230                // get() returns None → condition is false → no duplicate tombstone.
4231                if build_props_view(&self.props, &self.base)
4232                    .get(id, field)
4233                    .is_some()
4234                {
4235                    Arc::make_mut(&mut self.props).record_prop_tombstone(id, field);
4236                }
4237                let cursor = self.engine.pending_delta_count();
4238                let mut eng = std::mem::take(&mut self.engine);
4239                {
4240                    let mut gm = make_graph_mut(
4241                        &self.ids,
4242                        Arc::make_mut(&mut self.syms),
4243                        &self.labels,
4244                        build_props_view(&self.props, &self.base),
4245                        Arc::make_mut(&mut self.topo),
4246                        &self.base,
4247                        Arc::make_mut(&mut self.edge_props),
4248                    );
4249                    eng.on_node_changed(id, Some((field, old)), &mut gm);
4250                }
4251                self.engine = eng;
4252                // Derived-edge deltas → view updates.
4253                if !self.view_store.is_empty() {
4254                    #[cfg(test)]
4255                    DELTA_COPY_COUNT.with(|c| c.set(c.get() + 1));
4256                    let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
4257                    for d in &new_deltas {
4258                        self.view_store.on_edge_changed(
4259                            d.etype_sym,
4260                            d.src_id,
4261                            d.dst_id,
4262                            d.fired,
4263                            Arc::make_mut(&mut self.props),
4264                            &build_topo_view(&self.topo, &self.base),
4265                            &self.ids,
4266                            &self.syms,
4267                            &self.labels,
4268                            base_columns(&self.base),
4269                        );
4270                    }
4271                }
4272                // Neighbor-aggregate views that read `field` must also update.
4273                self.view_store.on_prop_changed(
4274                    id,
4275                    field,
4276                    Arc::make_mut(&mut self.props),
4277                    &build_topo_view(&self.topo, &self.base),
4278                    &self.ids,
4279                    &self.syms,
4280                    &self.labels,
4281                    base_columns(&self.base),
4282                );
4283                // Full-text index maintenance: remove tokens for this field.
4284                if self.fulltext.field_indexed(field) {
4285                    Arc::make_mut(&mut self.fulltext).remove_node_field(id, field);
4286                }
4287                // Property (equality) index maintenance: drop this node's entry.
4288                if self.prop_index.field_indexed(field) {
4289                    if let Some(label) = self.labels.get(id as usize).and_then(|&sym| {
4290                        (sym != u32::MAX).then(|| self.syms.resolve(sym)).flatten()
4291                    }) {
4292                        self.prop_index.remove_node(label, field, id);
4293                    }
4294                }
4295            }
4296            WalRecord::DeleteEdge {
4297                edge_type,
4298                src_key,
4299                dst_key,
4300            } => {
4301                // Recovery-safe: unknown keys, unknown etype, or already-
4302                // absent edge is a clean no-op (remove_edge returns false).
4303                let Some(src) = self.ids.get(src_key) else {
4304                    return Ok(());
4305                };
4306                let Some(dst) = self.ids.get(dst_key) else {
4307                    return Ok(());
4308                };
4309                let Some(etype) = self.syms.get(edge_type) else {
4310                    return Ok(());
4311                };
4312                // I3: phantom-tombstone guard.  When a V8 base is present, a
4313                // DeleteEdge WAL record for an edge that was already absorbed into
4314                // the new base (i.e. neither in overlay nor in base) must be skipped.
4315                // Without this guard, remove_edge records a tombstone for an edge
4316                // that no longer exists, incorrectly understating edge_count.
4317                if self.base.is_some()
4318                    && !self
4319                        .topo_view()
4320                        .neighbors(etype, core_storage::topology::Direction::Out, src)
4321                        .contains(&dst)
4322                {
4323                    return Ok(());
4324                }
4325                Arc::make_mut(&mut self.topo).remove_edge(etype, src, dst);
4326                Arc::make_mut(&mut self.edge_props).remove_edge(etype, src, dst);
4327                // View maintenance for manual edge delete (topo already updated above).
4328                self.view_store.on_edge_changed(
4329                    etype,
4330                    src,
4331                    dst,
4332                    false,
4333                    Arc::make_mut(&mut self.props),
4334                    &build_topo_view(&self.topo, &self.base),
4335                    &self.ids,
4336                    &self.syms,
4337                    &self.labels,
4338                    base_columns(&self.base),
4339                );
4340                // Rule engine: via-hop rules must retract when user via-edges are deleted.
4341                let cursor = self.engine.pending_delta_count();
4342                let mut eng = std::mem::take(&mut self.engine);
4343                {
4344                    let mut gm = make_graph_mut(
4345                        &self.ids,
4346                        Arc::make_mut(&mut self.syms),
4347                        &self.labels,
4348                        build_props_view(&self.props, &self.base),
4349                        Arc::make_mut(&mut self.topo),
4350                        &self.base,
4351                        Arc::make_mut(&mut self.edge_props),
4352                    );
4353                    eng.on_edge_changed(edge_type, src, dst, &mut gm);
4354                }
4355                self.engine = eng;
4356                if !self.view_store.is_empty() {
4357                    let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
4358                    for d in &new_deltas {
4359                        self.view_store.on_edge_changed(
4360                            d.etype_sym,
4361                            d.src_id,
4362                            d.dst_id,
4363                            d.fired,
4364                            Arc::make_mut(&mut self.props),
4365                            &build_topo_view(&self.topo, &self.base),
4366                            &self.ids,
4367                            &self.syms,
4368                            &self.labels,
4369                            base_columns(&self.base),
4370                        );
4371                    }
4372                }
4373            }
4374            WalRecord::DeleteNode { key } => {
4375                // Recovery-safe: already-tombstoned / unknown key is a clean
4376                // no-op. Crash-window replay over a snapshot that already
4377                // applied this record cannot recover the retired id from the
4378                // key (`IdMap::get` is None), so every subsequent step is
4379                // skipped. Each step is independently idempotent if invoked
4380                // twice on a still-live id: retraction is a no-op on empty
4381                // provenance, `remove_edge` returns false, `remove_all` is a
4382                // no-op, `ids.delete` returns None, label sentinel is sticky.
4383                let Some(n) = self.ids.get(key) else {
4384                    return Ok(());
4385                };
4386
4387                // (1) Retract derived edges + de-index while props/labels live.
4388                let cursor = self.engine.pending_delta_count();
4389                let mut eng = std::mem::take(&mut self.engine);
4390                {
4391                    let mut gm = make_graph_mut(
4392                        &self.ids,
4393                        Arc::make_mut(&mut self.syms),
4394                        &self.labels,
4395                        build_props_view(&self.props, &self.base),
4396                        Arc::make_mut(&mut self.topo),
4397                        &self.base,
4398                        Arc::make_mut(&mut self.edge_props),
4399                    );
4400                    eng.on_node_removed(n, &mut gm);
4401                }
4402                self.engine = eng;
4403                // Derived-edge retractions → view updates for neighbors.
4404                if !self.view_store.is_empty() {
4405                    #[cfg(test)]
4406                    DELTA_COPY_COUNT.with(|c| c.set(c.get() + 1));
4407                    let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
4408                    for d in &new_deltas {
4409                        self.view_store.on_edge_changed(
4410                            d.etype_sym,
4411                            d.src_id,
4412                            d.dst_id,
4413                            d.fired,
4414                            Arc::make_mut(&mut self.props),
4415                            &build_topo_view(&self.topo, &self.base),
4416                            &self.ids,
4417                            &self.syms,
4418                            &self.labels,
4419                            base_columns(&self.base),
4420                        );
4421                    }
4422                }
4423
4424                // (2) Sweep ALL remaining edges incident to n, both directions,
4425                // every etype. This cascade is intentionally mask-independent:
4426                // topology integrity requires removing every edge touching the
4427                // deleted node regardless of the caller's visibility scope.
4428                // (The mask limits which nodes a role's read phase can return;
4429                // the WAL delete always executes with full storage authority.)
4430                // Collect then remove so neighbor slices stay valid during
4431                // iteration. Remove from topo first, then call view maintenance
4432                // so Avg/Min/Max recompute sees the correct (reduced) neighbor set.
4433                let etypes: Vec<u32> = self.topo.etypes().collect();
4434                let mut doomed = Vec::new();
4435                for et in &etypes {
4436                    for &dst in self.topo.neighbors(*et, Direction::Out, n).as_ref() {
4437                        doomed.push((*et, n, dst));
4438                    }
4439                    for &src in self.topo.neighbors(*et, Direction::In, n).as_ref() {
4440                        doomed.push((*et, src, n));
4441                    }
4442                }
4443                for (et, s, d) in doomed {
4444                    Arc::make_mut(&mut self.topo).remove_edge(et, s, d);
4445                    Arc::make_mut(&mut self.edge_props).remove_edge(et, s, d);
4446                    // View maintenance: n's own view values will be cleared by
4447                    // remove_all below; only update surviving neighbors.
4448                    self.view_store.on_edge_changed(
4449                        et,
4450                        s,
4451                        d,
4452                        false,
4453                        Arc::make_mut(&mut self.props),
4454                        &build_topo_view(&self.topo, &self.base),
4455                        &self.ids,
4456                        &self.syms,
4457                        &self.labels,
4458                        base_columns(&self.base),
4459                    );
4460                }
4461
4462                // (3) Drop every remaining prop (`ColumnStore::remove_all`).
4463                Arc::make_mut(&mut self.props).remove_all(n);
4464                // Full-text index maintenance: remove all tokens for this node.
4465                Arc::make_mut(&mut self.fulltext).remove_node(n);
4466                // Property (equality) index maintenance: drop all entries for n.
4467                self.prop_index.remove_node_all(n);
4468
4469                // (4) Retire the dense id and stamp the label sentinel.
4470                Arc::make_mut(&mut self.ids).delete(key);
4471                if let Some(slot) = Arc::make_mut(&mut self.labels).get_mut(n as usize) {
4472                    *slot = u32::MAX;
4473                }
4474            }
4475            WalRecord::Batch(inner) => {
4476                // Apply each inner record in order through the same apply path.
4477                // Inner records are validated free of nested Batch by encode_record.
4478                for rec in inner {
4479                    self.apply(rec)?;
4480                }
4481            }
4482            WalRecord::RebuildRule { name } => {
4483                // Replay-over-snapshot idempotency: the snapshot may already
4484                // reflect a later delete_rule, so the rule is absent; skip.
4485                if !self.engine.rules().any(|r| r.name == *name) {
4486                    return Ok(());
4487                }
4488                let cursor = self.engine.pending_delta_count();
4489                let mut eng = std::mem::take(&mut self.engine);
4490                let result = {
4491                    let mut gm = make_graph_mut(
4492                        &self.ids,
4493                        Arc::make_mut(&mut self.syms),
4494                        &self.labels,
4495                        build_props_view(&self.props, &self.base),
4496                        Arc::make_mut(&mut self.topo),
4497                        &self.base,
4498                        Arc::make_mut(&mut self.edge_props),
4499                    );
4500                    eng.rebuild(name, &mut gm)
4501                };
4502                self.engine = eng;
4503                result.map_err(|_| GraphError::RuleNotFound { name: name.clone() })?;
4504                // Derived-edge delta changes → view updates.
4505                if !self.view_store.is_empty() {
4506                    #[cfg(test)]
4507                    DELTA_COPY_COUNT.with(|c| c.set(c.get() + 1));
4508                    let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
4509                    for d in &new_deltas {
4510                        self.view_store.on_edge_changed(
4511                            d.etype_sym,
4512                            d.src_id,
4513                            d.dst_id,
4514                            d.fired,
4515                            Arc::make_mut(&mut self.props),
4516                            &build_topo_view(&self.topo, &self.base),
4517                            &self.ids,
4518                            &self.syms,
4519                            &self.labels,
4520                            base_columns(&self.base),
4521                        );
4522                    }
4523                }
4524            }
4525            WalRecord::CreateView { def_bytes } => {
4526                let def: ViewDef =
4527                    bincode::deserialize(def_bytes).map_err(|e| GraphError::Corrupt {
4528                        detail: format!("CreateView def_bytes deserialize failed: {e}"),
4529                    })?;
4530                // Replay-over-snapshot idempotency: view already present → skip.
4531                if self.view_store.has_view(&def.name) {
4532                    return Ok(());
4533                }
4534                self.view_store
4535                    .create_view(
4536                        def,
4537                        Arc::make_mut(&mut self.props),
4538                        &build_topo_view(&self.topo, &self.base),
4539                        &self.ids,
4540                        &self.syms,
4541                        &self.labels,
4542                    )
4543                    .map_err(|e| GraphError::RuleInvalid { detail: e })?;
4544            }
4545            WalRecord::DeleteView { name } => {
4546                // Replay-over-snapshot idempotency: view already absent → skip.
4547                if !self.view_store.has_view(name) {
4548                    return Ok(());
4549                }
4550                self.view_store
4551                    .delete_view(
4552                        name,
4553                        Arc::make_mut(&mut self.props),
4554                        &self.ids,
4555                        &self.labels,
4556                        &self.syms,
4557                    )
4558                    .map_err(|_| GraphError::RuleNotFound { name: name.clone() })?;
4559            }
4560            WalRecord::EnableFulltext { label, field } => {
4561                // Replay-over-snapshot idempotency: already enabled → skip.
4562                if self.fulltext.is_enabled(label, field) {
4563                    return Ok(());
4564                }
4565                Arc::make_mut(&mut self.fulltext).enable(label, field);
4566                if self.fulltext_rebuild_follows {
4567                    // The open path rebuilds the whole index after replay, which
4568                    // clears every posting this scan would write. Doing it twice
4569                    // costs a full tokenise-and-stem pass over the corpus per
4570                    // enabled pair: measured at 456 ms against 3.8 ms for the
4571                    // same 30,000-entity store with no pair enabled.
4572                    return Ok(());
4573                }
4574                // Backfill: index all live nodes of this label that have the field.
4575                let n = self.ids.len() as u32;
4576                for id in 0..n {
4577                    let Some(&sym) = self.labels.get(id as usize) else {
4578                        continue;
4579                    };
4580                    if sym == u32::MAX {
4581                        continue; // tombstoned
4582                    }
4583                    let Some(lbl) = self.syms.resolve(sym) else {
4584                        continue;
4585                    };
4586                    if lbl != label {
4587                        continue;
4588                    }
4589                    if let Some(value) = build_props_view(&self.props, &self.base)
4590                        .get(id, field)
4591                        .map(|vr| vr.into_value())
4592                    {
4593                        Arc::make_mut(&mut self.fulltext).add_tokens(id, field, &value);
4594                    }
4595                }
4596            }
4597            WalRecord::DisableFulltext { label, field } => {
4598                // Replay-over-snapshot idempotency: already disabled → skip.
4599                if !self.fulltext.is_enabled(label, field) {
4600                    return Ok(());
4601                }
4602                // If another label still indexes this field, the postings column
4603                // is kept — but it must not contain node_ids from the now-disabled
4604                // label.  Remove them before calling disable() so the field_indexed
4605                // guard inside disable() sees the correct post-removal state.
4606                if self.fulltext.field_indexed_by_other(label, field) {
4607                    if let Some(label_sym) = self.syms.get(label) {
4608                        for (node_id, &lsym) in self.labels.iter().enumerate() {
4609                            if lsym == label_sym {
4610                                Arc::make_mut(&mut self.fulltext)
4611                                    .remove_node_field(node_id as u32, field);
4612                            }
4613                        }
4614                    }
4615                }
4616                Arc::make_mut(&mut self.fulltext).disable(label, field);
4617            }
4618            WalRecord::EnableIndex { label, field } => {
4619                // Replay-over-snapshot idempotency: already enabled → skip.
4620                if self.prop_index.is_enabled(label, field) {
4621                    return Ok(());
4622                }
4623                self.prop_index.enable(label, field);
4624                // Backfill: index all live nodes of this label that have the field.
4625                let n = self.ids.len() as u32;
4626                for id in 0..n {
4627                    let Some(&sym) = self.labels.get(id as usize) else {
4628                        continue;
4629                    };
4630                    if sym == u32::MAX {
4631                        continue; // tombstoned
4632                    }
4633                    let Some(lbl) = self.syms.resolve(sym) else {
4634                        continue;
4635                    };
4636                    if lbl != label {
4637                        continue;
4638                    }
4639                    if let Some(value) = build_props_view(&self.props, &self.base)
4640                        .get(id, field)
4641                        .map(|vr| vr.into_value())
4642                    {
4643                        self.prop_index.set(label, field, id, &value);
4644                    }
4645                }
4646            }
4647            WalRecord::DisableIndex { label, field } => {
4648                self.prop_index.disable(label, field);
4649            }
4650            // ── insert-count multiplicity (§5.13) ────────────────────────────
4651            //
4652            // Two shapes, told apart by `count`: the opt-in declaration, and an
4653            // absolute count for one triple. Absolute is what makes this
4654            // idempotent over a snapshot base — a pre-snapshot frame replayed
4655            // over a base that already folded it in lands on the same number
4656            // rather than adding to it, which is the failure a delta (or a count
4657            // derived from `InsertEdgeId` records) would have.
4658            WalRecord::SetEdgeCount {
4659                etype,
4660                src,
4661                dst,
4662                count,
4663            } => {
4664                if rec.is_multiplicity_decl() {
4665                    self.multiplicity = true;
4666                } else {
4667                    Arc::make_mut(&mut self.edge_props).set(
4668                        *etype,
4669                        *src,
4670                        *dst,
4671                        EDGE_COUNT_PROP,
4672                        Value::Int(*count as i64),
4673                    );
4674                }
4675            }
4676            // History markers carry no replay state — rules re-derive edges
4677            // deterministically on open/replay. Skip unconditionally.
4678            WalRecord::DerivedEdgeAdded { .. } | WalRecord::DerivedEdgeRetracted { .. } => {}
4679            // ── rename_node ──────────────────────────────────────────────────
4680            WalRecord::RenameNode { old_key, new_key } => {
4681                // Recovery-safe: if old_key is already gone (key was renamed
4682                // by a snapshot or a prior replay frame), skip cleanly.
4683                if self.ids.get(old_key).is_none() {
4684                    return Ok(());
4685                }
4686                // The rename only updates the key-table; the dense id, all
4687                // topo edges, props, labels, and rule state are id-indexed and
4688                // require no change.
4689                Arc::make_mut(&mut self.ids)
4690                    .rename(old_key, new_key)
4691                    .map_err(|e| GraphError::Corrupt {
4692                        detail: format!("wal replay RenameNode {old_key}→{new_key}: {e}"),
4693                    })?;
4694            }
4695        }
4696        Ok(())
4697    }
4698
4699    /// Intern `s` in `syms` and emit a WAL `Intern` record so `*Id` records
4700    /// replay on WAL-only `open_at` (no snapshot intern table). Apply is
4701    /// idempotent when the string is already bound. Always emit: after
4702    /// `snapshot()` the WAL is truncated and live intern is not on disk.
4703    fn intern_wal(&mut self, s: &str) -> (u32, WalRecord) {
4704        let id = if let Some(id) = self.syms.get(s) {
4705            id
4706        } else {
4707            Arc::make_mut(&mut self.syms).intern(s)
4708        };
4709        (
4710            id,
4711            WalRecord::Intern {
4712                id,
4713                text: s.to_string(),
4714            },
4715        )
4716    }
4717
4718    /// Rewrite user-facing records into dense-id records. On `Err`, no live
4719    /// state is left mutated: speculative interns made while building the
4720    /// output are rolled back, so a later successful mutation cannot log an
4721    /// `Intern` record whose id replay would never reproduce.
4722    fn rewrite_wal_dense(&mut self, recs: Vec<WalRecord>) -> Result<Vec<WalRecord>> {
4723        self.rewrite_wal_dense_planned(recs.into_iter().map(PlannedRec::Rec).collect())
4724    }
4725
4726    /// [`rewrite_wal_dense`](Self::rewrite_wal_dense) for a frame that still
4727    /// carries [`PlannedRec::DuplicateCount`] entries — the shape a batch
4728    /// produces, where a duplicate's count can only be named once this pass has
4729    /// assigned the frame's own ids.
4730    fn rewrite_wal_dense_planned(&mut self, recs: Vec<PlannedRec>) -> Result<Vec<WalRecord>> {
4731        let syms_checkpoint = self.syms.len();
4732        let result = self.rewrite_wal_dense_inner(recs);
4733        if result.is_err() {
4734            Arc::make_mut(&mut self.syms).truncate(syms_checkpoint);
4735        }
4736        result
4737    }
4738
4739    fn rewrite_wal_dense_inner(&mut self, recs: Vec<PlannedRec>) -> Result<Vec<WalRecord>> {
4740        let mut out = Vec::with_capacity(recs.len());
4741        // Node ids allocated by later apply(InsertNodeId) in this same batch.
4742        let mut pending: std::collections::HashMap<String, u32> = std::collections::HashMap::new();
4743        // Namespace of each node inserted earlier in this same frame, so a SET
4744        // on a node this frame created is measured against the namespace it was
4745        // created in rather than against the store, where it does not exist yet.
4746        let mut pending_ns: std::collections::HashMap<String, String> =
4747            std::collections::HashMap::new();
4748        let mut interned = std::collections::HashSet::<u32>::new();
4749        let mut next = u32::try_from(self.ids.len()).map_err(|_| GraphError::Corrupt {
4750            detail: "id space exhausted".into(),
4751        })?;
4752        // Insert counts this frame has already raised. `edge_insert_count`
4753        // reads committed state, which cannot see a count queued earlier in
4754        // this same frame, so N duplicates of one pair would otherwise all
4755        // compute `committed + 1` and the last would win.
4756        let mut pending_counts: HashMap<(u32, u32, u32), u64> = HashMap::new();
4757        let lookup = |ids: &IdMap,
4758                      pending: &std::collections::HashMap<String, u32>,
4759                      key: &str|
4760         -> Option<u32> { ids.get(key).or_else(|| pending.get(key).copied()) };
4761        for rec in recs {
4762            // A duplicate insert's count, resolved here and nowhere else.
4763            //
4764            // This is the only pass that knows the frame's own ids: a node
4765            // created earlier in the same frame has no dense id until the
4766            // `InsertNodeId` above allocates one, and an edge type first used in
4767            // this frame is not in `syms` until `intern_wal` puts it there.
4768            // Resolving the count in the batch's validate pass instead — where
4769            // it used to live — meant that a duplicate whose endpoints or type
4770            // were created in the same frame silently produced no count at all,
4771            // which is exactly the shape a mirror rebuild writes (defect #24).
4772            let rec = match rec {
4773                PlannedRec::Rec(rec) => rec,
4774                PlannedRec::DuplicateCount {
4775                    edge_type,
4776                    src_key,
4777                    dst_key,
4778                } => {
4779                    let (etype, intern) = self.intern_wal(&edge_type);
4780                    if interned.insert(etype) {
4781                        out.push(intern);
4782                    }
4783                    let src = lookup(&self.ids, &pending, &src_key).ok_or_else(|| {
4784                        GraphError::Corrupt {
4785                            detail: format!("dense WAL rewrite missing src {src_key}"),
4786                        }
4787                    })?;
4788                    let dst = lookup(&self.ids, &pending, &dst_key).ok_or_else(|| {
4789                        GraphError::Corrupt {
4790                            detail: format!("dense WAL rewrite missing dst {dst_key}"),
4791                        }
4792                    })?;
4793                    let count = pending_counts
4794                        .get(&(etype, src, dst))
4795                        .copied()
4796                        .unwrap_or_else(|| self.edge_insert_count(etype, src, dst))
4797                        .saturating_add(1);
4798                    pending_counts.insert((etype, src, dst), count);
4799                    out.push(WalRecord::SetEdgeCount {
4800                        etype,
4801                        src,
4802                        dst,
4803                        count,
4804                    });
4805                    continue;
4806                }
4807            };
4808            match rec {
4809                WalRecord::InsertNode { label, key, props } => {
4810                    // Namespace validation and normalisation, on the one seam
4811                    // every user-visible node insert passes through: insert_node,
4812                    // a batch, ingest, Cypher CREATE and MERGE all arrive here
4813                    // before the WAL append, and replay never does.
4814                    let (props, ns_name) = Self::normalise_insert_ns(&key, props)?;
4815                    pending_ns.insert(key.clone(), ns_name);
4816                    let (label_id, intern) = self.intern_wal(&label);
4817                    if interned.insert(label_id) {
4818                        out.push(intern);
4819                    }
4820                    let mut props_id = Vec::with_capacity(props.len());
4821                    for (field, value) in props {
4822                        let (field_id, intern) = self.intern_wal(&field);
4823                        if interned.insert(field_id) {
4824                            out.push(intern);
4825                        }
4826                        props_id.push((field_id, value));
4827                    }
4828                    if lookup(&self.ids, &pending, &key).is_none() {
4829                        pending.insert(key.clone(), next);
4830                        next = next.checked_add(1).ok_or_else(|| GraphError::Corrupt {
4831                            detail: "id space exhausted".into(),
4832                        })?;
4833                    }
4834                    out.push(WalRecord::InsertNodeId {
4835                        label: label_id,
4836                        key,
4837                        props: props_id,
4838                    });
4839                }
4840                WalRecord::SetProp { key, field, value } => {
4841                    // A namespace is set at insert and fixed after: the write is
4842                    // refused when it would move the node, and dropped when it
4843                    // names the namespace the node is already in. Checked here
4844                    // so set_prop, a batch, Cypher SET/MERGE and every upsert
4845                    // that merges props get the same answer.
4846                    if field == NS_PROP {
4847                        let Value::Str(ref to) = value else {
4848                            return Err(GraphError::RuleInvalid {
4849                                detail: format!(
4850                                    "node {key}: {NS_PROP} must be a string naming a namespace, \
4851                                     got {value:?}"
4852                                ),
4853                            });
4854                        };
4855                        let from = pending_ns
4856                            .get(&key)
4857                            .cloned()
4858                            .or_else(|| self.namespace_of(&key))
4859                            .unwrap_or_else(|| NS_DEFAULT.to_string());
4860                        let to = to.clone();
4861                        if to != from {
4862                            return Err(GraphError::NamespaceImmutable {
4863                                key: key.clone(),
4864                                from,
4865                                to,
4866                            });
4867                        }
4868                        continue;
4869                    }
4870                    let id =
4871                        lookup(&self.ids, &pending, &key).ok_or_else(|| GraphError::Corrupt {
4872                            detail: format!("dense WAL rewrite missing key {key}"),
4873                        })?;
4874                    let (field_id, intern) = self.intern_wal(&field);
4875                    if interned.insert(field_id) {
4876                        out.push(intern);
4877                    }
4878                    out.push(WalRecord::SetPropId {
4879                        id,
4880                        field: field_id,
4881                        value,
4882                    });
4883                }
4884                WalRecord::InsertEdge {
4885                    edge_type,
4886                    src_key,
4887                    dst_key,
4888                } => {
4889                    let (etype, intern) = self.intern_wal(&edge_type);
4890                    if interned.insert(etype) {
4891                        out.push(intern);
4892                    }
4893                    let src = lookup(&self.ids, &pending, &src_key).ok_or_else(|| {
4894                        GraphError::Corrupt {
4895                            detail: format!("dense WAL rewrite missing src {src_key}"),
4896                        }
4897                    })?;
4898                    let dst = lookup(&self.ids, &pending, &dst_key).ok_or_else(|| {
4899                        GraphError::Corrupt {
4900                            detail: format!("dense WAL rewrite missing dst {dst_key}"),
4901                        }
4902                    })?;
4903                    out.push(WalRecord::InsertEdgeId { etype, src, dst });
4904                }
4905                WalRecord::RenameNode {
4906                    ref old_key,
4907                    ref new_key,
4908                } => {
4909                    // Track the rename in `pending` so subsequent InsertEdge /
4910                    // SetProp records in this batch can resolve the new key.
4911                    let id = lookup(&self.ids, &pending, old_key).ok_or_else(|| {
4912                        GraphError::Corrupt {
4913                            detail: format!(
4914                                "dense WAL rewrite: RenameNode old key {old_key} not found"
4915                            ),
4916                        }
4917                    })?;
4918                    pending.remove(old_key.as_str());
4919                    pending.insert(new_key.clone(), id);
4920                    out.push(rec);
4921                }
4922                // # Symbol-order invariant (load-bearing)
4923                //
4924                // Write-time and replay-time symbol assignment must agree: every
4925                // symbol in a `Batch` frame has to receive the same dense id when
4926                // the frame's records are replayed in order as it received when
4927                // the frame was written.
4928                //
4929                // A rule's backfill interns its `edge_type` lazily
4930                // (`core_rules::engine`, every `g.syms.intern(&def.edge_type)`
4931                // site), and that backfill runs from `apply` — during the
4932                // `CreateRule` record itself, and again from any later
4933                // `InsertNodeId` in the same frame that makes the rule fire. At
4934                // write time the whole batch is rewritten before any of it is
4935                // applied, so a later `InsertEdge` in the same batch would win the
4936                // lower id for its edge type; on replay the rule's lazy intern
4937                // gets there first and steals it, and the `Intern` record fails at
4938                // the `wal intern assigned …` check in `apply`.
4939                //
4940                // Pre-interning the rule's `edge_type` here, and emitting its
4941                // `Intern` record ahead of the `CreateRule` record, makes both
4942                // orders identical. `weight_prop` needs no pre-intern:
4943                // `EdgeProps::set` keys props by `String`, never through the
4944                // interner. `via_edge` needs none either: via-hop rules resolve it
4945                // with `syms.get` and skip when it is absent.
4946                //
4947                // `RebuildRule` and `DeleteRule` need no such handling here:
4948                // `RebuildRule` has no `BatchOp` variant, so it never appears
4949                // inside a `Batch` today — it is only ever issued as its own
4950                // standalone commit (`rebuild_rule`, or the auto-rebuild path
4951                // that logs it as a second commit after the triggering op).
4952                // `DeleteRule` does have a `BatchOp` variant and can appear
4953                // inside a `Batch`, but it carries only a rule `name` — no
4954                // `edge_type` or other symbol that needs pre-interning — so
4955                // only `CreateRule` needs this arm.
4956                WalRecord::CreateRule { ref def_bytes } => {
4957                    let def = decode_rule_def(def_bytes).map_err(|e| GraphError::Corrupt {
4958                        detail: format!("CreateRule def_bytes deserialize failed: {e}"),
4959                    })?;
4960                    let (etype, intern) = self.intern_wal(&def.edge_type);
4961                    if interned.insert(etype) {
4962                        out.push(intern);
4963                    }
4964                    out.push(rec);
4965                }
4966                other => out.push(other),
4967            }
4968        }
4969        Ok(out)
4970    }
4971
4972    fn log_dense(&mut self, recs: Vec<WalRecord>) -> Result<()> {
4973        let recs = self.rewrite_wal_dense(recs)?;
4974        match recs.len() {
4975            0 => Ok(()),
4976            1 => self.log_then_apply(recs.into_iter().next().unwrap()),
4977            _ => self.log_then_apply(WalRecord::Batch(recs)),
4978        }
4979    }
4980
4981    /// Durable write, then notify the event sink. Replay (`apply` during
4982    /// `open`) never enters this function, so it is the replay-silent seam.
4983    /// Record that the commit occupying `frame_index` happened now, and append
4984    /// those 16 bytes to the sidecar.
4985    ///
4986    /// `frame_index` is the **global 0-based WAL frame index** of the commit's
4987    /// own record — the space every history surface addresses — taken from
4988    /// [`wal_frames_written`](GraphDb::wal_frames_written) before the append
4989    /// that puts the record there.
4990    ///
4991    /// **It is deliberately not derived from `commit_seq`.** A commit is not a
4992    /// frame: one whose rules fire appends a second frame for the derived-edge
4993    /// history marker, so `commit_seq - 1` falls one frame further behind per
4994    /// rule-firing commit and every date resolves to an ever-earlier graph.
4995    /// That was the shipped behaviour through v0.6.11 and it failed silently,
4996    /// because an older graph is a plausible answer rather than an error.
4997    ///
4998    /// Called from exactly one place — `log_then_apply_with`, immediately after
4999    /// `commit_seq` is incremented. Every write path in the engine funnels
5000    /// through that function, and replay deliberately does not: `apply_frames`
5001    /// re-applies commits that already happened, so stamping there would record
5002    /// replay time as commit time.
5003    ///
5004    /// **This is the engine's only wall-clock read.** Everything else uses
5005    /// `Instant`, which is monotonic and not a date.
5006    ///
5007    /// Failure is swallowed on purpose. The sidecar is not part of the WAL or
5008    /// the snapshot, so a failed append must not fail a commit that is already
5009    /// durable — it costs a date, not data. The map is marked poisoned so the
5010    /// gap is reported rather than resolved across.
5011    fn stamp_commit_time(&mut self, frame_index: u64) {
5012        if self.commit_times_poisoned {
5013            return;
5014        }
5015        let unix_ms = match self.commit_time_override {
5016            Some(ms) => ms,
5017            None => {
5018                let Ok(now) = std::time::SystemTime::now().duration_since(std::time::UNIX_EPOCH)
5019                else {
5020                    // A clock before 1970. Refuse to invent a timestamp.
5021                    self.commit_times_poisoned = true;
5022                    return;
5023                };
5024                now.as_millis() as i64
5025            }
5026        };
5027        let first = self.commit_times.is_empty();
5028        self.commit_times.push(frame_index, unix_ms);
5029
5030        let wrote = if first {
5031            self.fs.write_atomic(
5032                FileId::CommitTimes,
5033                &core_storage::commit_times::encode(&self.commit_times),
5034            )
5035        } else {
5036            self.fs.append(
5037                FileId::CommitTimes,
5038                &core_storage::commit_times::encode_entry(frame_index, unix_ms),
5039            )
5040        };
5041        if wrote.is_err() {
5042            self.commit_times_poisoned = true;
5043        }
5044    }
5045
5046    /// Read the time sidecar from disk into this handle.
5047    ///
5048    /// Absent is the normal case for any store written before v0.6.11 and is
5049    /// not an error; unreadable is recorded so date queries can say "damaged"
5050    /// rather than "none recorded".
5051    ///
5052    /// Called at open **and** by `refresh` when a peer's frames are absorbed.
5053    /// Both, because the map is a file another process appends to: a handle
5054    /// that raises its frame cursor to include a peer's commits while holding a
5055    /// stale map would answer dates from a prefix of the truth — and, if its
5056    /// own map were still empty, would rewrite the whole file with one entry
5057    /// and destroy the peer's.
5058    fn load_commit_times_from_fs(&mut self) {
5059        match self.fs.read(FileId::CommitTimes) {
5060            Ok(bytes) if bytes.is_empty() => {}
5061            Ok(bytes) => {
5062                // A map from an older format version is discarded, not
5063                // reported as damage and not reinterpreted. Its entries were
5064                // written correctly against a different meaning of the number
5065                // — see `COMMIT_TIMES_VERSION` — and reading them in this
5066                // build's space would resolve dates onto unrelated commits.
5067                // Leaving the map empty makes the store answer
5068                // `NoRecordedTime`, which is the truth: it records no times
5069                // this build can use, and the next commit starts a usable map.
5070                if core_storage::commit_times::superseded_version(&bytes).is_some() {
5071                    return;
5072                }
5073                match core_storage::commit_times::decode(&bytes) {
5074                    Ok(t) => self.commit_times = t,
5075                    Err(_) => self.commit_times_poisoned = true,
5076                }
5077            }
5078            Err(_) => {}
5079        }
5080    }
5081
5082    /// Rewrite the sidecar from memory. Used after truncation, which is the one
5083    /// operation that cannot be expressed as an append.
5084    fn rewrite_commit_times(&mut self) {
5085        if self.commit_times_poisoned {
5086            return;
5087        }
5088        if self
5089            .fs
5090            .write_atomic(
5091                FileId::CommitTimes,
5092                &core_storage::commit_times::encode(&self.commit_times),
5093            )
5094            .is_err()
5095        {
5096            self.commit_times_poisoned = true;
5097        }
5098    }
5099
5100    /// The greatest commit whose recorded time is at or before `unix_ms`.
5101    ///
5102    /// Errors name what they can answer instead of guessing a commit:
5103    /// `Corrupt` when the sidecar would not decode, `NoRecordedTime` when the
5104    /// store records none, `TimeBeforeFloor` when the instant predates the
5105    /// oldest entry, and `CommitOutOfRange` when the answer falls below the WAL
5106    /// horizon and so cannot be replayed.
5107    ///
5108    /// The answer is a **0-based frame index**, ready to hand to `edges_at` or
5109    /// `was_linked` without adjustment.
5110    pub fn resolve_instant(&self, unix_ms: i64) -> Result<u64> {
5111        if self.commit_times_poisoned {
5112            return Err(GraphError::Corrupt {
5113                detail: "commit_times.bin will not decode; date queries are \
5114                         unavailable on this store"
5115                    .into(),
5116            });
5117        }
5118        let at = self
5119            .commit_times
5120            .resolve_instant(unix_ms, self.wal_horizon_floor)?;
5121        // The map outlives the history it describes. A truncating snapshot folds
5122        // the WAL and discards it, so entries can name commits the engine can no
5123        // longer replay — the floor check above catches pruning, and this catches
5124        // discarding. Returning an index the caller's next call will reject is a
5125        // two-step error where one will do, and `resolve_date` is public: it
5126        // either hands back a usable index or refuses.
5127        let total = self.wal_total_commits()?;
5128        if at >= total {
5129            return Err(GraphError::CommitOutOfRange {
5130                commit: at,
5131                total,
5132                floor: self.wal_horizon_floor,
5133            });
5134        }
5135        Ok(at)
5136    }
5137
5138    /// Record subsequent commits as having happened at `unix_ms`, or pass
5139    /// `None` to go back to the system clock.
5140    ///
5141    /// For **backfilled history**: a mirror importing rows that already carry
5142    /// their own timestamps, or a replay of events that happened months ago.
5143    /// Without this every such commit is stamped "now", so a store holding a
5144    /// year of imported history answers every date question with
5145    /// `TimeBeforeFloor` — the data is there and no date reaches it.
5146    ///
5147    /// Sticky until changed or cleared, because a day of backfilled rows
5148    /// genuinely shares one instant.
5149    ///
5150    /// **Import in chronological order.** A supplied instant earlier than
5151    /// anything already recorded is refused with
5152    /// [`GraphError::CommitTimeNotMonotonic`], because resolution walks commit
5153    /// order: a later commit carrying an earlier time would silently widen every
5154    /// answer after it. Equal is allowed — that is what a shared day means. The
5155    /// live clock is never held to this, so an NTP step backwards still commits.
5156    ///
5157    /// Deliberately **not** exposed over HTTP or MCP: asserting when a commit
5158    /// happened rewrites the store's apparent history, which is not something a
5159    /// role token models. It is an embedding-caller's operation.
5160    pub fn record_commits_at(&mut self, unix_ms: Option<i64>) -> Result<()> {
5161        if self.read_only {
5162            return Err(GraphError::ReadOnly);
5163        }
5164        if let Some(ms) = unix_ms {
5165            if self.commit_times_poisoned {
5166                return Err(GraphError::Corrupt {
5167                    detail: "commit_times.bin will not decode; this store cannot \
5168                             record an asserted commit time"
5169                        .into(),
5170                });
5171            }
5172            if let Some(newest) = self.commit_times.max_ms() {
5173                if ms < newest {
5174                    return Err(GraphError::CommitTimeNotMonotonic {
5175                        supplied_ms: ms,
5176                        newest_ms: newest,
5177                    });
5178                }
5179            }
5180        }
5181        self.commit_time_override = unix_ms;
5182        Ok(())
5183    }
5184
5185    /// The instant subsequent commits are being recorded at, when one is set.
5186    pub fn commit_time_override(&self) -> Option<i64> {
5187        self.commit_time_override
5188    }
5189
5190    /// [`Self::edges_at`] addressed by an instant rather than a commit index.
5191    ///
5192    /// Resolves through [`Self::resolve_instant`] — the last commit at or
5193    /// before the instant — then answers exactly as the commit-indexed call
5194    /// does. A store that records no times refuses by name; it never guesses.
5195    pub fn edges_at_instant(&self, key: &str, unix_ms: i64) -> Result<Vec<EdgeAt>> {
5196        let commit = self.resolve_instant(unix_ms)?;
5197        self.edges_at(key, commit)
5198    }
5199
5200    /// [`Self::was_linked`] addressed by an instant rather than a commit index.
5201    pub fn was_linked_at_instant(
5202        &self,
5203        a: &str,
5204        b: &str,
5205        edge_type: &str,
5206        unix_ms: i64,
5207    ) -> Result<bool> {
5208        let commit = self.resolve_instant(unix_ms)?;
5209        self.was_linked(a, b, edge_type, commit)
5210    }
5211
5212    /// Parse an RFC 3339 instant (or a bare `YYYY-MM-DD`) and resolve it.
5213    ///
5214    /// The one place every caller-facing surface converts a date string, so
5215    /// HTTP, MCP, Python and the CLI cannot drift in what they accept.
5216    pub fn resolve_date(&self, s: &str) -> Result<u64> {
5217        // The **end** of what the string denotes. A bare date is a day, so it
5218        // resolves to the last commit at or before that day's end — resolving to
5219        // the midnight that starts it would exclude everything that happened on
5220        // the date the caller asked about.
5221        let ms = core_storage::commit_times::parse_rfc3339_end_ms(s).ok_or_else(|| {
5222            GraphError::QueryError {
5223                detail: format!(
5224                    "could not parse {s:?} as a date; expected RFC 3339 \
5225                     (2026-06-19, or 2026-06-19T12:00:00Z)"
5226                ),
5227            }
5228        })?;
5229        self.resolve_instant(ms)
5230    }
5231
5232    /// The recorded wall-clock time of `commit`, when the sidecar holds one.
5233    ///
5234    /// `commit` is a **0-based frame index** — the space `edges_at`,
5235    /// `was_linked` and the history events use, not the 1-based `commit_seq`.
5236    pub fn commit_time_ms(&self, commit: u64) -> Option<i64> {
5237        if self.commit_times_poisoned {
5238            return None;
5239        }
5240        self.commit_times.time_of(commit)
5241    }
5242
5243    fn log_then_apply(&mut self, rec: WalRecord) -> Result<()> {
5244        self.log_then_apply_with(rec, None, self.fsync)
5245    }
5246
5247    /// Whether this frame must fsync under `policy`.
5248    ///
5249    /// Batched contract: user-visible batches (>1 mutation) fsync; single
5250    /// mutations do not. The dense rewrite wraps a single mutation in a
5251    /// `Batch([Intern.., <one *Id record>])`, so `Intern` records are excluded
5252    /// from the count — removing that filter would make every single-op write
5253    /// fsync under Batched (or, if the threshold were raised instead, skip a
5254    /// needed fsync for real two-op batches).
5255    fn wal_needs_sync(policy: FsyncPolicy, rec: &WalRecord) -> bool {
5256        match policy {
5257            FsyncPolicy::Relaxed => false,
5258            FsyncPolicy::Strict => true,
5259            FsyncPolicy::Batched => match rec {
5260                // Intern + one mutation is the single-op rewrite, not a user batch.
5261                WalRecord::Batch(inner) => {
5262                    inner
5263                        .iter()
5264                        .filter(|r| !matches!(r, WalRecord::Intern { .. }))
5265                        .count()
5266                        > 1
5267                }
5268                _ => false,
5269            },
5270        }
5271    }
5272
5273    /// # Apply-infallibility invariant (load-bearing)
5274    ///
5275    /// The ordering is: WAL append → fsync → apply. If `apply` returned `Err`
5276    /// for a `Batch` frame after a successful WAL write, the WAL would contain
5277    /// the full frame while in-memory state would reflect only the ops before
5278    /// the failure. On reopen, WAL replay would then apply the entire batch —
5279    /// diverging permanently from what the pre-crash process had in memory.
5280    ///
5281    /// For `Batch` frames this situation cannot arise because:
5282    /// - All validation runs via `commit_logged_batch`/`MutPreview` **before**
5283    ///   the WAL write. `MutPreview` uses the same `&mut self` that apply will
5284    ///   use, with no concurrent mutation between validation exit and apply entry.
5285    /// - Every `apply` arm for a validated op is either infallible by construction
5286    ///   (`InsertNode`, `RemoveProp`, `DeleteEdge`, `DeleteNode`), has idempotency
5287    ///   guards that return `Ok(())` (`CreateRule`, `DeleteRule`), or is
5288    ///   guaranteed-present by validation (`InsertEdge`/`SetProp` key lookups).
5289    /// - `on_node_changed` and `on_node_removed` return `()` — never `Err`.
5290    ///
5291    /// A `debug_assert!` below fires in debug builds if `apply` ever returns
5292    /// `Err` for a `Batch` frame, making any future regression immediately visible
5293    /// in tests rather than silently diverging crash-recovery behaviour.
5294    fn log_then_apply_with(
5295        &mut self,
5296        rec: WalRecord,
5297        ingest: Option<(String, usize)>,
5298        policy: FsyncPolicy,
5299    ) -> Result<()> {
5300        // Read-only guard: as-of instances must never write the WAL.
5301        if self.read_only {
5302            return Err(GraphError::ReadOnly);
5303        }
5304        // Degraded guard: fsync failure left WAL truncated, or a refresh failed
5305        // partway; in-memory state is ahead of (or out of step with) the
5306        // on-disk WAL, so further mutations would deepen the divergence.
5307        // Reopen the database to recover.  Checked before the lock guard: this
5308        // is the more serious condition and the more useful error.
5309        if self.degraded {
5310            return Err(GraphError::Io(std::io::Error::other(
5311                "database degraded after group-commit fsync failure; reopen required",
5312            )));
5313        }
5314        // Cross-process guard: this write scope asked for the store's write
5315        // lock and did not get it. Writing anyway would append frames on top of
5316        // a WAL another process is extending, so refuse instead.
5317        if self.lock_denied {
5318            return Err(GraphError::Busy { holder: None });
5319        }
5320        // Ensure retained provenance bytes are decoded into the live mutable
5321        // fields before any mutation touches self.engine.provenance.  This is a
5322        // no-op if provenance was never stored (fresh store) or has already been
5323        // consumed (subsequent mutations).  WAL replay calls apply() directly
5324        // and is covered by consume_retained_state_eager before replay.
5325        self.ensure_v8_base_sections_loaded();
5326        self.engine.ensure_provenance_loaded_mut();
5327        // Invariant (I-1): no stale deltas may enter from a previous apply.
5328        // If any engine method ever accumulates deltas before erroring, they would
5329        // contaminate the *next* commit's event stream. This assert fires in debug
5330        // builds, making any future regression visible at the earliest point.
5331        debug_assert_eq!(
5332            self.engine.pending_delta_count(),
5333            0,
5334            "stale engine deltas at log_then_apply_with entry — \
5335             a previous apply arm may have accumulated deltas before erroring; \
5336             the caller must drain_deltas() on any error path before returning"
5337        );
5338        let frame = encode_record(&rec);
5339        self.fs.append(FileId::Wal, &frame)?;
5340        // The cursor advances by exactly the bytes appended: these frames are
5341        // ours and already applied, so a later refresh must not replay them.
5342        self.wal_consumed += frame.len() as u64;
5343        self.wal_frames_written += 1;
5344        if Self::wal_needs_sync(policy, &rec) {
5345            self.fs.sync(FileId::Wal)?;
5346        }
5347        // Marker writing always needs the engine deltas, but the engine only
5348        // accumulates them when emit_deltas is true (normally gated on subscribers
5349        // or views being present).  Enable emission for this apply if it is
5350        // currently off, then restore the original state unconditionally via an
5351        // RAII guard — this prevents a panic in apply() from leaking the flag.
5352        // The same guard resets the engine's transient chaining state. A panic
5353        // unwinding out of a rule hook would otherwise leave `chain_depth`
5354        // non-zero, which makes every later `begin_chain` decide chaining is
5355        // already running and silently switch it off for good.
5356        struct RestoreEmitDeltas(*mut RuleEngine, bool);
5357        impl Drop for RestoreEmitDeltas {
5358            fn drop(&mut self) {
5359                // SAFETY: pointer into self (GraphDb); guard is dropped within
5360                // this frame before log_then_apply_with returns.
5361                unsafe {
5362                    (*self.0).set_emit_deltas(self.1);
5363                    (*self.0).reset_chain_state();
5364                }
5365            }
5366        }
5367        let original_emit = self.engine.emit_deltas();
5368        if !original_emit {
5369            self.engine.set_emit_deltas(true);
5370        }
5371        // SAFETY: raw pointer into self; guard dropped within this frame.
5372        let _emit_guard = RestoreEmitDeltas(&mut self.engine as *mut _, original_emit);
5373
5374        let apply_result = self.apply(&rec);
5375        // For Batch frames, post-validation apply must be infallible (see above).
5376        // A debug_assert here catches any future change that makes apply fallible
5377        // before the caller notices via silent WAL/memory divergence.
5378        if matches!(&rec, WalRecord::Batch(_)) {
5379            debug_assert!(
5380                apply_result.is_ok(),
5381                "Batch apply returned Err after successful WAL write — \
5382                 the validate-then-apply invariant has been violated; \
5383                 see log_then_apply_with invariant doc"
5384            );
5385        }
5386        if apply_result.is_err() {
5387            // Discard any partial deltas accumulated by the failed apply.
5388            // They must not ride the next commit's event stream (I-1).
5389            // _emit_guard restores emit_deltas on drop automatically.
5390            let _ = self.engine.drain_deltas();
5391            let _ = self.engine.take_rebuild_needed();
5392            apply_result?;
5393        }
5394        self.commit_seq += 1;
5395        let seq = self.commit_seq;
5396        // Update per-node last-change map for the committed record.
5397        // Must happen after commit_seq is incremented so the seq is correct.
5398        self.update_last_change_from_rec(&rec, seq);
5399        // Drain engine deltas and distribute to subscribers before the existing
5400        // MutationEvent sink fires — both happen post-fsync, post-apply.
5401        // _emit_guard restores emit_deltas after this line when it drops.
5402        let engine_deltas = self.engine.drain_deltas();
5403
5404        // Append history-marker WAL records for any derived-edge changes so
5405        // that `edge_history` and `was_linked` can surface rule-attributed
5406        // events. Markers are STATE NO-OPS during replay; they are written
5407        // without an additional fsync (the triggering commit's sync already
5408        // happened; the next commit's sync covers these lazily).
5409        if !engine_deltas.is_empty() {
5410            let markers: Vec<WalRecord> = engine_deltas
5411                .iter()
5412                .map(|d| {
5413                    if d.fired {
5414                        WalRecord::DerivedEdgeAdded {
5415                            rule: d.rule.clone(),
5416                            edge_type: d.edge_type.clone(),
5417                            src_key: d.src_key.clone(),
5418                            dst_key: d.dst_key.clone(),
5419                        }
5420                    } else {
5421                        WalRecord::DerivedEdgeRetracted {
5422                            rule: d.rule.clone(),
5423                            edge_type: d.edge_type.clone(),
5424                            src_key: d.src_key.clone(),
5425                            dst_key: d.dst_key.clone(),
5426                        }
5427                    }
5428                })
5429                .collect();
5430            let marker_frame = if markers.len() == 1 {
5431                markers.into_iter().next().unwrap()
5432            } else {
5433                WalRecord::Batch(markers)
5434            };
5435            // Ignore append errors: markers are best-effort history
5436            // annotations. Losing them does not affect state correctness.
5437            // The cursor only advances when the bytes actually landed.
5438            let marker_bytes = encode_record(&marker_frame);
5439            if self.fs.append(FileId::Wal, &marker_bytes).is_ok() {
5440                self.wal_consumed += marker_bytes.len() as u64;
5441                // A marker is a state no-op during replay but it is not an
5442                // index no-op: it occupies a frame that every history surface
5443                // counts. Missing this increment is the whole of defect 1.
5444                self.wal_frames_written += 1;
5445            }
5446        }
5447
5448        // Stamp the commit against the **last** frame it wrote.
5449        //
5450        // `edges_at` and `was_linked` reconstruct derived edges by reading the
5451        // history markers out of the WAL, not by re-running rules over a
5452        // prefix. A commit's complete state — its record *and* the edges its
5453        // rules derived — is therefore only reached at its marker frame, so
5454        // that is the frame a date naming this commit must resolve to. Stamping
5455        // the record's own frame would answer every date with the graph as it
5456        // was one derivation short.
5457        //
5458        // This runs after the marker append for that reason, and it is still
5459        // the single stamping site: one commit, one entry.
5460        self.stamp_commit_time(self.wal_frames_written - 1);
5461
5462        // Record MVCC CommitDelta for the epoch reader.  The WAL record is
5463        // stored as-is (including any nested Batch / Intern records); the
5464        // ReaderSnapshot's apply_one function handles all variants.
5465        {
5466            let derived_inserts = engine_deltas
5467                .iter()
5468                .filter(|d| d.fired)
5469                .map(|d| (d.etype_sym, d.src_id, d.dst_id))
5470                .collect();
5471            let derived_deletes = engine_deltas
5472                .iter()
5473                .filter(|d| !d.fired)
5474                .map(|d| (d.etype_sym, d.src_id, d.dst_id))
5475                .collect();
5476            let delta = Arc::new(crate::reader::CommitDelta {
5477                records: vec![rec.clone()],
5478                derived_inserts,
5479                derived_deletes,
5480            });
5481            self.delta_tail.push(delta);
5482            self.commits_since_fold += 1;
5483            if self.commits_since_fold >= crate::reader::FOLD_EVERY_K {
5484                self.fold_now();
5485            }
5486        }
5487
5488        if self.defer_events {
5489            // Group-commit drain thread: hold events until after the group
5490            // fsync so subscribers only observe durable data (R2).
5491            self.deferred_events.push(DeferredEvent {
5492                rec: rec.clone(),
5493                engine_deltas,
5494                seq,
5495                ingest,
5496            });
5497        } else {
5498            self.distribute_events(&rec, &engine_deltas, seq);
5499            self.emit_committed(&rec, ingest);
5500        }
5501        // Drift is only known after apply, so auto-rebuild cannot join the
5502        // triggering op's WAL frame. Issue RebuildRule as a second commit.
5503        // Skip when `rec` is itself RebuildRule: rebuild resets drift, so a
5504        // retrigger loop is impossible if the fit succeeded, but we still
5505        // drain the flag so a leftover cannot re-enter.
5506        // One slice of any outstanding vector-index build rides here too, so a
5507        // store that is being written to finishes its build without anyone
5508        // calling `pump_index_build`. A rule that becomes whole joins the same
5509        // RebuildRule loop below.
5510        let mut rebuilds = self.engine.take_rebuild_needed();
5511        if !matches!(&rec, WalRecord::RebuildRule { .. }) {
5512            // Not after `CreateRule`: that record's own apply already did the
5513            // rule's first slice, and pumping again here would make one
5514            // `create_rule` call do two slices' work under one lock.
5515            // Nothing pending is the overwhelmingly common case and must cost
5516            // a map lookup, not an engine swap: a store being written to has
5517            // long since populated its indexes, so the `pump_index_build`
5518            // entry point owns the not-yet-populated case on its own.
5519            if !is_create_rule_frame(&rec) && !self.engine.builds_in_progress().is_empty() {
5520                rebuilds.extend(self.pump_one_slice().into_iter().map(|b| b.rule));
5521            }
5522            let mut failed = Vec::new();
5523            for name in rebuilds {
5524                if self.engine.rules().any(|r| r.name == name) {
5525                    // User op is already durable. A failed second commit must
5526                    // not surface as the caller's error.
5527                    if let Err(e) =
5528                        self.log_then_apply(WalRecord::RebuildRule { name: name.clone() })
5529                    {
5530                        eprintln!(
5531                            "auto-rebuild of rule {name:?} failed after durable user commit: {e}"
5532                        );
5533                        failed.push(name);
5534                    }
5535                }
5536            }
5537            for name in failed {
5538                self.engine.queue_rebuild_needed(name);
5539            }
5540        }
5541        Ok(())
5542    }
5543
5544    /// Install a post-commit hook. Replaces any previous sink.
5545    ///
5546    /// The sink runs inside `log_then_apply` after a successful
5547    /// durable commit, while the caller still holds `&mut self`. When this
5548    /// database is behind a [`crate::SharedDb`], that means the **write
5549    /// guard is held**. The sink must never call `read` / `write` (or any
5550    /// other method) on the same `SharedDb` — the `RwLock` is not
5551    /// re-entrant and doing so deadlocks. The sink is `Send + Sync`;
5552    /// `std::sync::mpsc::Sender` is not `Sync` and will not type-check.
5553    /// Intended examples: `std::sync::mpsc::SyncSender`,
5554    /// `tokio::sync::mpsc::Sender`, `tokio::sync::broadcast::Sender`
5555    /// (non-blocking `send`), or `Arc<Mutex<Vec<MutationEvent>>>`.
5556    pub fn set_event_sink(&mut self, sink: Box<dyn Fn(MutationEvent) + Send + Sync>) {
5557        self.event_sink = Some(sink);
5558    }
5559
5560    /// Whether a post-commit event sink is currently installed.
5561    pub fn has_event_sink(&self) -> bool {
5562        self.event_sink.is_some()
5563    }
5564
5565    /// Set WAL fsync cadence. Default [`FsyncPolicy::Strict`].
5566    pub fn set_fsync_policy(&mut self, p: FsyncPolicy) {
5567        self.fsync = p;
5568    }
5569
5570    /// Return the current WAL fsync cadence.
5571    pub fn fsync_policy(&self) -> FsyncPolicy {
5572        self.fsync
5573    }
5574
5575    // ── Group-commit event deferral ───────────────────────────────────────────
5576
5577    /// Enable or disable deferred event mode.
5578    ///
5579    /// When `true`, event notifications (subscription `DbEvent`s and legacy
5580    /// `MutationEvent` sink calls) are buffered rather than fired immediately.
5581    /// Call [`flush_deferred_events`] after the group fsync to deliver them,
5582    /// or [`discard_deferred_events`] if the fsync failed and the group must
5583    /// be treated as lost.
5584    pub fn set_deferred_events_mode(&mut self, defer: bool) {
5585        self.defer_events = defer;
5586    }
5587
5588    /// Fire all buffered events accumulated since [`set_deferred_events_mode`]
5589    /// was set to true.  Clears the buffer.
5590    ///
5591    /// Called by the drain thread AFTER a successful group fsync, so
5592    /// subscribers observe only data that is durably on disk.
5593    pub fn flush_deferred_events(&mut self) {
5594        let events = std::mem::take(&mut self.deferred_events);
5595        for de in events {
5596            self.distribute_events(&de.rec, &de.engine_deltas, de.seq);
5597            self.emit_committed(&de.rec, de.ingest);
5598        }
5599    }
5600
5601    /// Discard all buffered events without firing them.
5602    ///
5603    /// Called by the drain thread when a group fsync fails: the WAL has been
5604    /// truncated back to the pre-group offset, so the committed-but-unsynced
5605    /// ops must not be observable to subscribers.
5606    pub fn discard_deferred_events(&mut self) {
5607        self.deferred_events.clear();
5608    }
5609
5610    // ── Degraded state ────────────────────────────────────────────────────────
5611
5612    /// Mark this database as degraded.
5613    ///
5614    /// Called by the group-commit drain thread after a group fsync failure and
5615    /// WAL truncation: the in-memory state is now ahead of the on-disk WAL, so
5616    /// further mutations would deepen the divergence.  All subsequent calls to
5617    /// [`log_then_apply_with`] return `Err` until the database is reopened.
5618    pub fn set_degraded(&mut self) {
5619        self.degraded = true;
5620    }
5621
5622    fn emit(&self, ev: MutationEvent) {
5623        if let Some(sink) = &self.event_sink {
5624            sink(ev);
5625        }
5626    }
5627
5628    fn emit_committed(&self, rec: &WalRecord, ingest: Option<(String, usize)>) {
5629        match rec {
5630            WalRecord::Batch(inner) => {
5631                for r in inner {
5632                    if let Some(ev) = event_from_record(r, &self.syms, &self.ids) {
5633                        self.emit(ev);
5634                    }
5635                }
5636                match ingest {
5637                    Some((label, inserted)) => {
5638                        self.emit(MutationEvent::Ingested { label, inserted })
5639                    }
5640                    None => {
5641                        let ops = inner
5642                            .iter()
5643                            .filter(|r| !matches!(r, WalRecord::Intern { .. }))
5644                            .count();
5645                        if ops > 1 {
5646                            self.emit(MutationEvent::BatchApplied { ops });
5647                        }
5648                    }
5649                }
5650            }
5651            other => {
5652                if let Some(ev) = event_from_record(other, &self.syms, &self.ids) {
5653                    self.emit(ev);
5654                }
5655            }
5656        }
5657    }
5658
5659    // -----------------------------------------------------------------------
5660    // Subscription API
5661    // -----------------------------------------------------------------------
5662
5663    /// Distribute post-commit events to all live subscribers.
5664    ///
5665    /// Build a row-key → row-data map from a [`ResultSet`].
5666    ///
5667    /// Each row is serialized to JSON to form its key; a debug fallback is used
5668    /// if serialization fails. Used by both the initial-seed path in
5669    /// [`Self::subscribe_query`] and the per-commit diff path in
5670    /// [`Self::distribute_events`] to keep the two in sync.
5671    fn result_to_row_map(
5672        result: &core_query::ResultSet,
5673    ) -> std::collections::HashMap<String, Vec<Option<Value>>> {
5674        (0..result.len())
5675            .map(|i| {
5676                let row = result.row(i).to_vec();
5677                let key = serde_json::to_string(&row).unwrap_or_else(|_| format!("{row:?}"));
5678                (key, row)
5679            })
5680            .collect()
5681    }
5682
5683    /// Collect the set of label syms touched by a WAL record.
5684    ///
5685    /// Returns `Some(set)` when every record in this commit can be attributed to
5686    /// a known label sym. Returns `None` when the commit must not be skipped:
5687    /// edge records, unresolvable key→label lookups, or any record type not in
5688    /// the explicit handled set.
5689    ///
5690    /// Handled record types and their actions:
5691    /// - `InsertNode`   → look up label in interner (fails → None)
5692    /// - `InsertNodeId` → label sym is carried directly
5693    /// - `SetProp`      → resolve key→id→label (fails → None)
5694    /// - `DeleteNode`   → resolve key→id→label (fails → None)
5695    /// - `Batch`        → recurse into every inner record
5696    /// - `InsertEdge`, `DeleteEdge`, `InsertEdgeId` → always None (edge records)
5697    /// - everything else → None (conservative)
5698    fn commit_touched_labels(
5699        rec: &WalRecord,
5700        syms: &Interner,
5701        ids: &IdMap,
5702        labels: &[u32],
5703    ) -> Option<BTreeSet<u32>> {
5704        let mut out = BTreeSet::new();
5705        if Self::collect_touched_labels(rec, syms, ids, labels, &mut out) {
5706            Some(out)
5707        } else {
5708            None
5709        }
5710    }
5711
5712    fn collect_touched_labels(
5713        rec: &WalRecord,
5714        syms: &Interner,
5715        ids: &IdMap,
5716        labels: &[u32],
5717        out: &mut BTreeSet<u32>,
5718    ) -> bool {
5719        match rec {
5720            // String-key insert: the dense rewrite converts this to
5721            // [Intern, InsertNodeId], so this arm fires only for legacy WAL
5722            // records written before the dense path was added.
5723            WalRecord::InsertNode { label, .. } => {
5724                if let Some(sym) = syms.get(label) {
5725                    out.insert(sym);
5726                    true
5727                } else {
5728                    false
5729                }
5730            }
5731            // Dense-id insert (produced by rewrite_wal_dense for every
5732            // insert_node call in the current codebase).
5733            WalRecord::InsertNodeId { label, .. } => {
5734                out.insert(*label);
5735                true
5736            }
5737            // String-key prop set: dense path converts to [Intern, SetPropId].
5738            WalRecord::SetProp { key, .. } => {
5739                if let Some(sym) = Self::resolve_key_label_sym(key, ids, labels) {
5740                    out.insert(sym);
5741                    true
5742                } else {
5743                    false
5744                }
5745            }
5746            // Dense-id prop set (produced by rewrite_wal_dense for set_prop).
5747            WalRecord::SetPropId { id, .. } => {
5748                if let Some(sym) = labels.get(*id as usize).copied().filter(|&s| s != u32::MAX) {
5749                    out.insert(sym);
5750                    true
5751                } else {
5752                    false
5753                }
5754            }
5755            WalRecord::DeleteNode { key } => {
5756                if let Some(sym) = Self::resolve_key_label_sym(key, ids, labels) {
5757                    out.insert(sym);
5758                    true
5759                } else {
5760                    false
5761                }
5762            }
5763            WalRecord::Batch(inner) => inner
5764                .iter()
5765                .all(|r| Self::collect_touched_labels(r, syms, ids, labels, out)),
5766            // Intern is a pure metadata record — it does not touch any node's
5767            // label and is safe to skip for the label-skip predicate.
5768            WalRecord::Intern { .. } => true,
5769            // Edge records: always re-execute (edges can change join results).
5770            WalRecord::InsertEdge { .. }
5771            | WalRecord::DeleteEdge { .. }
5772            | WalRecord::InsertEdgeId { .. } => false,
5773            _ => false,
5774        }
5775    }
5776
5777    /// Resolve a node key to its label sym via the dense id table.
5778    /// Returns `None` if the key is unknown or the label is a tombstone sentinel.
5779    fn resolve_key_label_sym(key: &str, ids: &IdMap, labels: &[u32]) -> Option<u32> {
5780        let id = ids.get(key)?;
5781        let sym = labels.get(id as usize).copied()?;
5782        (sym != u32::MAX).then_some(sym)
5783    }
5784
5785    /// Distribute post-commit events to all live subscribers.
5786    ///
5787    /// Called from `log_then_apply_with` after apply + fsync, before the
5788    /// legacy MutationEvent sink. Prunes dead `Weak` entries in-place.
5789    ///
5790    /// Query subscriptions (subscribe_query) re-execute their plan on every
5791    /// call and diff the result against the previous run. Zero overhead when
5792    /// no query subscriptions are active.
5793    fn distribute_events(&mut self, rec: &WalRecord, engine_deltas: &[EngineEdgeDelta], seq: u64) {
5794        if self.subscriptions.is_empty() && self.query_subscriptions.is_empty() {
5795            return;
5796        }
5797
5798        if !self.subscriptions.is_empty() {
5799            // Build write events from the WAL record.
5800            let write_events: Vec<DbEvent> =
5801                Self::write_events_from_record(rec, seq, &self.syms, &self.ids);
5802
5803            // Build edge events from engine deltas.  Weight is looked up from
5804            // edge_props at distribution time (after apply), so it's always fresh.
5805            let edge_events: Vec<DbEvent> = engine_deltas
5806                .iter()
5807                .map(|d| {
5808                    if d.fired {
5809                        // The score lives under the rule's declared weight_prop,
5810                        // which is not always the literal "weight".
5811                        let prop = self
5812                            .engine
5813                            .rules()
5814                            .find(|r| r.name == d.rule)
5815                            .and_then(|r| r.weight_prop.as_deref());
5816                        let weight = prop.and_then(|p| {
5817                            self.edge_props
5818                                .get(d.etype_sym, d.src_id, d.dst_id, p)
5819                                .and_then(|v| {
5820                                    if let core_storage::Value::Float(f) = v {
5821                                        Some(*f)
5822                                    } else {
5823                                        None
5824                                    }
5825                                })
5826                        });
5827                        DbEvent::EdgeFired {
5828                            rule: d.rule.clone(),
5829                            src_key: d.src_key.clone(),
5830                            dst_key: d.dst_key.clone(),
5831                            edge_type: d.edge_type.clone(),
5832                            weight,
5833                            commit_seq: seq,
5834                        }
5835                    } else {
5836                        DbEvent::EdgeRetracted {
5837                            rule: d.rule.clone(),
5838                            src_key: d.src_key.clone(),
5839                            dst_key: d.dst_key.clone(),
5840                            edge_type: d.edge_type.clone(),
5841                            commit_seq: seq,
5842                        }
5843                    }
5844                })
5845                .collect();
5846
5847            // Prune dead entries; push matching events to live ones.
5848            self.subscriptions.retain(|entry| {
5849                let Some(inner) = entry.inner.upgrade() else {
5850                    return false;
5851                };
5852                for ev in &write_events {
5853                    if event_matches(ev, &entry.filter) {
5854                        inner.push(ev.clone());
5855                    }
5856                }
5857                for ev in &edge_events {
5858                    if event_matches(ev, &entry.filter) {
5859                        inner.push(ev.clone());
5860                    }
5861                }
5862                true
5863            });
5864
5865            // Turn off delta accumulation if all subscribers dropped and no views remain.
5866            if self.subscriptions.is_empty() && self.view_store.is_empty() {
5867                self.engine.set_emit_deltas(false);
5868            }
5869        }
5870
5871        // Query subscriptions: full re-run per commit, then diff rows.
5872        // IMPORTANT: full re-execution on every commit — use LIMIT to bound cost.
5873        // Differential evaluation is roadmap / Phase 5.
5874        if !self.query_subscriptions.is_empty() {
5875            // Take the list out so we can call self.view() without borrow conflict.
5876            let mut query_subs = std::mem::take(&mut self.query_subscriptions);
5877            let empty_params = BTreeMap::new();
5878            query_subs.retain_mut(|entry| {
5879                let Some(inner) = entry.inner.upgrade() else {
5880                    return false; // subscriber dropped — prune
5881                };
5882                // Label-skip: if the plan has a known scan label and this commit
5883                // can be proven to touch only different labels (and no rule-derived
5884                // edge deltas fired), the result set cannot have changed — skip.
5885                if let Some(scan_sym) = entry.scan_label {
5886                    if engine_deltas.is_empty() {
5887                        let touched =
5888                            Self::commit_touched_labels(rec, &self.syms, &self.ids, &self.labels);
5889                        if touched.map(|t| !t.contains(&scan_sym)).unwrap_or(false) {
5890                            return true; // safe to skip — result set unchanged
5891                        }
5892                    }
5893                }
5894                QUERY_SUB_EXECS_TL.with(|c| c.set(c.get() + 1));
5895                let result = match execute(&self.view(), &entry.ops, &Params(&empty_params)) {
5896                    Ok(r) => r,
5897                    Err(e) => {
5898                        // Keep the subscription alive; skip the diff for this commit.
5899                        // Re-run errors are transient (e.g., planner change) and
5900                        // self-heal when the next commit succeeds.
5901                        eprintln!("[mushroomdb] subscribe_query re-run failed: {e}");
5902                        return true;
5903                    }
5904                };
5905                // Build new row map: serialized-key → row data.
5906                let new_row_map = Self::result_to_row_map(&result);
5907                // Removed rows: in prev but not in new.
5908                for (key, row) in &entry.prev_row_map {
5909                    if !new_row_map.contains_key(key) {
5910                        inner.push(DbEvent::QueryRowRemoved {
5911                            columns: entry.columns.clone(),
5912                            row: row.clone(),
5913                        });
5914                    }
5915                }
5916                // Added rows: in new but not in prev.
5917                for (key, row) in &new_row_map {
5918                    if !entry.prev_row_map.contains_key(key) {
5919                        inner.push(DbEvent::QueryRowAdded {
5920                            columns: entry.columns.clone(),
5921                            row: row.clone(),
5922                        });
5923                    }
5924                }
5925                entry.prev_row_map = new_row_map;
5926                true
5927            });
5928            self.query_subscriptions = query_subs;
5929        }
5930    }
5931
5932    /// Returns `true` if any live subscriber or view definition requires delta
5933    /// accumulation. Used to set `engine.emit_deltas` on subscribe/view DDL.
5934    fn needs_emit_deltas(&self) -> bool {
5935        !self.view_store.is_empty()
5936            || self
5937                .subscriptions
5938                .iter()
5939                .any(|e| e.inner.upgrade().is_some())
5940    }
5941
5942    /// Convert a WAL record into `DbEvent` write events with the given seq.
5943    fn write_events_from_record(
5944        rec: &WalRecord,
5945        seq: u64,
5946        intern: &Interner,
5947        ids: &IdMap,
5948    ) -> Vec<DbEvent> {
5949        match rec {
5950            WalRecord::InsertNode { label, key, .. } => vec![DbEvent::NodeInserted {
5951                label: label.clone(),
5952                key: key.clone(),
5953                commit_seq: seq,
5954            }],
5955            // *Id arms run after a successful apply, so resolution can only
5956            // fail on a programming error. Skip the event rather than emit a
5957            // fabricated "" that clients can't tell from a real empty value
5958            // (mirrors event_from_record returning None).
5959            WalRecord::InsertNodeId { label, key, .. } => intern
5960                .resolve(*label)
5961                .map(|label| DbEvent::NodeInserted {
5962                    label: label.to_string(),
5963                    key: key.clone(),
5964                    commit_seq: seq,
5965                })
5966                .into_iter()
5967                .collect(),
5968            WalRecord::SetProp { key, field, .. } => vec![DbEvent::PropSet {
5969                key: key.clone(),
5970                field: field.clone(),
5971                commit_seq: seq,
5972            }],
5973            WalRecord::SetPropId { id, field, .. } => ids
5974                .key_of(*id)
5975                .zip(intern.resolve(*field))
5976                .map(|(key, field)| DbEvent::PropSet {
5977                    key: key.to_string(),
5978                    field: field.to_string(),
5979                    commit_seq: seq,
5980                })
5981                .into_iter()
5982                .collect(),
5983            WalRecord::RemoveProp { key, field } => vec![DbEvent::PropRemoved {
5984                key: key.clone(),
5985                field: field.clone(),
5986                commit_seq: seq,
5987            }],
5988            WalRecord::InsertEdge {
5989                edge_type,
5990                src_key,
5991                dst_key,
5992            } => vec![DbEvent::EdgeInserted {
5993                edge_type: edge_type.clone(),
5994                src: src_key.clone(),
5995                dst: dst_key.clone(),
5996                commit_seq: seq,
5997            }],
5998            WalRecord::InsertEdgeId { etype, src, dst } => (|| {
5999                Some(DbEvent::EdgeInserted {
6000                    edge_type: intern.resolve(*etype)?.to_string(),
6001                    src: ids.key_of(*src)?.to_string(),
6002                    dst: ids.key_of(*dst)?.to_string(),
6003                    commit_seq: seq,
6004                })
6005            })()
6006            .into_iter()
6007            .collect(),
6008            WalRecord::DeleteEdge {
6009                edge_type,
6010                src_key,
6011                dst_key,
6012            } => vec![DbEvent::EdgeDeleted {
6013                edge_type: edge_type.clone(),
6014                src: src_key.clone(),
6015                dst: dst_key.clone(),
6016                commit_seq: seq,
6017            }],
6018            WalRecord::DeleteNode { key } => vec![DbEvent::NodeDeleted {
6019                key: key.clone(),
6020                commit_seq: seq,
6021            }],
6022            WalRecord::Batch(inner) => inner
6023                .iter()
6024                .flat_map(|r| Self::write_events_from_record(r, seq, intern, ids))
6025                .collect(),
6026            WalRecord::CreateRule { .. }
6027            | WalRecord::DeleteRule { .. }
6028            | WalRecord::RebuildRule { .. }
6029            | WalRecord::CreateView { .. }
6030            | WalRecord::DeleteView { .. }
6031            | WalRecord::EnableFulltext { .. }
6032            | WalRecord::DisableFulltext { .. }
6033            | WalRecord::EnableIndex { .. }
6034            | WalRecord::DisableIndex { .. }
6035            | WalRecord::Intern { .. }
6036            // History markers produce no DbEvent — the engine delta already
6037            // fired the EdgeFired/EdgeRetracted subscription events.
6038            | WalRecord::DerivedEdgeAdded { .. }
6039            | WalRecord::DerivedEdgeRetracted { .. }
6040            // A count is not an edge event: the pair it counts already fired one
6041            // when it was first inserted.
6042            | WalRecord::SetEdgeCount { .. }
6043            | WalRecord::RenameNode { .. } => vec![],
6044        }
6045    }
6046
6047    /// Subscribe to edge-fire and edge-retract events for one named rule.
6048    ///
6049    /// Returns `Err(GraphError::RuleNotFound)` if `rule_name` is not
6050    /// currently registered. Dropping the returned [`Subscription`] handle
6051    /// unregisters the subscriber — no further events are queued, no
6052    /// resources leak.
6053    pub fn subscribe_rule(&mut self, rule_name: &str) -> core_storage::Result<Subscription> {
6054        if self.read_only {
6055            return Err(core_storage::GraphError::ReadOnly);
6056        }
6057        if !self.engine.rules().any(|r| r.name == rule_name) {
6058            return Err(core_storage::GraphError::RuleNotFound {
6059                name: rule_name.to_string(),
6060            });
6061        }
6062        let inner = SubInner::new(self.sub_capacity());
6063        self.subscriptions.push(SubEntry {
6064            filter: SubFilter::Rule(rule_name.to_string()),
6065            inner: std::sync::Arc::downgrade(&inner),
6066        });
6067        self.engine.set_emit_deltas(true);
6068        Ok(Subscription(inner))
6069    }
6070
6071    /// Subscribe to edge-fire and edge-retract events for **all** rules.
6072    ///
6073    /// Returns `Err(GraphError::ReadOnly)` if called on an as-of instance —
6074    /// as-of instances never commit, so `distribute_events` never runs and the
6075    /// subscription would never deliver events.
6076    pub fn subscribe_all_rules(&mut self) -> core_storage::Result<Subscription> {
6077        if self.read_only {
6078            return Err(core_storage::GraphError::ReadOnly);
6079        }
6080        let inner = SubInner::new(self.sub_capacity());
6081        self.subscriptions.push(SubEntry {
6082            filter: SubFilter::AllRules,
6083            inner: std::sync::Arc::downgrade(&inner),
6084        });
6085        self.engine.set_emit_deltas(true);
6086        Ok(Subscription(inner))
6087    }
6088
6089    /// Subscribe to write events: node insert/delete, prop set/remove.
6090    ///
6091    /// Does not include edge-fire / edge-retract (rule-derived edge events).
6092    ///
6093    /// Returns `Err(GraphError::ReadOnly)` if called on an as-of instance —
6094    /// as-of instances never commit, so `distribute_events` never runs and the
6095    /// subscription would never deliver events.
6096    pub fn subscribe_writes(&mut self) -> core_storage::Result<Subscription> {
6097        if self.read_only {
6098            return Err(core_storage::GraphError::ReadOnly);
6099        }
6100        let inner = SubInner::new(self.sub_capacity());
6101        self.subscriptions.push(SubEntry {
6102            filter: SubFilter::Writes,
6103            inner: std::sync::Arc::downgrade(&inner),
6104        });
6105        self.engine.set_emit_deltas(true);
6106        Ok(Subscription(inner))
6107    }
6108
6109    /// Subscribe to incremental Cypher query results.
6110    ///
6111    /// Parses and plans `cypher`; rejects the query if the plan is not in the
6112    /// allowlisted subset (see [`core_query::cypher::is_subscribable`]):
6113    ///   - `MATCH (n:Label) WHERE … RETURN … [LIMIT n]`
6114    ///   - `MATCH (a)-[r:TYPE]->(b) RETURN … [LIMIT n]`  (exactly one hop)
6115    ///
6116    /// SKIP is not supported — it shifts the result window on every commit,
6117    /// causing spurious Added/Removed churn for rows whose data never changed.
6118    /// Multi-hop Expand chains are not supported; each additional MATCH clause
6119    /// widens scope beyond the documented single-scan / single-hop subset.
6120    ///
6121    /// After each successful commit, the plan is **fully re-executed** and the
6122    /// result is diffed against the previous run. Added rows produce
6123    /// [`DbEvent::QueryRowAdded`]; removed rows produce
6124    /// [`DbEvent::QueryRowRemoved`].
6125    ///
6126    /// **Full re-run per commit; use LIMIT to bound execution cost.**
6127    /// The existing 1 M intermediate-row cap applies. Differential evaluation
6128    /// is roadmap / Phase 5.
6129    ///
6130    /// Returns `Err(GraphError::ReadOnly)` if called on an as-of instance —
6131    /// as-of instances never commit, so `distribute_events` never runs and the
6132    /// subscription would never deliver events.
6133    ///
6134    /// Returns `Err(GraphError::QueryError)` if the query fails to parse, plan,
6135    /// or if the plan shape is not in the allowlist.
6136    pub fn subscribe_query(&mut self, cypher: &str) -> Result<Subscription> {
6137        if self.read_only {
6138            return Err(GraphError::ReadOnly);
6139        }
6140        let tokens = lex(cypher).map_err(|e| GraphError::QueryError {
6141            detail: format!("lex: {e}"),
6142        })?;
6143        let ast = parse(&tokens).map_err(|e| GraphError::QueryError {
6144            detail: format!("parse: {e}"),
6145        })?;
6146        let ops = plan(&ast).map_err(|e| GraphError::QueryError {
6147            detail: format!("plan: {e}"),
6148        })?;
6149        if !is_subscribable(&ops) {
6150            return Err(GraphError::QueryError {
6151                detail: "subscribe_query only supports allowlisted plan shapes: \
6152                         MATCH (n:Label) WHERE … RETURN … [LIMIT n] or \
6153                         MATCH (a)-[r:TYPE]->(b) RETURN … [LIMIT n] (exactly one hop). \
6154                         Not supported: multi-hop Expand chains, SKIP (creates \
6155                         unstable offset windows), ORDER BY, DISTINCT, aggregates, \
6156                         variable-length paths, OPTIONAL MATCH, WITH, UNWIND. \
6157                         Use LIMIT to bound re-execution cost."
6158                    .to_string(),
6159            });
6160        }
6161        // Execute once to capture initial state (initial rows are not emitted as
6162        // events — the subscriber learns the baseline via the first query call).
6163        let empty_params = BTreeMap::new();
6164        let initial = execute(&self.view(), &ops, &Params(&empty_params)).map_err(|e| {
6165            GraphError::QueryError {
6166                detail: format!("execute: {e}"),
6167            }
6168        })?;
6169        let columns = initial.columns().to_vec();
6170        let prev_row_map = Self::result_to_row_map(&initial);
6171        let inner = SubInner::new(self.sub_capacity());
6172        // Derive the scan-label sym for the commit-skip fast-path.  Any Expand op
6173        // or unrecognized leading scan → None (always re-execute).
6174        let scan_label = extract_scan_label(&ops, Arc::make_mut(&mut self.syms));
6175        self.query_subscriptions.push(QuerySubEntry {
6176            ops,
6177            columns,
6178            prev_row_map,
6179            inner: std::sync::Arc::downgrade(&inner),
6180            scan_label,
6181        });
6182        Ok(Subscription(inner))
6183    }
6184
6185    /// Queue capacity used for new subscriptions.
6186    fn sub_capacity(&self) -> usize {
6187        self.sub_capacity
6188    }
6189
6190    /// Override per-subscriber queue capacity for subsequently created
6191    /// subscriptions on this db instance.
6192    ///
6193    /// Default is [`DEFAULT_SUB_CAPACITY`] (65,536 events). Use a smaller
6194    /// value in tests to exercise the [`DbEvent::Lagged`] path without
6195    /// generating tens of thousands of events.
6196    ///
6197    /// This is a test-support escape hatch. Calling it in production reduces
6198    /// subscriber reliability (more Lagged events). It is hidden from rustdoc
6199    /// to discourage accidental production use.
6200    #[doc(hidden)]
6201    pub fn set_sub_capacity(&mut self, capacity: usize) {
6202        self.sub_capacity = capacity;
6203    }
6204
6205    // -----------------------------------------------------------------------
6206
6207    /// Start an atomic batch.
6208    ///
6209    /// The returned [`BatchBuilder`] borrows `self` mutably until
6210    /// [`BatchBuilder::commit`]. Builder methods queue ops only — no
6211    /// validation, no WAL I/O. `commit` validates every queued op against
6212    /// live state plus preceding ops in this batch (duplicate key inside
6213    /// the batch is `Err`; an edge between two nodes created earlier in
6214    /// the batch is valid; `delete_node` then insert of the same key is a
6215    /// fresh identity). Validation never mutates the database. Any failure
6216    /// leaves WAL bytes and in-memory state identical to before `commit`.
6217    /// On success, one `WalRecord::Batch` frame is appended (one fsync)
6218    /// and each inner record is applied in order so rules fire per record.
6219    /// An empty batch, or a batch of only no-ops, writes zero WAL bytes.
6220    ///
6221    /// **Rule-window limitation:** batch validation cannot see edges that a
6222    /// rule created earlier in the *same* batch will derive at apply time, so
6223    /// a `delete_edge` / `insert_edge` in that window is silently no-oped
6224    /// where sequential calls would return `Err(RuleOwned)`. State integrity
6225    /// is unaffected (idempotent apply, provenance intact). Create rules in
6226    /// their own batch, or sequentially, when later ops may touch derived
6227    /// edges.
6228    pub fn batch(&mut self) -> BatchBuilder<'_, F> {
6229        BatchBuilder {
6230            db: self,
6231            ops: Vec::new(),
6232        }
6233    }
6234
6235    /// Closure-style atomic write batch.
6236    ///
6237    /// Equivalent to calling [`GraphDb::batch`], invoking `build` to queue ops,
6238    /// then committing. All ops queued inside `build` are validated in order and
6239    /// committed as a single `WalRecord::Batch` frame (one fsync). Rules fire
6240    /// once per inner record, in order, after commit — semantically identical to
6241    /// sequential single-op writes.
6242    ///
6243    /// **Error semantics — validate-then-apply.** `build` queues ops without
6244    /// touching the database. [`BatchBuilder::commit`] validates every op against
6245    /// live state plus earlier ops in this batch before writing anything. If op N
6246    /// fails validation (duplicate key, unknown key, rule-owned edge, …) the
6247    /// entire batch is rejected: no WAL bytes are written and no in-memory state
6248    /// changes. The database is identical to its state before `write_batch` was
6249    /// called.
6250    ///
6251    /// **Atomicity is crash-level, NOT isolation-level.** On replay after a crash,
6252    /// a partial (torn) `Batch` frame applies NONE of its ops — the frame is
6253    /// either fully applied or not at all. However, while applying a committed
6254    /// batch, concurrent readers may observe intermediate states as ops are applied
6255    /// sequentially in memory. There is no interactive transaction isolation in v1.
6256    /// This is documented as "crash-atomic write batches; no interactive
6257    /// transactions or read isolation."
6258    ///
6259    /// **Returns** `(nodes_inserted, edges_inserted)`. An empty or all-noop batch
6260    /// writes zero WAL bytes and returns `(0, 0)`.
6261    ///
6262    /// # Example
6263    ///
6264    /// ```rust,ignore
6265    /// let (nodes, edges) = db.write_batch(|b| {
6266    ///     b.insert_node("Person", "alice", vec![("age".into(), Value::Int(30))]);
6267    ///     b.insert_node("Person", "bob", vec![]);
6268    ///     b.insert_edge("KNOWS", "alice", "bob");
6269    ///     b.set_prop("alice", "role", Value::Str("admin".into()));
6270    ///     b.delete_node("old_key");
6271    /// })?;
6272    /// // One fsync; on crash replay: all five ops land or none do.
6273    /// ```
6274    pub fn write_batch<C>(&mut self, build: C) -> Result<(usize, usize)>
6275    where
6276        C: FnOnce(&mut BatchBuilder<'_, F>),
6277    {
6278        let mut b = self.batch();
6279        build(&mut b);
6280        b.commit()
6281    }
6282
6283    /// Insert `rows` as nodes of `label`. One call is one atomic batch:
6284    /// auto-declared KeyMatch rules (if any) first, then the accepted node
6285    /// inserts, so incremental fire sees the new rules. Per-row key problems
6286    /// are collected in [`IngestReport::row_errors`] and skipped; a commit
6287    /// `Err` means nothing was applied.
6288    ///
6289    /// Auto-FK rule names are `auto_fk_<src_label_lowercase>_<field>` so
6290    /// distinct source labels sharing an FK field each get their own rule.
6291    pub fn ingest(
6292        &mut self,
6293        label: &str,
6294        rows: Vec<BTreeMap<String, Value>>,
6295        opts: &IngestOptions,
6296    ) -> Result<IngestReport> {
6297        self.ingest_with_edges(label, rows, opts, &[])
6298    }
6299
6300    /// [`ingest`] plus user edges in the **same** previewed WAL batch.
6301    /// A failing edge rejects the whole request; nothing is applied.
6302    pub fn ingest_with_edges(
6303        &mut self,
6304        label: &str,
6305        rows: Vec<BTreeMap<String, Value>>,
6306        opts: &IngestOptions,
6307        edges: &[(String, String, String)],
6308    ) -> Result<IngestReport> {
6309        crate::ingest::run(self, label, rows, opts, edges)
6310    }
6311
6312    /// Parse `json` as an array of objects and ingest via [`GraphDb::ingest`].
6313    ///
6314    /// JSON `null` fields are silently omitted (not stored, not a row error).
6315    /// Nested objects and arrays-of-objects are a per-row error (row skipped).
6316    /// Parse failures and a top-level value that is not an array of objects
6317    /// return [`GraphError::IngestError`].
6318    pub fn ingest_json(
6319        &mut self,
6320        label: &str,
6321        json: &str,
6322        opts: &IngestOptions,
6323    ) -> Result<IngestReport> {
6324        crate::ingest::run_json(self, label, json, opts)
6325    }
6326
6327    fn commit_logged_batch(
6328        &mut self,
6329        ops: Vec<BatchOp>,
6330        ingest: Option<(String, usize)>,
6331        // Two-source rule: write_batch_authz threads authz here directly (never
6332        // touches pending_write_authz); query_write_authz sets the field instead
6333        // and passes None.  Only one source is non-None per call.
6334        param_authz: Option<WriteAuthz>,
6335    ) -> Result<BatchOutcome> {
6336        // Read-only guard: catches empty-batch calls before the early-return
6337        // that skips log_then_apply_with, ensuring all mutation entry points fail.
6338        if self.read_only {
6339            return Err(GraphError::ReadOnly);
6340        }
6341        // Ensure provenance is decoded before MutPreview accesses it
6342        // (note_delete_rule / is_rule_owned may call engine.provenance()).
6343        self.engine.ensure_provenance_loaded_mut();
6344
6345        // ── Authz pre-check ──────────────────────────────────────────────────
6346        // Evaluate the decision table per-op BEFORE MutPreview so that a denial
6347        // produces no WAL frame (all-or-nothing at the authz boundary extends
6348        // the existing validate-then-apply contract to role-scope checks).
6349        //
6350        // `batch_created` tracks key→label for nodes created by earlier ops in
6351        // THIS batch, so InsertEdgeUpsert can count same-batch placeholder nodes
6352        // as visible without needing to call `self.ids.get` on not-yet-committed
6353        // keys (they won't be there yet).
6354        //
6355        // Two-source rule: param_authz (write_batch_authz path) takes precedence;
6356        // fall back to self.pending_write_authz (query_write_authz/Cypher path).
6357        // Cloning the field copy avoids a simultaneous borrow of self.ids below.
6358        let authz_opt = param_authz.or_else(|| self.pending_write_authz.clone());
6359        if let Some(ref authz) = authz_opt {
6360            let mut batch_created: BTreeMap<String, String> = BTreeMap::new();
6361            for op in &ops {
6362                self.check_single_op_authz(authz, op, &batch_created)?;
6363                // Update batch_created after a passing authz check so that
6364                // subsequent ops in this batch see the nodes as "about to exist".
6365                match op {
6366                    BatchOp::InsertNode { label, key, .. } => {
6367                        // Only track genuinely new nodes (absent from the
6368                        // snapshot at authz-check time). A pre-existing visible
6369                        // key would be a DuplicateKey — not a real creation —
6370                        // so MutPreview handles it. Letting it into batch_created
6371                        // would allow a later SetProp to bypass update_labels
6372                        // via the "batch-created → always updatable" ruling
6373                        // (delete+recreate exploit, fix for I1 review round 2).
6374                        //
6375                        // Accepted edge: for a delete+recreate-with-different-
6376                        // label batch, node_status resolves the pre-delete
6377                        // (store) label for any subsequent update checks. This
6378                        // grants no net-new capability — a role that can delete+
6379                        // create can already place arbitrary props via
6380                        // InsertNode's own props field.
6381                        if self.ids.get(key.as_str()).is_none() {
6382                            batch_created.insert(key.clone(), label.clone());
6383                        }
6384                    }
6385                    BatchOp::InsertEdgeUpsert {
6386                        placeholder_label,
6387                        src_key,
6388                        dst_key,
6389                        ..
6390                    } => {
6391                        // Both endpoints will be created if not already in store.
6392                        for ep_key in [src_key, dst_key] {
6393                            if self.ids.get(ep_key.as_str()).is_none()
6394                                && !batch_created.contains_key(ep_key.as_str())
6395                            {
6396                                batch_created.insert(ep_key.clone(), placeholder_label.clone());
6397                            }
6398                        }
6399                    }
6400                    _ => {}
6401                }
6402            }
6403        }
6404
6405        let mut outcome = BatchOutcome::default();
6406        let recs = {
6407            let mut preview = MutPreview::new(self);
6408            let mut recs = Vec::with_capacity(ops.len());
6409            // Which node row we are on, counted over the node-insert ops only.
6410            // A caller that queues its rows in order reads this straight back
6411            // as the index into its own list.
6412            let mut node_row = 0usize;
6413            // Every field name the store knows, which a `Replace` needs to work
6414            // out what it removes. Resolved on the first `Replace` in the frame
6415            // and reused, so N replaces read the field list once, not N times.
6416            let mut store_fields: Option<Vec<String>> = None;
6417            // Duplicate inserts this frame has to count, each paired with the
6418            // position in `recs` it belongs at. The count itself is named in the
6419            // dense rewrite and not here: a duplicate's endpoints and edge type
6420            // may all be created by earlier ops in this same frame, and nothing
6421            // in the frame has a dense id yet. See [`PlannedRec`].
6422            let mut deferred_counts: Vec<(usize, String, String, String)> = Vec::new();
6423            for op in ops {
6424                match op {
6425                    BatchOp::InsertNode { label, key, props } => {
6426                        node_row += 1;
6427                        preview.check_insert_node(&key, &props)?;
6428                        preview.note_insert_node(&label, &key, &props);
6429                        recs.push(WalRecord::InsertNode { label, key, props });
6430                    }
6431                    BatchOp::InsertNodeOnConflict {
6432                        label,
6433                        key,
6434                        props,
6435                        on_conflict,
6436                    } => {
6437                        let row = node_row;
6438                        node_row += 1;
6439                        if !preview.has_key(&key) {
6440                            // No conflict: an ordinary insert on any policy —
6441                            // except that a supplied view-owned field is the
6442                            // same mistake here as on a taken key, and gets the
6443                            // same row error rather than a frame error. Without
6444                            // this, one op answered one request two ways
6445                            // depending on whether the store already had the
6446                            // key (defect #19).
6447                            if let Some(why) = preview.supplied_view_owned_prop(&key, &props) {
6448                                outcome.row_errors.push((row, why));
6449                                continue;
6450                            }
6451                            preview.note_insert_node(&label, &key, &props);
6452                            recs.push(WalRecord::InsertNode { label, key, props });
6453                            continue;
6454                        }
6455                        match on_conflict {
6456                            OnConflict::Error => {
6457                                return Err(GraphError::DuplicateKey { key });
6458                            }
6459                            OnConflict::Skip => outcome.skipped += 1,
6460                            OnConflict::Replace => {
6461                                if store_fields.is_none() {
6462                                    store_fields = Some(preview.db.props_view().field_names());
6463                                }
6464                                let fields = store_fields.as_deref().unwrap_or_default();
6465                                match preview.plan_replace(&label, &key, &props, fields) {
6466                                    Ok((writes, kept_view_owned)) => {
6467                                        outcome.kept_view_owned += kept_view_owned;
6468                                        for (field, value) in writes {
6469                                            match value {
6470                                                Some(value) => {
6471                                                    preview.note_set_prop(&key, &field, &value);
6472                                                    recs.push(WalRecord::SetProp {
6473                                                        key: key.clone(),
6474                                                        field,
6475                                                        value,
6476                                                    });
6477                                                }
6478                                                None => {
6479                                                    preview.note_remove_prop(&key, &field);
6480                                                    recs.push(WalRecord::RemoveProp {
6481                                                        key: key.clone(),
6482                                                        field,
6483                                                    });
6484                                                }
6485                                            }
6486                                        }
6487                                        outcome.replaced += 1;
6488                                    }
6489                                    Err(why) => outcome.row_errors.push((row, why)),
6490                                }
6491                            }
6492                        }
6493                    }
6494                    BatchOp::InsertEdge {
6495                        edge_type,
6496                        src_key,
6497                        dst_key,
6498                    } => {
6499                        if preview.prepare_insert_edge(&edge_type, &src_key, &dst_key)? {
6500                            preview.note_insert_edge(&edge_type, &src_key, &dst_key);
6501                            recs.push(WalRecord::InsertEdge {
6502                                edge_type,
6503                                src_key,
6504                                dst_key,
6505                            });
6506                        } else if preview.db.multiplicity {
6507                            // A duplicate inside a batch counts the way a
6508                            // duplicate through `insert_edge` does: `ingest` and
6509                            // Cypher `CREATE` reach this choke-point and not
6510                            // that one, and a count only one entry point keeps
6511                            // would be worse than no count at all.
6512                            //
6513                            // This is the one gate on discriminant 23 from the
6514                            // batch path: a store that never opted in queues
6515                            // nothing here and so writes no such record.
6516                            deferred_counts.push((recs.len(), edge_type, src_key, dst_key));
6517                        }
6518                    }
6519                    BatchOp::SetProp { key, field, value } => {
6520                        if let Some(view_name) = preview.db.view_store.view_for_prop(&field) {
6521                            return Err(GraphError::ViewPropReadOnly {
6522                                view_name: view_name.to_string(),
6523                            });
6524                        }
6525                        preview.check_live_key(&key)?;
6526                        preview.note_set_prop(&key, &field, &value);
6527                        recs.push(WalRecord::SetProp { key, field, value });
6528                    }
6529                    BatchOp::RemoveProp { key, field } => {
6530                        if preview.prepare_remove_prop(&key, &field)? {
6531                            preview.note_remove_prop(&key, &field);
6532                            recs.push(WalRecord::RemoveProp { key, field });
6533                        }
6534                    }
6535                    BatchOp::DeleteEdge {
6536                        edge_type,
6537                        src_key,
6538                        dst_key,
6539                    } => {
6540                        if preview.prepare_delete_edge(&edge_type, &src_key, &dst_key)? {
6541                            preview.note_delete_edge(&edge_type, &src_key, &dst_key);
6542                            recs.push(WalRecord::DeleteEdge {
6543                                edge_type,
6544                                src_key,
6545                                dst_key,
6546                            });
6547                        }
6548                    }
6549                    BatchOp::DeleteNode { key } => {
6550                        preview.check_live_key(&key)?;
6551                        preview.note_delete_node(&key);
6552                        recs.push(WalRecord::DeleteNode { key });
6553                    }
6554                    BatchOp::CreateRule(def) => {
6555                        preview.check_create_rule(&def)?;
6556                        let def_bytes =
6557                            bincode::serialize(&def).map_err(|e| GraphError::Corrupt {
6558                                detail: format!("serialize rule: {e}"),
6559                            })?;
6560                        preview.note_create_rule(&def);
6561                        recs.push(WalRecord::CreateRule { def_bytes });
6562                    }
6563                    BatchOp::DeleteRule { name } => {
6564                        preview.check_delete_rule(&name)?;
6565                        preview.note_delete_rule(&name);
6566                        recs.push(WalRecord::DeleteRule { name });
6567                    }
6568                    BatchOp::RenameNode { old_key, new_key } => {
6569                        preview.check_rename_node(&old_key, &new_key)?;
6570                        preview.note_rename_node(&old_key, &new_key);
6571                        recs.push(WalRecord::RenameNode { old_key, new_key });
6572                    }
6573                    BatchOp::InsertEdgeUpsert {
6574                        edge_type,
6575                        src_key,
6576                        dst_key,
6577                        placeholder_label,
6578                    } => {
6579                        // Auto-create any missing endpoints as plain InsertNode ops.
6580                        // Rules fire and last-change is updated for each created node.
6581                        for key in [&src_key, &dst_key] {
6582                            if !preview.has_key(key) {
6583                                // A placeholder endpoint carries no props, so
6584                                // the view-owned check has nothing to refuse.
6585                                preview.check_insert_node(key, &[])?;
6586                                preview.note_insert_node(&placeholder_label, key, &[]);
6587                                recs.push(WalRecord::InsertNode {
6588                                    label: placeholder_label.clone(),
6589                                    key: key.clone(),
6590                                    props: vec![],
6591                                });
6592                            }
6593                        }
6594                        if preview.prepare_insert_edge(&edge_type, &src_key, &dst_key)? {
6595                            preview.note_insert_edge(&edge_type, &src_key, &dst_key);
6596                            recs.push(WalRecord::InsertEdge {
6597                                edge_type,
6598                                src_key,
6599                                dst_key,
6600                            });
6601                        } else if preview.db.multiplicity {
6602                            // Same choke-point, same gate as `BatchOp::InsertEdge`
6603                            // above: an upsert that finds the pair already there
6604                            // is a duplicate insert and counts as one.
6605                            deferred_counts.push((recs.len(), edge_type, src_key, dst_key));
6606                        }
6607                    }
6608                }
6609            }
6610            // Splice the deferred counts back into the positions they were
6611            // raised at, so a count still sits exactly where the duplicate did
6612            // — before any later op in the frame that deletes the pair.
6613            let mut planned: Vec<PlannedRec> =
6614                Vec::with_capacity(recs.len() + deferred_counts.len());
6615            let mut deferred = deferred_counts.into_iter().peekable();
6616            for (i, rec) in recs.into_iter().enumerate() {
6617                while deferred.peek().is_some_and(|(at, ..)| *at == i) {
6618                    let (_, edge_type, src_key, dst_key) = deferred.next().expect("just peeked");
6619                    planned.push(PlannedRec::DuplicateCount {
6620                        edge_type,
6621                        src_key,
6622                        dst_key,
6623                    });
6624                }
6625                planned.push(PlannedRec::Rec(rec));
6626            }
6627            for (_, edge_type, src_key, dst_key) in deferred {
6628                planned.push(PlannedRec::DuplicateCount {
6629                    edge_type,
6630                    src_key,
6631                    dst_key,
6632                });
6633            }
6634            planned
6635        };
6636        // A frame that is nothing but skips or refused rows writes no WAL, but
6637        // it still has counts to report, so the early returns carry `outcome`
6638        // rather than zeros.
6639        if recs.is_empty() {
6640            return Ok(outcome);
6641        }
6642        // rewrite_wal_dense converts every InsertNode/InsertEdge into its
6643        // *Id form, so only the dense variants can appear in `recs` here.
6644        let recs = self.rewrite_wal_dense_planned(recs)?;
6645        // The rewrite can empty a non-empty batch: a `SET n.ns` naming the
6646        // namespace the node is already in is a no-op and is dropped there. An
6647        // empty `Batch` frame would still take a commit sequence and a WAL
6648        // record, so a batch that turns out to be nothing writes nothing.
6649        if recs.is_empty() {
6650            return Ok(outcome);
6651        }
6652        outcome.nodes_inserted = recs
6653            .iter()
6654            .filter(|r| matches!(r, WalRecord::InsertNodeId { .. }))
6655            .count();
6656        outcome.edges_inserted = recs
6657            .iter()
6658            .filter(|r| matches!(r, WalRecord::InsertEdgeId { .. }))
6659            .count();
6660        // Ingest / write_batch / query_write: one Batch frame, one fsync per call
6661        // under Strict.  Pass self.fsync directly so Strict stays Strict —
6662        // wal_needs_sync(Strict, _) always returns true regardless of op count.
6663        // Mapping Strict → Batched (the prior bug) caused wal_needs_sync to
6664        // short-circuit on single-op batches and silently skip the fsync.
6665        // Batched fsyncs only for multi-op batches; Relaxed always skips.
6666        self.log_then_apply_with(WalRecord::Batch(recs), ingest, self.fsync)?;
6667        Ok(outcome)
6668    }
6669
6670    fn commit_batch(&mut self, ops: Vec<BatchOp>) -> Result<(usize, usize)> {
6671        self.commit_logged_batch(ops, None, None).map(inserted_pair)
6672    }
6673
6674    /// Commit one submission WITHOUT an fsync — for use inside `commit_group`
6675    /// and the group-commit drain thread, which do a single group fsync later.
6676    fn commit_batch_nosync(&mut self, ops: Vec<BatchOp>) -> Result<(usize, usize)> {
6677        // Restore fsync policy even on panic via a raw-pointer drop guard.
6678        // A panic here would poison the RwLock anyway, but the correct policy
6679        // must be in place if the guard is ever unwrapped.
6680        struct RestoreFsync(*mut FsyncPolicy, FsyncPolicy);
6681        impl Drop for RestoreFsync {
6682            fn drop(&mut self) {
6683                // SAFETY: the pointer is valid for the full duration of
6684                // commit_batch_nosync; the guard is dropped before the frame
6685                // returns, and GraphDb outlives this frame.
6686                unsafe {
6687                    *self.0 = self.1;
6688                }
6689            }
6690        }
6691        let saved = self.fsync;
6692        // SAFETY: raw pointer into self; guard dropped within this frame.
6693        let _g = RestoreFsync(&mut self.fsync as *mut FsyncPolicy, saved);
6694        self.fsync = FsyncPolicy::Relaxed;
6695        self.commit_logged_batch(ops, None, None).map(inserted_pair)
6696    }
6697
6698    /// Commit multiple op-batches as a **group**: each submission gets its own
6699    /// WAL `Batch` frame, but there is exactly **one** `Fs::sync` for the whole
6700    /// group (under `Strict` / `Batched` policy; `Relaxed` skips all syncs).
6701    ///
6702    /// # Durability semantics
6703    ///
6704    /// A crash before the group fsync may lose **all** submissions in the group.
6705    /// A crash after the group fsync preserves all of them.  No submission is
6706    /// ever torn: each WAL frame is either fully applied on replay or dropped
6707    /// in its entirety (CRC-protected frame boundaries).
6708    ///
6709    /// Events and subscription notifications fire per-submission immediately
6710    /// after apply, which may be before the group fsync.  From a subscriber's
6711    /// perspective this is equivalent to the `Relaxed` durability window.
6712    /// Submitters using [`SharedDb::submit_batch`] only unblock after the group
6713    /// fsync, so from their perspective durability is fully guaranteed.
6714    ///
6715    /// # MVCC interplay
6716    ///
6717    /// Each submission records its own `CommitDelta`; the fold-every-K counter
6718    /// increments per submission (not per group), preserving existing reader
6719    /// snapshot semantics.
6720    ///
6721    /// # Returns
6722    ///
6723    /// One `Result<(nodes_inserted, edges_inserted)>` per input group element,
6724    /// in order.  Failures are per-submission (validation errors); the group
6725    /// fsync error (if any) is returned as the second tuple element.
6726    pub fn commit_group(
6727        &mut self,
6728        groups: Vec<Vec<BatchOp>>,
6729    ) -> (Vec<Result<(usize, usize)>>, Option<GraphError>) {
6730        let mut results = Vec::with_capacity(groups.len());
6731        for ops in groups {
6732            results.push(self.commit_batch_nosync(ops));
6733        }
6734        let any_ok = results.iter().any(|r| r.is_ok());
6735        let sync_err = if self.fsync != FsyncPolicy::Relaxed && any_ok {
6736            self.fs
6737                .sync(core_storage::fs::FileId::Wal)
6738                .map_err(GraphError::Io)
6739                .err()
6740        } else {
6741            None
6742        };
6743        (results, sync_err)
6744    }
6745
6746    /// Like [`commit_group`] but skips the group fsync entirely.
6747    ///
6748    /// Used by the drain thread to apply submissions under the write lock and
6749    /// then perform the single fsync OUTSIDE the lock (via
6750    /// `core_storage::sync_wal_at`), reducing the write-lock hold time visible
6751    /// to concurrent readers.
6752    pub fn commit_group_nosync(
6753        &mut self,
6754        groups: Vec<Vec<BatchOp>>,
6755    ) -> Vec<Result<(usize, usize)>> {
6756        let mut results = Vec::with_capacity(groups.len());
6757        for ops in groups {
6758            results.push(self.commit_batch_nosync(ops));
6759        }
6760        results
6761    }
6762
6763    pub fn insert_node(
6764        &mut self,
6765        label: &str,
6766        key: &str,
6767        props: Vec<(String, Value)>,
6768    ) -> Result<()> {
6769        if self.read_only {
6770            return Err(GraphError::ReadOnly);
6771        }
6772        MutPreview::new(self).check_insert_node(key, &props)?;
6773        self.log_dense(vec![WalRecord::InsertNode {
6774            label: label.into(),
6775            key: key.into(),
6776            props,
6777        }])
6778    }
6779
6780    /// Insert a user edge. `Ok(true)` when the pair was new, `Ok(false)` when it
6781    /// was already there — the question is "was this pair new", and a duplicate
6782    /// does not make it so.
6783    ///
6784    /// On a store that called [`enable_multiplicity`](Self::enable_multiplicity)
6785    /// a duplicate is no longer a total no-op: it raises the pair's insert count
6786    /// (§5.13). Adjacency is still a set, so [`degree`](Self::degree) is
6787    /// unchanged and the return value is still `Ok(false)`; the count is visible
6788    /// only through [`degree_multiplicity`](Self::degree_multiplicity) and the
6789    /// reserved [`EDGE_COUNT_PROP`]. On every other store a duplicate writes
6790    /// nothing at all, as it always has.
6791    pub fn insert_edge(&mut self, edge_type: &str, src_key: &str, dst_key: &str) -> Result<bool> {
6792        if self.read_only {
6793            return Err(GraphError::ReadOnly);
6794        }
6795        if !MutPreview::new(self).prepare_insert_edge(edge_type, src_key, dst_key)? {
6796            // The pair exists. The only thing left to record is that it was
6797            // asked for again, and only where the store asked to be told.
6798            if let Some(rec) = self.edge_count_record(edge_type, src_key, dst_key) {
6799                self.log_then_apply(rec)?;
6800            }
6801            return Ok(false);
6802        }
6803        self.log_dense(vec![WalRecord::InsertEdge {
6804            edge_type: edge_type.into(),
6805            src_key: src_key.into(),
6806            dst_key: dst_key.into(),
6807        }])?;
6808        Ok(true)
6809    }
6810
6811    pub fn set_prop(&mut self, key: &str, field: &str, value: Value) -> Result<()> {
6812        if self.read_only {
6813            return Err(GraphError::ReadOnly);
6814        }
6815        if let Some(view_name) = self.view_store.view_for_prop(field) {
6816            return Err(GraphError::ViewPropReadOnly {
6817                view_name: view_name.to_string(),
6818            });
6819        }
6820        MutPreview::new(self).check_live_key(key)?;
6821        self.log_dense(vec![WalRecord::SetProp {
6822            key: key.into(),
6823            field: field.into(),
6824            value,
6825        }])
6826    }
6827
6828    /// Set several properties on one live node in a single WAL commit.
6829    ///
6830    /// Every per-property check [`set_prop`](Self::set_prop) runs — view-owned
6831    /// names, live key, the `ns` immutability rule and its type — is evaluated
6832    /// for the whole list before any record is logged. The first refusal
6833    /// returns and the node is unchanged. An empty list writes nothing.
6834    pub fn set_props(&mut self, key: &str, props: Vec<(String, Value)>) -> Result<()> {
6835        if self.read_only {
6836            return Err(GraphError::ReadOnly);
6837        }
6838        MutPreview::new(self).check_live_key(key)?;
6839        for (field, _) in &props {
6840            if let Some(view_name) = self.view_store.view_for_prop(field) {
6841                return Err(GraphError::ViewPropReadOnly {
6842                    view_name: view_name.to_string(),
6843                });
6844            }
6845        }
6846        if props.is_empty() {
6847            return Ok(());
6848        }
6849        self.write_batch(|b| {
6850            for (field, value) in props {
6851                b.set_prop(key, &field, value);
6852            }
6853        })
6854        .map(|_| ())
6855    }
6856
6857    /// Remove a property. Returns `Ok(false)` (and does not log) if the field
6858    /// is already absent. Unknown or tombstoned keys are `Err(KeyNotFound)`.
6859    /// A field a view owns is `Err(ViewPropReadOnly)` — stated once, in
6860    /// [`MutPreview::prepare_remove_prop`], so that the batch ops reaching that
6861    /// same choke-point cannot miss it.
6862    pub fn remove_prop(&mut self, key: &str, field: &str) -> Result<bool> {
6863        if self.read_only {
6864            return Err(GraphError::ReadOnly);
6865        }
6866        if !MutPreview::new(self).prepare_remove_prop(key, field)? {
6867            return Ok(false);
6868        }
6869        self.log_then_apply(WalRecord::RemoveProp {
6870            key: key.into(),
6871            field: field.into(),
6872        })?;
6873        Ok(true)
6874    }
6875
6876    /// Delete a user edge. Returns `Ok(false)` (and does not log) if the edge
6877    /// is absent. Unknown keys are `Err(KeyNotFound)`. Rule-owned edges — in
6878    /// provenance, or a pair a live rule would derive — are `Err(RuleOwned)`
6879    /// (the rule would just put the edge back; delete or change the rule).
6880    pub fn delete_edge(&mut self, edge_type: &str, src_key: &str, dst_key: &str) -> Result<bool> {
6881        if self.read_only {
6882            return Err(GraphError::ReadOnly);
6883        }
6884        if !MutPreview::new(self).prepare_delete_edge(edge_type, src_key, dst_key)? {
6885            return Ok(false);
6886        }
6887        self.log_then_apply(WalRecord::DeleteEdge {
6888            edge_type: edge_type.into(),
6889            src_key: src_key.into(),
6890            dst_key: dst_key.into(),
6891        })?;
6892        Ok(true)
6893    }
6894
6895    /// Delete a live node. Unknown or already-tombstoned keys are
6896    /// `Err(KeyNotFound)` and are not logged. Validation runs before the WAL
6897    /// write; `apply` of a logged `DeleteNode` for an already-tombstoned key
6898    /// (crash window) is a clean no-op.
6899    ///
6900    /// Returns a [`DeleteReport`] with counts of manual and derived edges
6901    /// removed (computed from live state before the deletion is applied).
6902    pub fn delete_node(&mut self, key: &str) -> Result<DeleteReport> {
6903        if self.read_only {
6904            return Err(GraphError::ReadOnly);
6905        }
6906        // Provenance must be loaded before we query provenance_touching.
6907        self.engine.ensure_provenance_loaded_mut();
6908        let id = self
6909            .ids
6910            .get(key)
6911            .ok_or_else(|| GraphError::KeyNotFound { key: key.into() })?;
6912
6913        // Count edges before the delete is applied so we can report counts.
6914        let derived_set: BTreeSet<(u32, u32, u32)> = self
6915            .engine
6916            .provenance_touching(id)
6917            .map(|(_, etype, src, dst)| (etype, src, dst))
6918            .collect();
6919        let derived_edges = derived_set.len() as u64;
6920
6921        let mut total_topo = 0u64;
6922        let tv = self.topo_view();
6923        for et in tv.etypes() {
6924            total_topo += tv.neighbors(et, Direction::Out, id).len() as u64
6925                + tv.neighbors(et, Direction::In, id).len() as u64;
6926        }
6927        // For symmetric rules (e.g. Overlap), a→b and b→a are two separate directed
6928        // triples in both the topo scan (Out and In from id) and in provenance_touching.
6929        // The subtraction remains correct because both counts include both directions.
6930        let manual_edges = total_topo.saturating_sub(derived_edges);
6931
6932        self.log_then_apply(WalRecord::DeleteNode { key: key.into() })?;
6933        Ok(DeleteReport {
6934            manual_edges,
6935            derived_edges,
6936        })
6937    }
6938
6939    /// Rename a live node's key.  The dense id (and therefore all edges,
6940    /// props, history, and last-change tracking) is unaffected.
6941    ///
6942    /// Returns `Err(KeyNotFound)` if `old` is not a live key.
6943    /// Returns `Err(DuplicateKey)` if `new` is already live.
6944    pub fn rename_node(&mut self, old: &str, new: &str) -> Result<()> {
6945        if self.read_only {
6946            return Err(GraphError::ReadOnly);
6947        }
6948        MutPreview::new(self).check_rename_node(old, new)?;
6949        self.log_then_apply(WalRecord::RenameNode {
6950            old_key: old.into(),
6951            new_key: new.into(),
6952        })
6953    }
6954
6955    /// Return the IVF drift counter for the dst-side candidate index of `rule`.
6956    /// `None` if the rule does not exist or is not approximate.
6957    ///
6958    /// The drift counter increments on IVF insert/remove after the last fit.
6959    /// When dst-side drift exceeds [`core_rules::IVF_DRIFT_REBUILD`], apply
6960    /// WAL-logs `RebuildRule` as a second commit (rebuild resets the counter).
6961    pub fn ivf_dst_drift(&self, rule: &str) -> Option<u64> {
6962        // SideIvfExport = (centroids, node→cluster, drift)
6963        self.engine
6964            .export_ivf_state()
6965            .remove(rule)
6966            .map(|(_src, dst)| dst.2)
6967    }
6968
6969    /// Validate and WAL-log a new rule, then backfill derived edges inside apply.
6970    /// Validation and duplicate-name check run before logging so invalid rules
6971    /// never enter the WAL.
6972    pub fn create_rule(&mut self, def: RuleDef) -> Result<()> {
6973        if self.read_only {
6974            return Err(GraphError::ReadOnly);
6975        }
6976        MutPreview::new(self).check_create_rule(&def)?;
6977        let def_bytes = bincode::serialize(&def).map_err(|e| GraphError::Corrupt {
6978            detail: format!("serialize rule: {e}"),
6979        })?;
6980        self.log_dense(vec![WalRecord::CreateRule { def_bytes }])
6981    }
6982
6983    /// Override this handle's HNSW build-slice size, or `None` to restore
6984    /// [`core_rules::HNSW_BUILD_BATCH`].
6985    ///
6986    /// Exposed for tests that need a small slice without a large corpus; not
6987    /// part of the stable surface.
6988    #[doc(hidden)]
6989    pub fn set_hnsw_build_batch(&mut self, batch: Option<usize>) {
6990        self.engine.set_hnsw_build_batch(batch);
6991    }
6992
6993    /// Rules whose vector index is still being built, in name order.
6994    ///
6995    /// The same list [`GraphDb::stats`] reports per rule in `building`.
6996    /// After a clean open this includes a build a snapshot cut short, so
6997    /// `serve`'s ticker can pump it without a write.
6998    pub fn builds_in_progress(&self) -> Vec<BuildProgress> {
6999        self.engine.builds_in_progress()
7000    }
7001
7002    /// Advance any vector index still building and backfill each rule that
7003    /// finishes. Returns what is still outstanding.
7004    ///
7005    /// A map lookup when nothing is pending, so it is cheap to call on a timer.
7006    /// One write lock and at most [`core_rules::HNSW_BUILD_BATCH`] vector
7007    /// inserts per pending rule per call, so a caller can drive a large build
7008    /// to completion without ever holding the lock for more than a slice.
7009    ///
7010    /// A rule that finishes here is backfilled through the same
7011    /// `WalRecord::RebuildRule` second commit that IVF drift already uses, so
7012    /// its derived edges are produced by [`GraphDb::rebuild_rule`]'s code path
7013    /// and appear all at once.
7014    ///
7015    /// Every ordinary write pumps one slice on its own (see the post-commit
7016    /// hook in `log_then_apply_with`), so this is for quiescent stores and for
7017    /// operators who want the build finished before traffic arrives.
7018    pub fn pump_index_build(&mut self) -> Result<Vec<BuildProgress>> {
7019        Ok(self.pump_index_build_reporting()?.1)
7020    }
7021
7022    /// [`GraphDb::pump_index_build`], also reporting the builds that **this**
7023    /// call finished, so a progress display can say so.
7024    ///
7025    /// A build can be registered and completed inside a single call — that is
7026    /// what a mid-build snapshot looks like on reopen, where the index scan
7027    /// finishes the graph and only the backfill is outstanding — and the
7028    /// outstanding list alone cannot show that anything happened.
7029    pub fn pump_index_build_reporting(
7030        &mut self,
7031    ) -> Result<(Vec<BuildProgress>, Vec<BuildProgress>)> {
7032        // A read-only handle cannot issue the `RebuildRule` a finished build
7033        // needs, so it would advance the index and then silently fail to
7034        // produce the edges. Refusing is the honest answer.
7035        if self.read_only {
7036            return Err(GraphError::ReadOnly);
7037        }
7038        let finished = self.pump_one_slice();
7039        for done in &finished {
7040            // The index is whole but the rule still owns no edges. A failed
7041            // second commit must leave the rule re-pumpable rather than
7042            // silently edge-less, so the error is surfaced here — unlike the
7043            // post-commit hook, this call is not riding someone else's commit.
7044            self.log_then_apply(WalRecord::RebuildRule {
7045                name: done.rule.clone(),
7046            })?;
7047        }
7048        Ok((finished, self.engine.builds_in_progress()))
7049    }
7050
7051    /// Run the deferred candidate-index build, if it is still owed, against the
7052    /// graph as it stands *now* — before the caller applies anything.
7053    ///
7054    /// A no-op bool test once the indexes are populated, which is after the
7055    /// first write of the handle's life, and for a store with no rules at all.
7056    fn populate_indexes_before_write(&mut self) {
7057        if !self.engine.needs_index_population() {
7058            return;
7059        }
7060        // The retained snapshot blobs arrive with the V8 base sections; without
7061        // them the scan would rebuild every graph the snapshot already holds.
7062        self.ensure_v8_base_sections_loaded();
7063        if !self.engine.needs_index_population() {
7064            return;
7065        }
7066        let mut eng = std::mem::take(&mut self.engine);
7067        {
7068            let gm = make_graph_mut(
7069                &self.ids,
7070                Arc::make_mut(&mut self.syms),
7071                &self.labels,
7072                build_props_view(&self.props, &self.base),
7073                Arc::make_mut(&mut self.topo),
7074                &self.base,
7075                Arc::make_mut(&mut self.edge_props),
7076            );
7077            eng.populate_indexes(&gm);
7078        }
7079        self.engine = eng;
7080    }
7081
7082    /// One slice of build work for every pending rule. Returns the rules whose
7083    /// index just became whole, which the caller must `RebuildRule`.
7084    ///
7085    /// Goes through the engine even with nothing pending when the indexes have
7086    /// not been populated yet: that call adopts the persisted graphs and, for
7087    /// an incomplete blob already registered at open, leaves the remainder to
7088    /// this slice rather than inserting it inline.
7089    fn pump_one_slice(&mut self) -> Vec<BuildProgress> {
7090        // The retained snapshot blobs — and the id count an interrupted build
7091        // is recognised against — arrive with the V8 base sections, which a
7092        // clean open reads lazily. Without this a freshly opened handle pumps
7093        // against empty retained state and concludes there is nothing to do,
7094        // which is precisely the store `build-index` exists for.
7095        self.ensure_v8_base_sections_loaded();
7096        let mut eng = std::mem::take(&mut self.engine);
7097        let finished = {
7098            let mut gm = make_graph_mut(
7099                &self.ids,
7100                Arc::make_mut(&mut self.syms),
7101                &self.labels,
7102                build_props_view(&self.props, &self.base),
7103                Arc::make_mut(&mut self.topo),
7104                &self.base,
7105                Arc::make_mut(&mut self.edge_props),
7106            );
7107            eng.pump_index_build(&mut gm)
7108        };
7109        self.engine = eng;
7110        finished
7111    }
7112
7113    /// Register a sliced build a snapshot cut short, from blobs with
7114    /// `complete == false`.
7115    ///
7116    /// Peeks the V8 mmap for incomplete entries without copying complete
7117    /// graphs. V5–V7 already hold the blobs in the engine from restore.
7118    fn register_outstanding_index_builds(&mut self) {
7119        if self.engine.indexes_populated() {
7120            return;
7121        }
7122        let extra = self.collect_incomplete_hnsw_blobs();
7123        let mut eng = std::mem::take(&mut self.engine);
7124        {
7125            let gm = make_graph_mut(
7126                &self.ids,
7127                Arc::make_mut(&mut self.syms),
7128                &self.labels,
7129                build_props_view(&self.props, &self.base),
7130                Arc::make_mut(&mut self.topo),
7131                &self.base,
7132                Arc::make_mut(&mut self.edge_props),
7133            );
7134            eng.register_incomplete_hnsw_builds(&extra, &gm);
7135        }
7136        self.engine = eng;
7137    }
7138
7139    /// Incomplete `(src, dst)` HNSW blobs from the V8 mmap, copied only when
7140    /// `complete` is false. Empty when there is no mmap base (V5–V7 uses the
7141    /// engine's retained map instead).
7142    fn collect_incomplete_hnsw_blobs(&self) -> BTreeMap<String, (Vec<u8>, Vec<u8>)> {
7143        let Some(base) = &self.base else {
7144            return BTreeMap::new();
7145        };
7146        let Ok(archived) = base.hnsw_section() else {
7147            return BTreeMap::new();
7148        };
7149        archived
7150            .rules
7151            .iter()
7152            .filter_map(|e| {
7153                let src = e.src_blob.as_slice();
7154                let dst = e.dst_blob.as_slice();
7155                if core_rules::hnsw::hnsw_blob_complete(src) == Some(false)
7156                    || core_rules::hnsw::hnsw_blob_complete(dst) == Some(false)
7157                {
7158                    Some((e.name.as_str().to_string(), (src.to_vec(), dst.to_vec())))
7159                } else {
7160                    None
7161                }
7162            })
7163            .collect()
7164    }
7165
7166    /// WAL-log rule deletion. Returns RuleNotFound if the rule does not exist.
7167    pub fn delete_rule(&mut self, name: &str) -> Result<()> {
7168        if self.read_only {
7169            return Err(GraphError::ReadOnly);
7170        }
7171        MutPreview::new(self).check_delete_rule(name)?;
7172        self.log_then_apply(WalRecord::DeleteRule { name: name.into() })
7173    }
7174
7175    /// Return a snapshot of all registered rules.
7176    pub fn rules(&self) -> Vec<RuleDef> {
7177        self.engine.rules().cloned().collect()
7178    }
7179
7180    // -----------------------------------------------------------------------
7181    // Rule suggestion API
7182    // -----------------------------------------------------------------------
7183
7184    /// Profile the database and suggest linking rules with previewed edge counts.
7185    ///
7186    /// Uses the default seed ([`core_rules::SUGGEST_DEFAULT_SEED`]) for deterministic
7187    /// sampling. Suggestions are sorted by estimated edge count (descending).
7188    /// **NO auto-accept** — call [`GraphDb::create_rule`] explicitly to apply.
7189    pub fn suggest_rules(&self) -> Vec<core_rules::RuleSuggestion> {
7190        self.suggest_rules_seeded(core_rules::SUGGEST_DEFAULT_SEED)
7191    }
7192
7193    /// Like [`suggest_rules`] but with a caller-supplied RNG seed for
7194    /// reproducibility. Same seed + same data = identical output.
7195    pub fn suggest_rules_seeded(&self, seed: u64) -> Vec<core_rules::RuleSuggestion> {
7196        self.suggest_rules_with_config(&core_rules::suggest::SuggestConfig::default(), seed)
7197            .suggestions
7198    }
7199
7200    /// [`suggest_rules_seeded`] with a fully custom [`SuggestConfig`].
7201    ///
7202    /// Returns a [`core_rules::SuggestReport`] that includes both the candidate list
7203    /// and a `truncated` flag indicating whether the global budget fired before all
7204    /// candidates were evaluated.
7205    pub fn suggest_rules_with_config(
7206        &self,
7207        config: &core_rules::suggest::SuggestConfig,
7208        seed: u64,
7209    ) -> core_rules::SuggestReport {
7210        use std::collections::BTreeMap;
7211
7212        // Collect (node_id, key) pairs per label, skipping tombstoned nodes.
7213        let mut label_nodes: BTreeMap<String, Vec<(u32, String)>> = BTreeMap::new();
7214        for id in 0..self.ids.len() as u32 {
7215            let Some(key) = self.ids.key_of(id) else {
7216                continue;
7217            };
7218            let Some(&sym) = self.labels.get(id as usize) else {
7219                continue;
7220            };
7221            if sym == u32::MAX {
7222                continue; // tombstoned
7223            }
7224            let Some(label) = self.syms.resolve(sym) else {
7225                continue;
7226            };
7227            label_nodes
7228                .entry(label.to_string())
7229                .or_default()
7230                .push((id, key.to_string()));
7231        }
7232
7233        let existing = self.rules();
7234        let pv = build_props_view(&self.props, &self.base);
7235        let all_fields: Vec<String> = pv.field_names();
7236
7237        core_rules::suggest::suggest_rules(
7238            &label_nodes,
7239            &|id, field| pv.get(id, field).map(|vr| vr.into_value()),
7240            &all_fields,
7241            &existing,
7242            config,
7243            seed,
7244        )
7245    }
7246
7247    /// Recompute a rule's derived edges from scratch. WAL-logged so un-trip
7248    /// plus later mutations replay identically (rebuild is a pure function
7249    /// of state).
7250    ///
7251    /// Only exit from the tripped latch: if the full desired set fits the
7252    /// budget, it is applied completely and `tripped` clears; if it still
7253    /// exceeds the budget, provenance is left untouched and `tripped` stays
7254    /// true. Counts as a fire evaluation (see [`RuleStats::fires`]).
7255    /// Unknown rule → `RuleNotFound`, nothing logged.
7256    pub fn rebuild_rule(&mut self, name: &str) -> Result<()> {
7257        if self.read_only {
7258            return Err(GraphError::ReadOnly);
7259        }
7260        if !self.engine.rules().any(|r| r.name == name) {
7261            return Err(GraphError::RuleNotFound { name: name.into() });
7262        }
7263        self.log_then_apply(WalRecord::RebuildRule { name: name.into() })
7264    }
7265
7266    // -----------------------------------------------------------------------
7267    // Materialized view API
7268    // -----------------------------------------------------------------------
7269
7270    /// Register a new materialized property view, backfill its values for all
7271    /// existing nodes, and WAL-log the definition.
7272    ///
7273    /// # Errors
7274    /// - `ReadOnly`: called on an as-of instance.
7275    /// - `RuleInvalid`: name collision, view_prop collision, or invalid def.
7276    pub fn create_view(&mut self, def: ViewDef) -> Result<()> {
7277        if self.read_only {
7278            return Err(GraphError::ReadOnly);
7279        }
7280        // Pre-validate before WAL write.
7281        def.validate()
7282            .map_err(|e| GraphError::RuleInvalid { detail: e })?;
7283        if self.view_store.has_view(&def.name) {
7284            return Err(GraphError::RuleInvalid {
7285                detail: format!("view {:?} already exists", def.name),
7286            });
7287        }
7288        if let Some(existing) = self.view_store.view_for_prop(&def.view_prop) {
7289            return Err(GraphError::RuleInvalid {
7290                detail: format!(
7291                    "view_prop {:?} is already used by view {:?}",
7292                    def.view_prop, existing
7293                ),
7294            });
7295        }
7296        let def_bytes = bincode::serialize(&def).map_err(|e| GraphError::Corrupt {
7297            detail: format!("serialize view: {e}"),
7298        })?;
7299        // Enable delta accumulation before the view is registered so subsequent
7300        // incremental edge events reach view maintenance from this point onward.
7301        // (The backfill inside create_view reads topo directly; it does not rely
7302        // on pending deltas.)
7303        self.engine.set_emit_deltas(true);
7304        self.log_then_apply(WalRecord::CreateView { def_bytes })
7305    }
7306
7307    /// Remove a named view and delete its values from every node.
7308    ///
7309    /// # Errors
7310    /// - `ReadOnly`: called on an as-of instance.
7311    /// - `RuleNotFound`: view does not exist.
7312    pub fn delete_view(&mut self, name: &str) -> Result<()> {
7313        if self.read_only {
7314            return Err(GraphError::ReadOnly);
7315        }
7316        if !self.view_store.has_view(name) {
7317            return Err(GraphError::RuleNotFound { name: name.into() });
7318        }
7319        let result = self.log_then_apply(WalRecord::DeleteView { name: name.into() });
7320        // After deletion, disable accumulation if no listeners remain.
7321        if !self.needs_emit_deltas() {
7322            self.engine.set_emit_deltas(false);
7323        }
7324        result
7325    }
7326
7327    /// Snapshot of all registered view definitions.
7328    pub fn views(&self) -> Vec<ViewDef> {
7329        self.view_store.views().cloned().collect()
7330    }
7331
7332    // -----------------------------------------------------------------------
7333    // Full-text-lite API
7334    // -----------------------------------------------------------------------
7335
7336    /// Enable full-text indexing for all nodes of `label` on property `field`.
7337    ///
7338    /// After this call, every subsequent write to `(label, field)` is reflected
7339    /// in the index incrementally.  Existing nodes are backfilled immediately.
7340    /// The declaration is persisted as a WAL record; the index itself is rebuilt
7341    /// from scratch on re-open (no snapshot format changes).
7342    ///
7343    /// # Errors
7344    /// - [`GraphError::ReadOnly`]: called on an as-of instance.
7345    /// - [`GraphError::RuleInvalid`]: `(label, field)` is already indexed.
7346    pub fn enable_fulltext(&mut self, label: &str, field: &str) -> Result<()> {
7347        if self.read_only {
7348            return Err(GraphError::ReadOnly);
7349        }
7350        if self.fulltext.is_enabled(label, field) {
7351            return Err(GraphError::RuleInvalid {
7352                detail: format!("full-text index for ({label:?}, {field:?}) already enabled"),
7353            });
7354        }
7355        self.log_then_apply(WalRecord::EnableFulltext {
7356            label: label.into(),
7357            field: field.into(),
7358        })
7359    }
7360
7361    /// Disable full-text indexing for `(label, field)` and drop its postings.
7362    ///
7363    /// # Errors
7364    /// - [`GraphError::ReadOnly`]: called on an as-of instance.
7365    /// - [`GraphError::RuleNotFound`]: `(label, field)` is not currently indexed.
7366    pub fn disable_fulltext(&mut self, label: &str, field: &str) -> Result<()> {
7367        if self.read_only {
7368            return Err(GraphError::ReadOnly);
7369        }
7370        if !self.fulltext.is_enabled(label, field) {
7371            return Err(GraphError::RuleNotFound {
7372                name: format!("fulltext({label},{field})"),
7373            });
7374        }
7375        self.log_then_apply(WalRecord::DisableFulltext {
7376            label: label.into(),
7377            field: field.into(),
7378        })
7379    }
7380
7381    /// Whether `(label, field)` is currently indexed for full-text search.
7382    pub fn is_fulltext_enabled(&self, label: &str, field: &str) -> bool {
7383        self.fulltext.is_enabled(label, field)
7384    }
7385
7386    /// Every `(label, field)` pair with a live full-text index, sorted.
7387    ///
7388    /// Note that [`GraphDb::search`] is keyed by field alone — a pair only
7389    /// declares which nodes are *indexed*, so callers that want to search
7390    /// everything indexed should query each distinct field once.
7391    pub fn fulltext_pairs(&self) -> Vec<(String, String)> {
7392        let mut v: Vec<(String, String)> = self.fulltext.enabled_pairs().cloned().collect();
7393        v.sort();
7394        v
7395    }
7396
7397    /// Enable an equality index for all nodes of `label` on scalar property
7398    /// `field`. Subsequent `WHERE n.field = value` lookups become O(matches)
7399    /// instead of an O(N_label) scan. Existing nodes are backfilled; the
7400    /// declaration persists via WAL and the postings rebuild on re-open.
7401    ///
7402    /// # Errors
7403    /// - [`GraphError::ReadOnly`]: called on an as-of instance.
7404    /// - [`GraphError::RuleInvalid`]: `(label, field)` is already indexed.
7405    pub fn enable_index(&mut self, label: &str, field: &str) -> Result<()> {
7406        if self.read_only {
7407            return Err(GraphError::ReadOnly);
7408        }
7409        if self.prop_index.is_enabled(label, field) {
7410            return Err(GraphError::RuleInvalid {
7411                detail: format!("property index for ({label:?}, {field:?}) already enabled"),
7412            });
7413        }
7414        self.log_then_apply(WalRecord::EnableIndex {
7415            label: label.into(),
7416            field: field.into(),
7417        })
7418    }
7419
7420    /// Disable the equality index for `(label, field)` and drop its postings.
7421    ///
7422    /// # Errors
7423    /// - [`GraphError::ReadOnly`]: called on an as-of instance.
7424    /// - [`GraphError::RuleNotFound`]: `(label, field)` is not currently indexed.
7425    pub fn disable_index(&mut self, label: &str, field: &str) -> Result<()> {
7426        if self.read_only {
7427            return Err(GraphError::ReadOnly);
7428        }
7429        if !self.prop_index.is_enabled(label, field) {
7430            return Err(GraphError::RuleNotFound {
7431                name: format!("index({label},{field})"),
7432            });
7433        }
7434        self.log_then_apply(WalRecord::DisableIndex {
7435            label: label.into(),
7436            field: field.into(),
7437        })
7438    }
7439
7440    /// Whether `(label, field)` currently has an equality index.
7441    pub fn is_index_enabled(&self, label: &str, field: &str) -> bool {
7442        self.prop_index.is_enabled(label, field)
7443    }
7444
7445    /// Every `(label, field)` pair with a live equality index, sorted.
7446    ///
7447    /// The enumerator [`is_index_enabled`](Self::is_index_enabled) never had:
7448    /// the MCP `schema` tool lists a store's indexes rather than probing them
7449    /// one guess at a time.
7450    pub fn index_pairs(&self) -> Vec<(String, String)> {
7451        self.prop_index.enabled_pairs().cloned().collect()
7452    }
7453
7454    /// The dense id of a live key, `None` for an unknown or deleted one.
7455    ///
7456    /// Ids are allocated in insertion order and never reused, so the lowest
7457    /// live id among a set of keys is the oldest node — which is how identity
7458    /// resolution picks a canonical node without a timestamp field
7459    /// (`memory::identity`). Crate-private: an id is an engine detail, not
7460    /// something a caller should hold.
7461    pub(crate) fn dense_id(&self, key: &str) -> Option<u32> {
7462        self.ids.get(key)
7463    }
7464
7465    /// Start recording insert-count multiplicity on this store (§5.13).
7466    ///
7467    /// Adjacency stays a set and nothing about an existing read changes: a
7468    /// duplicate [`insert_edge`](Self::insert_edge) still returns `Ok(false)`
7469    /// and still leaves [`degree`](Self::degree) alone. What it gains is that
7470    /// the duplicate is *counted*, as the reserved edge property
7471    /// [`EDGE_COUNT_PROP`], readable through
7472    /// [`degree_multiplicity`](Self::degree_multiplicity).
7473    ///
7474    /// # This is a one-way step, and that is why it is a call
7475    ///
7476    /// The count is durable, so it is written to the WAL — as discriminant 23,
7477    /// which no release before v0.6.10 knows. A reader meeting an unknown WAL
7478    /// discriminant cannot know what the record would have changed, so it
7479    /// cannot degrade the way an unreadable index blob can. **After this call
7480    /// the store can no longer be read by an older binary, and there is no call
7481    /// that undoes it.** Gating the record behind this method is what keeps
7482    /// that step a decision an operator makes when they want the feature,
7483    /// rather than one everybody takes by upgrading.
7484    ///
7485    /// # It fails loudly, and that costs a snapshot
7486    ///
7487    /// An older binary does not refuse discriminant 23 — it truncates the WAL
7488    /// at it and, with `repair_wal`, persists the truncation. So this call also
7489    /// writes a **V10 snapshot**, a version no earlier release knows, and it
7490    /// writes it *first*: the snapshot is read before the WAL, so an older
7491    /// binary stops at `snapshot: unsupported version 10` with the WAL
7492    /// untouched. Taking the snapshot before appending the record is what makes
7493    /// the guard unconditional — the store is never, at any interruption point,
7494    /// carrying the record without the stamp that announces it.
7495    ///
7496    /// The snapshot keeps the WAL (`keep_wal: true`): opting in is not a
7497    /// compaction, and history reachable by [`open_at`](Self::open_at) stays
7498    /// reachable. On a large store the call therefore costs one full snapshot
7499    /// write.
7500    ///
7501    /// # What it costs a store that archives
7502    ///
7503    /// Writing `snapshot.bin` is also how the archive path decides whether the
7504    /// store may have a *genesis chain* — whether `open_at` can replay
7505    /// archive-resident commits from empty state. The rule is conservative: a
7506    /// snapshot that was already on disk might have been a truncating one, and
7507    /// once the handle that took it is gone this binary cannot tell. A
7508    /// `keep_wal` snapshot taken by **this** handle is the case where it can, so
7509    /// opting in and then archiving **in the same session** keeps the chain.
7510    ///
7511    /// Opting in, closing the store, and archiving in a *later* session does
7512    /// not — but that is the answer any store with a prior snapshot gets, not
7513    /// something this call causes. A store that wants the chain should take its
7514    /// first archive in the session that opted in.
7515    ///
7516    /// Calling it on a store that has already opted in writes nothing and
7517    /// returns `Ok(())`: an operator should not have to ask first.
7518    ///
7519    /// # This call is not atomic, and an `Err` does not undo it
7520    ///
7521    /// There is no rollback here, and there never was one. An `Err` means this
7522    /// handle stopped believing the store is opted in — `self.multiplicity` is
7523    /// reset, so this handle reports `false` from then on — and nothing more. It
7524    /// says nothing about what reached disk. Two reachable failures leave the
7525    /// opt-in standing:
7526    ///
7527    /// * **The declaration landed and only its fsync failed.** `log_then_apply`
7528    ///   appends, then syncs; a failed barrier leaves `MULTIPLICITY_ENABLED`
7529    ///   already in `wal.bin`. The next open replays it and the store is opted
7530    ///   in. No archive is involved — this one predates the recovery below.
7531    /// * **The declaration never landed, but the V10 snapshot did, on a store
7532    ///   that already had an archive.** The open-time recovery in
7533    ///   `load_from_disk` reads V10-beside-an-archive as an interrupted archive
7534    ///   sequence and opts the store in.
7535    ///
7536    /// So a failed call may leave the opt-in on disk immediately (the first
7537    /// case) or conjure it at the next open (the second), and nothing puts the
7538    /// store back out. Treat `Err` as "the outcome is unknown", not as "nothing
7539    /// happened".
7540    ///
7541    /// **This is safe, and the ordering is the reason.** The V10 stamp is
7542    /// written *before* the declaration, so every one of these intermediate
7543    /// states is one an older binary refuses by name rather than truncates at.
7544    /// The failure direction costs a refusal, never a commit. That ordering is
7545    /// the property worth protecting, not the atomicity this call never had.
7546    ///
7547    /// **To know where the store stands, ask the store.** Reopen it and call
7548    /// [`is_multiplicity_enabled`](Self::is_multiplicity_enabled); that is the
7549    /// only answer that accounts for what reached disk.
7550    ///
7551    /// The one case that really does leave the store opted out is a failure with
7552    /// no archive present and no record written: a stray V10 snapshot remains,
7553    /// costing an older reader a refusal it did not strictly need, and *that*
7554    /// store's next snapshot rewrites at V9.
7555    ///
7556    /// # Errors
7557    /// - [`GraphError::ReadOnly`]: called on an as-of instance.
7558    /// - Anything [`snapshot_with`](Self::snapshot_with) can return, and
7559    ///   anything the WAL append or its fsync can return. See the atomicity
7560    ///   section above for what the store is left holding.
7561    pub fn enable_multiplicity(&mut self) -> Result<()> {
7562        if self.read_only {
7563            return Err(GraphError::ReadOnly);
7564        }
7565        if self.multiplicity {
7566            return Ok(());
7567        }
7568        // The snapshot goes first, and the order is the guard.
7569        //
7570        // An older reader refuses a V10 snapshot by name and stops; it does not
7571        // refuse discriminant 23, it truncates the WAL at it. So the store must
7572        // never hold the record without the snapshot that announces it — not
7573        // even for the width of one fsync. Writing the snapshot before the
7574        // record makes the only reachable intermediate state "V10 snapshot, no
7575        // record", which is merely conservative: this binary reads it as a
7576        // store that has not opted in, and an older one refuses it.
7577        //
7578        // `keep_wal: true` because opting in is not a compaction: an operator
7579        // asking for multiplicity has not asked to lose the history `open_at`
7580        // can reach.
7581        self.multiplicity = true;
7582        let forced = self
7583            .snapshot_with(SnapshotOptions {
7584                keep_wal: true,
7585                ..SnapshotOptions::default()
7586            })
7587            .and_then(|()| self.log_then_apply(core_storage::wal::MULTIPLICITY_ENABLED));
7588        if forced.is_err() {
7589            // This handle stops believing it is opted in. That is all this line
7590            // does — it is not a rollback, and cannot be one: the declaration
7591            // may already be in `wal.bin` (the append succeeded and only the
7592            // fsync failed), and even when it is not, the V10 snapshot beside an
7593            // existing archive is enough for the open-time recovery to opt the
7594            // store in. See the "not atomic" section on this method.
7595            //
7596            // It fails in the safe direction either way: the V10 stamp reached
7597            // disk before anything a v0.6.9 reader would truncate at, so the
7598            // worst an interruption costs that reader is a refusal by name.
7599            self.multiplicity = false;
7600        }
7601        forced
7602    }
7603
7604    /// Whether this store records insert-count multiplicity.
7605    ///
7606    /// `false` on every store that has not called
7607    /// [`enable_multiplicity`](Self::enable_multiplicity) — which is every
7608    /// store that did not ask for it, including one upgraded from an earlier
7609    /// release.
7610    pub fn is_multiplicity_enabled(&self) -> bool {
7611        self.multiplicity
7612    }
7613
7614    /// How many times `(etype, src, dst)` has been inserted: the reserved
7615    /// `count` edge property, or 1 when it is absent.
7616    ///
7617    /// Answers 1 for a pair on a store that never opted in, which is the truth
7618    /// available there — the pair was inserted at least once, and the store
7619    /// kept no record of any second insert.
7620    fn edge_insert_count(&self, etype: u32, src: u32, dst: u32) -> u64 {
7621        match self.edge_props_view().get(etype, src, dst, EDGE_COUNT_PROP) {
7622            Some(Value::Int(n)) if n > 0 => n as u64,
7623            _ => 1,
7624        }
7625    }
7626
7627    /// The `SetEdgeCount` record a duplicate insert of `(edge_type, src_key,
7628    /// dst_key)` should log, or `None` when nothing should be written.
7629    ///
7630    /// `None` when the store has not opted in, so **no discriminant-23 record
7631    /// is written at all** — the gate the whole feature rests on.
7632    ///
7633    /// The other two `None`s are unreachable from the one caller. This is the
7634    /// single-mutation path, where `prepare_insert_edge` has already refused a
7635    /// missing endpoint and an existing pair's edge type is necessarily
7636    /// interned. A batch is the case where a pair's endpoints and type can all
7637    /// be created by the same frame, and it does not come through here: it
7638    /// queues a [`PlannedRec::DuplicateCount`] and names the count in the dense
7639    /// rewrite, which is the only pass that knows the frame's own ids.
7640    fn edge_count_record(
7641        &self,
7642        edge_type: &str,
7643        src_key: &str,
7644        dst_key: &str,
7645    ) -> Option<WalRecord> {
7646        if !self.multiplicity {
7647            return None;
7648        }
7649        let etype = self.syms.get(edge_type)?;
7650        let src = self.ids.get(src_key)?;
7651        let dst = self.ids.get(dst_key)?;
7652        Some(WalRecord::SetEdgeCount {
7653            etype,
7654            src,
7655            dst,
7656            count: self.edge_insert_count(etype, src, dst).saturating_add(1),
7657        })
7658    }
7659
7660    /// Search a full-text-indexed field.
7661    ///
7662    /// Returns `(node_key, match_count)` pairs sorted by match_count descending,
7663    /// ties broken by key (lexicographic).  Tombstoned nodes are excluded.
7664    ///
7665    /// **Query syntax:**
7666    /// - Space-separated terms are AND'd: `"foo bar"` requires both.
7667    /// - `OR` between terms forms disjunction: `"foo OR bar"` matches either.
7668    /// - Trailing `*` on a term is a prefix match: `"rust*"` matches `rustlang`, `rusty`.
7669    /// - `AND` keyword is accepted explicitly and is the default.
7670    /// - Tokenization is unicode-alphanumeric (same as index time); case-insensitive.
7671    ///
7672    /// **Unindexed field:** returns `Ok(vec![])` if `field` is not indexed.
7673    /// Pin: this is the documented, tested, stable behavior for v1.
7674    ///
7675    /// **Memory / performance:** O(postings) lookup; no scan.  The index is
7676    /// in-memory and proportional to total indexed text across all enabled fields.
7677    ///
7678    /// **v2 grammar:** supports `"phrase"`, `-negation`, `prefix*`, `OR`, `AND`.
7679    /// Results are BM25-scored (k1=1.2, b=0.75) and sorted by score descending,
7680    /// key ascending for deterministic tiebreaking.
7681    pub fn search(&self, field: &str, query: &str) -> Vec<(String, f64)> {
7682        // Resolve node_ids to keys (excluding tombstones) then re-sort by
7683        // (score DESC, key ASC) to give a deterministic, key-lexicographic
7684        // tiebreak.  FulltextIndex::search sorts by (score DESC, node_id ASC)
7685        // which diverges from key order when nodes were not inserted in key-lex order.
7686        let mut results: Vec<(String, f64)> = self
7687            .fulltext
7688            .search(field, query, 0)
7689            .into_iter()
7690            .filter_map(|(id, score)| self.ids.key_of(id).map(|key| (key.to_string(), score)))
7691            .collect();
7692        results.sort_by(|a, b| {
7693            b.1.partial_cmp(&a.1)
7694                .unwrap_or(std::cmp::Ordering::Equal)
7695                .then(a.0.cmp(&b.0))
7696        });
7697        results
7698    }
7699
7700    /// [`search`](Self::search), stopping at the `k` best hits.
7701    ///
7702    /// Same ranking and the same deterministic tiebreak, but the index drops
7703    /// everything past `k` before any key is resolved, so a caller that wants
7704    /// the top few out of a field that matched thousands does not pay to
7705    /// materialise and re-sort the tail. `k == 0` means no limit, exactly as
7706    /// [`search`](Self::search) behaves.
7707    ///
7708    /// The BM25 scoring itself is not bounded by `k` — every candidate is
7709    /// scored either way — so this trims the resolve and the sort, not the
7710    /// search.
7711    pub fn search_top(&self, field: &str, query: &str, k: usize) -> Vec<(String, f64)> {
7712        // A tombstoned id resolves to nothing, so asking the index for exactly
7713        // `k` could return fewer. Over-fetching a little and truncating after
7714        // the filter keeps the count right without unbounding the call.
7715        let want = if k == 0 { 0 } else { k.saturating_mul(2) };
7716        let mut results: Vec<(String, f64)> = self
7717            .fulltext
7718            .search(field, query, want)
7719            .into_iter()
7720            .filter_map(|(id, score)| self.ids.key_of(id).map(|key| (key.to_string(), score)))
7721            .collect();
7722        results.sort_by(|a, b| {
7723            b.1.partial_cmp(&a.1)
7724                .unwrap_or(std::cmp::Ordering::Equal)
7725                .then(a.0.cmp(&b.0))
7726        });
7727        if k > 0 {
7728            results.truncate(k);
7729        }
7730        results
7731    }
7732
7733    /// Hybrid search: Reciprocal Rank Fusion (RRF) over fulltext + vector results.
7734    ///
7735    /// Takes up to `4*k` fulltext hits for `(text_field, query_text)` and up to
7736    /// `4*k` vector hits for `(vector_field, query_vec, min=0.0)`, then fuses
7737    /// them with RRF using a fixed constant of 60.
7738    ///
7739    /// ```text
7740    /// score(d) = Σ  1 / (60 + rank_i(d))    (rank 1-based per list)
7741    /// ```
7742    ///
7743    /// Returns the top `k` nodes by fused score, ties broken by node key
7744    /// ascending (deterministic).
7745    ///
7746    /// # Vector leg fallback
7747    ///
7748    /// When `query_vec` is empty the vector leg is skipped entirely and
7749    /// results are ranked by the text list alone through the same RRF path
7750    /// (each text result scores `1/(60 + rank)` from that single list).
7751    ///
7752    /// When `label` is `None`, the vector leg **always** returns empty results.
7753    /// Internally `label` is mapped to `""`, which does not match any rule-created
7754    /// HNSW index (all such indexes are keyed to a specific non-empty label), and
7755    /// the brute-force fallback finds no nodes with an empty label.  The fused
7756    /// ranking is therefore text-only in this case.
7757    pub fn search_hybrid(
7758        &self,
7759        text_field: &str,
7760        query_text: &str,
7761        vector_field: &str,
7762        query_vec: &[f64],
7763        label: Option<&str>,
7764        k: usize,
7765    ) -> Vec<(String, f64)> {
7766        self.search_hybrid_inner(
7767            text_field,
7768            query_text,
7769            vector_field,
7770            query_vec,
7771            label,
7772            k,
7773            None,
7774        )
7775    }
7776
7777    /// [`search_hybrid`](Self::search_hybrid) with **each leg** filtered to the
7778    /// mask before the fusion.
7779    ///
7780    /// Filtering the fused list afterwards would quietly return fewer than `k`.
7781    /// Each leg over-fetches `4*k` candidates, so when the visible nodes rank
7782    /// below `4*k` hidden ones neither leg carries them into the fusion at all
7783    /// and the post-filter has nothing left to keep. Filtering first spends the
7784    /// `4*k` on **visible** hits, so a scoped call is as long as the corpus it
7785    /// can see allows.
7786    ///
7787    /// The ranks that enter RRF are therefore the ranks of the visible corpus,
7788    /// not the visible entries of the store-wide ranking. The constant stays 60
7789    /// and the tiebreak stays key-ascending.
7790    #[allow(clippy::too_many_arguments)]
7791    pub fn search_hybrid_scoped(
7792        &self,
7793        text_field: &str,
7794        query_text: &str,
7795        vector_field: &str,
7796        query_vec: &[f64],
7797        label: Option<&str>,
7798        k: usize,
7799        mask: &crate::mask::NodeMask,
7800    ) -> Vec<(String, f64)> {
7801        self.search_hybrid_inner(
7802            text_field,
7803            query_text,
7804            vector_field,
7805            query_vec,
7806            label,
7807            k,
7808            Some(mask),
7809        )
7810    }
7811
7812    /// The body shared by [`search_hybrid`](Self::search_hybrid) and
7813    /// [`search_hybrid_scoped`](Self::search_hybrid_scoped). `mask = None` is
7814    /// the unscoped contract unchanged: the filter below is then a no-op and
7815    /// the vector leg is the same unmasked call it has always been.
7816    #[allow(clippy::too_many_arguments)]
7817    fn search_hybrid_inner(
7818        &self,
7819        text_field: &str,
7820        query_text: &str,
7821        vector_field: &str,
7822        query_vec: &[f64],
7823        label: Option<&str>,
7824        k: usize,
7825        mask: Option<&crate::mask::NodeMask>,
7826    ) -> Vec<(String, f64)> {
7827        use std::collections::HashMap;
7828
7829        const RRF_K: f64 = 60.0;
7830        let pool = 4 * k;
7831
7832        // Accumulate per-node RRF scores.
7833        let mut scores: HashMap<String, f64> = HashMap::new();
7834
7835        // Text leg. The mask bites on the candidates, before `take(pool)`, so
7836        // the over-fetch is a budget of visible hits rather than one a hidden
7837        // prefix can exhaust.
7838        let text_hits = self.search(text_field, query_text);
7839        let visible_text = text_hits
7840            .into_iter()
7841            .filter(|(key, _count)| mask.is_none_or(|m| m.contains_node(self, key)));
7842        for (rank0, (key, _count)) in visible_text.take(pool).enumerate() {
7843            let rank = (rank0 + 1) as f64;
7844            *scores.entry(key).or_insert(0.0) += 1.0 / (RRF_K + rank);
7845        }
7846
7847        // Vector leg (skipped when query_vec is empty). The masked variant
7848        // applies the mask before its own k-truncation, for the same reason.
7849        if !query_vec.is_empty() {
7850            // `ExactnessCaller::Hybrid`: the leg is the same one
7851            // `find_similar_vector_masked` runs, but the advice its warning
7852            // gives has to fit *this* signature, which has no `exact`.
7853            let vec_hits = self
7854                .find_similar_vector_as(
7855                    vector_field,
7856                    label,
7857                    query_vec,
7858                    pool,
7859                    0.0,
7860                    mask,
7861                    None,
7862                    false,
7863                    ExactnessCaller::Hybrid,
7864                )
7865                .expect("find_similar_vector_as is infallible without where_");
7866            for (rank0, (key, _sim)) in vec_hits.into_iter().enumerate() {
7867                let rank = (rank0 + 1) as f64;
7868                *scores.entry(key).or_insert(0.0) += 1.0 / (RRF_K + rank);
7869            }
7870        }
7871
7872        // Sort: score DESC, then key ASC for deterministic tie-breaking.
7873        let mut ranked: Vec<(String, f64)> = scores.into_iter().collect();
7874        ranked.sort_by(|a, b| {
7875            b.1.partial_cmp(&a.1)
7876                .unwrap_or(std::cmp::Ordering::Equal)
7877                .then(a.0.cmp(&b.0))
7878        });
7879        ranked.truncate(k);
7880        ranked
7881    }
7882
7883    /// For DST/testing: scratch BM25 search over live nodes without the index.
7884    /// Walks every live node, re-stems field tokens, computes corpus stats, and
7885    /// returns BM25-ranked results.
7886    ///
7887    /// The oracle: the ordered key list of `search(field, q)` must equal that of
7888    /// `scratch_search(field, q)` at every quiescent state.
7889    #[doc(hidden)]
7890    pub fn scratch_search(&self, field: &str, query: &str) -> Vec<(String, f64)> {
7891        use core_storage::fulltext::{parse_query, value_tokens_stemmed_with_positions};
7892        use std::collections::BTreeMap;
7893
7894        let groups = parse_query(query);
7895        if groups.is_empty() {
7896            return vec![];
7897        }
7898
7899        // --- Pass 1: collect all live indexed nodes with stemmed token data ---
7900        struct NodeData {
7901            key: String,
7902            /// stemmed_token → positions (sorted)
7903            tokens: BTreeMap<String, Vec<u32>>,
7904            dl: u32,
7905        }
7906
7907        let mut nodes: Vec<NodeData> = Vec::new();
7908        for id in 0..self.ids.len() as u32 {
7909            let Some(key) = self.ids.key_of(id) else {
7910                continue;
7911            };
7912            let Some(&sym) = self.labels.get(id as usize) else {
7913                continue;
7914            };
7915            if sym == u32::MAX {
7916                continue;
7917            }
7918            let label = match self.syms.resolve(sym) {
7919                Some(l) => l,
7920                None => continue,
7921            };
7922            if !self.fulltext.is_enabled(label, field) {
7923                continue;
7924            }
7925            let Some(value) = self.props_view().get(id, field).map(|vr| vr.into_value()) else {
7926                continue;
7927            };
7928            // Use value_tokens_stemmed_with_positions so list elements are
7929            // separated by POSITION_GAP — identical to the index path, which
7930            // prevents phrase queries from matching across element boundaries.
7931            let stemmed_with_pos = match &value {
7932                Value::Str(_) | Value::List(_) => value_tokens_stemmed_with_positions(&value),
7933                _ => continue,
7934            };
7935            let dl = stemmed_with_pos.len() as u32;
7936            let mut tok_map: BTreeMap<String, Vec<u32>> = BTreeMap::new();
7937            for (tok, pos) in stemmed_with_pos {
7938                tok_map.entry(tok).or_default().push(pos);
7939            }
7940            nodes.push(NodeData {
7941                key: key.to_string(),
7942                tokens: tok_map,
7943                dl,
7944            });
7945        }
7946
7947        if nodes.is_empty() {
7948            return vec![];
7949        }
7950
7951        // --- BM25 corpus stats ---
7952        let n = nodes.len() as f64;
7953        let avg_dl: f64 = nodes.iter().map(|nd| nd.dl as f64).sum::<f64>() / n;
7954        // df per stemmed token across all live indexed nodes.
7955        let mut df_map: BTreeMap<&str, f64> = BTreeMap::new();
7956        for nd in &nodes {
7957            for tok in nd.tokens.keys() {
7958                *df_map.entry(tok.as_str()).or_insert(0.0) += 1.0;
7959            }
7960        }
7961
7962        const K1: f64 = 1.2;
7963        const B: f64 = 0.75;
7964
7965        // --- Pass 2: score each node against each OR-group ---
7966        let mut results: Vec<(String, f64)> = Vec::new();
7967        for nd in &nodes {
7968            let dl = nd.dl as f64;
7969            let mut total_score = 0.0f64;
7970
7971            'group: for group in &groups {
7972                let mut group_score = 0.0f64;
7973
7974                for term in group {
7975                    if term.negated {
7976                        // Negated: if doc has this stemmed token → group fails.
7977                        let present = if term.prefix {
7978                            nd.tokens.keys().any(|t| t.starts_with(term.token.as_str()))
7979                        } else {
7980                            nd.tokens.contains_key(term.token.as_str())
7981                        };
7982                        if present {
7983                            continue 'group;
7984                        }
7985                        continue;
7986                    }
7987                    if term.prefix {
7988                        // Prefix: sum BM25 for all matching stemmed tokens.
7989                        let mut prefix_matched = false;
7990                        for (tok, positions) in &nd.tokens {
7991                            if tok.starts_with(term.token.as_str()) {
7992                                let tf = positions.len() as f64;
7993                                let df = df_map.get(tok.as_str()).copied().unwrap_or(1.0);
7994                                let idf = ((n - df + 0.5) / (df + 0.5) + 1.0).ln();
7995                                let tf_norm =
7996                                    tf * (K1 + 1.0) / (tf + K1 * (1.0 - B + B * dl / avg_dl));
7997                                group_score += idf * tf_norm;
7998                                prefix_matched = true;
7999                            }
8000                        }
8001                        if !prefix_matched {
8002                            continue 'group;
8003                        }
8004                    } else {
8005                        // term.token is already stemmed by parse_query; use directly.
8006                        match nd.tokens.get(term.token.as_str()) {
8007                            None => continue 'group,
8008                            Some(positions) => {
8009                                let tf = positions.len() as f64;
8010                                let df = df_map.get(term.token.as_str()).copied().unwrap_or(1.0);
8011                                let idf = ((n - df + 0.5) / (df + 0.5) + 1.0).ln();
8012                                let tf_norm =
8013                                    tf * (K1 + 1.0) / (tf + K1 * (1.0 - B + B * dl / avg_dl));
8014                                group_score += idf * tf_norm;
8015                            }
8016                        }
8017                    }
8018                }
8019
8020                if group_score > 0.0 {
8021                    total_score += group_score;
8022                }
8023            }
8024
8025            if total_score > 0.0 {
8026                results.push((nd.key.clone(), total_score));
8027            }
8028        }
8029
8030        results.sort_by(|a, b| {
8031            b.1.partial_cmp(&a.1)
8032                .unwrap_or(std::cmp::Ordering::Equal)
8033                .then(a.0.cmp(&b.0))
8034        });
8035        results
8036    }
8037
8038    /// Return the current view-maintained value of `view_prop` for node `key`.
8039    /// Equivalent to `get_prop` but documents that it reads a view-managed column.
8040    pub fn get_view_prop(&self, key: &str, view_prop: &str) -> Option<Value> {
8041        let id = self.ids.get(key)?;
8042        self.props_view()
8043            .get(id, view_prop)
8044            .map(|vr| vr.into_value())
8045    }
8046
8047    /// For testing / DST oracle: scratch recompute of a view value for one node.
8048    ///
8049    /// Returns `None` if the node does not exist, the view does not exist, or
8050    /// the view has no result for the node (e.g. Avg with no qualifying neighbors).
8051    #[doc(hidden)]
8052    pub fn scratch_view_value(&self, key: &str, view_name: &str) -> Option<Value> {
8053        let node = self.ids.get(key)?;
8054        let def = self.view_store.views().find(|v| v.name == view_name)?;
8055        // Use TopologyView so that NeighborAgg sees base + overlay edges
8056        // without materialising a temporary Topology (I1).
8057        let topo_view = self.topo_view();
8058        core_rules::views::compute_view_value(
8059            def,
8060            node,
8061            self.props_view(),
8062            &topo_view,
8063            &self.ids,
8064            &self.syms,
8065            &self.labels,
8066        )
8067    }
8068
8069    // -----------------------------------------------------------------------
8070    // Graph algorithm API
8071    // -----------------------------------------------------------------------
8072
8073    /// Run PageRank over the unified topology (manual + derived edges).
8074    ///
8075    /// Returns a [`PageRankReport`] with scores sorted descending (ties: key
8076    /// ascending).  Set `config.edge_type` to restrict to one edge type.
8077    /// `config.converged` is `true` only when the power iteration converged
8078    /// within `config.max_iters` and within any time budget.
8079    pub fn pagerank(&self, config: &crate::algo::PageRankConfig) -> crate::algo::PageRankReport {
8080        let topo = build_topo_view(&self.topo, &self.base);
8081        let edge_props = self.edge_props_view();
8082        crate::algo::pagerank(
8083            &topo,
8084            &self.ids,
8085            &self.syms,
8086            &self.labels,
8087            &edge_props,
8088            config,
8089        )
8090    }
8091
8092    /// Weakly-connected components over the unified topology (treated as
8093    /// undirected regardless of how edges were inserted).
8094    ///
8095    /// Component IDs are the key of the smallest member in the component
8096    /// (deterministic).  Result sorted by (component_id, key).
8097    pub fn connected_components(&self, config: &crate::algo::WccConfig) -> crate::algo::WccReport {
8098        let topo = build_topo_view(&self.topo, &self.base);
8099        let edge_props = self.edge_props_view();
8100        crate::algo::wcc(
8101            &topo,
8102            &self.ids,
8103            &self.syms,
8104            &self.labels,
8105            &edge_props,
8106            config,
8107        )
8108    }
8109
8110    /// Degree centrality for every live node.
8111    ///
8112    /// `direction`: `AlgoDir::Out` = out-degree, `AlgoDir::In` = in-degree,
8113    /// `AlgoDir::Both` = out + in (total directed degree).
8114    ///
8115    /// For one-shot ranking use this; for a live property updated on every
8116    /// write, create a Degree materialized view instead (see `docs/site/algorithms.md`).
8117    pub fn degree_centrality(
8118        &self,
8119        config: &crate::algo::DegreeConfig,
8120    ) -> crate::algo::DegreeReport {
8121        let topo = build_topo_view(&self.topo, &self.base);
8122        let edge_props = self.edge_props_view();
8123        crate::algo::degree_centrality(
8124            &topo,
8125            &self.ids,
8126            &self.syms,
8127            &self.labels,
8128            &edge_props,
8129            config,
8130        )
8131    }
8132
8133    /// Louvain community detection over the unified topology (undirected).
8134    ///
8135    /// See [`crate::algo::LouvainConfig`] for edge-type/weight/label
8136    /// restriction and [`crate::algo::CommunityReport`] for the shape of the
8137    /// result (communities sorted size-desc, then smallest member key asc).
8138    pub fn communities(&self, config: &crate::algo::LouvainConfig) -> crate::algo::CommunityReport {
8139        let topo = build_topo_view(&self.topo, &self.base);
8140        let edge_props = self.edge_props_view();
8141        crate::algo::louvain(
8142            &topo,
8143            &self.ids,
8144            &self.syms,
8145            &self.labels,
8146            &edge_props,
8147            config,
8148        )
8149    }
8150
8151    /// Write a vector of `(node_key, score)` pairs as `prop_name` on each node,
8152    /// atomically via a single write-batch (one WAL frame, one fsync).
8153    ///
8154    /// # Errors
8155    /// - [`GraphError::ReadOnly`]: called on an as-of instance.
8156    /// - [`GraphError::RuleInvalid`]: `prop_name` is managed by an existing view
8157    ///   (collision check mirrors `create_view`).
8158    /// - [`GraphError::KeyNotFound`]: a key in `scores` does not exist as a live node.
8159    pub fn write_scores(&mut self, prop_name: &str, scores: &[(String, f64)]) -> Result<()> {
8160        if self.read_only {
8161            return Err(GraphError::ReadOnly);
8162        }
8163        // Collision check: refuse if prop_name is view-managed.
8164        if let Some(view_name) = self.view_store.view_for_prop(prop_name) {
8165            return Err(GraphError::RuleInvalid {
8166                detail: format!(
8167                    "prop {:?} is managed by view {:?} and cannot be written as scores",
8168                    prop_name, view_name
8169                ),
8170            });
8171        }
8172        // Refuse if prop_name is a view name itself (confusing namespace collision).
8173        if self.view_store.has_view(prop_name) {
8174            return Err(GraphError::RuleInvalid {
8175                detail: format!(
8176                    "prop_name {:?} collides with an existing view name",
8177                    prop_name
8178                ),
8179            });
8180        }
8181        // Write all scores in a single crash-atomic batch.
8182        self.write_batch(|b| {
8183            for (key, score) in scores {
8184                b.set_prop(key, prop_name, Value::Float(*score));
8185            }
8186        })?;
8187        Ok(())
8188    }
8189
8190    /// Return the value of `field` for the node with key `key`, or `None` if
8191    /// the node or field is absent.  Reads through the overlay-over-base
8192    /// `ColumnsView`, materialising base values on demand (zero heap cost for
8193    /// overlay hits; one clone per base hit).
8194    pub fn get_prop(&self, key: &str, field: &str) -> Option<Value> {
8195        let id = self.ids.get(key)?;
8196        self.props_view().get(id, field).map(|vr| vr.into_value())
8197    }
8198
8199    pub fn has_node(&self, key: &str) -> bool {
8200        self.ids.get(key).is_some()
8201    }
8202
8203    /// One human line for a node: the first of `text`, `summary` or `name` it
8204    /// carries, truncated. Used by the recall digest.
8205    pub fn node_summary_line(&self, key: &str) -> Option<String> {
8206        for field in ["text", "summary", "name"] {
8207            if let Some(Value::Str(s)) = self.get_prop(key, field) {
8208                let s = s.trim();
8209                if !s.is_empty() {
8210                    return Some(s.chars().take(120).collect());
8211                }
8212            }
8213        }
8214        None
8215    }
8216
8217    /// Borrow the raw id map. Used by `NodeMask::from_keys` to resolve keys.
8218    pub(crate) fn ids(&self) -> &IdMap {
8219        &self.ids
8220    }
8221
8222    // -----------------------------------------------------------------------
8223    // Namespaces
8224    // -----------------------------------------------------------------------
8225
8226    /// The index `name` already has in `ns_names`, if any.
8227    fn ns_index_of(&self, name: &str) -> Option<u32> {
8228        self.ns_names
8229            .iter()
8230            .position(|n| n == name)
8231            .map(|i| i as u32)
8232    }
8233
8234    /// The index for `name`, appending it to `ns_names` when it is new.
8235    ///
8236    /// The table holds one entry per distinct namespace in the store — a
8237    /// tenant count, not a node count — so the linear scan is cheaper than a
8238    /// map and keeps `namespaces()` allocation-free of a second index.
8239    fn ns_index_for(&mut self, name: &str) -> u32 {
8240        match self.ns_index_of(name) {
8241            Some(i) => i,
8242            None => {
8243                self.ns_names.push(name.to_string());
8244                (self.ns_names.len() - 1) as u32
8245            }
8246        }
8247    }
8248
8249    /// The namespace name at `idx`, or [`NS_DEFAULT`] for an index this handle
8250    /// does not know (unreachable; the default is the narrowing answer).
8251    fn ns_name(&self, idx: u32) -> &str {
8252        self.ns_names
8253            .get(idx as usize)
8254            .map(String::as_str)
8255            .unwrap_or(NS_DEFAULT)
8256    }
8257
8258    /// The namespace index of dense node `id`, defaulting for an id with no
8259    /// entry (a node inserted before this handle rebuilt the array cannot
8260    /// exist: every insert path maintains it).
8261    fn node_ns_idx(&self, id: u32) -> u32 {
8262        self.node_ns
8263            .get(id as usize)
8264            .copied()
8265            .unwrap_or(NS_DEFAULT_IDX)
8266    }
8267
8268    /// File node `id` under namespace `name`, growing `node_ns` as `labels`
8269    /// grows. Called from `apply` for every node insert, live and replayed.
8270    fn set_node_ns(&mut self, id: u32, name: &str) {
8271        let idx = if name == NS_DEFAULT {
8272            NS_DEFAULT_IDX
8273        } else {
8274            self.ns_index_for(name)
8275        };
8276        if self.node_ns.len() <= id as usize {
8277            self.node_ns.resize(id as usize + 1, NS_DEFAULT_IDX);
8278        }
8279        self.node_ns[id as usize] = idx;
8280    }
8281
8282    /// Rebuild `node_ns` from the `ns` column — one pass, at the end of an
8283    /// open or a reload, after the snapshot is restored and the WAL replayed.
8284    ///
8285    /// A store with no `ns` column reads nothing: the column-name check fails
8286    /// and the vector is filled with one constant.
8287    fn rebuild_node_ns(&mut self) {
8288        let total = self.ids.len();
8289        self.ns_names.truncate(1);
8290        self.node_ns.clear();
8291        self.node_ns.resize(total, NS_DEFAULT_IDX);
8292        let has_ns_column = {
8293            let cv = self.props_view();
8294            cv.field_names().iter().any(|f| f == NS_PROP)
8295        };
8296        if !has_ns_column {
8297            return;
8298        }
8299        // Collected first so the props view is released before `ns_index_for`
8300        // takes `&mut self`.
8301        let named: Vec<(u32, String)> = {
8302            let cv = self.props_view();
8303            (0..total as u32)
8304                .filter_map(|id| match cv.get(id, NS_PROP).map(|vr| vr.into_value()) {
8305                    Some(Value::Str(s)) if s != NS_DEFAULT => Some((id, s)),
8306                    _ => None,
8307                })
8308                .collect()
8309        };
8310        for (id, name) in named {
8311            let idx = self.ns_index_for(&name);
8312            self.node_ns[id as usize] = idx;
8313        }
8314    }
8315
8316    /// Every namespace with at least one live node, in name order.
8317    ///
8318    /// `["default"]` on any store that has never named a namespace, including
8319    /// an empty one: a store is always at least its default namespace.
8320    pub fn namespaces(&self) -> Vec<String> {
8321        let mut out: BTreeSet<&str> = BTreeSet::new();
8322        out.insert(NS_DEFAULT);
8323        for (id, &idx) in self.node_ns.iter().enumerate() {
8324            if idx == NS_DEFAULT_IDX || !self.is_live_node(id as u32) {
8325                continue;
8326            }
8327            out.insert(self.ns_name(idx));
8328        }
8329        out.into_iter().map(str::to_string).collect()
8330    }
8331
8332    /// The namespace of `key`, or `None` when the key names no live node.
8333    pub fn namespace_of(&self, key: &str) -> Option<String> {
8334        let id = self.ids.get(key)?;
8335        if !self.is_live_node(id) {
8336            return None;
8337        }
8338        Some(self.ns_name(self.node_ns_idx(id)).to_string())
8339    }
8340
8341    /// Every live node in `namespace`, as a visibility mask.
8342    ///
8343    /// Built off `node_ns` on whichever handle this is, so on a temporal handle
8344    /// it is the namespace's membership at that commit. A name no node uses
8345    /// gives an empty mask — a namespace scope never widens.
8346    pub fn mask_for_namespace(&self, namespace: &str) -> crate::mask::NodeMask {
8347        let Some(idx) = self.ns_index_of(namespace) else {
8348            return crate::mask::NodeMask::from_ids(std::collections::HashSet::new());
8349        };
8350        let visible: std::collections::HashSet<u32> = (0..self.ids.len() as u32)
8351            .filter(|&id| self.node_ns_idx(id) == idx && self.is_live_node(id))
8352            .collect();
8353        crate::mask::NodeMask::from_ids(visible)
8354    }
8355
8356    /// Live-node test used by the namespace accessors: a deleted node keeps its
8357    /// dense id and its `node_ns` slot, and the label sentinel is what marks it
8358    /// gone — the same test `mask_for_role`'s label leg applies implicitly.
8359    fn is_live_node(&self, id: u32) -> bool {
8360        self.labels
8361            .get(id as usize)
8362            .is_some_and(|&sym| sym != u32::MAX)
8363            && self.ids.key_of(id).is_some()
8364    }
8365
8366    /// Per-namespace live node counts for [`Stats`], in name order.
8367    fn namespace_stats(&self) -> Vec<NamespaceStats> {
8368        let mut counts: BTreeMap<&str, usize> = BTreeMap::new();
8369        counts.insert(NS_DEFAULT, 0);
8370        for id in 0..self.ids.len() as u32 {
8371            if !self.is_live_node(id) {
8372                continue;
8373            }
8374            *counts
8375                .entry(self.ns_name(self.node_ns_idx(id)))
8376                .or_insert(0) += 1;
8377        }
8378        counts
8379            .into_iter()
8380            .filter(|&(name, n)| n > 0 || name == NS_DEFAULT)
8381            .map(|(name, nodes_live)| NamespaceStats {
8382                name: name.to_string(),
8383                nodes_live,
8384            })
8385            .collect()
8386    }
8387
8388    /// The namespace a create-class op would put its node in: the `ns` entry of
8389    /// the props it carries, normalised, with absent meaning [`NS_DEFAULT`].
8390    fn created_namespace<'a>(key: &str, props: &'a [(String, Value)]) -> Result<&'a str> {
8391        Ok(namespace_of_value(Self::sole_ns_entry(key, props)?))
8392    }
8393
8394    /// The one `ns` entry in a node's props, or `None` when it carries none.
8395    ///
8396    /// A props list naming `ns` twice is refused. Without that refusal the
8397    /// write path and the authorisation path can read the same list
8398    /// differently — one taking the first entry, the other the last — and
8399    /// `CREATE (n:L {ns: 'mine', ns: 'theirs'})` lands a node in a namespace
8400    /// the role was checked against the other of. One entry is the only shape
8401    /// where "the node's namespace" is a single fact, so it is the only shape
8402    /// accepted, and every reader of it agrees by construction.
8403    fn sole_ns_entry<'a>(key: &str, props: &'a [(String, Value)]) -> Result<Option<&'a Value>> {
8404        let mut found: Option<&'a Value> = None;
8405        for (field, value) in props {
8406            if field != NS_PROP {
8407                continue;
8408            }
8409            if found.is_some() {
8410                return Err(GraphError::RuleInvalid {
8411                    detail: format!(
8412                        "node {key}: {NS_PROP} is given more than once; a node has exactly \
8413                         one namespace"
8414                    ),
8415                });
8416            }
8417            found = Some(value);
8418        }
8419        Ok(found)
8420    }
8421
8422    /// The definition of the role a write authorisation names.
8423    ///
8424    /// `None` when `roles.json` was corrupt at open or the role has since been
8425    /// removed — neither can reach a write, because the authorisation carries a
8426    /// mask `mask_for_role` already resolved for that name.
8427    fn role_def_for(&self, role: &str) -> Option<&RoleDef> {
8428        self.roles.as_ref()?.iter().find(|r| r.name == role)
8429    }
8430
8431    /// Validate the `ns` entry of a node's props and drop an explicit default.
8432    ///
8433    /// Runs on the write path only (see `rewrite_wal_dense`), never on replay:
8434    /// a record that reached the WAL was already accepted here.
8435    fn normalise_insert_ns(
8436        key: &str,
8437        props: Vec<(String, Value)>,
8438    ) -> Result<(Vec<(String, Value)>, String)> {
8439        // One `ns` or none: this is where that is enforced, so every later
8440        // reader of the list — the authorisation gate, the two `apply` arms,
8441        // `node_ns` — is looking at a single entry and cannot disagree about
8442        // which one counts.
8443        Self::sole_ns_entry(key, &props)?;
8444        let mut name = NS_DEFAULT.to_string();
8445        let mut out = Vec::with_capacity(props.len());
8446        for (field, value) in props {
8447            if field != NS_PROP {
8448                out.push((field, value));
8449                continue;
8450            }
8451            let Value::Str(ref s) = value else {
8452                return Err(GraphError::RuleInvalid {
8453                    detail: format!(
8454                        "node {key}: {NS_PROP} must be a string naming a namespace, \
8455                         got {value:?}"
8456                    ),
8457                });
8458            };
8459            if !valid_namespace(s) {
8460                return Err(GraphError::RuleInvalid {
8461                    detail: format!(
8462                        "node {key}: {s:?} is not a valid namespace name — 1 to {NS_MAX_LEN} \
8463                         characters of [A-Za-z0-9_.-]"
8464                    ),
8465                });
8466            }
8467            name = s.clone();
8468            // An explicit default stores nothing, so a single-tenant store
8469            // never grows an `ns` column.
8470            if name != NS_DEFAULT {
8471                out.push((field, value));
8472            }
8473        }
8474        Ok((out, name))
8475    }
8476
8477    // -----------------------------------------------------------------------
8478    // RBAC role resolution
8479    // -----------------------------------------------------------------------
8480
8481    /// Parse `roles.json` bytes from `fs`.
8482    ///
8483    /// Return values:
8484    ///   `Ok(Some(roles))` — file absent (returns `vec![]`) **or** file present
8485    ///                       and valid; in both cases `mask_for_role` uses the
8486    ///                       list normally (an absent file means no roles defined).
8487    ///   `Ok(None)`        — file present but corrupt or unrecognised version
8488    ///                       → poisoned state; `mask_for_role` returns `Err` for
8489    ///                       any role name until the file is fixed and the DB
8490    ///                       re-opened (or `apply_schema` is called to repair it).
8491    ///
8492    /// Note: `None` signals corruption, not absence — the opposite of what an
8493    /// optional "file missing" convention would suggest.  The open path stores
8494    /// this result on `db.roles` directly.
8495    fn load_roles_from_fs(fs: &F) -> Result<Option<Vec<RoleDef>>> {
8496        let bytes = fs.read(FileId::Roles).map_err(GraphError::Io)?;
8497        if bytes.is_empty() {
8498            // Empty bytes means either the file is absent or zero-byte — both
8499            // are treated identically as "no roles defined".  A zero-byte
8500            // roles.json does NOT widen access: an absent file and a zero-byte
8501            // file both resolve to an empty role list (sees nothing by default).
8502            return Ok(Some(vec![]));
8503        }
8504        match serde_json::from_slice::<RolesFile>(&bytes) {
8505            Ok(f) if matches!(f.version, 1..=4) => Ok(Some(f.roles)),
8506            // Corrupt or unrecognised version (>4): poison the roles state.
8507            // Never widen: a version this binary does not know may carry a
8508            // narrowing this binary would not apply.
8509            _ => Ok(None),
8510        }
8511    }
8512
8513    /// Resolve a role to a node-visibility mask against the current graph state.
8514    ///
8515    /// Returns `Err` when:
8516    /// - `roles.json` was present but corrupt at open (poisoned state), or
8517    /// - `role` does not match any defined role name.
8518    ///
8519    /// The mask union is: explicit `keys` (unknown keys silently ignored) plus
8520    /// all live nodes carrying any label in `labels` that also pass the role's
8521    /// [`visible_where`](crate::roles::RoleDef::visible_where) predicate, if it
8522    /// has one.  Label resolution is live — new nodes of an allowed label are
8523    /// visible without re-applying the schema, and a property edited out of the
8524    /// predicate takes its node out of the mask on the next read.  An empty
8525    /// union = empty mask = sees nothing.
8526    ///
8527    /// This is the one resolver every read path calls, live and as-of alike, so
8528    /// the predicate applies everywhere at once.  On an as-of handle the role
8529    /// *definition* is the current one and the graph is the historical one: the
8530    /// predicate is evaluated against the property values at the commit being
8531    /// read.
8532    ///
8533    /// The result is memoised per `(role, commit_seq)`, so a scoped reader
8534    /// between two writes resolves the role once.  See
8535    /// [`RoleMaskCache`](crate::mask::RoleMaskCache) for why that cannot go
8536    /// stale.
8537    pub fn mask_for_role(&self, role: &str) -> Result<crate::mask::NodeMask> {
8538        self.role_masks
8539            .get_or_build(role, self.commit_seq, || self.build_mask_for_role(role))
8540            .map(|m| (*m).clone())
8541    }
8542
8543    /// The mask an [`AsOfScope`] names, resolved against this handle.
8544    ///
8545    /// Shared by [`GraphDb::query_at_scoped`] and
8546    /// [`GraphDb::query_at_scoped_in_namespace`] so one scope resolves one way
8547    /// however the namespace leg is added.
8548    fn mask_at_scope(&self, scope: AsOfScope<'_>) -> Result<crate::mask::NodeMask> {
8549        // One resolver answers "what may this role see" — `mask_for_role` — and
8550        // it runs against this handle, so on a temporal one the answer is the
8551        // as-of one.
8552        Ok(match scope {
8553            AsOfScope::Role(role) => self.mask_for_role(role)?,
8554            AsOfScope::Keys(keys) => {
8555                crate::mask::NodeMask::from_keys(self, keys.iter().map(String::as_str))
8556            }
8557            AsOfScope::RoleAndKeys(role, keys) => {
8558                self.mask_for_role(role)?
8559                    .intersect(&crate::mask::NodeMask::from_keys(
8560                        self,
8561                        keys.iter().map(String::as_str),
8562                    ))
8563            }
8564            AsOfScope::Namespace(namespace) => self.mask_for_namespace(namespace),
8565        })
8566    }
8567
8568    /// Resolve `role` against the current graph, ignoring the memo.
8569    fn build_mask_for_role(&self, role: &str) -> Result<crate::mask::NodeMask> {
8570        let roles = self.roles.as_ref().ok_or_else(roles_poisoned)?;
8571        let def = roles
8572            .iter()
8573            .find(|r| r.name == role)
8574            .ok_or_else(|| GraphError::KeyNotFound {
8575                key: format!("role:{role}"),
8576            })?;
8577
8578        let mut visible = std::collections::HashSet::new();
8579
8580        // Key leg: resolve explicit keys to dense ids (unknown keys ignored).
8581        // An administrative grant, never narrowed by the predicate.
8582        for key in &def.keys {
8583            if let Some(id) = self.ids.get(key) {
8584                visible.insert(id);
8585            }
8586        }
8587
8588        // Label leg: live scan — iterate labels vec for matching symbol, and
8589        // when the role carries a predicate, test the property as well.  The
8590        // property comes from the store's own merged view (overlay over the
8591        // mmap'd base), so an as-of handle reads the values of its own commit.
8592        let props = def.visible_where.as_ref().map(|_| self.props_view());
8593        for label_name in &def.labels {
8594            if let Some(sym) = self.syms.get(label_name) {
8595                for (i, &s) in self.labels.iter().enumerate() {
8596                    if s != sym {
8597                        continue;
8598                    }
8599                    let id = i as u32;
8600                    match (&def.visible_where, &props) {
8601                        (Some(pred), Some(view)) => {
8602                            let value = view.get(id, &pred.field).map(|vr| vr.into_value());
8603                            if pred.holds(value.as_ref()) {
8604                                visible.insert(id);
8605                            }
8606                        }
8607                        _ => {
8608                            visible.insert(id);
8609                        }
8610                    }
8611                }
8612            }
8613        }
8614
8615        // Namespace leg: an intersection over the whole union, the key leg
8616        // included. A namespace is a tenancy boundary, so a key naming a node in
8617        // another tenant's namespace is not an administrative grant — and
8618        // `apply_schema` has already refused that role, so this only has to be
8619        // right about the node that moved into existence afterwards.
8620        if def.namespaces.is_some() {
8621            visible.retain(|&id| def.sees_namespace(self.ns_name(self.node_ns_idx(id))));
8622        }
8623
8624        Ok(crate::mask::NodeMask::from_ids(visible))
8625    }
8626
8627    /// Return the current list of role definitions.
8628    ///
8629    /// Returns an empty list when no roles are defined or when `roles.json`
8630    /// was corrupt at open (check [`mask_for_role`](Self::mask_for_role) for
8631    /// the fail-loud error in that case, or call
8632    /// [`roles_checked`](Self::roles_checked), which is this readout with that
8633    /// error in it).
8634    pub fn roles(&self) -> Vec<RoleDef> {
8635        self.roles.as_deref().unwrap_or(&[]).to_vec()
8636    }
8637
8638    /// The role definitions, or the poison error when `roles.json` was corrupt
8639    /// at open.
8640    ///
8641    /// [`roles`](Self::roles) answers `[]` both for a store that defines no
8642    /// roles and for one whose sidecar did not parse, and a caller validating a
8643    /// role name at boot cannot tell those apart. The wrong reading of the pair
8644    /// is the dangerous one: a store with no roles at all is an unrestricted
8645    /// store, so a poisoned file would read as "nothing is restricted here".
8646    ///
8647    /// This is the same answer, for the same cause, that
8648    /// [`mask_for_role`](Self::mask_for_role) gives on the first read.
8649    pub fn roles_checked(&self) -> Result<Vec<RoleDef>> {
8650        match self.roles.as_deref() {
8651            Some(roles) => Ok(roles.to_vec()),
8652            None => Err(roles_poisoned()),
8653        }
8654    }
8655
8656    // ── Role-scoped write authz ───────────────────────────────────────────────
8657
8658    /// Execute `ops` with optional role-scoped write authorization.
8659    ///
8660    /// - `None` → full authority, identical to [`write_batch`](Self::write_batch)
8661    ///   (zero-cost bypass of all authz checks).
8662    /// - `Some(authz)` → the decision table is evaluated per-op BEFORE any WAL
8663    ///   record is built.  A denial returns an error with no WAL frame written
8664    ///   (all-or-nothing at the authz boundary, then at the MutPreview boundary).
8665    ///
8666    /// See the plan's "authz decision table" section for the full semantics.
8667    pub fn write_batch_authz(
8668        &mut self,
8669        authz: Option<&WriteAuthz>,
8670        ops: Vec<BatchOp>,
8671    ) -> Result<(usize, usize)> {
8672        // Thread authz as a direct parameter — never touches pending_write_authz.
8673        self.commit_logged_batch(ops, None, authz.cloned())
8674            .map(inserted_pair)
8675    }
8676
8677    /// Execute a Cypher write statement with role-scoped write authorization.
8678    ///
8679    /// Resolves scope + mask from `self.roles` inside the call (same write-guard
8680    /// lifetime as execution, satisfying §5 lock discipline).  The resolved
8681    /// `WriteAuthz` is stored as `pending_write_authz` for the duration of the
8682    /// call so that all inner `batch.commit()` calls are authz-checked.
8683    ///
8684    /// MERGE is handled specially: the MERGE scope precondition (§3.3) is
8685    /// checked in `exec_merge` BEFORE `has_node` to close the §6.2
8686    /// timing-oracle item (hidden ≡ absent for unscoped roles).
8687    ///
8688    /// Roles with `write: None` (v1 behavior) → `RoleWriteDenied` with
8689    /// "this endpoint is not permitted".
8690    pub fn query_write_authz(
8691        &mut self,
8692        role: &str,
8693        cypher: &str,
8694        params: &BTreeMap<String, Value>,
8695    ) -> Result<ResultSet> {
8696        // Resolve scope (fails fast if role has no write scope).
8697        // write:None → byte-identical v1 blanket-403 body (plan §v1-sidecar mandate).
8698        let scope =
8699            {
8700                let roles = self.roles.as_deref().ok_or_else(|| GraphError::Corrupt {
8701                    detail: "roles.json was corrupt at open; re-open to restore role access".into(),
8702                })?;
8703                let def = roles.iter().find(|r| r.name == role).ok_or_else(|| {
8704                    GraphError::KeyNotFound {
8705                        key: format!("role:{role}"),
8706                    }
8707                })?;
8708                def.write
8709                    .clone()
8710                    .ok_or_else(|| GraphError::RoleWriteDenied {
8711                        reason: "role-bound token: writes are not permitted".into(),
8712                    })?
8713            };
8714        // Resolve mask inside the call (same guard, §5 coherence).
8715        let mask = self.mask_for_role(role)?;
8716        self.pending_write_authz = Some(WriteAuthz {
8717            role: role.into(),
8718            scope,
8719            mask,
8720        });
8721        // RAII guard: always clears pending_write_authz on scope exit, including
8722        // on panic or early-return, mirroring the RestoreEmitDeltas precedent.
8723        struct ClearPendingAuthzOnDrop(*mut Option<WriteAuthz>);
8724        impl Drop for ClearPendingAuthzOnDrop {
8725            fn drop(&mut self) {
8726                // SAFETY: pointer into the owning GraphDb; guard is dropped
8727                // within this function's frame before it returns.
8728                unsafe { *self.0 = None };
8729            }
8730        }
8731        // SAFETY: raw pointer into self; guard dropped before this fn returns.
8732        let _authz_guard = ClearPendingAuthzOnDrop(&mut self.pending_write_authz as *mut _);
8733        let tokens = lex(cypher).map_err(|e| GraphError::QueryError {
8734            detail: format!("lex: {e}"),
8735        })?;
8736        let stmt = parse_write(&tokens).map_err(|e| GraphError::QueryError {
8737            detail: format!("parse: {e}"),
8738        })?;
8739        self.exec_write_stmt(stmt, params)
8740    }
8741
8742    /// Execute `ops` with optional role-scoped write authorization, suppressing
8743    /// fsync (for use inside the group-commit drain thread, which performs one
8744    /// group fsync after releasing the write lock).
8745    ///
8746    /// Identical to [`write_batch_authz`] except the fsync policy is temporarily
8747    /// forced to `Relaxed` for the duration of the call, matching the drain-thread
8748    /// contract established by [`commit_batch_nosync`].
8749    pub(crate) fn write_batch_authz_nosync(
8750        &mut self,
8751        authz: Option<&WriteAuthz>,
8752        ops: Vec<BatchOp>,
8753    ) -> Result<(usize, usize)> {
8754        let saved = self.fsync;
8755        struct RestoreFsync(*mut FsyncPolicy, FsyncPolicy);
8756        impl Drop for RestoreFsync {
8757            fn drop(&mut self) {
8758                // SAFETY: pointer into the owning GraphDb; guard is dropped
8759                // within the enclosing function's frame before it returns.
8760                unsafe { *self.0 = self.1 };
8761            }
8762        }
8763        // SAFETY: raw pointer into self; guard dropped before this fn returns.
8764        let _g = RestoreFsync(&mut self.fsync as *mut FsyncPolicy, saved);
8765        self.fsync = FsyncPolicy::Relaxed;
8766        self.commit_logged_batch(ops, None, authz.cloned())
8767            .map(inserted_pair)
8768    }
8769
8770    /// Execute a `/ingest` request with role-scoped write authorization.
8771    ///
8772    /// Resolves the role's `WriteScope` and `NodeMask` inside this call (same
8773    /// write-guard lifetime as the mutation, satisfying §5 lock discipline).
8774    /// Sets `pending_write_authz` for the duration of the call so that the
8775    /// `commit_ingest` → `commit_logged_batch` path picks up the authz context
8776    /// and evaluates the decision table per-op before any WAL write.
8777    ///
8778    /// §7.3: roles with empty `create_labels` will see every `InsertNode` op
8779    /// denied by the decision table with the appropriate §4.3 scope reason;
8780    /// no special HTTP-layer check is needed.
8781    ///
8782    /// Roles with `write: None` return `RoleWriteDenied` with
8783    /// "writes are not permitted" (byte-identical to v1 blanket 403).
8784    pub fn ingest_with_edges_authz(
8785        &mut self,
8786        role: &str,
8787        label: &str,
8788        rows: Vec<std::collections::BTreeMap<String, Value>>,
8789        opts: &crate::ingest::IngestOptions,
8790        edges: &[(String, String, String)],
8791    ) -> Result<crate::ingest::IngestReport> {
8792        // Resolve scope (fails fast if role has no write scope).
8793        // write:None → byte-identical v1 blanket-403 body (plan §v1-sidecar mandate).
8794        let scope =
8795            {
8796                let roles = self.roles.as_deref().ok_or_else(|| GraphError::Corrupt {
8797                    detail: "roles.json was corrupt at open; re-open to restore role access".into(),
8798                })?;
8799                let def = roles.iter().find(|r| r.name == role).ok_or_else(|| {
8800                    GraphError::KeyNotFound {
8801                        key: format!("role:{role}"),
8802                    }
8803                })?;
8804                def.write
8805                    .clone()
8806                    .ok_or_else(|| GraphError::RoleWriteDenied {
8807                        reason: "role-bound token: writes are not permitted".into(),
8808                    })?
8809            };
8810        let mask = self.mask_for_role(role)?;
8811        self.pending_write_authz = Some(WriteAuthz {
8812            role: role.into(),
8813            scope,
8814            mask,
8815        });
8816        // RAII guard: always clears pending_write_authz on scope exit, including
8817        // on panic or early-return, mirroring the RestoreEmitDeltas precedent.
8818        struct ClearPendingAuthzOnDrop(*mut Option<WriteAuthz>);
8819        impl Drop for ClearPendingAuthzOnDrop {
8820            fn drop(&mut self) {
8821                // SAFETY: pointer into the owning GraphDb; guard is dropped
8822                // within this function's frame before it returns.
8823                unsafe { *self.0 = None };
8824            }
8825        }
8826        // SAFETY: raw pointer into self; guard dropped before this fn returns.
8827        let _authz_guard = ClearPendingAuthzOnDrop(&mut self.pending_write_authz as *mut _);
8828        self.ingest_with_edges(label, rows, opts, edges)
8829    }
8830
8831    /// Evaluate the write-authz decision table for one `BatchOp`.
8832    ///
8833    /// Called by `commit_logged_batch` for each op when `pending_write_authz`
8834    /// is `Some`, BEFORE MutPreview.  A denial returns an error immediately;
8835    /// the remaining ops are not evaluated and no WAL frame is written.
8836    ///
8837    /// `batch_created` carries the key→label pairs of nodes that earlier ops in
8838    /// THIS batch will create.  Used by `InsertEdgeUpsert` to count same-batch
8839    /// placeholder nodes as visible (spec: "a placeholder endpoint the SAME
8840    /// batch creates counts as visible if its label passed the create-class gate").
8841    fn check_single_op_authz(
8842        &self,
8843        authz: &WriteAuthz,
8844        op: &BatchOp,
8845        batch_created: &BTreeMap<String, String>,
8846    ) -> Result<()> {
8847        // Helper: 3-way node status under the authz mask.
8848        //
8849        // Batch-created nodes (from earlier InsertNode in THIS batch) are treated
8850        // as Visible with their recorded label — their create gate already passed
8851        // and they are not yet in self.ids (not committed).  This fixes the
8852        // MERGE+ON CREATE SET case where InsertNode + SetProp arrive together:
8853        // the SetProp must not see the node as Absent.
8854        let node_status = |key: &str| -> NodeAuthzStatus {
8855            if let Some(label) = batch_created.get(key) {
8856                return NodeAuthzStatus::Visible(label.clone());
8857            }
8858            match self.ids.get(key) {
8859                None => NodeAuthzStatus::Absent,
8860                Some(id) if !authz.mask.contains_id(id) => NodeAuthzStatus::Hidden,
8861                Some(id) => {
8862                    let label = self
8863                        .labels
8864                        .get(id as usize)
8865                        .and_then(|&sym| {
8866                            if sym == u32::MAX {
8867                                None
8868                            } else {
8869                                self.syms.resolve(sym).map(str::to_string)
8870                            }
8871                        })
8872                        .unwrap_or_default();
8873                    NodeAuthzStatus::Visible(label)
8874                }
8875            }
8876        };
8877
8878        // Helper: is an InsertEdgeUpsert endpoint visible?
8879        // A same-batch placeholder counts as visible if its label passed
8880        // the create-class gate (spec "upsert placeholder-counts-as-visible").
8881        let upsert_ep_visible = |ep_key: &str, placeholder_label: &str| -> bool {
8882            // In store and visible?
8883            if let Some(id) = self.ids.get(ep_key) {
8884                return authz.mask.contains_id(id);
8885            }
8886            // Created by an earlier op in this batch?
8887            if let Some(created_label) = batch_created.get(ep_key) {
8888                return authz.scope.create_labels.contains(created_label);
8889            }
8890            // Will be created by THIS InsertEdgeUpsert: placeholder_label
8891            // must pass the create-class gate.
8892            authz
8893                .scope
8894                .create_labels
8895                .contains(&placeholder_label.to_string())
8896        };
8897
8898        match op {
8899            // RenameNode / CreateRule / DeleteRule: defense-in-depth gate.
8900            // These ops are never routed to role-scoped paths by the HTTP layer,
8901            // but we 403 them here to close any future bypass route.
8902            //
8903            // InsertNodeOnConflict joins them: it is reachable only from the
8904            // embedded Python binding, which has no role token, and `Replace`
8905            // is a create and an update at once. Rather than split the decision
8906            // table for an op no role-scoped path constructs, refuse it — a
8907            // role-scoped caller writes through the ops that are already in the
8908            // table.
8909            BatchOp::RenameNode { .. }
8910            | BatchOp::CreateRule(_)
8911            | BatchOp::DeleteRule { .. }
8912            | BatchOp::InsertNodeOnConflict { .. } => {
8913                return Err(GraphError::RoleWriteDenied {
8914                    reason: "role-bound token: this endpoint is not permitted".into(),
8915                });
8916            }
8917
8918            // ── CREATE-class: InsertNode ─────────────────────────────────────
8919            //
8920            // Decision table row 1 (scope-before-lookup): check label in
8921            // create_labels BEFORE any key lookup.  This is the structural
8922            // closure of the §6.2 timing-oracle item — the denial fires even
8923            // when the store is EMPTY (see test_create_scope_denied_empty_store).
8924            BatchOp::InsertNode { label, key, props } => {
8925                if !authz.scope.create_labels.contains(label) {
8926                    return Err(GraphError::RoleWriteDenied {
8927                        reason: format!(
8928                            "role-bound token: label '{}' not in write scope (create_labels)",
8929                            label
8930                        ),
8931                    });
8932                }
8933                // A role bound to namespaces may only create inside them. The
8934                // never-widen rule is about what a write makes visible to *any*
8935                // party, not only to the writer: a node this role could never
8936                // read back is a write into somebody else's tenancy. Also a
8937                // scope check, so it runs before the key lookup — it discloses
8938                // nothing about the store. Covers Cypher `CREATE` and the node
8939                // `MERGE` creates, both of which arrive as this op.
8940                // Resolved before the role lookup so a props list naming `ns`
8941                // twice is refused for every role, scoped or not: it is the same
8942                // malformed write the seam refuses, and leaving it to the seam
8943                // would mean the gate had already read one of the two.
8944                let target = Self::created_namespace(key, props)?;
8945                if let Some(def) = self.role_def_for(&authz.role) {
8946                    if !def.sees_namespace(target) {
8947                        return Err(GraphError::RoleWriteDenied {
8948                            reason: format!(
8949                                "role-bound token: namespace '{target}' not in the role's \
8950                                 namespaces"
8951                            ),
8952                        });
8953                    }
8954                }
8955                // Row 2/3: key lookup.
8956                match self.ids.get(key.as_str()) {
8957                    Some(id) if authz.mask.contains_id(id) => {
8958                        // Visible: DuplicateKey — let MutPreview handle this.
8959                    }
8960                    Some(_) => {
8961                        // Hidden: indistinguishable from absent to the role.
8962                        return Err(GraphError::RoleWriteDenied {
8963                            reason: "role-bound token: target node not visible".into(),
8964                        });
8965                    }
8966                    None => {
8967                        // Absent: proceed (create).
8968                    }
8969                }
8970            }
8971
8972            // ── UPDATE-class: SetProp, RemoveProp ────────────────────────────
8973            BatchOp::SetProp { key, .. } | BatchOp::RemoveProp { key, .. } => {
8974                if batch_created.contains_key(key.as_str()) {
8975                    // Batch-created node: create gate already passed this batch.
8976                    // Updating it in the same batch is always allowed, regardless
8977                    // of update_labels (ruling §3.5: "writer just created it").
8978                } else {
8979                    let label = match node_status(key) {
8980                        NodeAuthzStatus::Visible(lbl) => lbl,
8981                        _ => {
8982                            return Err(GraphError::RoleWriteDenied {
8983                                reason: "role-bound token: target node not visible".into(),
8984                            });
8985                        }
8986                    };
8987                    if !authz.scope.update_labels.contains(&label) {
8988                        return Err(GraphError::RoleWriteDenied {
8989                            reason: format!(
8990                                "role-bound token: label '{}' not in write scope (update_labels)",
8991                                label
8992                            ),
8993                        });
8994                    }
8995                }
8996            }
8997
8998            // ── DELETE-class: DeleteNode ─────────────────────────────────────
8999            BatchOp::DeleteNode { key } => {
9000                let label = match node_status(key) {
9001                    NodeAuthzStatus::Visible(lbl) => lbl,
9002                    _ => {
9003                        return Err(GraphError::RoleWriteDenied {
9004                            reason: "role-bound token: target node not visible".into(),
9005                        });
9006                    }
9007                };
9008                if !authz.scope.delete_labels.contains(&label) {
9009                    return Err(GraphError::RoleWriteDenied {
9010                        reason: format!(
9011                            "role-bound token: label '{}' not in write scope (delete_labels)",
9012                            label
9013                        ),
9014                    });
9015                }
9016            }
9017
9018            // ── DELETE-class: DeleteEdge ─────────────────────────────────────
9019            //
9020            // Derived-edge rejection runs BEFORE the delete_edge_types scope
9021            // check (spec §3.5: "existing derived-edge rejection precedes
9022            // delete_edge_types check").
9023            BatchOp::DeleteEdge {
9024                edge_type,
9025                src_key,
9026                dst_key,
9027            } => {
9028                // Check provenance ownership BEFORE scope (spec §3.5 ordering).
9029                if let (Some(src_id), Some(dst_id), Some(et_sym)) = (
9030                    self.ids.get(src_key.as_str()),
9031                    self.ids.get(dst_key.as_str()),
9032                    self.syms.get(edge_type.as_str()),
9033                ) {
9034                    if self.engine.is_owned(et_sym, src_id, dst_id) {
9035                        return Err(GraphError::RuleOwned {
9036                            detail: format!(
9037                                "edge {edge_type} {src_key}→{dst_key} is rule-owned; \
9038                                 delete or change the owning rule"
9039                            ),
9040                        });
9041                    }
9042                    // Also check would_derive via MutPreview (empty overlay, pre-batch).
9043                    let preview = MutPreview::new(self);
9044                    if preview.would_derive(edge_type, src_key, dst_key) {
9045                        return Err(GraphError::RuleOwned {
9046                            detail: format!(
9047                                "edge {edge_type} {src_key}→{dst_key} is rule-owned; \
9048                                 delete or change the owning rule, or a live rule would \
9049                                 re-derive it"
9050                            ),
9051                        });
9052                    }
9053                }
9054                // Scope check (AFTER derived-edge check, BEFORE endpoint visibility).
9055                if !authz.scope.delete_edge_types.contains(edge_type) {
9056                    return Err(GraphError::RoleWriteDenied {
9057                        reason: format!(
9058                            "role-bound token: edge type '{}' not in write scope (delete_edge_types)",
9059                            edge_type
9060                        ),
9061                    });
9062                }
9063                // Both endpoints must be visible.
9064                for ep_key in [src_key.as_str(), dst_key.as_str()] {
9065                    match self.ids.get(ep_key) {
9066                        None => {
9067                            return Err(GraphError::RoleWriteDenied {
9068                                reason: "role-bound token: edge endpoint not visible".into(),
9069                            });
9070                        }
9071                        Some(id) if !authz.mask.contains_id(id) => {
9072                            return Err(GraphError::RoleWriteDenied {
9073                                reason: "role-bound token: edge endpoint not visible".into(),
9074                            });
9075                        }
9076                        _ => {}
9077                    }
9078                }
9079            }
9080
9081            // ── EDGE-CREATE: InsertEdge ──────────────────────────────────────
9082            //
9083            // Scope check BEFORE endpoint lookup (preserves timing symmetry).
9084            BatchOp::InsertEdge {
9085                edge_type,
9086                src_key,
9087                dst_key,
9088            } => {
9089                if !authz.scope.create_edge_types.contains(edge_type) {
9090                    return Err(GraphError::RoleWriteDenied {
9091                        reason: format!(
9092                            "role-bound token: edge type '{}' not in write scope (create_edge_types)",
9093                            edge_type
9094                        ),
9095                    });
9096                }
9097                // Both endpoints must be visible. A node created by an earlier
9098                // InsertNode in the same batch (tracked in batch_created) counts
9099                // as visible if its label passed the create-class gate.
9100                for ep_key in [src_key.as_str(), dst_key.as_str()] {
9101                    if batch_created.contains_key(ep_key) {
9102                        // Created earlier this batch — already scope-checked.
9103                        continue;
9104                    }
9105                    match self.ids.get(ep_key) {
9106                        None => {
9107                            return Err(GraphError::RoleWriteDenied {
9108                                reason: "role-bound token: edge endpoint not visible".into(),
9109                            });
9110                        }
9111                        Some(id) if !authz.mask.contains_id(id) => {
9112                            return Err(GraphError::RoleWriteDenied {
9113                                reason: "role-bound token: edge endpoint not visible".into(),
9114                            });
9115                        }
9116                        _ => {}
9117                    }
9118                }
9119            }
9120
9121            // ── EDGE-CREATE: InsertEdgeUpsert ────────────────────────────────
9122            //
9123            // Scope check first; then endpoint visibility using same-batch
9124            // placeholder awareness (spec: "a placeholder endpoint the SAME
9125            // batch creates counts as visible if its label passed the
9126            // create-class gate").
9127            BatchOp::InsertEdgeUpsert {
9128                edge_type,
9129                src_key,
9130                dst_key,
9131                placeholder_label,
9132            } => {
9133                if !authz.scope.create_edge_types.contains(edge_type) {
9134                    return Err(GraphError::RoleWriteDenied {
9135                        reason: format!(
9136                            "role-bound token: edge type '{}' not in write scope (create_edge_types)",
9137                            edge_type
9138                        ),
9139                    });
9140                }
9141                // Check placeholder label against create_labels (create-class gate).
9142                // This ensures the auto-created endpoints are scope-allowed.
9143                for ep_key in [src_key.as_str(), dst_key.as_str()] {
9144                    if !upsert_ep_visible(ep_key, placeholder_label) {
9145                        return Err(GraphError::RoleWriteDenied {
9146                            reason: "role-bound token: edge endpoint not visible".into(),
9147                        });
9148                    }
9149                }
9150                // A placeholder is created with no props, so it lands in the
9151                // default namespace. A role that cannot read `default` must not
9152                // create one there, for the same reason it may not create a node
9153                // there outright.
9154                //
9155                // The refusal is byte-identical to the hidden-endpoint one above,
9156                // and deliberately so: this arm fires only for an endpoint that
9157                // does **not** exist, and the one above only for an endpoint that
9158                // does. Two different strings would make the pair an existence
9159                // oracle — ask for an upsert and read off whether the key is
9160                // taken. Hidden ≡ absent is the rule everywhere else in this
9161                // table and it holds here too.
9162                if let Some(def) = self.role_def_for(&authz.role) {
9163                    if !def.sees_namespace(NS_DEFAULT) {
9164                        for ep_key in [src_key.as_str(), dst_key.as_str()] {
9165                            if self.ids.get(ep_key).is_none() && !batch_created.contains_key(ep_key)
9166                            {
9167                                return Err(GraphError::RoleWriteDenied {
9168                                    reason: "role-bound token: edge endpoint not visible".into(),
9169                                });
9170                            }
9171                        }
9172                    }
9173                }
9174            }
9175        }
9176        Ok(())
9177    }
9178
9179    /// Write `roles` to `roles.json` atomically and update the in-memory list.
9180    ///
9181    /// Called by `apply_schema` when roles change. Never called on unchanged
9182    /// re-apply — this preserves byte-identical idempotency.
9183    pub(crate) fn commit_roles(&mut self, roles: Vec<RoleDef>) -> Result<()> {
9184        let file = RolesFile::new_versioned(roles.clone());
9185        let bytes = serde_json::to_vec(&file).map_err(|e| GraphError::Corrupt {
9186            detail: format!("roles serialization: {e}"),
9187        })?;
9188        self.fs
9189            .write_atomic(FileId::Roles, &bytes)
9190            .map_err(GraphError::Io)?;
9191        self.roles = Some(roles);
9192        // Rewriting the sidecar is not a commit, so `commit_seq` does not move
9193        // and a memoised mask would still match its version. Install a fresh
9194        // cache instead of clearing the shared one: a reader snapshot frozen
9195        // against the old definitions keeps the old `Arc` to itself and can
9196        // never publish an answer this handle would read back.
9197        self.role_masks = Arc::new(crate::mask::RoleMaskCache::new());
9198        // Refresh the MVCC frozen overlay so that reader() immediately sees the
9199        // updated role definitions without waiting for the next K-commit fold.
9200        self.fold_now();
9201        Ok(())
9202    }
9203
9204    fn view(&self) -> GraphView<'_> {
9205        GraphView {
9206            ids: &self.ids,
9207            syms: &self.syms,
9208            labels: &self.labels,
9209            props: self.props_view(),
9210            topo: self.topo_view(),
9211            edge_props: self.edge_props_view(),
9212            mask: None,
9213            prop_index: Some(&self.prop_index),
9214        }
9215    }
9216
9217    fn view_masked<'a>(&'a self, mask: &'a crate::mask::NodeMask) -> GraphView<'a> {
9218        GraphView {
9219            ids: &self.ids,
9220            syms: &self.syms,
9221            labels: &self.labels,
9222            props: self.props_view(),
9223            topo: self.topo_view(),
9224            edge_props: self.edge_props_view(),
9225            mask: Some(&mask.visible),
9226            prop_index: Some(&self.prop_index),
9227        }
9228    }
9229
9230    /// Execute a read-only Cypher query with a node visibility mask.
9231    ///
9232    /// Only nodes whose key is in `mask` are accessible: label scans, key
9233    /// lookups, and neighbor expansions all respect the mask. Edges where
9234    /// either endpoint is hidden are silently dropped.
9235    ///
9236    /// Returns `Err` with a "masked queries are read-only" message when
9237    /// `cypher` is a write statement (CREATE / MERGE / MATCH…SET / DELETE).
9238    pub fn query_masked(
9239        &self,
9240        cypher: &str,
9241        params: &std::collections::BTreeMap<String, Value>,
9242        mask: &crate::mask::NodeMask,
9243    ) -> Result<ResultSet> {
9244        // Reject write statements up front.
9245        let tokens = lex(cypher).map_err(|e| GraphError::QueryError {
9246            detail: format!("lex: {e}"),
9247        })?;
9248        if is_write_tokens(&tokens) {
9249            return Err(GraphError::MaskedReadOnly);
9250        }
9251        let union = parse_read(&tokens).map_err(|e| GraphError::QueryError {
9252            detail: format!("parse: {e}"),
9253        })?;
9254        // Each UNION part executes against the same masked view, so the mask
9255        // applies uniformly across the chain.
9256        execute_union(&self.view_masked(mask), &union, &Params(params)).map_err(|e| {
9257            GraphError::QueryError {
9258                detail: format!("execute: {e}"),
9259            }
9260        })
9261    }
9262
9263    pub fn node_ref(&self, key: &str) -> Option<NodeRef<'_, F>> {
9264        let id = self.ids.get(key)?;
9265        Some(NodeRef { db: self, id })
9266    }
9267
9268    /// BFS neighborhood expansion restricted to visible nodes in `mask`.
9269    ///
9270    /// Hidden nodes are never used as traversal intermediaries in either
9271    /// [`MaskMode::Omit`] or [`MaskMode::Stub`] — a visible node reachable
9272    /// only through a hidden node will not appear in results.
9273    ///
9274    /// In [`MaskMode::Stub`] mode, hidden nodes that are direct neighbours of
9275    /// a visited visible node are appended to the result as stub rows
9276    /// (`label` column is `null`, same key+depth columns as visible rows).
9277    /// They are NOT added to the BFS frontier.
9278    ///
9279    /// Returns `None` when `key` does not exist (caller should 404).
9280    ///
9281    /// **SECURITY**: role-token callers always pass an Omit-mode mask, so
9282    /// stub rows are never produced on the role path.
9283    pub fn neighborhood_masked(
9284        &self,
9285        key: &str,
9286        depth: u32,
9287        edge_types: Option<&[&str]>,
9288        dir: Dir,
9289        mask: &crate::mask::NodeMask,
9290    ) -> Option<ResultSet> {
9291        let start_id = self.ids.get(key)?;
9292        let view = self.view_masked(mask);
9293        let resolved: Option<Vec<u32>> = edge_types.map(|names| {
9294            names
9295                .iter()
9296                .filter_map(|name| view.syms.get(name))
9297                .collect()
9298        });
9299        let nb = neighborhood(&view, start_id, depth, resolved.as_deref(), dir);
9300        let mut rs = ResultSet::new(vec!["key".into(), "label".into(), "depth".into()]);
9301        // Collect visible BFS results (start_id at depth 0, BFS nodes after).
9302        let mut visited: Vec<(u32, u32)> = Vec::with_capacity(nb.nodes.len() + 1);
9303        visited.push((start_id, 0));
9304        for (nid, d) in &nb.nodes {
9305            let k = view.key_of(*nid);
9306            let label = view
9307                .label_of(*nid)
9308                .expect("real nodes always have a label; u32::MAX sentinel cannot occur");
9309            rs.push_row(vec![
9310                Some(Value::Str(k.to_string())),
9311                Some(Value::Str(label.to_string())),
9312                Some(Value::Int(*d as i64)),
9313            ]);
9314            visited.push((*nid, *d));
9315        }
9316        // Stub mode: add hidden direct neighbours of each visited node as stubs.
9317        // Hidden nodes are edge-endpoints only — they are not added to the BFS
9318        // frontier, so the BFS never expands through them.
9319        if mask.mode() == crate::mask::MaskMode::Stub {
9320            let raw_view = self.view();
9321            let mut seen: std::collections::HashSet<u32> =
9322                visited.iter().map(|(id, _)| *id).collect();
9323            for (node_id, node_depth) in &visited {
9324                if *node_depth >= depth {
9325                    continue;
9326                }
9327                for e in expand(&raw_view, *node_id, resolved.as_deref(), dir) {
9328                    let nbr = if e.src == *node_id { e.dst } else { e.src };
9329                    if !mask.contains_id(nbr) && seen.insert(nbr) {
9330                        if let Some(k) = self.ids.key_of(nbr) {
9331                            rs.push_row(vec![
9332                                Some(Value::Str(k.to_string())),
9333                                None,
9334                                Some(Value::Int((*node_depth + 1) as i64)),
9335                            ]);
9336                        }
9337                    }
9338                }
9339            }
9340        }
9341        Some(rs)
9342    }
9343
9344    /// [`neighborhood_masked`](Self::neighborhood_masked) with the **subject
9345    /// check** a scoped caller needs: a start key the mask hides answers exactly
9346    /// as an absent one does.
9347    ///
9348    /// `neighborhood_masked` expands from any existing key, hidden or not,
9349    /// because a full-token caller supplying a client mask already knows which
9350    /// keys exist. A scoped caller does not, so telling it apart a hidden key
9351    /// from an absent one would be an existence oracle.
9352    ///
9353    /// Expansion itself is unchanged: hidden nodes are neither returned nor used
9354    /// as traversal intermediaries, so a visible node reachable only through a
9355    /// hidden one stays out of the result.
9356    ///
9357    /// Hidden or unknown `key` → [`GraphError::KeyNotFound`].
9358    pub fn neighborhood_scoped(
9359        &self,
9360        key: &str,
9361        depth: u32,
9362        edge_types: Option<&[&str]>,
9363        dir: Dir,
9364        mask: &crate::mask::NodeMask,
9365    ) -> Result<ResultSet> {
9366        if !mask.contains_node(self, key) {
9367            return Err(GraphError::KeyNotFound { key: key.into() });
9368        }
9369        self.neighborhood_masked(key, depth, edge_types, dir, mask)
9370            .ok_or_else(|| GraphError::KeyNotFound { key: key.into() })
9371    }
9372
9373    /// Live node's key, label, and columnar props. Unknown or tombstoned → `None`.
9374    pub fn node_info(&self, key: &str) -> Option<NodeInfo> {
9375        let n = self.node_ref(key)?;
9376        Some(NodeInfo {
9377            key: n.key().to_string(),
9378            label: n.label().to_string(),
9379            props: n.props(),
9380        })
9381    }
9382
9383    /// Look up a node with mask awareness.
9384    ///
9385    /// | Key state         | Omit mode       | Stub mode              |
9386    /// |-------------------|-----------------|------------------------|
9387    /// | does not exist    | `None` (→ 404)  | `None` (→ 404)         |
9388    /// | exists, visible   | `Some(Visible)` | `Some(Visible)`        |
9389    /// | exists, hidden    | `None` (→ 404)  | `Some(Restricted)`     |
9390    ///
9391    /// **SECURITY**: only call from client-mask (full-token) paths.
9392    /// Role-token paths must use [`node_info`] after an explicit visibility check.
9393    pub fn node_info_masked(
9394        &self,
9395        key: &str,
9396        mask: &crate::mask::NodeMask,
9397    ) -> Option<MaskedNodeResult> {
9398        let id = self.ids.get(key)?;
9399        if mask.contains_id(id) {
9400            Some(MaskedNodeResult::Visible(self.node_info(key)?))
9401        } else {
9402            match mask.mode() {
9403                crate::mask::MaskMode::Stub => Some(MaskedNodeResult::Restricted),
9404                crate::mask::MaskMode::Omit => None,
9405            }
9406        }
9407    }
9408
9409    /// Get edges for `key` with mask-aware hidden-endpoint handling.
9410    ///
9411    /// - Omit mode: edges to hidden endpoints are excluded (same as role-path filtering).
9412    /// - Stub mode: edges to hidden endpoints are included; `src_restricted`/`dst_restricted`
9413    ///   is `true` for each hidden endpoint.
9414    ///
9415    /// Unknown key → [`GraphError::KeyNotFound`].
9416    ///
9417    /// **SECURITY**: only call from client-mask (full-token) paths.
9418    pub fn node_edges_masked(
9419        &self,
9420        key: &str,
9421        mask: &crate::mask::NodeMask,
9422    ) -> Result<Vec<MaskedEdge>> {
9423        self.ensure_v8_base_sections_loaded();
9424        let id = self
9425            .ids
9426            .get(key)
9427            .ok_or_else(|| GraphError::KeyNotFound { key: key.into() })?;
9428        let derived: BTreeSet<(u32, u32, u32)> = self
9429            .engine
9430            .provenance_touching(id)
9431            .map(|(_rule, etype, src, dst)| (etype, src, dst))
9432            .collect();
9433        let mut edges = Vec::new();
9434        let tv = self.topo_view();
9435        for etype in tv.etypes() {
9436            // etype comes from the archived CSR (access_unchecked, no eager CRC).
9437            // A bit-flip in the large TOPOLOGY section can produce an etype id
9438            // that is not in the interner.  Return Corrupt rather than panic.
9439            let edge_type = self
9440                .syms
9441                .resolve(etype)
9442                .ok_or_else(|| GraphError::Corrupt {
9443                    detail: format!("v8: topology etype {etype} not in interner"),
9444                })?
9445                .to_string();
9446            for dir in [Direction::Out, Direction::In] {
9447                for &nbr in tv.neighbors(etype, dir, id).as_ref() {
9448                    let nbr_restricted = !mask.contains_id(nbr);
9449                    if nbr_restricted && mask.mode() == crate::mask::MaskMode::Omit {
9450                        continue;
9451                    }
9452                    let nbr_key = self
9453                        .ids
9454                        .key_of(nbr)
9455                        .ok_or_else(|| GraphError::Corrupt {
9456                            detail: format!("topology id {nbr} has no key"),
9457                        })?
9458                        .to_string();
9459                    let (src_id, dst_id, src_key, dst_key, src_restricted, dst_restricted) =
9460                        match dir {
9461                            Direction::Out => {
9462                                (id, nbr, key.to_string(), nbr_key, false, nbr_restricted)
9463                            }
9464                            Direction::In => {
9465                                (nbr, id, nbr_key, key.to_string(), nbr_restricted, false)
9466                            }
9467                        };
9468                    edges.push(MaskedEdge {
9469                        edge_type: edge_type.clone(),
9470                        src_key,
9471                        src_restricted,
9472                        dst_key,
9473                        dst_restricted,
9474                        derived: derived.contains(&(etype, src_id, dst_id)),
9475                    });
9476                }
9477            }
9478        }
9479        edges.sort_by(|a, b| {
9480            a.edge_type
9481                .cmp(&b.edge_type)
9482                .then(a.src_key.cmp(&b.src_key))
9483                .then(a.dst_key.cmp(&b.dst_key))
9484        });
9485        edges.dedup_by(|a, b| {
9486            a.edge_type == b.edge_type && a.src_key == b.src_key && a.dst_key == b.dst_key
9487        });
9488        Ok(edges)
9489    }
9490
9491    /// [`node_edges_masked`](Self::node_edges_masked) with the **subject check**
9492    /// a scoped caller needs, and a plain [`EdgeInfo`] list.
9493    ///
9494    /// `node_edges_masked` raises [`GraphError::KeyNotFound`] only when `key` is
9495    /// unknown; a key that exists but is hidden still yields its (filtered) edge
9496    /// list, which is correct for a full-token client mask and an existence
9497    /// oracle for a scoped one. Here a hidden subject answers exactly as an
9498    /// absent one does.
9499    ///
9500    /// Every edge naming a hidden endpoint is dropped, whatever the mask's
9501    /// [`MaskMode`](crate::mask::MaskMode): a scoped caller never sees a
9502    /// restricted stub, so there is nothing for it to render.
9503    ///
9504    /// Hidden or unknown `key` → [`GraphError::KeyNotFound`].
9505    pub fn node_edges_scoped(
9506        &self,
9507        key: &str,
9508        mask: &crate::mask::NodeMask,
9509    ) -> Result<Vec<EdgeInfo>> {
9510        if !mask.contains_node(self, key) {
9511            return Err(GraphError::KeyNotFound { key: key.into() });
9512        }
9513        Ok(self
9514            .node_edges_masked(key, mask)?
9515            .into_iter()
9516            .filter(|e| !e.src_restricted && !e.dst_restricted)
9517            .map(|e| EdgeInfo {
9518                edge_type: e.edge_type,
9519                src_key: e.src_key,
9520                dst_key: e.dst_key,
9521                derived: e.derived,
9522            })
9523            .collect())
9524    }
9525
9526    /// Every directed edge incident on `key`, both directions, every etype.
9527    ///
9528    /// Walk is `topology.etypes()` × `{Out, In}` × `neighbors()`. `derived` is
9529    /// membership in [`RuleEngine::provenance_touching`] (O(degree) via the
9530    /// Plan-8 `by_node` index). Sorted by `(edge_type, src_key, dst_key)`.
9531    /// Unknown key → [`GraphError::KeyNotFound`].
9532    pub fn node_edges(&self, key: &str) -> Result<Vec<EdgeInfo>> {
9533        self.ensure_v8_base_sections_loaded();
9534        let id = self
9535            .ids
9536            .get(key)
9537            .ok_or_else(|| GraphError::KeyNotFound { key: key.into() })?;
9538        let derived: BTreeSet<(u32, u32, u32)> = self
9539            .engine
9540            .provenance_touching(id)
9541            .map(|(_rule, etype, src, dst)| (etype, src, dst))
9542            .collect();
9543        let mut edges = Vec::new();
9544        let tv = self.topo_view();
9545        for etype in tv.etypes() {
9546            // Same guard as node_edges_masked: etype from unchecked-CRC CSR.
9547            let edge_type = self
9548                .syms
9549                .resolve(etype)
9550                .ok_or_else(|| GraphError::Corrupt {
9551                    detail: format!("v8: topology etype {etype} not in interner"),
9552                })?
9553                .to_string();
9554            for dir in [Direction::Out, Direction::In] {
9555                for &nbr in tv.neighbors(etype, dir, id).as_ref() {
9556                    let (src, dst, src_key, dst_key) = match dir {
9557                        Direction::Out => (
9558                            id,
9559                            nbr,
9560                            key.to_string(),
9561                            self.ids
9562                                .key_of(nbr)
9563                                .ok_or_else(|| GraphError::Corrupt {
9564                                    detail: format!("topology id {nbr} has no key"),
9565                                })?
9566                                .to_string(),
9567                        ),
9568                        Direction::In => (
9569                            nbr,
9570                            id,
9571                            self.ids
9572                                .key_of(nbr)
9573                                .ok_or_else(|| GraphError::Corrupt {
9574                                    detail: format!("topology id {nbr} has no key"),
9575                                })?
9576                                .to_string(),
9577                            key.to_string(),
9578                        ),
9579                    };
9580                    edges.push(EdgeInfo {
9581                        edge_type: edge_type.clone(),
9582                        src_key,
9583                        dst_key,
9584                        derived: derived.contains(&(etype, src, dst)),
9585                    });
9586                }
9587            }
9588        }
9589        edges.sort_by(|a, b| {
9590            a.edge_type
9591                .cmp(&b.edge_type)
9592                .then(a.src_key.cmp(&b.src_key))
9593                .then(a.dst_key.cmp(&b.dst_key))
9594        });
9595        // Self-loops appear in both Out and In; sort makes the pair adjacent
9596        // (sort key matches PartialEq for this case) so one pass drops the dup.
9597        edges.dedup();
9598        Ok(edges)
9599    }
9600
9601    // ── Backup ────────────────────────────────────────────────────────────────
9602
9603    /// Copy this store to `dest` as a consistent, verified snapshot.
9604    ///
9605    /// Copies every durable file in the database directory — `snapshot.bin`,
9606    /// `wal.bin`, all `wal.<N>.archive` files, `wal.floor`, `wal.genesis`, and
9607    /// `roles.json` — into a freshly created `dest` directory using OS-level
9608    /// `copy` calls (no large in-process buffers).
9609    ///
9610    /// # Consistency guarantee
9611    ///
9612    /// The guarantee is **process-local**: the caller holds `&self`, which
9613    /// prevents any concurrent writer in the **same process** from modifying
9614    /// the files during the copy.  Running `mushroomdb backup` against a
9615    /// directory that is **concurrently being written by another process** (e.g.
9616    /// `mushroomdb serve`) is **unsafe** — the copy can be torn.  The post-copy
9617    /// `verified: true` result reduces but does not eliminate the risk of a
9618    /// silent corrupt backup (CRC catches many bit-flips; it cannot catch a
9619    /// consistent mid-write snapshot).
9620    ///
9621    /// **The safe path for a live-served store is `POST /backup` on the HTTP
9622    /// server.** That handler acquires the read lock on the shared database
9623    /// before calling this method, which is the correct cross-process
9624    /// synchronisation point because the server is the single process writing
9625    /// the files.
9626    ///
9627    /// After copying, opens the destination read-only and runs the CRC section
9628    /// verifier (`verify_snapshot`) to confirm byte-for-byte integrity.
9629    /// `BackupReport::verified` reflects whether both checks passed.
9630    ///
9631    /// Returns `Err` when `self` is not backed by a `RealFs` (e.g. `SimFs`).
9632    pub fn backup_to(&self, dest: &std::path::Path) -> Result<BackupReport> {
9633        // Derive source directory from snapshot_path (RealFs only).
9634        let src_dir = match self.fs.snapshot_path() {
9635            Some(p) => p.parent().map(|d| d.to_path_buf()).ok_or_else(|| {
9636                GraphError::Io(std::io::Error::other("snapshot has no parent dir"))
9637            })?,
9638            None => {
9639                return Err(GraphError::Io(std::io::Error::other(
9640                    "backup_to requires a real filesystem (RealFs)",
9641                )))
9642            }
9643        };
9644
9645        std::fs::create_dir_all(dest)?;
9646
9647        let mut files: Vec<String> = Vec::new();
9648        let mut bytes: u64 = 0;
9649
9650        // Helper: copy src_dir/name → dest/name if the file exists.
9651        let mut try_copy = |name: &str| -> std::io::Result<()> {
9652            let src_path = src_dir.join(name);
9653            if src_path.exists() {
9654                let n = std::fs::copy(&src_path, dest.join(name))?;
9655                bytes += n;
9656                files.push(name.to_string());
9657            }
9658            Ok(())
9659        };
9660
9661        try_copy("snapshot.bin")?;
9662        try_copy("snapshot.bin.bak")?;
9663        try_copy("wal.bin")?;
9664        try_copy("wal.floor")?;
9665        try_copy("wal.genesis")?;
9666        try_copy("roles.json")?;
9667
9668        // Copy WAL archives.
9669        let archives = self.fs.list_archives()?;
9670        for n in &archives {
9671            let name = format!("wal.{n}.archive");
9672            let n_bytes = std::fs::copy(src_dir.join(&name), dest.join(&name))?;
9673            bytes += n_bytes;
9674            files.push(name);
9675        }
9676
9677        files.sort();
9678
9679        // Post-copy verification: open dest and run CRC checks.
9680        let snap_in_dest = dest.join("snapshot.bin").exists();
9681        let crc_ok = if snap_in_dest {
9682            crate::verify_snapshot(dest)
9683                .map(|results| results.iter().all(|(_, _, _, r)| r.is_ok()))
9684                .unwrap_or(false)
9685        } else {
9686            true // WAL-only store: nothing to CRC-check in snapshot
9687        };
9688        let opens_ok = GraphDb::<core_storage::fs::RealFs>::open(dest).is_ok();
9689        let verified = crc_ok && opens_ok;
9690
9691        Ok(BackupReport {
9692            files,
9693            bytes,
9694            verified,
9695        })
9696    }
9697
9698    // ── Export helpers ────────────────────────────────────────────────────────
9699
9700    /// All live nodes, sorted by key (deterministic).
9701    ///
9702    /// Reads base + WAL overlay. Tombstoned nodes are excluded.
9703    pub fn all_nodes_for_export(&self) -> Vec<NodeInfo> {
9704        self.ensure_v8_base_sections_loaded();
9705        let pv = self.props_view();
9706        let mut nodes = Vec::new();
9707        for id in 0..self.ids.len() as u32 {
9708            let Some(key) = self.ids.key_of(id) else {
9709                continue;
9710            };
9711            let Some(&sym) = self.labels.get(id as usize) else {
9712                continue;
9713            };
9714            if sym == u32::MAX {
9715                continue; // tombstoned
9716            }
9717            let Some(label) = self.syms.resolve(sym) else {
9718                continue;
9719            };
9720            let mut props = BTreeMap::new();
9721            for field in pv.field_names() {
9722                if let Some(vr) = pv.get(id, &field) {
9723                    props.insert(field, vr.into_value());
9724                }
9725            }
9726            nodes.push(NodeInfo {
9727                key: key.to_string(),
9728                label: label.to_string(),
9729                props,
9730            });
9731        }
9732        nodes.sort_by(|a, b| a.key.cmp(&b.key));
9733        nodes
9734    }
9735
9736    /// All directed edges, sorted by `(edge_type, src, dst)`. Each edge appears once.
9737    ///
9738    /// Derived edges carry `derived: true` and the creating rule's name in `rule`.
9739    /// Manual edges carry `derived: false` and `rule: None`.
9740    /// `weight` is the creating rule's `weight_prop` value read off the edge
9741    /// (numeric only), mirroring the convention used by [`GraphDb::explain`]
9742    /// and [`GraphDb::weighted_edges`]. Deterministic across runs on the same
9743    /// store state.
9744    pub fn all_edges_for_export(&self) -> Vec<ExportEdge> {
9745        self.ensure_v8_base_sections_loaded();
9746
9747        // Build (etype_sym, src_id, dst_id) → rule_name for O(1) derivation lookup.
9748        let mut prov: HashMap<(u32, u32, u32), String> = HashMap::new();
9749        for (rule_name, triples) in self.engine.provenance() {
9750            for &(etype, src, dst) in triples {
9751                prov.insert((etype, src, dst), rule_name.clone());
9752            }
9753        }
9754
9755        // rule_name → weight_prop, for O(1) lookup per derived edge.
9756        let weight_props: HashMap<&str, Option<&str>> = self
9757            .engine
9758            .rules()
9759            .map(|r| (r.name.as_str(), r.weight_prop.as_deref()))
9760            .collect();
9761
9762        let tv = self.topo_view();
9763        let ep = self.edge_props_view();
9764        let mut edges = Vec::new();
9765
9766        for id in 0..self.ids.len() as u32 {
9767            let Some(key) = self.ids.key_of(id) else {
9768                continue;
9769            };
9770            let Some(&lsym) = self.labels.get(id as usize) else {
9771                continue;
9772            };
9773            if lsym == u32::MAX {
9774                continue; // tombstoned
9775            }
9776
9777            for etype_sym in tv.etypes() {
9778                // etype from archived CSR (access_unchecked, no eager CRC).
9779                // Skip edges whose etype is not in the interner; this can only
9780                // occur with a corrupt large TOPOLOGY section (bit-flip on an
9781                // etype field in the archived data).  The function returns Vec,
9782                // not Result, so we continue rather than propagate.
9783                let Some(edge_type) = self.syms.resolve(etype_sym) else {
9784                    continue;
9785                };
9786                let edge_type = edge_type.to_string();
9787                for &nbr in tv.neighbors(etype_sym, Direction::Out, id).as_ref() {
9788                    let Some(dst_key) = self.ids.key_of(nbr) else {
9789                        continue; // skip corrupt entries
9790                    };
9791                    let prov_key = (etype_sym, id, nbr);
9792                    let rule = prov.get(&prov_key).cloned();
9793                    let derived = rule.is_some();
9794                    let weight = rule
9795                        .as_deref()
9796                        .and_then(|rn| weight_props.get(rn).copied().flatten())
9797                        .and_then(|prop| match ep.get(etype_sym, id, nbr, prop) {
9798                            Some(Value::Float(f)) => Some(f),
9799                            Some(Value::Int(i)) => Some(i as f64),
9800                            _ => None,
9801                        });
9802                    edges.push(ExportEdge {
9803                        edge_type: edge_type.clone(),
9804                        src: key.to_string(),
9805                        dst: dst_key.to_string(),
9806                        derived,
9807                        rule,
9808                        weight,
9809                    });
9810                }
9811            }
9812        }
9813
9814        edges.sort_by(|a, b| {
9815            a.edge_type
9816                .cmp(&b.edge_type)
9817                .then(a.src.cmp(&b.src))
9818                .then(a.dst.cmp(&b.dst))
9819        });
9820        edges
9821    }
9822
9823    /// What each edge type *is*, without building one record per edge.
9824    ///
9825    /// [`all_edges_for_export`](Self::all_edges_for_export) answers the same
9826    /// question by materialising every edge — three `String`s apiece, a
9827    /// provenance `HashMap` over every derived edge, and a final sort. That is
9828    /// the right shape for an export, and the wrong one for a summary: on a
9829    /// store with 1.3 M derived edges it allocates hundreds of megabytes to
9830    /// produce nine lines. This walks the topology instead, summing neighbour
9831    /// slice lengths and collecting *label symbols* rather than label strings,
9832    /// so the per-edge cost is an integer add and a set insert on a set with
9833    /// as many members as the store has labels.
9834    ///
9835    /// The rule names come off the rule *definitions*, which each declare the
9836    /// `edge_type` they derive, so naming them costs one pass over the rules
9837    /// rather than one provenance lookup per edge. That is also why `rules`
9838    /// is a list: two rules may derive the same type — the association store
9839    /// derives `INDUSTRY_ALIGNMENT` from both a talent→company and a
9840    /// talent→job rule — and naming only one of them would be a half-truth.
9841    /// A type with no rules is one written by hand.
9842    ///
9843    /// `sample` is the first edge of the type in the store's own id order,
9844    /// which is insertion order: deterministic for a given store, and not the
9845    /// same as key order, which cannot be had without resolving a key per
9846    /// edge. Sorted by `edge_type`.
9847    pub fn edge_type_census(&self) -> Vec<EdgeTypeCensus> {
9848        self.ensure_v8_base_sections_loaded();
9849
9850        let mut rules_by_type: BTreeMap<&str, BTreeSet<&str>> = BTreeMap::new();
9851        for r in self.engine.rules() {
9852            rules_by_type
9853                .entry(r.edge_type.as_str())
9854                .or_default()
9855                .insert(r.name.as_str());
9856        }
9857
9858        let tv = self.topo_view();
9859        let node_count = self.ids.len() as u32;
9860        let mut out = Vec::new();
9861        for etype_sym in tv.etypes() {
9862            // An etype the interner cannot resolve means a corrupt TOPOLOGY
9863            // section; skip it rather than name it, as `all_edges_for_export`
9864            // does for the same reason.
9865            let Some(edge_type) = self.syms.resolve(etype_sym) else {
9866                continue;
9867            };
9868            let mut edges: u64 = 0;
9869            let mut src_syms: BTreeSet<u32> = BTreeSet::new();
9870            let mut dst_syms: BTreeSet<u32> = BTreeSet::new();
9871            let mut sample: Option<(u32, u32)> = None;
9872            for id in 0..node_count {
9873                let Some(&lsym) = self.labels.get(id as usize) else {
9874                    continue;
9875                };
9876                if lsym == u32::MAX {
9877                    continue; // tombstoned
9878                }
9879                let nbrs = tv.neighbors(etype_sym, Direction::Out, id);
9880                let nbrs = nbrs.as_ref();
9881                if nbrs.is_empty() {
9882                    continue;
9883                }
9884                edges += nbrs.len() as u64;
9885                src_syms.insert(lsym);
9886                for &nbr in nbrs {
9887                    if let Some(&dsym) = self.labels.get(nbr as usize) {
9888                        if dsym != u32::MAX {
9889                            dst_syms.insert(dsym);
9890                        }
9891                    }
9892                }
9893                if sample.is_none() {
9894                    sample = Some((id, nbrs[0]));
9895                }
9896            }
9897            let resolve = |syms: &BTreeSet<u32>| -> Vec<String> {
9898                syms.iter()
9899                    .filter_map(|&s| self.syms.resolve(s))
9900                    .map(ToString::to_string)
9901                    .collect()
9902            };
9903            out.push(EdgeTypeCensus {
9904                edge_type: edge_type.to_string(),
9905                edges,
9906                src_labels: resolve(&src_syms),
9907                dst_labels: resolve(&dst_syms),
9908                rules: rules_by_type
9909                    .get(edge_type)
9910                    .map(|rs| rs.iter().map(ToString::to_string).collect())
9911                    .unwrap_or_default(),
9912                sample: sample.and_then(|(s, d)| {
9913                    Some((
9914                        self.ids.key_of(s)?.to_string(),
9915                        self.ids.key_of(d)?.to_string(),
9916                    ))
9917                }),
9918            });
9919        }
9920        out.sort_by(|a, b| a.edge_type.cmp(&b.edge_type));
9921        out
9922    }
9923
9924    /// All directed edges of `edge_type`, with the raw value of `weight_prop`
9925    /// on each edge when given.
9926    ///
9927    /// `weight` is `Some(f)` only when `weight_prop` is set and the edge
9928    /// carries that property with a numeric (`Int`/`Float`) value; otherwise
9929    /// `None` — callers that want a default weight (e.g. `1.0` for missing
9930    /// props) apply it themselves, matching the convention used internally
9931    /// by [`GraphDb::pagerank`], [`GraphDb::connected_components`],
9932    /// [`GraphDb::degree_centrality`], and [`GraphDb::communities`].
9933    ///
9934    /// Sorted by `(src, dst)` for determinism. Reads the unified topology
9935    /// (manual + rule-derived edges).  An unknown `edge_type` returns an
9936    /// empty vec.
9937    pub fn weighted_edges(
9938        &self,
9939        edge_type: &str,
9940        weight_prop: Option<&str>,
9941    ) -> Vec<(String, String, Option<f64>)> {
9942        let Some(etype_sym) = self.syms.get(edge_type) else {
9943            return Vec::new();
9944        };
9945        let tv = self.topo_view();
9946        let ep = self.edge_props_view();
9947        let mut out = Vec::new();
9948        for id in 0..self.ids.len() as u32 {
9949            let Some(key) = self.ids.key_of(id) else {
9950                continue;
9951            };
9952            let Some(&sym) = self.labels.get(id as usize) else {
9953                continue;
9954            };
9955            if sym == u32::MAX {
9956                continue; // tombstoned
9957            }
9958            for &nbr in tv.neighbors(etype_sym, Direction::Out, id).as_ref() {
9959                let Some(dst_key) = self.ids.key_of(nbr) else {
9960                    continue;
9961                };
9962                let weight = weight_prop.and_then(|prop| match ep.get(etype_sym, id, nbr, prop) {
9963                    Some(Value::Float(f)) => Some(f),
9964                    Some(Value::Int(i)) => Some(i as f64),
9965                    _ => None,
9966                });
9967                out.push((key.to_string(), dst_key.to_string(), weight));
9968            }
9969        }
9970        out.sort_by(|a, b| a.0.cmp(&b.0).then(a.1.cmp(&b.1)));
9971        out
9972    }
9973
9974    pub fn nodes_with_label(&self, label: &str) -> Vec<NodeRef<'_, F>> {
9975        self.view()
9976            .nodes_with_label(label)
9977            .into_iter()
9978            .map(|id| NodeRef { db: self, id })
9979            .collect()
9980    }
9981
9982    pub fn find_nodes(&self, label: &str, filter: &Filter) -> Vec<NodeRef<'_, F>> {
9983        let view = self.view();
9984        view.nodes_with_label(label)
9985            .into_iter()
9986            .filter(|&id| {
9987                eval_filter(filter, &|field| {
9988                    view.prop(id, field).map(|vr| vr.into_value())
9989                })
9990            })
9991            .map(|id| NodeRef { db: self, id })
9992            .collect()
9993    }
9994
9995    /// Returns `true` if any approximate (HNSW) VectorSimilar rule covers
9996    /// `field`.  Use as a capability probe: when `true`, `find_similar_vector`
9997    /// with `label = None` will use the native ANN path rather than the O(n)
9998    /// brute-force scan.
9999    pub fn has_vector_rule(&self, field: &str) -> bool {
10000        self.engine.hnsw_has_rule(field)
10001    }
10002
10003    /// How many HNSW graphs this handle has built from scratch since it was
10004    /// opened (one per side of an approximate rule).
10005    ///
10006    /// An open that restored every graph from the snapshot reports `0`.
10007    /// Exposed for tests that assert the open path reuses the persisted index
10008    /// rather than rebuilding it; not part of the stable surface.
10009    #[doc(hidden)]
10010    pub fn hnsw_build_count(&self) -> u64 {
10011        self.engine.hnsw_build_count()
10012    }
10013
10014    /// How many rules this handle still holds a lazily-decoded HNSW graph for.
10015    ///
10016    /// Zero before the first ANN query on a clean open, and again once the
10017    /// live indexes own the graphs. See [`core_rules::RuleEngine::lazy_hnsw_len`].
10018    /// Exposed for tests that assert the lazy copies are released; not part of
10019    /// the stable surface.
10020    #[doc(hidden)]
10021    pub fn lazy_hnsw_len(&self) -> usize {
10022        self.engine.lazy_hnsw_len()
10023    }
10024
10025    /// Find nodes whose `field` vector is most similar to `q` (cosine
10026    /// similarity), returning up to `k` results with similarity ≥ `min`,
10027    /// sorted descending.
10028    ///
10029    /// When `label` is `None` the search spans all labels (via
10030    /// `hnsw_search_any_dst` or a full brute-force scan); when `label` is
10031    /// `Some(lbl)` it restricts to nodes with that label.
10032    ///
10033    /// Uses the HNSW index when one is available (fast path); otherwise falls
10034    /// back to an O(n) brute-force scan.
10035    ///
10036    /// **The index supplies candidates, never scores.** Its own distances are
10037    /// `f32` (accurate to ~1e-6, so an exact duplicate scores 0.9999999), so
10038    /// every candidate is re-scored from the `f64` property vectors by
10039    /// [`exact_vector_similarity`] before `min`, the ordering and the reported
10040    /// score are decided. `k + VECTOR_RESCORE_MARGIN` candidates are fetched so
10041    /// the re-ordering cannot drop a true top-`k` member; see that constant for
10042    /// the rule. The score a caller receives is therefore the same number the
10043    /// brute-force path would have produced, to `f64` precision, and `min = 1.0`
10044    /// finds an exact duplicate.
10045    pub fn find_similar_vector(
10046        &self,
10047        field: &str,
10048        label: Option<&str>,
10049        q: &[f64],
10050        k: usize,
10051        min: f64,
10052    ) -> Vec<(String, f64)> {
10053        self.find_similar_vector_filtered(field, label, q, k, min, None, None, false)
10054            .expect("find_similar_vector_filtered is infallible without where_")
10055    }
10056
10057    /// Like [`find_similar_vector`] but restricts results to nodes visible in
10058    /// `mask`. Hidden nodes never appear in results; the mask is applied
10059    /// **before** k-truncation so a caller still receives up to `k` visible
10060    /// hits.
10061    ///
10062    /// # HNSW path (widening beam)
10063    ///
10064    /// When an HNSW index covers the request, the beam starts at an over-fetch
10065    /// of `k × n / |visible|` (plus the rescore margin) when the mask's
10066    /// selectivity is known from the index length, otherwise at `k` plus that
10067    /// margin. If fewer than `k` visible candidates remain after the mask and
10068    /// `min` filter, the beam doubles — the same ×2 loop exact `VectorSimilar`
10069    /// rules use, capped at `ef_max()` (`EF_MAX` = 4,096). Reaching the cap,
10070    /// or a beam that comes back short of its own width, falls through to the
10071    /// exhaustive masked scan rather than returning a short result.
10072    ///
10073    /// Every surviving candidate is re-scored from the `f64` property vectors,
10074    /// exactly as [`find_similar_vector`] does and for the same reason.
10075    ///
10076    /// # Brute-force path
10077    ///
10078    /// When no HNSW index covers the request, or the beam cannot admit `k`
10079    /// hits, the function builds a masked [`GraphView`] so that `nodes_all` /
10080    /// `nodes_with_label` return only visible nodes, guaranteeing exact `k`
10081    /// results (or all visible nodes if fewer than `k` exist).
10082    pub fn find_similar_vector_masked(
10083        &self,
10084        field: &str,
10085        label: Option<&str>,
10086        q: &[f64],
10087        k: usize,
10088        min: f64,
10089        mask: &crate::mask::NodeMask,
10090    ) -> Vec<(String, f64)> {
10091        self.find_similar_vector_filtered(field, label, q, k, min, Some(mask), None, false)
10092            .expect("find_similar_vector_filtered is infallible without where_")
10093    }
10094
10095    /// Exact or ANN kNN with optional key-list `mask` and property `where_`.
10096    ///
10097    /// `where_` present and failing [`PropPredicate::validate_named`] `"where"`
10098    /// → `QueryError`. `exact=true` or `where_=Some` skip HNSW and GEMM-brute
10099    /// the candidate set (`label ∩ mask ∩ holds(where)`). `mask` alone still
10100    /// uses HNSW when an index covers the field.
10101    #[allow(clippy::too_many_arguments)]
10102    pub fn find_similar_vector_filtered(
10103        &self,
10104        field: &str,
10105        label: Option<&str>,
10106        q: &[f64],
10107        k: usize,
10108        min: f64,
10109        mask: Option<&crate::mask::NodeMask>,
10110        where_: Option<&PropPredicate>,
10111        exact: bool,
10112    ) -> Result<Vec<(String, f64)>> {
10113        self.find_similar_vector_as(
10114            field,
10115            label,
10116            q,
10117            k,
10118            min,
10119            mask,
10120            where_,
10121            exact,
10122            ExactnessCaller::Vector,
10123        )
10124    }
10125
10126    /// [`find_similar_vector_filtered`](Self::find_similar_vector_filtered)
10127    /// with the caller shape named, so the exactness warning can advise the
10128    /// signature that actually reached it. Everything else is identical.
10129    #[allow(clippy::too_many_arguments)]
10130    fn find_similar_vector_as(
10131        &self,
10132        field: &str,
10133        label: Option<&str>,
10134        q: &[f64],
10135        k: usize,
10136        min: f64,
10137        mask: Option<&crate::mask::NodeMask>,
10138        where_: Option<&PropPredicate>,
10139        exact: bool,
10140        caller: ExactnessCaller,
10141    ) -> Result<Vec<(String, f64)>> {
10142        if let Some(pred) = where_ {
10143            pred.validate_named("where")
10144                .map_err(|detail| GraphError::QueryError { detail })?;
10145        }
10146
10147        // Ensure any HNSW blobs retained from the snapshot are deserialized
10148        // before the first ANN query on a clean-open (no-WAL) path.  The
10149        // section read has to come first: on a clean open nothing else has
10150        // called it, so without it `retained_hnsw_blobs` is empty,
10151        // `ensure_hnsw_loaded` caches an empty map in its `OnceLock`, and every
10152        // approximate query on the handle runs brute force — correct results,
10153        // silently off the index.  Both calls are idempotent and cheap once hot.
10154        self.ensure_v8_base_sections_loaded();
10155        self.engine.ensure_hnsw_loaded();
10156        let norm: f64 = q.iter().map(|x| x * x).sum::<f64>().sqrt();
10157        if norm == 0.0 {
10158            return Ok(vec![]);
10159        }
10160        if let Some(m) = mask {
10161            if k == 0 || m.is_empty() {
10162                return Ok(vec![]);
10163            }
10164        }
10165        let q_unit: Vec<f64> = q.iter().map(|x| x / norm).collect();
10166
10167        // `where` implies exact: a predicate must not ride a silent ANN.
10168        let skip_hnsw = exact || where_.is_some();
10169        if !skip_hnsw {
10170            if let Some(mask) = mask {
10171                if let Some(out) =
10172                    self.find_similar_hnsw_masked(field, label, &q_unit, k, min, mask, caller)
10173                {
10174                    return Ok(out);
10175                }
10176            } else if let Some(out) = self.find_similar_hnsw(field, label, &q_unit, k, min) {
10177                return Ok(out);
10178            }
10179        }
10180
10181        let view = match mask {
10182            Some(m) => self.view_masked(m),
10183            None => self.view(),
10184        };
10185        let candidate_ids = Self::vector_candidates(&view, label, where_);
10186        Ok(self.brute_vector_hits(&view, candidate_ids, field, &q_unit, k, min))
10187    }
10188
10189    /// Unmasked HNSW path. `None` when no populated index covers the request.
10190    fn find_similar_hnsw(
10191        &self,
10192        field: &str,
10193        label: Option<&str>,
10194        q_unit: &[f64],
10195        k: usize,
10196        min: f64,
10197    ) -> Option<Vec<(String, f64)>> {
10198        // Try HNSW fast path.
10199        // `None` label searches across all VectorSimilar rules covering `field`
10200        // (merging their results); `Some(lbl)` restricts to rules whose
10201        // dst_label matches.  Returns `None` when no populated HNSW index
10202        // covers the request — the O(n) brute-force fallback handles that case.
10203        let over_k = k.saturating_add(VECTOR_RESCORE_MARGIN);
10204        let hits = match label {
10205            Some(lbl) => self.engine.hnsw_search_dst(field, lbl, q_unit, over_k)?,
10206            None => self.engine.hnsw_search_any_dst(field, q_unit, over_k)?,
10207        };
10208        // Candidates only: the index's `f32` similarity is discarded and
10209        // each hit is re-scored against the `f64` vectors.
10210        let view = self.view();
10211        let mut out: Vec<(String, f64)> = hits
10212            .into_iter()
10213            .filter_map(|(id, _)| {
10214                let sim = exact_vector_similarity(&view, id, field, q_unit)?;
10215                if sim < min {
10216                    return None;
10217                }
10218                Some((self.ids.key_of(id)?.to_string(), sim))
10219            })
10220            .collect();
10221        out.sort_by(|a, b| {
10222            b.1.partial_cmp(&a.1)
10223                .unwrap_or(std::cmp::Ordering::Equal)
10224                .then_with(|| a.0.cmp(&b.0))
10225        });
10226        out.truncate(k);
10227        Some(out)
10228    }
10229
10230    /// Masked HNSW widening beam. `None` when no index covers the request or
10231    /// the beam cannot admit `k` visible hits (caller falls through to brute).
10232    #[allow(clippy::too_many_arguments)]
10233    fn find_similar_hnsw_masked(
10234        &self,
10235        field: &str,
10236        label: Option<&str>,
10237        q_unit: &[f64],
10238        k: usize,
10239        min: f64,
10240        mask: &crate::mask::NodeMask,
10241        caller: ExactnessCaller,
10242    ) -> Option<Vec<(String, f64)>> {
10243        let index_len = match label {
10244            Some(lbl) => self.engine.hnsw_dst_len(field, lbl, q_unit.len()),
10245            None => self.engine.hnsw_any_dst_len(field, q_unit.len()),
10246        };
10247        let n = index_len?;
10248        // The `?` above is the coverage test: past it, an index exists and this
10249        // masked, non-exact call is about to ride it.
10250        self.note_ambiguous_exactness(field, label, caller);
10251        // Same ceiling the exact-rule widening loop in `hnsw_candidates`
10252        // consults — including the `with_ef_max` test hook.
10253        let cap = ef_max();
10254        let visible = mask.len();
10255        let mut ef = k.saturating_add(VECTOR_RESCORE_MARGIN);
10256        if visible > 0 && n > 0 {
10257            let over = k
10258                .saturating_mul(n)
10259                .div_ceil(visible)
10260                .saturating_add(VECTOR_RESCORE_MARGIN);
10261            ef = ef.max(over);
10262        }
10263        loop {
10264            let hits = match label {
10265                Some(lbl) => self
10266                    .engine
10267                    .hnsw_search_dst_with_ef(field, lbl, q_unit, ef, ef),
10268                None => self
10269                    .engine
10270                    .hnsw_search_any_dst_with_ef(field, q_unit, ef, ef),
10271            };
10272            let hits = hits?;
10273            let full = hits.len() == ef;
10274            let mut out = self.score_masked_hnsw_hits(&hits, field, q_unit, min, mask);
10275            if out.len() >= k {
10276                out.truncate(k);
10277                return Some(out);
10278            }
10279            // Short of its width (frontier exhausted) or at the ceiling:
10280            // a wider beam reaches nothing new, so the scan answers.
10281            if !full || ef >= cap {
10282                return None;
10283            }
10284            ef = ef.saturating_mul(2);
10285        }
10286    }
10287
10288    /// Say once, per `(field, label)` index and caller shape, that a masked
10289    /// search is answering approximately.
10290    ///
10291    /// A mask narrows *which nodes may be returned*. It does not choose a
10292    /// kernel — `exact=true` and a `where=` predicate do, and nothing else
10293    /// does. A caller who needed exact answers, passed `mask=` alone, and read
10294    /// the mask as a promise of exhaustiveness gets a correct-looking
10295    /// approximate answer and no signal at all; that is a silent wrong answer,
10296    /// and it has cost an integration team real time.
10297    ///
10298    /// The fix is a question, not a behaviour change. Making a mask imply
10299    /// `exact` would turn every existing masked caller's ANN into an O(n) GEMM
10300    /// without asking them, which is a worse trade than the ambiguity.
10301    ///
10302    /// Printed once per index for the reason the dimension-mismatch skip in
10303    /// `core_rules::hnsw` is: a line on every call is a line callers learn to
10304    /// scroll past.
10305    ///
10306    /// `caller` decides the advice. The same leg is reached from two signatures
10307    /// and only one of them has an `exact` argument to pass; see
10308    /// [`ExactnessCaller`].
10309    fn note_ambiguous_exactness(&self, field: &str, label: Option<&str>, caller: ExactnessCaller) {
10310        let entry = (field.to_string(), label.unwrap_or("").to_string(), caller);
10311        let first = match self.warned_ambiguous_exactness.lock() {
10312            Ok(mut seen) => seen.insert(entry),
10313            Err(poisoned) => poisoned.into_inner().insert(entry),
10314        };
10315        if !first {
10316            return;
10317        }
10318        AMBIGUOUS_EXACTNESS_WARNS.with(|c| c.set(c.get().saturating_add(1)));
10319        let which = match label {
10320            Some(lbl) => format!(" (label `{lbl}`)"),
10321            None => String::new(),
10322        };
10323        let subject = caller.subject();
10324        let advice = caller.advice();
10325        let line = format!(
10326            "mushroomdb: {subject} on field `{field}`{which} is answering \
10327             approximately. A mask narrows which nodes may be returned; it does not \
10328             change which kernel runs, and an index covers this field. For an exact \
10329             answer over the same visible candidate set, {advice} Further masked \
10330             searches of this shape on this index are silent."
10331        );
10332        AMBIGUOUS_EXACTNESS_LAST.with(|c| *c.borrow_mut() = Some(line.clone()));
10333        eprintln!("{line}");
10334    }
10335
10336    /// `label ∩ mask ∩ holds(where)`. Index fast path when `label` is `Some`
10337    /// and `(label, where.field)` is enabled; otherwise scan with `visible()`.
10338    fn vector_candidates(
10339        view: &GraphView<'_>,
10340        label: Option<&str>,
10341        where_: Option<&PropPredicate>,
10342    ) -> Vec<u32> {
10343        if let (Some(lbl), Some(pred)) = (label, where_) {
10344            let indexed = view
10345                .prop_index
10346                .is_some_and(|idx| idx.is_enabled(lbl, &pred.field));
10347            if indexed {
10348                match (&pred.eq, &pred.in_) {
10349                    (Some(eq), None) => {
10350                        if let Some(ids) = view.nodes_with_prop(lbl, &pred.field, eq) {
10351                            return ids;
10352                        }
10353                    }
10354                    (None, Some(allowed)) => {
10355                        let mut seen = HashSet::new();
10356                        let mut out = Vec::new();
10357                        for v in allowed {
10358                            if let Some(ids) = view.nodes_with_prop(lbl, &pred.field, v) {
10359                                for id in ids {
10360                                    if seen.insert(id) {
10361                                        out.push(id);
10362                                    }
10363                                }
10364                            }
10365                        }
10366                        return out;
10367                    }
10368                    _ => {}
10369                }
10370            }
10371        }
10372
10373        let mut ids: Vec<u32> = match label {
10374            Some(lbl) => view
10375                .nodes_with_label(lbl)
10376                .into_iter()
10377                .filter(|&id| view.visible(id))
10378                .collect(),
10379            None => view.nodes_all(),
10380        };
10381        if let Some(pred) = where_ {
10382            ids.retain(|&id| match view.prop(id, &pred.field) {
10383                None => pred.holds(None),
10384                Some(vr) => pred.holds(Some(vr.as_value())),
10385            });
10386        }
10387        ids
10388    }
10389
10390    /// Exact brute kNN: pack candidates at `q_unit`'s dim, GEMV, keep
10391    /// `score >= min`, sort `(sim desc, key asc)`, truncate to `k`.
10392    fn brute_vector_hits(
10393        &self,
10394        view: &GraphView<'_>,
10395        candidate_ids: impl IntoIterator<Item = u32>,
10396        field: &str,
10397        q_unit: &[f64],
10398        k: usize,
10399        min: f64,
10400    ) -> Vec<(String, f64)> {
10401        let rows: Vec<(u32, std::borrow::Cow<'_, [f64]>)> = candidate_ids
10402            .into_iter()
10403            .filter_map(|id| crate::exact_knn::vector_f64(view, id, field).map(|v| (id, v)))
10404            .collect();
10405        let packed =
10406            crate::exact_knn::pack(rows.iter().map(|(id, v)| (*id, v.as_ref())), q_unit.len());
10407        let scores = crate::exact_knn::gemv(&packed, q_unit);
10408        let mut scored: Vec<(String, f64)> = packed
10409            .ids
10410            .iter()
10411            .zip(scores.iter())
10412            .filter_map(|(&id, &sim)| {
10413                if sim < min {
10414                    return None;
10415                }
10416                let key = self.ids.key_of(id)?.to_string();
10417                Some((key, sim))
10418            })
10419            .collect();
10420        scored.sort_by(|a, b| {
10421            b.1.partial_cmp(&a.1)
10422                .unwrap_or(std::cmp::Ordering::Equal)
10423                .then_with(|| a.0.cmp(&b.0))
10424        });
10425        scored.truncate(k);
10426        scored
10427    }
10428
10429    /// Exact cosine top-k for each key in `keys`, scored only against `keys`.
10430    ///
10431    /// `min` is cosine similarity in [-1, 1], inclusive (`score >= min`), the
10432    /// same unit and inequality as `find_similar_vector`. Self-matches are
10433    /// excluded. Unknown keys, keys with no `field`, zero-norm or wrong-dim
10434    /// embeddings are omitted as both query and candidate. Duplicate keys are
10435    /// collapsed, first-seen order. Empty `keys` → empty `Ok(vec![])`. Never
10436    /// uses HNSW. `n > PAIRWISE_MAX_N` → `QueryError`.
10437    #[allow(clippy::type_complexity)]
10438    pub fn pairwise_similar(
10439        &self,
10440        keys: &[&str],
10441        field: &str,
10442        k: usize,
10443        min: f64,
10444    ) -> Result<Vec<(String, Vec<(String, f64)>)>> {
10445        let mut seen = HashSet::new();
10446        let mut unique_ids = Vec::new();
10447        for key in keys {
10448            let Some(id) = self.ids.get(key) else {
10449                continue;
10450            };
10451            if seen.insert(id) {
10452                unique_ids.push(id);
10453            }
10454        }
10455        let max_n = crate::exact_knn::pairwise_max_n();
10456        if unique_ids.len() > max_n {
10457            return Err(GraphError::QueryError {
10458                detail: format!(
10459                    "pairwise_similar: n={} exceeds PAIRWISE_MAX_N ({max_n})",
10460                    unique_ids.len()
10461                ),
10462            });
10463        }
10464        if unique_ids.is_empty() {
10465            return Ok(Vec::new());
10466        }
10467
10468        let view = self.view();
10469        let mut rows: Vec<(u32, std::borrow::Cow<'_, [f64]>)> = Vec::new();
10470        let mut counts: HashMap<usize, usize> = HashMap::new();
10471        for id in unique_ids {
10472            let Some(v) = crate::exact_knn::vector_f64(&view, id, field) else {
10473                continue;
10474            };
10475            let norm: f64 = v.iter().map(|x| x * x).sum::<f64>().sqrt();
10476            if norm == 0.0 {
10477                continue;
10478            }
10479            *counts.entry(v.len()).or_default() += 1;
10480            rows.push((id, v));
10481        }
10482        if rows.is_empty() {
10483            return Ok(Vec::new());
10484        }
10485        let dim = counts
10486            .into_iter()
10487            .max_by_key(|&(d, c)| (c, d))
10488            .map(|(d, _)| d)
10489            .expect("rows non-empty");
10490        let packed = crate::exact_knn::pack(rows.iter().map(|(id, v)| (*id, v.as_ref())), dim);
10491        let n = packed.ids.len();
10492        if n == 0 {
10493            return Ok(Vec::new());
10494        }
10495        let src_keys: Vec<String> = packed
10496            .ids
10497            .iter()
10498            .map(|&id| self.ids.key_of(id).unwrap_or("").to_string())
10499            .collect();
10500
10501        let mut out = Vec::with_capacity(n);
10502        if n <= crate::exact_knn::pairwise_gram_max() {
10503            let sims = crate::exact_knn::gram(&packed);
10504            for i in 0..n {
10505                out.push(Self::topk_from_row(
10506                    &src_keys,
10507                    i,
10508                    &sims[i * n..(i + 1) * n],
10509                    k,
10510                    min,
10511                ));
10512            }
10513        } else {
10514            for i in 0..n {
10515                let row = &packed.data[i * packed.dim..(i + 1) * packed.dim];
10516                let scores = crate::exact_knn::gemv(&packed, row);
10517                out.push(Self::topk_from_row(&src_keys, i, &scores, k, min));
10518            }
10519        }
10520        Ok(out)
10521    }
10522
10523    /// [`pairwise_similar`](Self::pairwise_similar) over the keys the mask
10524    /// admits — intersected **before** the matmul, never filtered after it.
10525    ///
10526    /// A hidden vector packed into the Gram is a row every visible key is
10527    /// scored against. It can take a visible neighbour's place in the top-`k`,
10528    /// and because the packed dimension is a majority vote over the candidate
10529    /// rows it can decide whether a visible pair is scored at all. Dropping
10530    /// hidden names from the finished answer leaves both effects standing, so
10531    /// the intersection happens first and the answer is byte-for-byte the one
10532    /// `pairwise_similar` gives for the visible keys alone.
10533    ///
10534    /// The caps therefore measure the **post-filter** count: a key set over
10535    /// [`PAIRWISE_MAX_N`](crate::PAIRWISE_MAX_N) unscoped can come under it
10536    /// scoped and succeed, because the work the cap refuses is work this call
10537    /// no longer does. A filtered count still over the cap is still refused.
10538    #[allow(clippy::type_complexity)]
10539    pub fn pairwise_similar_scoped(
10540        &self,
10541        keys: &[&str],
10542        field: &str,
10543        k: usize,
10544        min: f64,
10545        mask: &crate::mask::NodeMask,
10546    ) -> Result<Vec<(String, Vec<(String, f64)>)>> {
10547        let visible: Vec<&str> = keys
10548            .iter()
10549            .copied()
10550            .filter(|key| mask.contains_node(self, key))
10551            .collect();
10552        self.pairwise_similar(&visible, field, k, min)
10553    }
10554
10555    /// Neighbours of packed row `i`: drop self, keep `score >= min`, sort
10556    /// `(sim desc, key asc)`, truncate to `k`. Packed srcs with no survivors
10557    /// still appear as `(src, [])`.
10558    fn topk_from_row(
10559        src_keys: &[String],
10560        i: usize,
10561        scores: &[f64],
10562        k: usize,
10563        min: f64,
10564    ) -> (String, Vec<(String, f64)>) {
10565        let mut neigh: Vec<(String, f64)> = scores
10566            .iter()
10567            .enumerate()
10568            .filter_map(|(j, &sim)| {
10569                if i == j || sim < min {
10570                    return None;
10571                }
10572                Some((src_keys[j].clone(), sim))
10573            })
10574            .collect();
10575        neigh.sort_by(|a, b| {
10576            b.1.partial_cmp(&a.1)
10577                .unwrap_or(std::cmp::Ordering::Equal)
10578                .then_with(|| a.0.cmp(&b.0))
10579        });
10580        neigh.truncate(k);
10581        (src_keys[i].clone(), neigh)
10582    }
10583
10584    /// Re-score HNSW candidates from the `f64` vectors, drop hidden / below-`min`
10585    /// hits, order by score then key. The index's own `f32` similarity is discarded.
10586    fn score_masked_hnsw_hits(
10587        &self,
10588        hits: &[(u32, f64)],
10589        field: &str,
10590        q_unit: &[f64],
10591        min: f64,
10592        mask: &crate::mask::NodeMask,
10593    ) -> Vec<(String, f64)> {
10594        let view = self.view_masked(mask);
10595        let mut out: Vec<(String, f64)> = hits
10596            .iter()
10597            .copied()
10598            .filter(|&(id, _)| mask.contains_id(id))
10599            .filter_map(|(id, _)| {
10600                let sim = exact_vector_similarity(&view, id, field, q_unit)?;
10601                if sim < min {
10602                    return None;
10603                }
10604                Some((self.ids.key_of(id)?.to_string(), sim))
10605            })
10606            .collect();
10607        out.sort_by(|a, b| {
10608            b.1.partial_cmp(&a.1)
10609                .unwrap_or(std::cmp::Ordering::Equal)
10610                .then_with(|| a.0.cmp(&b.0))
10611        });
10612        out
10613    }
10614
10615    /// Read a single property from an edge.
10616    ///
10617    /// Returns `None` when the edge does not exist, the field is absent, or any
10618    /// of the string keys cannot be resolved to interned ids.  Only edge props
10619    /// written by rules (weight fields) are accessible without a `set_edge_prop`
10620    /// binding; topology-only edges (no props set) return `None` for every field.
10621    pub fn get_edge_prop(
10622        &self,
10623        edge_type: &str,
10624        src_key: &str,
10625        dst_key: &str,
10626        field: &str,
10627    ) -> Option<Value> {
10628        let etype = self.syms.get(edge_type)?;
10629        let src = self.ids.get(src_key)?;
10630        let dst = self.ids.get(dst_key)?;
10631        self.edge_props_view().get(etype, src, dst, field)
10632    }
10633
10634    /// Lex → parse → plan → execute `cypher` over a read-only view.
10635    /// Every pipeline `Err(String)` becomes `GraphError::QueryError` with a
10636    /// stage prefix (`lex:` / `parse:` / `plan:` / `execute:`).
10637    pub fn query(&self, cypher: &str, params: &BTreeMap<String, Value>) -> Result<ResultSet> {
10638        let tokens = lex(cypher).map_err(|e| GraphError::QueryError {
10639            detail: format!("lex: {e}"),
10640        })?;
10641        let union = parse_read(&tokens).map_err(|e| GraphError::QueryError {
10642            detail: format!("parse: {e}"),
10643        })?;
10644        let t0 = std::time::Instant::now();
10645        let result = execute_union(&self.view(), &union, &Params(params)).map_err(|e| {
10646            GraphError::QueryError {
10647                detail: format!("execute: {e}"),
10648            }
10649        });
10650        let elapsed_ms = t0.elapsed().as_millis() as u64;
10651        let threshold = self.slow_query_threshold_ms;
10652        if threshold > 0 && elapsed_ms >= threshold {
10653            eprintln!("[mushroomdb] slow query ({elapsed_ms}ms): {cypher}");
10654            let entry = SlowQueryEntry {
10655                ms: elapsed_ms,
10656                query: cypher.to_string(),
10657                at_commit: self.commit_seq,
10658            };
10659            if let Ok(mut log) = self.slow_queries.lock() {
10660                if log.entries.len() == SLOW_QUERY_RING_CAP {
10661                    log.entries.pop_front();
10662                }
10663                log.entries.push_back(entry);
10664                log.total += 1;
10665            }
10666        }
10667        result
10668    }
10669
10670    /// Convenience entry-point that accepts a slice of `(name, value)` pairs
10671    /// instead of a pre-built `BTreeMap`.  Equivalent to building the map and
10672    /// calling [`GraphDb::query`].
10673    pub fn query_with_params(&self, cypher: &str, params: &[(&str, Value)]) -> Result<ResultSet> {
10674        let map: BTreeMap<String, Value> = params
10675            .iter()
10676            .map(|(k, v)| (k.to_string(), v.clone()))
10677            .collect();
10678        self.query(cypher, &map)
10679    }
10680
10681    /// Execute a Cypher write statement (CREATE / MATCH…SET / MATCH…DELETE / MERGE).
10682    ///
10683    /// All mutations flow through the same `insert_node` / `set_prop` /
10684    /// `delete_edge` / `insert_edge` path as the Rust API so the rule engine
10685    /// fires and the WAL captures everything with one fsync per statement.
10686    ///
10687    /// Returns a one-row [`ResultSet`] with columns `created`, `properties_set`,
10688    /// and `deleted` matching the write-result contract.
10689    ///
10690    /// **Mutation routing**: mutations are collected into a single
10691    /// [`BatchBuilder`] and committed atomically (one WAL `Batch` frame, one
10692    /// fsync). The MATCH phase for SET/DELETE uses a read-only `execute` call
10693    /// over `self.view()` — the borrow is dropped before the batch is opened.
10694    ///
10695    /// **Limitations (v1)**:
10696    /// - SET RHS must be a literal, `$param`, or arithmetic; bare property copy → named error.
10697    /// - `DETACH DELETE n` → calls `delete_node` for each matched node (removes all edges).
10698    /// - Bare `DELETE n` → error if n has any incident edges; succeeds for isolated nodes.
10699    /// - MERGE supports `ON CREATE SET` / `ON MATCH SET` in the same write batch.
10700    /// - Deleting a derived edge → named error "cannot delete derived edge".
10701    pub fn query_write(
10702        &mut self,
10703        cypher: &str,
10704        params: &BTreeMap<String, Value>,
10705    ) -> Result<ResultSet> {
10706        let tokens = lex(cypher).map_err(|e| GraphError::QueryError {
10707            detail: format!("lex: {e}"),
10708        })?;
10709        let stmt = parse_write(&tokens).map_err(|e| GraphError::QueryError {
10710            detail: format!("parse: {e}"),
10711        })?;
10712        self.exec_write_stmt(stmt, params)
10713    }
10714
10715    fn exec_write_stmt(
10716        &mut self,
10717        stmt: WriteStatement,
10718        params: &BTreeMap<String, Value>,
10719    ) -> Result<ResultSet> {
10720        match stmt {
10721            WriteStatement::Create(s) => self.exec_create(s, params),
10722            WriteStatement::MatchSet(s) => self.exec_match_set(s, params),
10723            WriteStatement::MatchDelete(s) => self.exec_match_delete(s, params),
10724            WriteStatement::MatchDeleteNode(s) => self.exec_match_delete_node(s, params),
10725            WriteStatement::Merge(s) => self.exec_merge(s, params),
10726        }
10727    }
10728
10729    fn exec_create(
10730        &mut self,
10731        stmt: core_query::cypher::CreateStmt,
10732        params: &BTreeMap<String, Value>,
10733    ) -> Result<ResultSet> {
10734        // Extract the node key from props: require a string-valued `id` field.
10735        let mut var_to_key: BTreeMap<String, String> = BTreeMap::new();
10736        for node in &stmt.nodes {
10737            let var = node.var.as_deref().unwrap_or("_cn0");
10738            let key = node
10739                .props
10740                .iter()
10741                .find(|(f, _)| f == "id")
10742                .and_then(|(_, v)| {
10743                    if let Value::Str(s) = v {
10744                        Some(s.clone())
10745                    } else {
10746                        None
10747                    }
10748                })
10749                .ok_or_else(|| GraphError::QueryError {
10750                    detail: format!(
10751                        "CREATE node ({}:{}) requires a string 'id' property",
10752                        var, node.label
10753                    ),
10754                })?;
10755            var_to_key.insert(var.to_string(), key);
10756        }
10757
10758        let mut batch = self.batch();
10759        let mut created: usize = 0;
10760        for node in &stmt.nodes {
10761            let var = node.var.as_deref().unwrap_or("_cn0");
10762            let key = &var_to_key[var];
10763            batch.insert_node(&node.label, key, node.props.clone());
10764            created += 1;
10765        }
10766        for edge in &stmt.edges {
10767            let src_key = var_to_key
10768                .get(&edge.src_var)
10769                .ok_or_else(|| GraphError::QueryError {
10770                    detail: format!("CREATE edge src variable '{}' is not bound", edge.src_var),
10771                })?;
10772            let dst_key = var_to_key
10773                .get(&edge.dst_var)
10774                .ok_or_else(|| GraphError::QueryError {
10775                    detail: format!("CREATE edge dst variable '{}' is not bound", edge.dst_var),
10776                })?;
10777            batch.insert_edge(&edge.etype, src_key, dst_key);
10778        }
10779        batch.commit()?;
10780
10781        // Optional RETURN clause: project created bindings as a read result.
10782        if let Some(returns) = stmt.returns {
10783            // Each created node is looked up by its key via a separate MATCH pattern.
10784            // Multiple single-node patterns cross-join to produce 1 output row with
10785            // all variables bound (each pattern returns exactly 1 row).
10786            let patterns: Vec<Pattern> = stmt
10787                .nodes
10788                .iter()
10789                .map(|node| {
10790                    let var = node.var.as_deref().unwrap_or("_cn0");
10791                    let key = var_to_key[var].clone();
10792                    Pattern {
10793                        start: NodePat {
10794                            var: Some(var.to_string()),
10795                            label: Some(node.label.clone()),
10796                            props: vec![("id".to_string(), Operand::Lit(Value::Str(key)))],
10797                        },
10798                        chain: vec![],
10799                        shortest: false,
10800                    }
10801                })
10802                .collect();
10803            let q = Query {
10804                matches: patterns,
10805                optional_clauses: vec![],
10806                where_expr: None,
10807                unwinds: vec![],
10808                post_unwind_where: None,
10809                stages: vec![],
10810                returns,
10811                distinct: false,
10812                order_by: vec![],
10813                skip: None,
10814                limit: None,
10815            };
10816            let ops = plan(&q).map_err(|e| GraphError::QueryError {
10817                detail: format!("plan: {e}"),
10818            })?;
10819            return execute(&self.view(), &ops, &Params(params)).map_err(|e| {
10820                GraphError::QueryError {
10821                    detail: format!("execute: {e}"),
10822                }
10823            });
10824        }
10825
10826        let mut rs = write_result_set();
10827        rs.push_row(vec![
10828            Some(Value::Int(created as i64)),
10829            Some(Value::Int(0)),
10830            Some(Value::Int(0)),
10831        ]);
10832        Ok(rs)
10833    }
10834
10835    fn exec_match_set(
10836        &mut self,
10837        stmt: core_query::cypher::MatchSetStmt,
10838        params: &BTreeMap<String, Value>,
10839    ) -> Result<ResultSet> {
10840        let project_returns = stmt.returns.clone();
10841        // Collect unique node vars targeted by SET clauses, plus RETURN bindings
10842        // so the post-write projection can look them up by key.
10843        let mut set_vars: Vec<String> = Vec::new();
10844        for s in &stmt.sets {
10845            if !set_vars.contains(&s.var) {
10846                set_vars.push(s.var.clone());
10847            }
10848        }
10849        let rel_vars = pattern_rel_vars(&stmt.matches);
10850        // `count` is the engine's, on an edge: it is the insert-count §5.13
10851        // maintains, and a `SET` that overwrote it would make the number mean
10852        // whatever the last writer said rather than how many times the pair was
10853        // inserted. Refused by name here, before the match runs, so the caller
10854        // is told what is actually wrong instead of meeting the executor's
10855        // generic "did not resolve to a node key" — and so the answer does not
10856        // depend on whether the pattern happened to match a row. The same name
10857        // on a *node* is an ordinary property and is untouched.
10858        for s in &stmt.sets {
10859            if s.field == EDGE_COUNT_PROP && rel_vars.iter().any(|r| r == &s.var) {
10860                return Err(GraphError::QueryError {
10861                    detail: format!(
10862                        "cannot SET {}.{EDGE_COUNT_PROP}: `{EDGE_COUNT_PROP}` is a reserved edge \
10863                         property holding the pair's insert count",
10864                        s.var
10865                    ),
10866                });
10867            }
10868        }
10869        let mut lookup_vars = set_vars.clone();
10870        for v in pattern_node_vars(&stmt.matches) {
10871            add_var(&mut lookup_vars, &v);
10872        }
10873        if let Some(ref returns) = project_returns {
10874            for v in ret_node_vars(returns) {
10875                if !rel_vars.iter().any(|r| r == &v) {
10876                    add_var(&mut lookup_vars, &v);
10877                }
10878            }
10879        }
10880
10881        // Synthesize a read query: MATCH … WHERE … RETURN <lookup_vars>, <set_values…>
10882        // SET values are projected as ScalarExpr items so that arithmetic expressions
10883        // (e.g. `SET n.score = n.score * 1.5`) are evaluated in the matched-row context.
10884        let mut set_returns: Vec<RetItem> = lookup_vars
10885            .iter()
10886            .map(|v| RetItem {
10887                value: RetVal::Var(v.clone()),
10888                alias: None,
10889            })
10890            .collect();
10891        // One computed column per SET clause; alias is `__sv_<i>`.
10892        let set_val_cols: Vec<String> = stmt
10893            .sets
10894            .iter()
10895            .enumerate()
10896            .map(|(i, _)| format!("__sv_{i}"))
10897            .collect();
10898        for (sc, col) in stmt.sets.iter().zip(&set_val_cols) {
10899            set_returns.push(RetItem {
10900                value: RetVal::ScalarExpr(sc.value.clone()),
10901                alias: Some(col.clone()),
10902            });
10903        }
10904        // Capture relationship types while r is bound; SET does not change them.
10905        for r in &rel_vars {
10906            set_returns.push(RetItem {
10907                value: RetVal::FuncCall {
10908                    name: "type".into(),
10909                    args: vec![Operand::Var(r.clone())],
10910                },
10911                alias: Some(rel_type_alias(r)),
10912            });
10913        }
10914
10915        let read_q = Query {
10916            matches: stmt.matches.clone(),
10917            optional_clauses: vec![],
10918            where_expr: stmt.where_expr.clone(),
10919            unwinds: vec![],
10920            post_unwind_where: None,
10921            stages: vec![],
10922            returns: set_returns,
10923            distinct: false,
10924            order_by: vec![],
10925            skip: None,
10926            limit: None,
10927        };
10928        let ops = plan(&read_q).map_err(|e| GraphError::QueryError {
10929            detail: format!("plan: {e}"),
10930        })?;
10931        // MATCH phase is read-only; borrow ends before batch opens.
10932        //
10933        // When a role-scoped write is in flight, run the MATCH read through
10934        // view_masked so hidden nodes are invisible → hidden ≡ absent ≡
10935        // zero-rows (no SetProp ops generated, no existence-oracle 403).
10936        // Full-authority writes (pending_write_authz=None) keep view().
10937        let match_rs = {
10938            let mask_opt = self.pending_write_authz.as_ref().map(|a| a.mask.clone());
10939            if let Some(ref mask) = mask_opt {
10940                execute(&self.view_masked(mask), &ops, &Params(params))
10941            } else {
10942                execute(&self.view(), &ops, &Params(params))
10943            }
10944        }
10945        .map_err(|e| GraphError::QueryError {
10946            detail: format!("execute: {e}"),
10947        })?;
10948
10949        // Collect (key, field, value) for each matched row × each SET clause.
10950        let mut set_ops: Vec<(String, String, Value)> = Vec::new();
10951        for row_i in 0..match_rs.len() {
10952            for (sc, col) in stmt.sets.iter().zip(&set_val_cols) {
10953                let key = match match_rs.get(row_i, &sc.var) {
10954                    Some(Value::Str(k)) => k.clone(),
10955                    _ => {
10956                        return Err(GraphError::QueryError {
10957                            detail: format!(
10958                                "SET variable '{}' did not resolve to a node key",
10959                                sc.var
10960                            ),
10961                        })
10962                    }
10963                };
10964                // The SET value was already evaluated by the executor.
10965                let value = match match_rs.get(row_i, col) {
10966                    Some(v) => v.clone(),
10967                    None => {
10968                        return Err(GraphError::QueryError {
10969                            detail: format!(
10970                                "SET value for {}.{} evaluated to null",
10971                                sc.var, sc.field
10972                            ),
10973                        })
10974                    }
10975                };
10976                set_ops.push((key, sc.field.clone(), value));
10977            }
10978        }
10979
10980        // Apply as one atomic batch.
10981        let props_set = set_ops.len();
10982        let mut batch = self.batch();
10983        for (key, field, value) in set_ops {
10984            batch.set_prop(&key, &field, value);
10985        }
10986        batch.commit()?;
10987
10988        if let Some(returns) = project_returns {
10989            return project_set_return_rows(self, &rel_vars, &match_rs, &returns, params);
10990        }
10991
10992        let mut rs = write_result_set();
10993        rs.push_row(vec![
10994            Some(Value::Int(0)),
10995            Some(Value::Int(props_set as i64)),
10996            Some(Value::Int(0)),
10997        ]);
10998        Ok(rs)
10999    }
11000
11001    fn exec_match_delete(
11002        &mut self,
11003        stmt: core_query::cypher::MatchDeleteStmt,
11004        params: &BTreeMap<String, Value>,
11005    ) -> Result<ResultSet> {
11006        // Collect unique node vars needed to identify edge endpoints.
11007        let mut node_vars: Vec<String> = Vec::new();
11008        for ed in &stmt.deletes {
11009            if !node_vars.contains(&ed.src_var) {
11010                node_vars.push(ed.src_var.clone());
11011            }
11012            if !node_vars.contains(&ed.dst_var) {
11013                node_vars.push(ed.dst_var.clone());
11014            }
11015        }
11016
11017        // Synthesize read query.
11018        let returns: Vec<RetItem> = node_vars
11019            .iter()
11020            .map(|v| RetItem {
11021                value: RetVal::Var(v.clone()),
11022                alias: None,
11023            })
11024            .collect();
11025        let read_q = Query {
11026            matches: stmt.matches,
11027            optional_clauses: vec![],
11028            where_expr: stmt.where_expr,
11029            unwinds: vec![],
11030            post_unwind_where: None,
11031            stages: vec![],
11032            returns,
11033            distinct: false,
11034            order_by: vec![],
11035            skip: None,
11036            limit: None,
11037        };
11038        let ops = plan(&read_q).map_err(|e| GraphError::QueryError {
11039            detail: format!("plan: {e}"),
11040        })?;
11041        // Role-scoped writes: mask the MATCH read phase so hidden nodes are
11042        // invisible → hidden ≡ absent ≡ zero-rows (spec §3.1, hidden ≡ absent).
11043        let match_rs = {
11044            let mask_opt = self.pending_write_authz.as_ref().map(|a| a.mask.clone());
11045            if let Some(ref mask) = mask_opt {
11046                execute(&self.view_masked(mask), &ops, &Params(params))
11047            } else {
11048                execute(&self.view(), &ops, &Params(params))
11049            }
11050        }
11051        .map_err(|e| GraphError::QueryError {
11052            detail: format!("execute: {e}"),
11053        })?;
11054
11055        // Collect (etype, src_key, dst_key) for each row × each delete target.
11056        let mut del_ops: Vec<(String, String, String)> = Vec::new();
11057        for row_i in 0..match_rs.len() {
11058            for ed in &stmt.deletes {
11059                let src_key = match match_rs.get(row_i, &ed.src_var) {
11060                    Some(Value::Str(k)) => k.clone(),
11061                    _ => {
11062                        return Err(GraphError::QueryError {
11063                            detail: format!(
11064                                "DELETE src variable '{}' did not resolve to a node key",
11065                                ed.src_var
11066                            ),
11067                        })
11068                    }
11069                };
11070                let dst_key = match match_rs.get(row_i, &ed.dst_var) {
11071                    Some(Value::Str(k)) => k.clone(),
11072                    _ => {
11073                        return Err(GraphError::QueryError {
11074                            detail: format!(
11075                                "DELETE dst variable '{}' did not resolve to a node key",
11076                                ed.dst_var
11077                            ),
11078                        })
11079                    }
11080                };
11081                del_ops.push((ed.etype.clone(), src_key, dst_key));
11082            }
11083        }
11084
11085        // Apply as one atomic batch.
11086        let deleted = del_ops.len();
11087        let mut batch = self.batch();
11088        for (etype, src_key, dst_key) in del_ops {
11089            batch.delete_edge(&etype, &src_key, &dst_key);
11090        }
11091        batch.commit().map_err(|e| match e {
11092            GraphError::RuleOwned { .. } => GraphError::QueryError {
11093                detail: "cannot delete derived edge; retract via the rule or change the property"
11094                    .to_string(),
11095            },
11096            other => other,
11097        })?;
11098
11099        let mut rs = write_result_set();
11100        rs.push_row(vec![
11101            Some(Value::Int(0)),
11102            Some(Value::Int(0)),
11103            Some(Value::Int(deleted as i64)),
11104        ]);
11105        Ok(rs)
11106    }
11107
11108    /// Execute `MATCH … [DETACH] DELETE <node_var> [, …]`.
11109    ///
11110    /// Collects the matching node keys via an ephemeral read query, then calls
11111    /// `delete_node` on each one.  When `stmt.detach` is `false` (bare DELETE)
11112    /// the executor first checks that the node has no incident edges; if any
11113    /// remain it returns a named error matching openCypher semantics.
11114    fn exec_match_delete_node(
11115        &mut self,
11116        stmt: MatchDeleteNodeStmt,
11117        params: &BTreeMap<String, Value>,
11118    ) -> Result<ResultSet> {
11119        // Build a read query returning only the node keys we need.
11120        let returns: Vec<RetItem> = stmt
11121            .node_vars
11122            .iter()
11123            .map(|v| RetItem {
11124                value: RetVal::Var(v.clone()),
11125                alias: None,
11126            })
11127            .collect();
11128        let read_q = Query {
11129            matches: stmt.matches,
11130            optional_clauses: vec![],
11131            where_expr: stmt.where_expr,
11132            unwinds: vec![],
11133            post_unwind_where: None,
11134            stages: vec![],
11135            returns,
11136            distinct: false,
11137            order_by: vec![],
11138            skip: None,
11139            limit: None,
11140        };
11141        let ops = plan(&read_q).map_err(|e| GraphError::QueryError {
11142            detail: format!("plan: {e}"),
11143        })?;
11144        // Role-scoped writes: mask the MATCH read phase so hidden nodes are
11145        // invisible → hidden ≡ absent ≡ zero-rows (spec §3.1, hidden ≡ absent).
11146        let match_rs = {
11147            let mask_opt = self.pending_write_authz.as_ref().map(|a| a.mask.clone());
11148            if let Some(ref mask) = mask_opt {
11149                execute(&self.view_masked(mask), &ops, &Params(params))
11150            } else {
11151                execute(&self.view(), &ops, &Params(params))
11152            }
11153        }
11154        .map_err(|e| GraphError::QueryError {
11155            detail: format!("execute: {e}"),
11156        })?;
11157
11158        // Collect unique node keys to delete (deduplicate across rows × vars).
11159        let mut keys: Vec<String> = Vec::new();
11160        for row_i in 0..match_rs.len() {
11161            for var in &stmt.node_vars {
11162                if let Some(Value::Str(k)) = match_rs.get(row_i, var) {
11163                    if !keys.contains(k) {
11164                        keys.push(k.clone());
11165                    }
11166                }
11167            }
11168        }
11169
11170        if !stmt.detach {
11171            // openCypher bare DELETE: error if any matched node has incident edges.
11172            for key in &keys {
11173                if let Some(id) = self.ids.get(key) {
11174                    let tv = self.topo_view();
11175                    let has_edges = tv.etypes().any(|et| {
11176                        !tv.neighbors(et, Direction::Out, id).is_empty()
11177                            || !tv.neighbors(et, Direction::In, id).is_empty()
11178                    });
11179                    if has_edges {
11180                        return Err(GraphError::QueryError {
11181                            detail: format!(
11182                                "Cannot delete node `{key}` because it still has incident edges. \
11183                                 Use DETACH DELETE to remove the node and all its edges."
11184                            ),
11185                        });
11186                    }
11187                }
11188            }
11189        }
11190
11191        let mut nodes_deleted = 0i64;
11192        let mut edges_deleted = 0i64;
11193        for key in keys {
11194            match self.delete_node(&key) {
11195                Ok(report) => {
11196                    nodes_deleted += 1;
11197                    edges_deleted += (report.manual_edges + report.derived_edges) as i64;
11198                }
11199                Err(GraphError::KeyNotFound { .. }) => {
11200                    // Node may have been deleted by an earlier iteration (e.g., via
11201                    // multiple MATCH rows for the same node).  Safe to skip.
11202                }
11203                Err(e) => return Err(e),
11204            }
11205        }
11206
11207        let mut rs = write_result_set();
11208        rs.push_row(vec![
11209            Some(Value::Int(0)),
11210            Some(Value::Int(0)),
11211            Some(Value::Int(nodes_deleted + edges_deleted)),
11212        ]);
11213        Ok(rs)
11214    }
11215
11216    /// Props the MERGE create arm inserts: the identifying key, plus `ns` when
11217    /// the pattern named one, or the executing role's sole namespace when it
11218    /// did not. A role bound to two or more namespaces cannot choose, and is
11219    /// refused with [`MERGE_CREATE_NEEDS_ONE_NAMESPACE`]. The authorizer still
11220    /// refuses a named `ns` the role cannot write.
11221    fn merge_create_props(
11222        &self,
11223        key_field: &str,
11224        key_value: &Value,
11225        named_ns: Option<&Value>,
11226    ) -> Result<Vec<(String, Value)>> {
11227        let mut props = vec![(key_field.to_string(), key_value.clone())];
11228        if let Some(ns) = named_ns {
11229            props.push((NS_PROP.to_string(), ns.clone()));
11230            return Ok(props);
11231        }
11232        if let Some(ns) = self.merge_create_stamp_ns()? {
11233            props.push((NS_PROP.to_string(), Value::Str(ns)));
11234        }
11235        Ok(props)
11236    }
11237
11238    /// The namespace a role-scoped MERGE create stamps when the pattern does
11239    /// not name `ns`. `None` = unscoped / full authority, so the node lands in
11240    /// `default`.
11241    fn merge_create_stamp_ns(&self) -> Result<Option<String>> {
11242        let Some(authz) = self.pending_write_authz.as_ref() else {
11243            return Ok(None);
11244        };
11245        let Some(def) = self.role_def_for(&authz.role) else {
11246            return Ok(None);
11247        };
11248        match def.namespaces.as_deref() {
11249            Some([only]) => Ok(Some(only.clone())),
11250            Some(_) => Err(GraphError::RoleWriteDenied {
11251                reason: MERGE_CREATE_NEEDS_ONE_NAMESPACE.to_string(),
11252            }),
11253            None => Ok(None),
11254        }
11255    }
11256
11257    fn exec_merge(
11258        &mut self,
11259        stmt: core_query::cypher::MergeStmt,
11260        params: &BTreeMap<String, Value>,
11261    ) -> Result<ResultSet> {
11262        // MERGE: check if a node with the given key already exists.
11263        let key = match &stmt.key_value {
11264            Value::Str(s) => s.clone(),
11265            _ => {
11266                return Err(GraphError::QueryError {
11267                    detail: format!(
11268                        "MERGE key value must be a string (got {:?})",
11269                        stmt.key_value
11270                    ),
11271                })
11272            }
11273        };
11274
11275        if let Some(var) = stmt.var.as_deref() {
11276            for sc in stmt.on_create.iter().chain(&stmt.on_match) {
11277                if sc.var != var {
11278                    return Err(GraphError::QueryError {
11279                        detail: format!(
11280                            "SET variable '{}' does not match MERGE variable '{var}'",
11281                            sc.var
11282                        ),
11283                    });
11284                }
11285            }
11286        }
11287
11288        // ── MERGE authz pre-check (when role-scoped) ─────────────────────────
11289        //
11290        // MERGE scope precondition: check create OR update scope for the
11291        // declared label BEFORE calling `has_node` (timing-oracle closure,
11292        // spec §6.2 "MERGE visibility oracle" item: hidden ≡ absent for
11293        // unscoped roles — the scope denial fires without touching the key store).
11294        //
11295        // Clone to avoid holding a borrow on `self.pending_write_authz` while
11296        // also calling `self.ids.get(key)`.
11297        let merge_existed: bool = if let Some(authz) = self.pending_write_authz.clone() {
11298            let has_create = authz.scope.create_labels.contains(&stmt.label);
11299            let has_update = authz.scope.update_labels.contains(&stmt.label);
11300            if !has_create && !has_update {
11301                // Scope-before-lookup: 403 without has_node call (timing oracle
11302                // closure — see test_merge_unscoped_no_key_lookup).
11303                return Err(GraphError::RoleWriteDenied {
11304                    reason: format!(
11305                        "role-bound token: label '{}' not in write scope (create_labels)",
11306                        stmt.label
11307                    ),
11308                });
11309            }
11310            // Key lookup under mask.
11311            match self.ids.get(key.as_str()) {
11312                Some(id) if authz.mask.contains_id(id) => {
11313                    // Visible: must have update scope to proceed to match arm.
11314                    if !has_update {
11315                        return Err(GraphError::RoleWriteDenied {
11316                            reason: format!(
11317                                "role-bound token: label '{}' not in write scope (update_labels)",
11318                                stmt.label
11319                            ),
11320                        });
11321                    }
11322                    true // existed = true → match arm
11323                }
11324                Some(_) => {
11325                    // Hidden: same error as absent to the role (spec §3.1/§3.3).
11326                    return Err(GraphError::RoleWriteDenied {
11327                        reason: "role-bound token: target node not visible".into(),
11328                    });
11329                }
11330                None => {
11331                    // Absent: must have create scope to proceed to the create arm.
11332                    //
11333                    // Update-only roles (create_labels empty, update_labels set):
11334                    // return the SAME "not visible" error as the hidden-key branch
11335                    // so hidden ≡ absent — no distinguishing oracle (spec §6.1
11336                    // "confirm existence of hidden nodes: No").
11337                    //
11338                    // Create-scoped roles (has_create=true): absent → create arm
11339                    // as before.  The accepted structural key-existence disclosure
11340                    // (§THREAT-MODEL) applies only when the role holds create scope.
11341                    if !has_create {
11342                        return Err(GraphError::RoleWriteDenied {
11343                            reason: "role-bound token: target node not visible".into(),
11344                        });
11345                    }
11346                    false // existed = false → create arm
11347                }
11348            }
11349        } else {
11350            // Full authority: use the existing non-masked has_node check.
11351            self.has_node(&key)
11352        };
11353
11354        let existed = merge_existed;
11355        let create_props = if existed {
11356            None
11357        } else {
11358            Some(self.merge_create_props(&stmt.key_field, &stmt.key_value, stmt.ns.as_ref())?)
11359        };
11360        let mut created = 0i64;
11361        if create_props.is_some() || !stmt.on_match.is_empty() {
11362            let mut batch = self.batch();
11363            if let Some(props) = create_props {
11364                batch.insert_node(&stmt.label, &key, props);
11365                for sc in &stmt.on_create {
11366                    let value = resolve_merge_set_value(&sc.value, params)?;
11367                    batch.set_prop(&key, &sc.field, value);
11368                }
11369                created = 1;
11370            } else {
11371                for sc in &stmt.on_match {
11372                    let value = resolve_merge_set_value(&sc.value, params)?;
11373                    batch.set_prop(&key, &sc.field, value);
11374                }
11375            }
11376            batch.commit()?;
11377        }
11378
11379        // Refresh the role mask so the just-created node is visible to this
11380        // statement's RETURN (read-after-write). Safe: create_labels ⊆ read labels
11381        // (apply_schema subset rule), so the new node's label is already in the
11382        // role's read scope — this never widens beyond the role's declared labels.
11383        if !existed {
11384            if let Some(role) = self.pending_write_authz.as_ref().map(|a| a.role.clone()) {
11385                let new_mask = self.mask_for_role(&role)?;
11386                if let Some(a) = self.pending_write_authz.as_mut() {
11387                    a.mask = new_mask;
11388                }
11389            }
11390        }
11391
11392        // Optional RETURN clause: project the node (created or matched) as a read result.
11393        if let Some(returns) = stmt.returns {
11394            let var = stmt.var.as_deref().unwrap_or("_mn0");
11395            let q = Query {
11396                matches: vec![Pattern {
11397                    start: NodePat {
11398                        var: Some(var.to_string()),
11399                        label: Some(stmt.label.clone()),
11400                        props: vec![("id".to_string(), Operand::Lit(stmt.key_value.clone()))],
11401                    },
11402                    chain: vec![],
11403                    shortest: false,
11404                }],
11405                optional_clauses: vec![],
11406                where_expr: None,
11407                unwinds: vec![],
11408                post_unwind_where: None,
11409                stages: vec![],
11410                returns,
11411                distinct: false,
11412                order_by: vec![],
11413                skip: None,
11414                limit: None,
11415            };
11416            let ops = plan(&q).map_err(|e| GraphError::QueryError {
11417                detail: format!("plan: {e}"),
11418            })?;
11419            // Use view_masked when a role-scoped write is in flight so the
11420            // post-merge projection is consistent with the masked read phase.
11421            let mask_opt = self.pending_write_authz.as_ref().map(|a| a.mask.clone());
11422            return (if let Some(ref mask) = mask_opt {
11423                execute(&self.view_masked(mask), &ops, &Params(params))
11424            } else {
11425                execute(&self.view(), &ops, &Params(params))
11426            })
11427            .map_err(|e| GraphError::QueryError {
11428                detail: format!("execute: {e}"),
11429            });
11430        }
11431
11432        let mut rs = write_result_set();
11433        rs.push_row(vec![
11434            Some(Value::Int(created)),
11435            Some(Value::Int(0)),
11436            Some(Value::Int(0)),
11437        ]);
11438        Ok(rs)
11439    }
11440
11441    /// Return all rule-owned edges between `key_a` and `key_b` (either direction),
11442    /// annotated with rule name, edge type, direction, and weight.
11443    /// Results are sorted by (rule, edge_type).
11444    /// Returns `Err(KeyNotFound)` if either key is unknown.
11445    pub fn explain(&self, key_a: &str, key_b: &str) -> Result<Vec<Explanation>> {
11446        self.ensure_v8_base_sections_loaded();
11447        let id_a = self
11448            .ids
11449            .get(key_a)
11450            .ok_or_else(|| GraphError::KeyNotFound { key: key_a.into() })?;
11451        let id_b = self
11452            .ids
11453            .get(key_b)
11454            .ok_or_else(|| GraphError::KeyNotFound { key: key_b.into() })?;
11455
11456        let mut results = Vec::new();
11457
11458        // Walk the smaller incident set so explain is O(min(deg(a), deg(b)))
11459        // rather than O(total provenance).
11460        let scan = if self.engine.provenance_touching_len(id_a)
11461            <= self.engine.provenance_touching_len(id_b)
11462        {
11463            id_a
11464        } else {
11465            id_b
11466        };
11467        for (rule_name, etype, src, dst) in self.engine.provenance_touching(scan) {
11468            if !((src == id_a && dst == id_b) || (src == id_b && dst == id_a)) {
11469                continue;
11470            }
11471            let Some(rule_def) = self.engine.rules().find(|r| r.name == rule_name) else {
11472                continue;
11473            };
11474            let edge_type = match self.syms.resolve(etype) {
11475                Some(s) => s.to_string(),
11476                None => continue,
11477            };
11478            // Provenance (src, dst) ids come from the archived PROVENANCE section
11479            // (large, no eager CRC).  A corrupt section can produce ids that are
11480            // out of range; return Corrupt rather than panic.
11481            let src_key = self
11482                .ids
11483                .key_of(src)
11484                .ok_or_else(|| GraphError::Corrupt {
11485                    detail: format!("v8: provenance src id {src} not in id table"),
11486                })?
11487                .to_string();
11488            let dst_key = self
11489                .ids
11490                .key_of(dst)
11491                .ok_or_else(|| GraphError::Corrupt {
11492                    detail: format!("v8: provenance dst id {dst} not in id table"),
11493                })?
11494                .to_string();
11495            let stored = rule_def.weight_prop.as_deref().and_then(|prop| {
11496                self.edge_props_view()
11497                    .get(etype, src, dst, prop)
11498                    .and_then(|v| {
11499                        if let Value::Float(f) = v {
11500                            Some(f)
11501                        } else {
11502                            None
11503                        }
11504                    })
11505            });
11506            // Rules that store no weight (KeyMatch/FieldEqual defaults, auto-FK)
11507            // still have a score: recompute it from the predicate so explain
11508            // never reports "no score" for an edge the engine scored.  Via-hop
11509            // rules score over their via set, not over (src, dst), so leave
11510            // those None rather than report a number the rule did not produce.
11511            let weight = stored.or_else(|| {
11512                if rule_def.via_edge.is_some() {
11513                    return None;
11514                }
11515                let props_view = build_props_view(&self.props, &self.base);
11516                let src_get = |field: &str| props_view.get(src, field).map(|vr| vr.into_value());
11517                let dst_get = |field: &str| props_view.get(dst, field).map(|vr| vr.into_value());
11518                let src_view = NodeView {
11519                    key: &src_key,
11520                    props: &src_get,
11521                };
11522                let dst_view = NodeView {
11523                    key: &dst_key,
11524                    props: &dst_get,
11525                };
11526                evaluate(&rule_def.predicate, &src_view, &dst_view)
11527            });
11528            results.push(Explanation {
11529                rule: rule_name.to_string(),
11530                edge_type,
11531                src_key,
11532                dst_key,
11533                weight,
11534                predicate: PredicateSummary {
11535                    approximate: rule_def.approximate,
11536                    ..PredicateSummary::from(&rule_def.predicate)
11537                },
11538                via_edge: rule_def.via_edge.clone(),
11539            });
11540        }
11541
11542        results.sort_by(|a, b| a.rule.cmp(&b.rule).then(a.edge_type.cmp(&b.edge_type)));
11543        Ok(results)
11544    }
11545
11546    /// [`explain`](Self::explain) with the scoped read contract (§5.3): both
11547    /// endpoints are subject-checked, and any explanation whose evidence runs
11548    /// through a hidden node is **dropped entirely, not redacted**.
11549    ///
11550    /// A plain two-node rule's evidence is the pair itself, so once both
11551    /// subjects are visible there is nothing left to hide. A **via-hop** rule is
11552    /// different: it fires `src → dst` because some node carrying `via_label`
11553    /// sits between them, and [`Explanation`] carries the hop's edge *type*
11554    /// (`via_edge`) and never the hop's key. There is no field to blank, so a
11555    /// redacted explanation would still say "these two are linked through
11556    /// something you cannot see" — which discloses that the something exists.
11557    /// The explanation is therefore kept only when at least one **visible** via
11558    /// node satisfies the rule on its own.
11559    ///
11560    /// The weight is the **visible corpus's** number, not the store's: a via-hop
11561    /// rule stores the max over every via it hopped through, so the stored value
11562    /// can be a score only a hidden via produced. It is recomputed over the
11563    /// visible vias alone.
11564    ///
11565    /// Hidden or unknown `key_a` or `key_b` → [`GraphError::KeyNotFound`].
11566    pub fn explain_scoped(
11567        &self,
11568        key_a: &str,
11569        key_b: &str,
11570        mask: &crate::mask::NodeMask,
11571    ) -> Result<Vec<Explanation>> {
11572        for key in [key_a, key_b] {
11573            if !mask.contains_node(self, key) {
11574                return Err(GraphError::KeyNotFound { key: key.into() });
11575            }
11576        }
11577        Ok(self
11578            .explain(key_a, key_b)?
11579            .into_iter()
11580            .filter_map(|e| self.scoped_explanation(e, mask))
11581            .collect())
11582    }
11583
11584    /// `e` as a caller limited to `mask` may have it, or `None` when it must be
11585    /// dropped entirely.
11586    ///
11587    /// Every non-via-hop explanation passes through untouched: its only nodes
11588    /// are the two subjects, which [`explain_scoped`](Self::explain_scoped) has
11589    /// already checked, and its weight is scored over that pair alone.
11590    ///
11591    /// A via-hop explanation is kept only when some via node the caller may see
11592    /// satisfies the rule on its own — and then its weight is recomputed as the
11593    /// max over exactly those vias. The engine writes the max over **all** of
11594    /// them (`core-rules::engine`, `best = prev.max(score)`), so passing the
11595    /// stored number through would let a hidden node set a figure the caller
11596    /// reads: the same disclosure dropping the explanation exists to prevent.
11597    ///
11598    /// A rule that stores no weight still reports none. The recomputed score is
11599    /// a sanitised version of a number `explain` already returned, never a new
11600    /// one — a scoped read must not say more than the unscoped read it narrows.
11601    fn scoped_explanation(
11602        &self,
11603        e: Explanation,
11604        mask: &crate::mask::NodeMask,
11605    ) -> Option<Explanation> {
11606        let Some(via_edge) = e.via_edge.clone() else {
11607            return Some(e);
11608        };
11609        let Some(rule_def) = self.engine.rules().find(|r| r.name == e.rule) else {
11610            // The rule is gone but its provenance is not; nothing can vouch for
11611            // the hop, so nothing is shown.
11612            return None;
11613        };
11614        let Some(via_label) = rule_def.via_label.as_deref() else {
11615            return Some(e);
11616        };
11617        let (Some(src), Some(dst)) = (self.ids.get(&e.src_key), self.ids.get(&e.dst_key)) else {
11618            return None;
11619        };
11620        let (Some(via_etype), Some(via_sym)) = (self.syms.get(&via_edge), self.syms.get(via_label))
11621        else {
11622            return None;
11623        };
11624        let via_dir = rule_def.via_dir.unwrap_or(Direction::Out);
11625        let props_view = build_props_view(&self.props, &self.base);
11626        let dst_get = |field: &str| props_view.get(dst, field).map(|vr| vr.into_value());
11627        let dst_view = NodeView {
11628            key: &e.dst_key,
11629            props: &dst_get,
11630        };
11631        // The rule's own namespace test, the one the engine applies to each via
11632        // candidate (`core-rules::engine::rule_sees_node`). Without it a
11633        // visible, out-of-namespace via — right label, satisfying predicate —
11634        // vouches for a hop the engine never made, and an explanation whose real
11635        // evidence is a hidden in-namespace node is kept.
11636        let rule_sees = |id: u32| match rule_def.namespace.as_deref() {
11637            None => true,
11638            Some(ns) => {
11639                let value = props_view.get(id, NS_PROP).map(|vr| vr.into_value());
11640                namespace_of_value(value.as_ref()) == ns
11641            }
11642        };
11643        let best = self
11644            .topo_view()
11645            .neighbors(via_etype, via_dir, src)
11646            .iter()
11647            .copied()
11648            .filter_map(|via| {
11649                if !mask.contains_id(via) {
11650                    return None;
11651                }
11652                if self.labels.get(via as usize).copied() != Some(via_sym) {
11653                    return None;
11654                }
11655                if !rule_sees(via) {
11656                    return None;
11657                }
11658                let via_key = self.ids.key_of(via)?;
11659                let via_get = |field: &str| props_view.get(via, field).map(|vr| vr.into_value());
11660                let via_view = NodeView {
11661                    key: via_key,
11662                    props: &via_get,
11663                };
11664                evaluate(&rule_def.predicate, &via_view, &dst_view)
11665            })
11666            .fold(None::<f64>, |best, score| {
11667                Some(match best {
11668                    None => score,
11669                    Some(prev) => prev.max(score),
11670                })
11671            })?;
11672        let weight = e.weight.map(|_| best);
11673        Some(Explanation { weight, ..e })
11674    }
11675
11676    pub fn neighbors(&self, key: &str, edge_type: &str, dir: Direction) -> Result<Vec<String>> {
11677        let id = self
11678            .ids
11679            .get(key)
11680            .ok_or_else(|| GraphError::KeyNotFound { key: key.into() })?;
11681        let Some(sym) = self.syms.get(edge_type) else {
11682            return Ok(Vec::new());
11683        };
11684        self.topo_view()
11685            .neighbors(sym, dir, id)
11686            .iter()
11687            .map(|&n| {
11688                self.ids
11689                    .key_of(n)
11690                    .map(|k| k.to_string())
11691                    .ok_or_else(|| GraphError::Corrupt {
11692                        detail: format!("topology id {n} has no key"),
11693                    })
11694            })
11695            .collect::<Result<Vec<_>>>()
11696    }
11697
11698    /// Unique directed degree of `key`. Unknown key → [`GraphError::KeyNotFound`].
11699    /// Unknown `edge_type` → 0. [`crate::algo::AlgoDir::Both`] is out + in (sum).
11700    pub fn degree(
11701        &self,
11702        key: &str,
11703        edge_type: Option<&str>,
11704        direction: crate::algo::AlgoDir,
11705    ) -> Result<u64> {
11706        let id = self
11707            .ids
11708            .get(key)
11709            .ok_or_else(|| GraphError::KeyNotFound { key: key.into() })?;
11710        let topo = self.topo_view();
11711        Ok(Self::unique_directed_degree(
11712            &topo, &self.syms, id, edge_type, direction,
11713        ))
11714    }
11715
11716    /// [`degree`](Self::degree) summing each pair's **insert count** instead of
11717    /// counting each pair once (§5.13).
11718    ///
11719    /// The unique degree asks how many neighbours there are; this asks how many
11720    /// times they were inserted. A pair with no recorded count contributes 1,
11721    /// so on a store that never called
11722    /// [`enable_multiplicity`](Self::enable_multiplicity) this returns exactly
11723    /// what [`degree`](Self::degree) returns rather than erroring — the
11724    /// distinction is a readout preference, not a demand the store cannot meet.
11725    ///
11726    /// `AlgoDir::Both` still sums out + in, so a pair visible on both sides
11727    /// still contributes twice: multiplicity changes what a pair is worth, never
11728    /// how a direction is counted.
11729    pub fn degree_multiplicity(
11730        &self,
11731        key: &str,
11732        edge_type: Option<&str>,
11733        direction: crate::algo::AlgoDir,
11734    ) -> Result<u64> {
11735        let id = self
11736            .ids
11737            .get(key)
11738            .ok_or_else(|| GraphError::KeyNotFound { key: key.into() })?;
11739        Ok(Self::multiplicity_directed_degree(
11740            &self.topo_view(),
11741            &self.edge_props_view(),
11742            &self.syms,
11743            id,
11744            edge_type,
11745            direction,
11746            None,
11747        ))
11748    }
11749
11750    /// [`degree_multiplicity`](Self::degree_multiplicity) under a scope.
11751    ///
11752    /// The sum covers **visible pairs only**. A hidden neighbour's inserts stay
11753    /// out of it for the reason
11754    /// [`degree_scoped`](Self::degree_scoped) documents, and more sharply: an
11755    /// unscoped multiplicity count discloses not only that a hidden neighbour
11756    /// exists but how often it was written.
11757    ///
11758    /// Hidden or unknown `key` → [`GraphError::KeyNotFound`].
11759    pub fn degree_scoped_multiplicity(
11760        &self,
11761        key: &str,
11762        edge_type: Option<&str>,
11763        direction: crate::algo::AlgoDir,
11764        mask: &crate::mask::NodeMask,
11765    ) -> Result<u64> {
11766        let id = self
11767            .ids
11768            .get(key)
11769            .filter(|&id| mask.contains_id(id))
11770            .ok_or_else(|| GraphError::KeyNotFound { key: key.into() })?;
11771        Ok(Self::multiplicity_directed_degree(
11772            &self.topo_view(),
11773            &self.edge_props_view(),
11774            &self.syms,
11775            id,
11776            edge_type,
11777            direction,
11778            Some(mask),
11779        ))
11780    }
11781
11782    /// [`degree`](Self::degree) counting **only neighbours the mask admits**.
11783    ///
11784    /// The filter is a correctness requirement, not an optimisation: an
11785    /// unfiltered count discloses the existence of a hidden neighbour to a
11786    /// caller who cannot see it, which is the same leak
11787    /// [`node_edges_scoped`](Self::node_edges_scoped) exists to prevent —
11788    /// reached by arithmetic instead of by name.
11789    ///
11790    /// Hidden or unknown `key` → [`GraphError::KeyNotFound`]. Unknown
11791    /// `edge_type` is still 0, as it is unscoped.
11792    pub fn degree_scoped(
11793        &self,
11794        key: &str,
11795        edge_type: Option<&str>,
11796        direction: crate::algo::AlgoDir,
11797        mask: &crate::mask::NodeMask,
11798    ) -> Result<u64> {
11799        let id = self
11800            .ids
11801            .get(key)
11802            .filter(|&id| mask.contains_id(id))
11803            .ok_or_else(|| GraphError::KeyNotFound { key: key.into() })?;
11804        let topo = self.topo_view();
11805        Ok(Self::visible_directed_degree(
11806            &topo, &self.syms, id, edge_type, direction, mask,
11807        ))
11808    }
11809
11810    /// Unique directed degree for a subset or a label scan.
11811    ///
11812    /// Unknown keys in `keys` are omitted (mask-like). `keys = Some(&[])` →
11813    /// empty `Ok(vec![])`. `limit` is applied after sorting degree desc, key
11814    /// asc, and only when `Some`. Invalid `where_` → `QueryError`.
11815    #[allow(clippy::too_many_arguments)]
11816    pub fn degrees(
11817        &self,
11818        keys: Option<&[String]>,
11819        label: Option<&str>,
11820        where_: Option<&PropPredicate>,
11821        edge_type: Option<&str>,
11822        direction: crate::algo::AlgoDir,
11823        limit: Option<usize>,
11824    ) -> Result<Vec<(String, u64)>> {
11825        self.degrees_inner(
11826            keys, label, where_, edge_type, direction, limit, None, false,
11827        )
11828    }
11829
11830    /// [`degrees`](Self::degrees) reporting each row's **insert-count** sum
11831    /// instead of its unique neighbour count, as
11832    /// [`degree_multiplicity`](Self::degree_multiplicity) does for one key.
11833    ///
11834    /// The sort is still degree descending, key ascending — over the counts this
11835    /// reading produces — and `limit` still applies after it.
11836    #[allow(clippy::too_many_arguments)]
11837    pub fn degrees_multiplicity(
11838        &self,
11839        keys: Option<&[String]>,
11840        label: Option<&str>,
11841        where_: Option<&PropPredicate>,
11842        edge_type: Option<&str>,
11843        direction: crate::algo::AlgoDir,
11844        limit: Option<usize>,
11845    ) -> Result<Vec<(String, u64)>> {
11846        self.degrees_inner(keys, label, where_, edge_type, direction, limit, None, true)
11847    }
11848
11849    /// [`degrees_scoped`](Self::degrees_scoped) reporting insert counts.
11850    ///
11851    /// Both filters apply: a hidden key stays out of the result, and every
11852    /// row's sum covers its **visible** pairs only.
11853    #[allow(clippy::too_many_arguments)]
11854    pub fn degrees_scoped_multiplicity(
11855        &self,
11856        keys: Option<&[String]>,
11857        label: Option<&str>,
11858        where_: Option<&PropPredicate>,
11859        edge_type: Option<&str>,
11860        direction: crate::algo::AlgoDir,
11861        limit: Option<usize>,
11862        mask: &crate::mask::NodeMask,
11863    ) -> Result<Vec<(String, u64)>> {
11864        self.degrees_inner(
11865            keys,
11866            label,
11867            where_,
11868            edge_type,
11869            direction,
11870            limit,
11871            Some(mask),
11872            true,
11873        )
11874    }
11875
11876    /// [`degrees`](Self::degrees) with the scope applied on both sides: a hidden
11877    /// key is omitted from the input — whether it arrived in `keys` or came out
11878    /// of the `label`/`where_` scan — and every row's count is the count of its
11879    /// **visible** neighbours, for the reason
11880    /// [`degree_scoped`](Self::degree_scoped) documents.
11881    ///
11882    /// Unlike `degree_scoped`, a hidden key here is not
11883    /// [`GraphError::KeyNotFound`]: `degrees` already drops unknown keys
11884    /// silently, so hidden and absent stay one answer by staying out of the
11885    /// result. `limit` still applies after the sort, and so counts visible rows.
11886    #[allow(clippy::too_many_arguments)]
11887    pub fn degrees_scoped(
11888        &self,
11889        keys: Option<&[String]>,
11890        label: Option<&str>,
11891        where_: Option<&PropPredicate>,
11892        edge_type: Option<&str>,
11893        direction: crate::algo::AlgoDir,
11894        limit: Option<usize>,
11895        mask: &crate::mask::NodeMask,
11896    ) -> Result<Vec<(String, u64)>> {
11897        self.degrees_inner(
11898            keys,
11899            label,
11900            where_,
11901            edge_type,
11902            direction,
11903            limit,
11904            Some(mask),
11905            false,
11906        )
11907    }
11908
11909    /// The body shared by [`degrees`](Self::degrees) and
11910    /// [`degrees_scoped`](Self::degrees_scoped). `mask = None` is the unscoped
11911    /// contract unchanged.
11912    #[allow(clippy::too_many_arguments)]
11913    fn degrees_inner(
11914        &self,
11915        keys: Option<&[String]>,
11916        label: Option<&str>,
11917        where_: Option<&PropPredicate>,
11918        edge_type: Option<&str>,
11919        direction: crate::algo::AlgoDir,
11920        limit: Option<usize>,
11921        mask: Option<&crate::mask::NodeMask>,
11922        multiplicity: bool,
11923    ) -> Result<Vec<(String, u64)>> {
11924        if let Some(pred) = where_ {
11925            pred.validate_named("where")
11926                .map_err(|detail| GraphError::QueryError { detail })?;
11927        }
11928        if matches!(keys, Some(ks) if ks.is_empty()) {
11929            return Ok(Vec::new());
11930        }
11931        let view = self.view();
11932        let ids: Vec<u32> = match keys {
11933            Some(ks) => {
11934                let mut seen = HashSet::new();
11935                let mut out = Vec::new();
11936                for k in ks {
11937                    let Some(id) = view.ids.get(k) else {
11938                        continue;
11939                    };
11940                    if !seen.insert(id) {
11941                        continue;
11942                    }
11943                    if let Some(pred) = where_ {
11944                        let holds = match view.prop(id, &pred.field) {
11945                            None => pred.holds(None),
11946                            Some(vr) => pred.holds(Some(vr.as_value())),
11947                        };
11948                        if !holds {
11949                            continue;
11950                        }
11951                    }
11952                    out.push(id);
11953                }
11954                out
11955            }
11956            None => Self::vector_candidates(&view, label, where_),
11957        };
11958        let mut out: Vec<(String, u64)> = ids
11959            .into_iter()
11960            // A hidden candidate leaves as quietly as an unknown key does.
11961            .filter(|&id| mask.is_none_or(|m| m.contains_id(id)))
11962            .filter_map(|id| {
11963                let key = self.ids.key_of(id)?.to_string();
11964                let deg = match (multiplicity, mask) {
11965                    // The same `view` the unique arms read, so the per-row
11966                    // rebuild F9 measured is gone and all three arms agree on
11967                    // the state they are reading.
11968                    (true, m) => Self::multiplicity_directed_degree(
11969                        &view.topo,
11970                        &view.edge_props,
11971                        view.syms,
11972                        id,
11973                        edge_type,
11974                        direction,
11975                        m,
11976                    ),
11977                    (false, Some(m)) => Self::visible_directed_degree(
11978                        &view.topo, view.syms, id, edge_type, direction, m,
11979                    ),
11980                    (false, None) => Self::unique_directed_degree(
11981                        &view.topo, view.syms, id, edge_type, direction,
11982                    ),
11983                };
11984                Some((key, deg))
11985            })
11986            .collect();
11987        out.sort_by(|a, b| b.1.cmp(&a.1).then_with(|| a.0.cmp(&b.0)));
11988        if let Some(lim) = limit {
11989            out.truncate(lim);
11990        }
11991        Ok(out)
11992    }
11993
11994    /// Unique neighbour count for `id` across `edge_type` (or all types) and
11995    /// `direction`. Unknown `edge_type` → 0. `Both` sums out + in.
11996    fn unique_directed_degree(
11997        topo: &TopologyView<'_>,
11998        syms: &Interner,
11999        id: u32,
12000        edge_type: Option<&str>,
12001        direction: crate::algo::AlgoDir,
12002    ) -> u64 {
12003        let dirs: &[Direction] = match direction {
12004            crate::algo::AlgoDir::Out => &[Direction::Out],
12005            crate::algo::AlgoDir::In => &[Direction::In],
12006            crate::algo::AlgoDir::Both => &[Direction::Out, Direction::In],
12007        };
12008        match edge_type {
12009            Some(name) => {
12010                let Some(et) = syms.get(name) else {
12011                    return 0;
12012                };
12013                dirs.iter().map(|&d| topo.degree(et, d, id) as u64).sum()
12014            }
12015            None => topo
12016                .etypes()
12017                .map(|et| {
12018                    dirs.iter()
12019                        .map(|&d| topo.degree(et, d, id) as u64)
12020                        .sum::<u64>()
12021                })
12022                .sum(),
12023        }
12024    }
12025
12026    /// [`unique_directed_degree`](Self::unique_directed_degree) counting only
12027    /// neighbours `mask` admits.
12028    ///
12029    /// Same shape, one substitution: `topo.degree` is a length, so it cannot be
12030    /// filtered; the neighbour list it measures can. `Both` still sums out + in,
12031    /// so a node visible on both sides still counts twice — the filter changes
12032    /// which neighbours are counted, never how a degree is defined.
12033    fn visible_directed_degree(
12034        topo: &TopologyView<'_>,
12035        syms: &Interner,
12036        id: u32,
12037        edge_type: Option<&str>,
12038        direction: crate::algo::AlgoDir,
12039        mask: &crate::mask::NodeMask,
12040    ) -> u64 {
12041        let dirs: &[Direction] = match direction {
12042            crate::algo::AlgoDir::Out => &[Direction::Out],
12043            crate::algo::AlgoDir::In => &[Direction::In],
12044            crate::algo::AlgoDir::Both => &[Direction::Out, Direction::In],
12045        };
12046        let visible = |et: u32| -> u64 {
12047            dirs.iter()
12048                .map(|&d| {
12049                    topo.neighbors(et, d, id)
12050                        .iter()
12051                        .filter(|&&n| mask.contains_id(n))
12052                        .count() as u64
12053                })
12054                .sum()
12055        };
12056        match edge_type {
12057            Some(name) => syms.get(name).map_or(0, visible),
12058            None => topo.etypes().map(visible).sum(),
12059        }
12060    }
12061
12062    /// Sum of the insert counts of `id`'s pairs (§5.13), over `edge_type` (or
12063    /// all types) and `direction`, restricted to what `mask` admits when one is
12064    /// given.
12065    ///
12066    /// The same neighbour lists the unique reading measures, with each entry
12067    /// worth its pair's count rather than worth 1 — so the filter decides which
12068    /// pairs are in the sum and the count decides what each contributes. A
12069    /// direction decides which way round the pair is addressed: an `In`
12070    /// neighbour `n` of `id` is the pair `(et, n, id)`.
12071    ///
12072    /// Takes its views as parameters, exactly as the unique helpers do, because
12073    /// it is called once per row from a label scan. `edge_props_view()` reaches
12074    /// into the mmap'd base's rkyv section on every call, so building the two
12075    /// views inside made an N-row `degrees(multiplicity=True)` do N section
12076    /// accesses where the unique reading does one: worth 2.57 ms of 16.68 ms
12077    /// over 20 000 rows, about 0.13 us per row (defect #29,
12078    /// `tests/f9_bench.rs`). Most of that call's cost is the per-neighbour
12079    /// count lookup and is inherent, so this is a hoist, not a rescue.
12080    ///
12081    /// The views are exactly `self.view()`'s own `topo` and `edge_props`, so a
12082    /// caller that already has a view passes its halves and reads the same
12083    /// state it reads everything else from.
12084    fn multiplicity_directed_degree(
12085        topo: &TopologyView<'_>,
12086        edge_props: &EdgePropsView<'_>,
12087        syms: &Interner,
12088        id: u32,
12089        edge_type: Option<&str>,
12090        direction: crate::algo::AlgoDir,
12091        mask: Option<&crate::mask::NodeMask>,
12092    ) -> u64 {
12093        let dirs: &[Direction] = match direction {
12094            crate::algo::AlgoDir::Out => &[Direction::Out],
12095            crate::algo::AlgoDir::In => &[Direction::In],
12096            crate::algo::AlgoDir::Both => &[Direction::Out, Direction::In],
12097        };
12098        let count_of = |et: u32, src: u32, dst: u32| -> u64 {
12099            match edge_props.get(et, src, dst, EDGE_COUNT_PROP) {
12100                Some(Value::Int(n)) if n > 0 => n as u64,
12101                _ => 1,
12102            }
12103        };
12104        let per_etype = |et: u32| -> u64 {
12105            dirs.iter()
12106                .map(|&d| {
12107                    topo.neighbors(et, d, id)
12108                        .iter()
12109                        .filter(|&&n| mask.is_none_or(|m| m.contains_id(n)))
12110                        .map(|&n| match d {
12111                            Direction::Out => count_of(et, id, n),
12112                            Direction::In => count_of(et, n, id),
12113                        })
12114                        .sum::<u64>()
12115                })
12116                .sum()
12117        };
12118        match edge_type {
12119            Some(name) => syms.get(name).map_or(0, per_etype),
12120            None => topo.etypes().map(per_etype).sum(),
12121        }
12122    }
12123
12124    /// Return the last-change commit sequence for `key`, or `None` if the node
12125    /// does not exist or has never been mutated since the last V5-V7 snapshot
12126    /// (horizon-bounded for legacy stores).
12127    ///
12128    /// The returned sequence is a monotonically increasing counter that starts
12129    /// at 1 for the first commit after `open` and increments with every
12130    /// successful write.  WAL replay at open also assigns sequences (1..N for N
12131    /// replayed frames), so sequences are consistent across snapshot+WAL cycles.
12132    ///
12133    /// For V5-V7 stores opened without a V8 snapshot, nodes that were present
12134    /// in the snapshot but not touched by any WAL frame will return `None`
12135    /// (horizon-bounded: CAS against such nodes is only safe after the first
12136    /// V8 snapshot or after the node is next mutated).
12137    pub fn last_changed(&self, key: &str) -> Option<u64> {
12138        let id = self.ids.get(key)?;
12139        self.last_change.get(&id).copied()
12140    }
12141
12142    /// Which loaded store this handle is.
12143    ///
12144    /// Paired with [`commit_seq`](GraphDb::commit_seq) it identifies a graph
12145    /// state outright, which `commit_seq` alone does not: two stores of the same
12146    /// age share a sequence, and a reload can return to one. Memos of dense
12147    /// node ids that the handle does not own are stamped with both; see
12148    /// [`StoreStamp`](crate::mask::StoreStamp).
12149    pub(crate) fn store_id(&self) -> crate::mask::StoreId {
12150        self.store_id
12151    }
12152
12153    /// The current commit sequence (number of successful commits since open,
12154    /// including WAL replay frames).  Useful for recording a baseline before
12155    /// a read-modify-write cycle.
12156    pub fn commit_seq(&self) -> u64 {
12157        self.commit_seq
12158    }
12159
12160    /// Check that all `preconds` are satisfied against the current db state.
12161    /// Returns `Err(GraphError::CasConflict)` on the first failing precondition.
12162    pub(crate) fn check_preconditions(&self, preconds: &[Precondition]) -> Result<()> {
12163        for precond in preconds {
12164            match precond {
12165                Precondition::NodeUnchangedSince { key, expected } => {
12166                    // Missing entry means the node predates the WAL window or
12167                    // does not exist; treat as 0 (before any commit).
12168                    let actual = self.last_changed(key).unwrap_or_default();
12169                    if actual != *expected {
12170                        return Err(GraphError::CasConflict {
12171                            key: key.clone(),
12172                            expected: *expected,
12173                            actual,
12174                        });
12175                    }
12176                }
12177                Precondition::NodeAbsent { key } => {
12178                    // Node must not exist (not live).
12179                    if self.ids.get(key).is_some() {
12180                        let actual = self.last_changed(key).unwrap_or(0);
12181                        return Err(GraphError::CasConflict {
12182                            key: key.clone(),
12183                            expected: u64::MAX,
12184                            actual,
12185                        });
12186                    }
12187                }
12188            }
12189        }
12190        Ok(())
12191    }
12192
12193    /// Apply a batch of mutations with compare-and-set preconditions.
12194    ///
12195    /// All preconditions are checked atomically before any operation is applied.
12196    /// If any precondition fails, the entire batch is rejected with
12197    /// [`GraphError::CasConflict`] and no WAL frame is written.
12198    ///
12199    /// # Returns
12200    /// `(nodes_inserted, edges_inserted)` on success, same as [`write_batch`].
12201    ///
12202    /// # Errors
12203    /// - [`GraphError::CasConflict`] if any precondition is not satisfied.
12204    /// - Any error that [`write_batch`] would return for the ops themselves.
12205    pub fn write_batch_cas(
12206        &mut self,
12207        preconds: Vec<Precondition>,
12208        ops: Vec<BatchOp>,
12209    ) -> Result<(usize, usize)> {
12210        self.check_preconditions(&preconds)?;
12211        self.commit_logged_batch(ops, None, None).map(inserted_pair)
12212    }
12213
12214    /// Update the per-node last-change map for a WAL record at commit `seq`.
12215    ///
12216    /// Called after a successful apply to record which nodes were touched.
12217    /// For replay, called with the WAL-frame's replayed seq.
12218    ///
12219    /// Touch definition (see [`Precondition`] doc):
12220    /// - InsertNode / InsertNodeId / SetProp / SetPropId / RemoveProp → the node.
12221    /// - InsertEdge / InsertEdgeId / DeleteEdge → both src and dst.
12222    /// - DeleteNode → node tombstoned; last_changed() returns None so no update needed.
12223    /// - DerivedEdge markers, Intern, rule/view records → no-ops.
12224    /// - Batch → recurse into inner records.
12225    fn update_last_change_from_rec(&mut self, rec: &WalRecord, seq: u64) {
12226        match rec {
12227            WalRecord::InsertNode { key, .. }
12228            | WalRecord::SetProp { key, .. }
12229            | WalRecord::RemoveProp { key, .. } => {
12230                if let Some(id) = self.ids.get(key) {
12231                    self.last_change.insert(id, seq);
12232                }
12233            }
12234            WalRecord::InsertNodeId { key, .. } => {
12235                if let Some(id) = self.ids.get(key) {
12236                    self.last_change.insert(id, seq);
12237                }
12238            }
12239            WalRecord::SetPropId { id, .. } => {
12240                self.last_change.insert(*id, seq);
12241            }
12242            WalRecord::InsertEdge {
12243                src_key, dst_key, ..
12244            }
12245            | WalRecord::DeleteEdge {
12246                src_key, dst_key, ..
12247            } => {
12248                if let Some(src_id) = self.ids.get(src_key) {
12249                    self.last_change.insert(src_id, seq);
12250                }
12251                if let Some(dst_id) = self.ids.get(dst_key) {
12252                    self.last_change.insert(dst_id, seq);
12253                }
12254            }
12255            WalRecord::InsertEdgeId { src, dst, .. } => {
12256                self.last_change.insert(*src, seq);
12257                self.last_change.insert(*dst, seq);
12258            }
12259            // A count record touches the pair, so it touches both endpoints —
12260            // the same reading `InsertEdgeId` gets, because a duplicate insert
12261            // that raises the count *is* a mutation of that pair. The opt-in
12262            // declaration touches nothing.
12263            WalRecord::SetEdgeCount { src, dst, .. } if !rec.is_multiplicity_decl() => {
12264                self.last_change.insert(*src, seq);
12265                self.last_change.insert(*dst, seq);
12266            }
12267            WalRecord::SetEdgeCount { .. } => {}
12268            // DeleteNode: node is tombstoned; last_changed(key) returns None for
12269            // deleted keys (ids.get() returns None post-tombstone), so no update needed.
12270            // History markers: state no-ops; the underlying mutation already
12271            // touched the relevant nodes' last_change entries.
12272            WalRecord::DeleteNode { .. }
12273            | WalRecord::DerivedEdgeAdded { .. }
12274            | WalRecord::DerivedEdgeRetracted { .. }
12275            | WalRecord::Intern { .. }
12276            | WalRecord::CreateRule { .. }
12277            | WalRecord::DeleteRule { .. }
12278            | WalRecord::RebuildRule { .. }
12279            | WalRecord::CreateView { .. }
12280            | WalRecord::DeleteView { .. }
12281            | WalRecord::EnableFulltext { .. }
12282            | WalRecord::DisableFulltext { .. }
12283            | WalRecord::EnableIndex { .. }
12284            | WalRecord::DisableIndex { .. } => {}
12285            // RenameNode: node id is stable; update last_change via the new key.
12286            // Called after apply(), so ids already reflects new_key.
12287            WalRecord::RenameNode { new_key, .. } => {
12288                if let Some(id) = self.ids.get(new_key) {
12289                    self.last_change.insert(id, seq);
12290                }
12291            }
12292            WalRecord::Batch(inner) => {
12293                for inner_rec in inner {
12294                    self.update_last_change_from_rec(inner_rec, seq);
12295                }
12296            }
12297        }
12298    }
12299
12300    pub fn node_count(&self) -> usize {
12301        self.ids.len()
12302    }
12303
12304    /// Configure archive retention: keep the `N` newest WAL archives at each
12305    /// [`snapshot_with`] call when `archive_wal: true`.
12306    ///
12307    /// `Some(N)` where N > 0 → prune oldest archives keeping the newest N.
12308    /// `Some(0)` or `None` → unlimited (no pruning).
12309    ///
12310    /// Pruning only ever happens inside [`snapshot_with`]; this method only
12311    /// stores the policy.  Archives below the retention limit are deleted
12312    /// oldest-first.  The horizon floor is updated so that
12313    /// [`was_linked`] / history APIs return `CommitOutOfRange` for commits
12314    /// in pruned archives rather than silently returning wrong data.
12315    pub fn set_wal_archive_retention(&mut self, keep: Option<u32>) {
12316        self.wal_archive_retention = keep;
12317    }
12318
12319    /// Delete any WAL archives that are fully below the current horizon floor.
12320    ///
12321    /// Orphaned archives arise when the floor is written first during retention
12322    /// pruning and then a crash interrupts the archive-delete sequence.  The
12323    /// opening cleanup ensures no subsequent read path sees stale data.
12324    ///
12325    /// Under the monotonic naming scheme, the archive name N equals the
12326    /// cumulative end-frame index of the archive in global commit space (i.e.
12327    /// the archive covers global frames `[prev_n, N)`).  An archive is
12328    /// fully orphaned when `N <= wal_horizon_floor`: all of its frames fall
12329    /// below the floor and have already been counted in it.
12330    fn cleanup_orphaned_archives(&mut self) -> Result<()> {
12331        if self.wal_horizon_floor == 0 {
12332            // Floor at 0 means no pruning has ever occurred; nothing to clean.
12333            return Ok(());
12334        }
12335        let archive_ns = self.fs.list_archives()?;
12336        for n in archive_ns {
12337            if n <= self.wal_horizon_floor {
12338                // Archive N ends at global frame N; all its frames are below
12339                // the floor (floor already accounts for them) → orphaned.
12340                self.fs.delete_archive(n).map_err(GraphError::Io)?;
12341            } else {
12342                // Archives are sorted ascending; first one above floor stops scan.
12343                break;
12344            }
12345        }
12346        Ok(())
12347    }
12348
12349    /// Collect all WAL frames from surviving archives (oldest-first) then the
12350    /// live WAL into one flat list, and return the total along with the number
12351    /// of archive frames at the front of the list.
12352    ///
12353    /// Commit indices into the returned list are LOCAL (0 = first frame of
12354    /// oldest surviving archive).  To obtain the GLOBAL index add
12355    /// `self.wal_horizon_floor`.
12356    /// How many frames the surviving archives hold, without materialising them.
12357    ///
12358    /// The same count `all_frames` puts at the front of its list. Used to seed
12359    /// [`wal_frames_written`](GraphDb::wal_frames_written) at open without
12360    /// decoding the live WAL a second time; free on a store with no archives,
12361    /// which is most of them.
12362    fn archive_frame_count(&self) -> Result<u64> {
12363        let mut n = 0u64;
12364        for a in self.fs.list_archives()? {
12365            let bytes = self.fs.read_archive(a)?;
12366            let (frames, _) = decode_all(&bytes);
12367            n += frames.len() as u64;
12368        }
12369        Ok(n)
12370    }
12371
12372    fn all_frames(&self) -> Result<(Vec<WalRecord>, u64)> {
12373        let archive_ns = self.fs.list_archives()?;
12374        let mut all: Vec<WalRecord> = Vec::new();
12375        for n in archive_ns {
12376            let bytes = self.fs.read_archive(n)?;
12377            let (frames, _) = decode_all(&bytes);
12378            all.extend(frames);
12379        }
12380        let archive_count = all.len() as u64;
12381        let live_bytes = self.fs.read(FileId::Wal)?;
12382        let (live_frames, _) = decode_all(&live_bytes);
12383        all.extend(live_frames);
12384        Ok((all, archive_count))
12385    }
12386
12387    /// Return the total number of committed WAL frames visible in the current
12388    /// horizon window, including frames in surviving WAL archives.
12389    ///
12390    /// This is the exclusive upper bound for valid `at_commit` indices in
12391    /// `was_linked`. Valid indices are `wal_horizon_floor()..wal_total_commits()`.
12392    ///
12393    /// Returns the horizon floor when all surviving history is empty.
12394    pub fn wal_total_commits(&self) -> Result<u64> {
12395        let (frames, _) = self.all_frames()?;
12396        Ok(self.wal_horizon_floor + frames.len() as u64)
12397    }
12398
12399    /// The global frame index of the first commit reachable through surviving
12400    /// archives (0 when no archives have been pruned).
12401    pub fn wal_horizon_floor(&self) -> u64 {
12402        self.wal_horizon_floor
12403    }
12404
12405    /// Return the per-node change history for `key` by scanning the on-disk WAL.
12406    ///
12407    /// ## Horizon
12408    ///
12409    /// History reaches back only to the last WAL-truncating snapshot, exactly like `open_at`.
12410    /// Snapshots written with `keep_wal: true` preserve deeper history. This is the honest,
12411    /// zero-cost contract; a durable history log is out of scope.
12412    ///
12413    /// ## Derived edges
12414    ///
12415    /// Rule-created (derived) edges are **not** in the WAL and therefore do not appear in
12416    /// history. Only edges written directly by the application are recorded.
12417    ///
12418    /// ## Deleted nodes
12419    ///
12420    /// For nodes that have been deleted, dense-id records (SetPropId, InsertEdgeId) that
12421    /// predate the deletion may not resolve (the id is tombstoned in the live map). The
12422    /// string-keyed `DeleteNode` record still matches and produces a `NodeDeleted` entry.
12423    /// Prop/edge history of a deleted node may therefore be partially unresolvable.
12424    ///
12425    /// ## Dense-id edge entries and tombstoned partners
12426    ///
12427    /// Edge entries from dense-id WAL records (`InsertEdgeId`) are omitted when the partner
12428    /// endpoint's dense id is tombstoned. As a result, a live node's history can contain an
12429    /// `EdgeRemoved` (string-keyed, always resolves) without a corresponding `EdgeAdded`.
12430    /// Build commit-bounded alias intervals for `queried_key`.
12431    ///
12432    /// Returns a list of `(key, valid_from_inclusive, valid_until_exclusive)` tuples.
12433    /// A record written under `key` at commit `c` matches the queried identity iff
12434    /// `c >= valid_from && (valid_until.is_none() || c < valid_until)`.
12435    ///
12436    /// Each alias entry carries both a lower and an upper bound so that key-reuse
12437    /// after a rename is handled correctly: if "a" is renamed to "b" at commit 5,
12438    /// then a NEW node is created as "a" at commit 7 and renamed to "c" at commit 10,
12439    /// querying "c" must NOT surface identity-1's events (commits 0–4 under "a");
12440    /// only identity-2's events (commits 7–9 under "a") are in scope.
12441    ///
12442    /// Only **forward aliasing**: querying the *new* key surfaces events written
12443    /// under the *old* key.  The reverse direction is not supported.
12444    fn build_key_alias_intervals(
12445        &self,
12446        frames: &[core_storage::wal::WalRecord],
12447        queried_key: &str,
12448    ) -> Vec<(String, u64, Option<u64>)> {
12449        use core_storage::wal::WalRecord;
12450
12451        // Pre-pass: build reverse_rename and key_starts maps.
12452        let mut reverse_rename: HashMap<String, (String, u64)> = HashMap::new();
12453        let mut key_starts: HashMap<String, Vec<u64>> = HashMap::new();
12454
12455        for (local_i, frame) in frames.iter().enumerate() {
12456            let commit = self.wal_horizon_floor + local_i as u64;
12457            let records: &[WalRecord] = match frame {
12458                WalRecord::Batch(inner) => inner.as_slice(),
12459                single => std::slice::from_ref(single),
12460            };
12461            for rec in records {
12462                match rec {
12463                    WalRecord::InsertNode { key, .. } | WalRecord::InsertNodeId { key, .. } => {
12464                        key_starts.entry(key.clone()).or_default().push(commit);
12465                    }
12466                    WalRecord::RenameNode { old_key, new_key } => {
12467                        // new_key came into existence at this commit.
12468                        key_starts.entry(new_key.clone()).or_default().push(commit);
12469                        // Record the reverse rename: new_key was introduced by renaming old_key.
12470                        reverse_rename.insert(new_key.clone(), (old_key.clone(), commit));
12471                    }
12472                    _ => {}
12473                }
12474            }
12475        }
12476
12477        // Build alias intervals by following the reverse rename chain.
12478        let mut result: Vec<(String, u64, Option<u64>)> = Vec::new();
12479        let mut current_key = queried_key.to_string();
12480        let mut current_valid_until: Option<u64> = None;
12481
12482        loop {
12483            // valid_from: the most recent commit where current_key was assigned to this
12484            // identity.  For aliases (valid_until = Some(vu)), find the last start event
12485            // for the key strictly before vu — this is where the alias's occupancy by
12486            // this identity began, correctly excluding prior identities that reused the key.
12487            let valid_from = if let Some(vu) = current_valid_until {
12488                key_starts
12489                    .get(&current_key)
12490                    .and_then(|starts| starts.iter().rev().find(|&&s| s < vu).copied())
12491                    .unwrap_or(self.wal_horizon_floor)
12492            } else {
12493                // Queried key — no upper bound; may have been introduced at any commit.
12494                self.wal_horizon_floor
12495            };
12496
12497            result.push((current_key.clone(), valid_from, current_valid_until));
12498
12499            match reverse_rename.get(&current_key) {
12500                Some((old_key, rename_commit)) => {
12501                    current_valid_until = Some(*rename_commit);
12502                    current_key = old_key.clone();
12503                }
12504                None => break,
12505            }
12506        }
12507
12508        result
12509    }
12510
12511    /// Returns true if `record_key` matches any alias interval that covers `commit`.
12512    fn aliases_match(
12513        intervals: &[(String, u64, Option<u64>)],
12514        record_key: &str,
12515        commit: u64,
12516    ) -> bool {
12517        intervals
12518            .iter()
12519            .any(|(k, vf, vu)| k == record_key && commit >= *vf && vu.is_none_or(|u| commit < u))
12520    }
12521
12522    /// Return the change history of node `key` by scanning the on-disk WAL.
12523    ///
12524    /// ## Horizon
12525    ///
12526    /// History reaches back only as far as the retained WAL. The returned
12527    /// [`HistoryResult`](crate::history::HistoryResult) carries `total_commits`
12528    /// (the exclusive upper bound for valid commit indices) and `horizon` (the
12529    /// oldest commit still reachable). When `horizon > 0`, older events were
12530    /// pruned and are not in `items`.
12531    pub fn node_history(
12532        &self,
12533        key: &str,
12534    ) -> Result<crate::history::HistoryResult<crate::history::HistoryEntry>> {
12535        use crate::history::{HistoryChange, HistoryEntry, HistoryResult};
12536        use core_storage::wal::WalRecord;
12537
12538        let (frames, _) = self.all_frames()?;
12539        let total_commits = self.wal_horizon_floor + frames.len() as u64;
12540
12541        // Resolve commit-bounded alias intervals for `key` (handles renames in the WAL).
12542        let alias_intervals = self.build_key_alias_intervals(&frames, key);
12543
12544        let mut out: Vec<HistoryEntry> = Vec::new();
12545
12546        for (local_i, frame) in frames.iter().enumerate() {
12547            let commit = self.wal_horizon_floor + local_i as u64;
12548            // Collect the inner records to process — Batch is one commit, single records are one commit.
12549            let records: &[WalRecord] = match frame {
12550                WalRecord::Batch(inner) => inner.as_slice(),
12551                single => std::slice::from_ref(single),
12552            };
12553
12554            for rec in records {
12555                let change = match rec {
12556                    WalRecord::InsertNode { label, key: k, .. }
12557                        if Self::aliases_match(&alias_intervals, k, commit) =>
12558                    {
12559                        Some(HistoryChange::NodeInserted {
12560                            label: label.clone(),
12561                        })
12562                    }
12563                    WalRecord::InsertNodeId { label, key: k, .. }
12564                        if Self::aliases_match(&alias_intervals, k, commit) =>
12565                    {
12566                        let label_str = match self.syms.resolve(*label) {
12567                            Some(s) => s.to_string(),
12568                            None => continue,
12569                        };
12570                        Some(HistoryChange::NodeInserted { label: label_str })
12571                    }
12572                    WalRecord::SetProp {
12573                        key: k,
12574                        field,
12575                        value,
12576                    } if Self::aliases_match(&alias_intervals, k, commit) => {
12577                        Some(HistoryChange::PropSet {
12578                            field: field.clone(),
12579                            value: value.clone(),
12580                        })
12581                    }
12582                    WalRecord::SetPropId { id, field, value } => {
12583                        // Use key_of_historical (not key_of) so a node's prop_set
12584                        // events remain visible after the node is later deleted:
12585                        // key_of returns None for a tombstoned id, which would
12586                        // silently drop every PropSet between insert and delete.
12587                        // Mirrors the InsertEdgeId arm below and edge_history's
12588                        // own id-keyed arms.
12589                        match self.ids.key_of_historical(*id) {
12590                            // key_of_historical returns the last-known (possibly
12591                            // post-rename, possibly post-delete) key; compare to queried key.
12592                            Some(resolved) if resolved == key => {
12593                                let field_str = match self.syms.resolve(*field) {
12594                                    Some(s) => s.to_string(),
12595                                    None => continue,
12596                                };
12597                                Some(HistoryChange::PropSet {
12598                                    field: field_str,
12599                                    value: value.clone(),
12600                                })
12601                            }
12602                            _ => None,
12603                        }
12604                    }
12605                    WalRecord::RemoveProp { key: k, field }
12606                        if Self::aliases_match(&alias_intervals, k, commit) =>
12607                    {
12608                        Some(HistoryChange::PropRemoved {
12609                            field: field.clone(),
12610                        })
12611                    }
12612                    WalRecord::InsertEdge {
12613                        edge_type,
12614                        src_key,
12615                        dst_key,
12616                    } => {
12617                        if Self::aliases_match(&alias_intervals, src_key, commit) {
12618                            Some(HistoryChange::EdgeAdded {
12619                                edge_type: edge_type.clone(),
12620                                other: dst_key.clone(),
12621                                outgoing: true,
12622                            })
12623                        } else if Self::aliases_match(&alias_intervals, dst_key, commit) {
12624                            Some(HistoryChange::EdgeAdded {
12625                                edge_type: edge_type.clone(),
12626                                other: src_key.clone(),
12627                                outgoing: false,
12628                            })
12629                        } else {
12630                            None
12631                        }
12632                    }
12633                    WalRecord::InsertEdgeId { etype, src, dst } => {
12634                        let etype_str = match self.syms.resolve(*etype) {
12635                            Some(s) => s.to_string(),
12636                            None => continue,
12637                        };
12638                        // key_of_historical (not key_of): an edge added before
12639                        // either endpoint was later deleted must still resolve —
12640                        // see the SetPropId arm above and edge_history's
12641                        // InsertEdgeId arm, which use the same lookup for the
12642                        // same reason.
12643                        let src_key = self.ids.key_of_historical(*src);
12644                        let dst_key = self.ids.key_of_historical(*dst);
12645                        if src_key == Some(key) {
12646                            let other = match dst_key {
12647                                Some(s) => s.to_string(),
12648                                None => continue,
12649                            };
12650                            Some(HistoryChange::EdgeAdded {
12651                                edge_type: etype_str,
12652                                other,
12653                                outgoing: true,
12654                            })
12655                        } else if dst_key == Some(key) {
12656                            let other = match src_key {
12657                                Some(s) => s.to_string(),
12658                                None => continue,
12659                            };
12660                            Some(HistoryChange::EdgeAdded {
12661                                edge_type: etype_str,
12662                                other,
12663                                outgoing: false,
12664                            })
12665                        } else {
12666                            None
12667                        }
12668                    }
12669                    WalRecord::DeleteEdge {
12670                        edge_type,
12671                        src_key,
12672                        dst_key,
12673                    } => {
12674                        if Self::aliases_match(&alias_intervals, src_key, commit) {
12675                            Some(HistoryChange::EdgeRemoved {
12676                                edge_type: edge_type.clone(),
12677                                other: dst_key.clone(),
12678                                outgoing: true,
12679                            })
12680                        } else if Self::aliases_match(&alias_intervals, dst_key, commit) {
12681                            Some(HistoryChange::EdgeRemoved {
12682                                edge_type: edge_type.clone(),
12683                                other: src_key.clone(),
12684                                outgoing: false,
12685                            })
12686                        } else {
12687                            None
12688                        }
12689                    }
12690                    WalRecord::DeleteNode { key: k }
12691                        if Self::aliases_match(&alias_intervals, k, commit) =>
12692                    {
12693                        Some(HistoryChange::NodeDeleted)
12694                    }
12695                    // Skip: rule/view/fulltext/intern metadata; Batch wrapper handled above.
12696                    _ => None,
12697                };
12698
12699                if let Some(change) = change {
12700                    out.push(HistoryEntry { commit, change });
12701                }
12702            }
12703        }
12704
12705        Ok(HistoryResult {
12706            items: out,
12707            total_commits,
12708            horizon: self.wal_horizon_floor,
12709        })
12710    }
12711
12712    /// Return the per-edge change history between nodes `a` and `b` by scanning
12713    /// the on-disk WAL.
12714    ///
12715    /// ## Horizon
12716    ///
12717    /// History reaches back only to the last WAL-truncating snapshot, exactly
12718    /// like `node_history` and `open_at`. The returned [`HistoryResult`] carries
12719    /// `total_commits` (= number of WAL frames), which is the exclusive upper
12720    /// bound for valid commit indices.
12721    ///
12722    /// ## Derived edges
12723    ///
12724    /// Rule-derived edges appear via `DerivedEdgeAdded` / `DerivedEdgeRetracted`
12725    /// WAL markers written by `log_then_apply_with` after each rule-firing
12726    /// mutation. The `rule` field of those events carries the rule name.
12727    ///
12728    /// ## DeleteNode
12729    ///
12730    /// When a node is deleted, its manual incident edges are swept inline without
12731    /// individual `DeleteEdge` WAL records. `edge_history` detects `DeleteNode`
12732    /// events for either endpoint and synthesises `Retracted(rule:None)` events
12733    /// for each manual edge that was active at that point. Derived edges active at
12734    /// the time of deletion are handled by the `DerivedEdgeRetracted` marker that
12735    /// the engine appends immediately after the `DeleteNode` record; those events
12736    /// carry correct rule attribution and are emitted by the marker arm, not the
12737    /// synthetic sweep.
12738    ///
12739    /// ## Masks
12740    ///
12741    /// Like `node_history`, this method has no mask parameter and returns WAL
12742    /// history regardless of any role mask. For masked history semantics, apply
12743    /// the mask at the caller level.
12744    pub fn edge_history(
12745        &self,
12746        a: &str,
12747        b: &str,
12748    ) -> Result<crate::history::HistoryResult<crate::history::EdgeHistoryEvent>> {
12749        use crate::history::{EdgeEvent, EdgeHistoryEvent, HistoryResult};
12750        use core_storage::wal::WalRecord;
12751
12752        let (frames, _) = self.all_frames()?;
12753        let total_commits = self.wal_horizon_floor + frames.len() as u64;
12754
12755        // Resolve all historical names for a and b (handles RenameNode in the WAL).
12756        // Intervals are commit-bounded so recycled keys don't contaminate histories.
12757        let alias_a = self.build_key_alias_intervals(&frames, a);
12758        let alias_b = self.build_key_alias_intervals(&frames, b);
12759
12760        // Active edges between a and b tracked as (edge_type, src_key, dst_key, is_derived).
12761        // The is_derived flag is used by the DeleteNode sweep: manual edges are
12762        // swept with a synthetic Retracted(rule:None); derived edges are skipped
12763        // because the engine writes a DerivedEdgeRetracted marker immediately after
12764        // the DeleteNode record, which carries the correct rule attribution.
12765        let mut active: Vec<(String, String, String, bool)> = Vec::new();
12766        let mut out: Vec<EdgeHistoryEvent> = Vec::new();
12767
12768        for (local_i, frame) in frames.iter().enumerate() {
12769            let commit = self.wal_horizon_floor + local_i as u64;
12770            let records: &[WalRecord] = match frame {
12771                WalRecord::Batch(inner) => inner.as_slice(),
12772                single => std::slice::from_ref(single),
12773            };
12774
12775            for rec in records {
12776                match rec {
12777                    WalRecord::InsertEdge {
12778                        edge_type,
12779                        src_key,
12780                        dst_key,
12781                    } => {
12782                        let is_ab = Self::aliases_match(&alias_a, src_key, commit)
12783                            && Self::aliases_match(&alias_b, dst_key, commit);
12784                        let is_ba = Self::aliases_match(&alias_b, src_key, commit)
12785                            && Self::aliases_match(&alias_a, dst_key, commit);
12786                        if is_ab || is_ba {
12787                            active.push((
12788                                edge_type.clone(),
12789                                src_key.clone(),
12790                                dst_key.clone(),
12791                                false,
12792                            ));
12793                            out.push(EdgeHistoryEvent {
12794                                edge_type: edge_type.clone(),
12795                                commit,
12796                                event: EdgeEvent::Added,
12797                                rule: None,
12798                            });
12799                        }
12800                    }
12801                    WalRecord::InsertEdgeId { etype, src, dst } => {
12802                        let etype_str = match self.syms.resolve(*etype) {
12803                            Some(s) => s.to_string(),
12804                            None => continue,
12805                        };
12806                        // Use key_of_historical so tombstoned nodes (deleted
12807                        // later in the WAL) still resolve during the scan.
12808                        let src_key = self.ids.key_of_historical(*src);
12809                        let dst_key = self.ids.key_of_historical(*dst);
12810                        let is_ab = src_key == Some(a) && dst_key == Some(b);
12811                        let is_ba = src_key == Some(b) && dst_key == Some(a);
12812                        if is_ab || is_ba {
12813                            let src_str = src_key.unwrap().to_string();
12814                            let dst_str = dst_key.unwrap().to_string();
12815                            active.push((etype_str.clone(), src_str, dst_str, false));
12816                            out.push(EdgeHistoryEvent {
12817                                edge_type: etype_str,
12818                                commit,
12819                                event: EdgeEvent::Added,
12820                                rule: None,
12821                            });
12822                        }
12823                    }
12824                    WalRecord::DeleteEdge {
12825                        edge_type,
12826                        src_key,
12827                        dst_key,
12828                    } => {
12829                        let is_ab = Self::aliases_match(&alias_a, src_key, commit)
12830                            && Self::aliases_match(&alias_b, dst_key, commit);
12831                        let is_ba = Self::aliases_match(&alias_b, src_key, commit)
12832                            && Self::aliases_match(&alias_a, dst_key, commit);
12833                        if is_ab || is_ba {
12834                            // Remove the first matching active entry (flag ignored).
12835                            if let Some(pos) = active.iter().position(|(et, s, d, _)| {
12836                                et == edge_type && s == src_key && d == dst_key
12837                            }) {
12838                                active.remove(pos);
12839                            }
12840                            out.push(EdgeHistoryEvent {
12841                                edge_type: edge_type.clone(),
12842                                commit,
12843                                event: EdgeEvent::Retracted,
12844                                rule: None,
12845                            });
12846                        }
12847                    }
12848                    WalRecord::DeleteNode { key: k }
12849                        if Self::aliases_match(&alias_a, k, commit)
12850                            || Self::aliases_match(&alias_b, k, commit) =>
12851                    {
12852                        // Sweep: implicitly retract only MANUAL active edges.
12853                        // Derived active edges are skipped here because the rule
12854                        // engine appends a DerivedEdgeRetracted marker immediately
12855                        // after this DeleteNode record; that marker produces the
12856                        // single correctly-attributed Retracted event.  Derived
12857                        // entries are dropped from `active` (the marker arm's
12858                        // idempotent retain finds nothing to remove).
12859                        for (et, _, _, is_derived) in active.drain(..) {
12860                            if !is_derived {
12861                                out.push(EdgeHistoryEvent {
12862                                    edge_type: et,
12863                                    commit,
12864                                    event: EdgeEvent::Retracted,
12865                                    rule: None,
12866                                });
12867                            }
12868                            // Derived: drop silently; marker carries the Retracted event.
12869                        }
12870                    }
12871                    WalRecord::DerivedEdgeAdded {
12872                        rule,
12873                        edge_type: et,
12874                        src_key,
12875                        dst_key,
12876                    } => {
12877                        let is_ab = Self::aliases_match(&alias_a, src_key, commit)
12878                            && Self::aliases_match(&alias_b, dst_key, commit);
12879                        let is_ba = Self::aliases_match(&alias_b, src_key, commit)
12880                            && Self::aliases_match(&alias_a, dst_key, commit);
12881                        if is_ab || is_ba {
12882                            active.push((et.clone(), src_key.clone(), dst_key.clone(), true));
12883                            out.push(EdgeHistoryEvent {
12884                                edge_type: et.clone(),
12885                                commit,
12886                                event: EdgeEvent::Added,
12887                                rule: Some(rule.clone()),
12888                            });
12889                        }
12890                    }
12891                    WalRecord::DerivedEdgeRetracted {
12892                        rule,
12893                        edge_type: et,
12894                        src_key,
12895                        dst_key,
12896                    } => {
12897                        let is_ab = Self::aliases_match(&alias_a, src_key, commit)
12898                            && Self::aliases_match(&alias_b, dst_key, commit);
12899                        let is_ba = Self::aliases_match(&alias_b, src_key, commit)
12900                            && Self::aliases_match(&alias_a, dst_key, commit);
12901                        if is_ab || is_ba {
12902                            // Push unconditionally: a derived edge whose Added marker
12903                            // predates the history horizon has no `active` entry, but
12904                            // the retraction is still a real in-window event.
12905                            // Remove from active idempotently if present.
12906                            active.retain(|(aet, s, d, _)| {
12907                                !(aet == et && s == src_key && d == dst_key)
12908                            });
12909                            out.push(EdgeHistoryEvent {
12910                                edge_type: et.clone(),
12911                                commit,
12912                                event: EdgeEvent::Retracted,
12913                                rule: Some(rule.clone()),
12914                            });
12915                        }
12916                    }
12917                    // All other records (InsertNode, SetProp, CreateRule, etc.)
12918                    // do not affect edges between a and b.
12919                    _ => {}
12920                }
12921            }
12922        }
12923
12924        Ok(HistoryResult {
12925            items: out,
12926            total_commits,
12927            horizon: self.wal_horizon_floor,
12928        })
12929    }
12930
12931    /// Return `true` iff an edge of `edge_type` existed between `a` and `b`
12932    /// (in either direction) at the WAL commit `at_commit`.
12933    ///
12934    /// ## Horizon
12935    ///
12936    /// Valid commit indices are `0..total_commits` where `total_commits` is the
12937    /// number of WAL frames. An `at_commit >= total_commits` is outside the
12938    /// visible horizon and returns [`GraphError::CommitOutOfRange`].
12939    ///
12940    /// ## Derived edges
12941    ///
12942    /// Rule-derived edges are tracked via `DerivedEdgeAdded` / `DerivedEdgeRetracted`
12943    /// WAL markers appended at firing time (Task 1). `was_linked` reads these markers
12944    /// and therefore includes derived edges in its point-in-time evaluation,
12945    /// matching `edge_history`'s fidelity.
12946    pub fn was_linked(&self, a: &str, b: &str, edge_type: &str, at_commit: u64) -> Result<bool> {
12947        use core_storage::wal::WalRecord;
12948
12949        let (frames, _) = self.all_frames()?;
12950        let total_commits = self.wal_horizon_floor + frames.len() as u64;
12951
12952        // Horizon floor: commits in pruned archives are unreachable.
12953        if at_commit < self.wal_horizon_floor {
12954            return Err(GraphError::CommitOutOfRange {
12955                commit: at_commit,
12956                total: total_commits,
12957                floor: self.wal_horizon_floor,
12958            });
12959        }
12960        if at_commit >= total_commits {
12961            return Err(GraphError::CommitOutOfRange {
12962                commit: at_commit,
12963                total: total_commits,
12964                floor: self.wal_horizon_floor,
12965            });
12966        }
12967
12968        // Resolve all historical names for a and b (handles RenameNode in the WAL).
12969        // Intervals are commit-bounded so recycled keys don't contaminate point-in-time reads.
12970        let alias_a = self.build_key_alias_intervals(&frames, a);
12971        let alias_b = self.build_key_alias_intervals(&frames, b);
12972
12973        // Local index into surviving frames (0 = first frame of oldest archive).
12974        let local_commit = at_commit - self.wal_horizon_floor;
12975
12976        // Replay local frames 0..=local_commit, tracking active edges.
12977        let mut active: BTreeSet<(String, String, String)> = BTreeSet::new();
12978
12979        for (local_i, frame) in frames.iter().enumerate().take((local_commit + 1) as usize) {
12980            let commit = self.wal_horizon_floor + local_i as u64;
12981            let records: &[WalRecord] = match frame {
12982                WalRecord::Batch(inner) => inner.as_slice(),
12983                single => std::slice::from_ref(single),
12984            };
12985
12986            for rec in records {
12987                match rec {
12988                    WalRecord::InsertEdge {
12989                        edge_type: et,
12990                        src_key,
12991                        dst_key,
12992                    } => {
12993                        let is_ab = Self::aliases_match(&alias_a, src_key, commit)
12994                            && Self::aliases_match(&alias_b, dst_key, commit);
12995                        let is_ba = Self::aliases_match(&alias_b, src_key, commit)
12996                            && Self::aliases_match(&alias_a, dst_key, commit);
12997                        if is_ab || is_ba {
12998                            active.insert((et.clone(), src_key.clone(), dst_key.clone()));
12999                        }
13000                    }
13001                    WalRecord::InsertEdgeId { etype, src, dst } => {
13002                        let etype_str = match self.syms.resolve(*etype) {
13003                            Some(s) => s.to_string(),
13004                            None => continue,
13005                        };
13006                        // Use key_of_historical so tombstoned nodes resolve.
13007                        let src_key = self.ids.key_of_historical(*src);
13008                        let dst_key = self.ids.key_of_historical(*dst);
13009                        let is_ab = src_key == Some(a) && dst_key == Some(b);
13010                        let is_ba = src_key == Some(b) && dst_key == Some(a);
13011                        if is_ab || is_ba {
13012                            active.insert((
13013                                etype_str,
13014                                src_key.unwrap().to_string(),
13015                                dst_key.unwrap().to_string(),
13016                            ));
13017                        }
13018                    }
13019                    WalRecord::DeleteEdge {
13020                        edge_type: et,
13021                        src_key,
13022                        dst_key,
13023                    } => {
13024                        let is_ab = Self::aliases_match(&alias_a, src_key, commit)
13025                            && Self::aliases_match(&alias_b, dst_key, commit);
13026                        let is_ba = Self::aliases_match(&alias_b, src_key, commit)
13027                            && Self::aliases_match(&alias_a, dst_key, commit);
13028                        if is_ab || is_ba {
13029                            active.remove(&(et.clone(), src_key.clone(), dst_key.clone()));
13030                        }
13031                    }
13032                    WalRecord::DeleteNode { key: k }
13033                        if Self::aliases_match(&alias_a, k, commit)
13034                            || Self::aliases_match(&alias_b, k, commit) =>
13035                    {
13036                        // All edges touching the deleted node are gone.
13037                        active.retain(|(_, s, d)| s != k && d != k);
13038                    }
13039                    WalRecord::DerivedEdgeAdded {
13040                        edge_type: et,
13041                        src_key,
13042                        dst_key,
13043                        ..
13044                    } => {
13045                        let is_ab = Self::aliases_match(&alias_a, src_key, commit)
13046                            && Self::aliases_match(&alias_b, dst_key, commit);
13047                        let is_ba = Self::aliases_match(&alias_b, src_key, commit)
13048                            && Self::aliases_match(&alias_a, dst_key, commit);
13049                        if is_ab || is_ba {
13050                            active.insert((et.clone(), src_key.clone(), dst_key.clone()));
13051                        }
13052                    }
13053                    WalRecord::DerivedEdgeRetracted {
13054                        edge_type: et,
13055                        src_key,
13056                        dst_key,
13057                        ..
13058                    } => {
13059                        let is_ab = Self::aliases_match(&alias_a, src_key, commit)
13060                            && Self::aliases_match(&alias_b, dst_key, commit);
13061                        let is_ba = Self::aliases_match(&alias_b, src_key, commit)
13062                            && Self::aliases_match(&alias_a, dst_key, commit);
13063                        if is_ab || is_ba {
13064                            active.remove(&(et.clone(), src_key.clone(), dst_key.clone()));
13065                        }
13066                    }
13067                    _ => {}
13068                }
13069            }
13070        }
13071
13072        Ok(active.iter().any(|(et, _, _)| et == edge_type))
13073    }
13074
13075    /// Every edge incident to `key` — either endpoint — that existed at WAL
13076    /// commit `commit`, from ONE scan of the WAL.
13077    ///
13078    /// This is the bulk form of [`was_linked`](GraphDb::was_linked): answering
13079    /// "what did K's relationships look like at commit C" with one call instead
13080    /// of one [`edge_history`](GraphDb::edge_history) per candidate partner.
13081    /// The two agree edge for edge.
13082    ///
13083    /// Results are sorted by `(edge_type, src_key, dst_key)`.
13084    ///
13085    /// ## Horizon
13086    ///
13087    /// Valid commit indices are `wal_horizon_floor()..wal_total_commits()`;
13088    /// anything outside is [`GraphError::CommitOutOfRange`], exactly like
13089    /// `was_linked`. An unknown key is not an error — it simply had no edges.
13090    ///
13091    /// ## Derived edges
13092    ///
13093    /// `DerivedEdgeAdded` / `DerivedEdgeRetracted` markers carry rule
13094    /// attribution, so a rule-owned edge comes back with `derived: true` and
13095    /// `rule: Some(name)`.
13096    ///
13097    /// ## Renames
13098    ///
13099    /// `key` is matched through the same commit-bounded alias intervals
13100    /// `edge_history` uses, so querying a node's *current* key surfaces edges
13101    /// written under an earlier name. Endpoint keys in the result are reported
13102    /// under the name the node carries today, so they can be fed straight back
13103    /// into `node_info`, `explain` or another `edges_at`.
13104    ///
13105    /// ## Masks
13106    ///
13107    /// Like `edge_history` and `node_history`, this reads the WAL regardless of
13108    /// any role mask. Apply masking at the caller level.
13109    pub fn edges_at(&self, key: &str, commit: u64) -> Result<Vec<EdgeAt>> {
13110        use core_storage::wal::WalRecord;
13111
13112        let (frames, _) = self.all_frames()?;
13113        let total_commits = self.wal_horizon_floor + frames.len() as u64;
13114
13115        // Horizon floor: commits in pruned archives are unreachable.
13116        if commit < self.wal_horizon_floor || commit >= total_commits {
13117            return Err(GraphError::CommitOutOfRange {
13118                commit,
13119                total: total_commits,
13120                floor: self.wal_horizon_floor,
13121            });
13122        }
13123
13124        // Commit-bounded historical names of `key` (handles RenameNode).
13125        let alias = self.build_key_alias_intervals(&frames, key);
13126
13127        // Forward rename chain, for reporting endpoints under their current
13128        // names: old key → [(commit, new key)] in ascending commit order.
13129        // Built over the whole WAL, not just the prefix up to `commit`, because
13130        // a rename after `commit` still changes what the node is called today.
13131        let mut renames: HashMap<String, Vec<(u64, String)>> = HashMap::new();
13132        for (local_i, frame) in frames.iter().enumerate() {
13133            let c = self.wal_horizon_floor + local_i as u64;
13134            let records: &[WalRecord] = match frame {
13135                WalRecord::Batch(inner) => inner.as_slice(),
13136                single => std::slice::from_ref(single),
13137            };
13138            for rec in records {
13139                if let WalRecord::RenameNode { old_key, new_key } = rec {
13140                    renames
13141                        .entry(old_key.clone())
13142                        .or_default()
13143                        .push((c, new_key.clone()));
13144                }
13145            }
13146        }
13147
13148        // The name a node written as `k` at commit `from` carries today.
13149        // Follows the first rename at or after `from`, then keeps going. The
13150        // iteration cap bounds a rename cycle inside a single batch.
13151        let canon = |k: &str, from: u64| -> String {
13152            if renames.is_empty() {
13153                return k.to_string();
13154            }
13155            let mut cur = k.to_string();
13156            let mut at = from;
13157            for _ in 0..64 {
13158                match renames
13159                    .get(&cur)
13160                    .and_then(|v| v.iter().find(|(c, _)| *c >= at))
13161                {
13162                    Some((c, new)) => {
13163                        at = *c;
13164                        cur = new.clone();
13165                    }
13166                    None => break,
13167                }
13168            }
13169            cur
13170        };
13171
13172        let local_commit = commit - self.wal_horizon_floor;
13173        // (edge_type, src_key, dst_key) → (derived, rule)
13174        let mut active: BTreeMap<(String, String, String), (bool, Option<String>)> =
13175            BTreeMap::new();
13176
13177        for (local_i, frame) in frames.iter().enumerate().take((local_commit + 1) as usize) {
13178            let c = self.wal_horizon_floor + local_i as u64;
13179            let records: &[WalRecord] = match frame {
13180                WalRecord::Batch(inner) => inner.as_slice(),
13181                single => std::slice::from_ref(single),
13182            };
13183
13184            for rec in records {
13185                match rec {
13186                    WalRecord::InsertEdge {
13187                        edge_type,
13188                        src_key,
13189                        dst_key,
13190                    } => {
13191                        if Self::aliases_match(&alias, src_key, c)
13192                            || Self::aliases_match(&alias, dst_key, c)
13193                        {
13194                            active.insert(
13195                                (edge_type.clone(), canon(src_key, c), canon(dst_key, c)),
13196                                (false, None),
13197                            );
13198                        }
13199                    }
13200                    WalRecord::InsertEdgeId { etype, src, dst } => {
13201                        let Some(etype_str) = self.syms.resolve(*etype) else {
13202                            continue;
13203                        };
13204                        // `key_of_historical` resolves tombstoned ids too, and
13205                        // already returns the node's current key — no rename
13206                        // canonicalisation needed on this arm.
13207                        let (Some(src_key), Some(dst_key)) = (
13208                            self.ids.key_of_historical(*src),
13209                            self.ids.key_of_historical(*dst),
13210                        ) else {
13211                            continue;
13212                        };
13213                        if src_key == key || dst_key == key {
13214                            active.insert(
13215                                (
13216                                    etype_str.to_string(),
13217                                    src_key.to_string(),
13218                                    dst_key.to_string(),
13219                                ),
13220                                (false, None),
13221                            );
13222                        }
13223                    }
13224                    WalRecord::DeleteEdge {
13225                        edge_type,
13226                        src_key,
13227                        dst_key,
13228                    } => {
13229                        if Self::aliases_match(&alias, src_key, c)
13230                            || Self::aliases_match(&alias, dst_key, c)
13231                        {
13232                            active.remove(&(
13233                                edge_type.clone(),
13234                                canon(src_key, c),
13235                                canon(dst_key, c),
13236                            ));
13237                        }
13238                    }
13239                    WalRecord::DeleteNode { key: k } => {
13240                        if active.is_empty() {
13241                            continue;
13242                        }
13243                        if Self::aliases_match(&alias, k, c) {
13244                            // Our node is gone; every incident edge goes with it.
13245                            active.clear();
13246                        } else {
13247                            // A partner is gone; its edges to us go with it.
13248                            let ck = canon(k, c);
13249                            active.retain(|(_, s, d), _| *s != ck && *d != ck);
13250                        }
13251                    }
13252                    WalRecord::DerivedEdgeAdded {
13253                        rule,
13254                        edge_type,
13255                        src_key,
13256                        dst_key,
13257                    } => {
13258                        if Self::aliases_match(&alias, src_key, c)
13259                            || Self::aliases_match(&alias, dst_key, c)
13260                        {
13261                            active.insert(
13262                                (edge_type.clone(), canon(src_key, c), canon(dst_key, c)),
13263                                (true, Some(rule.clone())),
13264                            );
13265                        }
13266                    }
13267                    WalRecord::DerivedEdgeRetracted {
13268                        edge_type,
13269                        src_key,
13270                        dst_key,
13271                        ..
13272                    } => {
13273                        if Self::aliases_match(&alias, src_key, c)
13274                            || Self::aliases_match(&alias, dst_key, c)
13275                        {
13276                            active.remove(&(
13277                                edge_type.clone(),
13278                                canon(src_key, c),
13279                                canon(dst_key, c),
13280                            ));
13281                        }
13282                    }
13283                    // InsertNode, SetProp, CreateRule, … do not move edges.
13284                    _ => {}
13285                }
13286            }
13287        }
13288
13289        // BTreeMap iteration is already (edge_type, src, dst) order.
13290        Ok(active
13291            .into_iter()
13292            .map(|((edge_type, src_key, dst_key), (derived, rule))| EdgeAt {
13293                edge_type,
13294                src_key,
13295                dst_key,
13296                derived,
13297                rule,
13298            })
13299            .collect())
13300    }
13301
13302    /// The derived edges that would be retracted and derived if `key.field`
13303    /// were set to `value` — computed WITHOUT writing anything.
13304    ///
13305    /// Nothing is committed and nothing on `self` is mutated: the rule engine's
13306    /// provenance, its candidate indexes, the topology and the property columns
13307    /// are all cloned first, the change is applied to the clone, and the real
13308    /// per-node re-derivation (`RuleEngine::on_node_changed` — the same call
13309    /// `set_prop` makes during apply) runs against it. The derived-edge deltas
13310    /// it emits are the answer, so rule semantics — predicates, top-k,
13311    /// via-hops, chaining, weights — are the engine's, not a re-implementation.
13312    ///
13313    /// Works on a read-only handle.
13314    ///
13315    /// **While a rule's vector index is still building** (`RuleStats::building`)
13316    /// the clone carries no pending-build state, so this reports the edges that
13317    /// rule would derive — which the live store will not derive until its
13318    /// backfill runs. Right about the end state, early about the timing.
13319    ///
13320    /// Returns `Err(KeyNotFound)` for an unknown or tombstoned key and
13321    /// `Err(ViewPropReadOnly)` for a field a view owns — matching
13322    /// [`set_prop`](GraphDb::set_prop)'s validation. A change with no effect
13323    /// (the node already holds `value`, or no rule watches `field`) returns
13324    /// empty lists.
13325    ///
13326    /// ## Cost
13327    ///
13328    /// One clone of the property columns, the topology overlay, the symbol
13329    /// interner, the edge properties and the provenance map, plus one candidate
13330    /// re-index (O(nodes × rules)). That is much cheaper than copying the store
13331    /// directory, but it is not free — this is an interactive "what if", not a
13332    /// hot path.
13333    pub fn what_if_set_prop(&self, key: &str, field: &str, value: Value) -> Result<WhatIf> {
13334        // The engine's provenance, HNSW and IVF state live in the mmap'd base
13335        // until something asks for them. On a store opened cold from a snapshot
13336        // this is the first ask, and without it the clone below starts from an
13337        // empty provenance map: nothing to retract, so `lost` comes back empty.
13338        self.ensure_v8_base_sections_loaded();
13339
13340        let empty = WhatIf {
13341            lost: Vec::new(),
13342            gained: Vec::new(),
13343        };
13344
13345        if let Some(view_name) = self.view_store.view_for_prop(field) {
13346            return Err(GraphError::ViewPropReadOnly {
13347                view_name: view_name.to_string(),
13348            });
13349        }
13350        MutPreview::new(self).check_live_key(key)?;
13351        let id = self
13352            .ids
13353            .get(key)
13354            .ok_or_else(|| GraphError::KeyNotFound { key: key.into() })?;
13355
13356        let rules: Vec<RuleDef> = self.engine.rules().cloned().collect();
13357        if rules.is_empty() {
13358            return Ok(empty);
13359        }
13360
13361        // No rule watches this field → no derivation can change.
13362        if !rules.iter().any(|r| r.watched_fields().contains(field)) {
13363            return Ok(empty);
13364        }
13365
13366        let old_value = build_props_view(&self.props, &self.base)
13367            .get(id, field)
13368            .map(|vr| vr.into_value());
13369        if old_value.as_ref() == Some(&value) {
13370            return Ok(empty);
13371        }
13372
13373        // --- Clone every piece of state the re-derivation writes to. ---
13374        // `what_if` mutates its own throwaway copies and writes nothing, so it
13375        // takes owned clones rather than sharing the `Arc`s. Deref-clone, not
13376        // `Arc::clone`: sharing here would make `make_mut` copy on first touch
13377        // anyway, and an owned local keeps the rest of this function unchanged.
13378        let mut props = (*self.props).clone();
13379        let mut topo = (*self.topo).clone();
13380        let mut syms = (*self.syms).clone();
13381        let mut edge_props = (*self.edge_props).clone();
13382
13383        let mut tripped: BTreeMap<String, bool> = BTreeMap::new();
13384        let mut fires: BTreeMap<String, u64> = BTreeMap::new();
13385        for r in &rules {
13386            tripped.insert(r.name.clone(), self.engine.is_tripped(&r.name));
13387            fires.insert(r.name.clone(), self.engine.fire_count(&r.name));
13388        }
13389        // `provenance()` decodes retained snapshot bytes on first use; the
13390        // engine clone needs the real map, not an empty one.
13391        let provenance = self.engine.provenance().clone();
13392        let mut engine = core_rules::RuleEngine::from_persist(rules, provenance, tripped, fires);
13393
13394        // Build the candidate indexes from the state BEFORE the change, exactly
13395        // as apply() sees them: `on_node_changed` withdraws the node under its
13396        // old value and refiles it under the new one, so the index must not
13397        // already reflect the change.
13398        engine.reindex_all_load_state(
13399            &self.ids,
13400            &syms,
13401            &self.labels,
13402            build_props_view(&self.props, &self.base),
13403            self.engine.export_ivf_state(),
13404            self.engine.export_hnsw_state_passthrough(),
13405        );
13406        engine.set_emit_deltas(true);
13407
13408        // --- Apply the hypothetical change and re-derive. ---
13409        props.set(id, field, value);
13410        {
13411            let mut gm = make_graph_mut(
13412                &self.ids,
13413                &mut syms,
13414                &self.labels,
13415                build_props_view(&props, &self.base),
13416                &mut topo,
13417                &self.base,
13418                &mut edge_props,
13419            );
13420            engine.on_node_changed(id, Some((field, old_value)), &mut gm);
13421        }
13422
13423        let mut lost: BTreeSet<EdgeAt> = BTreeSet::new();
13424        let mut gained: BTreeSet<EdgeAt> = BTreeSet::new();
13425        for d in engine.drain_deltas() {
13426            let edge = EdgeAt {
13427                edge_type: d.edge_type,
13428                src_key: d.src_key,
13429                dst_key: d.dst_key,
13430                derived: true,
13431                rule: Some(d.rule),
13432            };
13433            if d.fired {
13434                gained.insert(edge);
13435            } else {
13436                lost.insert(edge);
13437            }
13438        }
13439        // An edge retracted and re-derived within the same re-derivation (top-k
13440        // churn) is not a change the caller would see.
13441        let churn: Vec<EdgeAt> = lost.intersection(&gained).cloned().collect();
13442        for e in churn {
13443            lost.remove(&e);
13444            gained.remove(&e);
13445        }
13446
13447        Ok(WhatIf {
13448            lost: lost.into_iter().collect(),
13449            gained: gained.into_iter().collect(),
13450        })
13451    }
13452
13453    pub fn edge_count(&self) -> u64 {
13454        self.topo_view().edge_count()
13455    }
13456
13457    /// Live/tombstone/edge counts plus per-rule provenance size, trip latch,
13458    /// and fire counter (includes rebuild evaluations). Rules are sorted by name.
13459    pub fn stats(&self) -> Stats {
13460        self.ensure_v8_base_sections_loaded();
13461        let building = self.engine.builds_in_progress();
13462        let rules: Vec<RuleStats> = self
13463            .engine
13464            .rules()
13465            .map(|r| RuleStats {
13466                name: r.name.clone(),
13467                edges: self
13468                    .engine
13469                    .provenance()
13470                    .get(&r.name)
13471                    .map(|s| s.len() as u64)
13472                    .unwrap_or(0),
13473                tripped: self.engine.is_tripped(&r.name),
13474                fires: self.engine.fire_count(&r.name),
13475                approximate: r.approximate,
13476                building: building.iter().find(|b| b.rule == r.name).cloned(),
13477            })
13478            .collect();
13479        Stats {
13480            nodes_live: self.ids.live_len(),
13481            nodes_tombstoned: self.ids.len() - self.ids.live_len(),
13482            edges: self.topo_view().edge_count(),
13483            rules,
13484            chain_truncations: self.engine.chain_truncations(),
13485            history_floor: self.wal_horizon_floor,
13486            namespaces: self.namespace_stats(),
13487        }
13488    }
13489
13490    /// On-disk size of the WAL file in bytes.
13491    ///
13492    /// Reads file metadata without loading WAL contents.  Returns `Err` for
13493    /// in-memory (`SimFs`) databases where no WAL file exists on disk.
13494    pub fn wal_size_bytes(&self) -> std::io::Result<u64> {
13495        let path = self.fs.wal_path().ok_or_else(|| {
13496            std::io::Error::new(
13497                std::io::ErrorKind::Unsupported,
13498                "wal_path not available for this Fs implementation",
13499            )
13500        })?;
13501        Ok(std::fs::metadata(path)?.len())
13502    }
13503
13504    /// Set the slow-query threshold.  Queries whose execution time equals or
13505    /// exceeds `ms` milliseconds are logged.  Pass `0` to disable.
13506    ///
13507    /// Use this setter in tests — the environment variable
13508    /// `MUSHROOMDB_SLOW_QUERY_MS` is process-global and races parallel test
13509    /// threads.
13510    pub fn set_slow_query_threshold_ms(&mut self, ms: u64) {
13511        self.slow_query_threshold_ms = ms;
13512    }
13513
13514    /// Snapshot of the slow-query ring buffer and lifetime counter.
13515    pub fn slow_query_snapshot(&self) -> SlowQuerySnapshot {
13516        let log = self.slow_queries.lock().unwrap_or_else(|e| e.into_inner());
13517        SlowQuerySnapshot {
13518            threshold_ms: self.slow_query_threshold_ms,
13519            count: log.total,
13520            last: log.entries.iter().cloned().collect(),
13521        }
13522    }
13523
13524    /// Instant the database was opened.  Used by consumers (e.g. `/metrics`)
13525    /// to compute uptime.
13526    pub fn started_at(&self) -> std::time::Instant {
13527        self.started_at
13528    }
13529
13530    /// The on-disk snapshot version a store that has opted in to nothing
13531    /// writes — the **floor**, not the whole answer.
13532    ///
13533    /// It is not "the version this binary writes", and it is not "the version
13534    /// this binary reads". Since v0.6.10 this binary writes 9 **or** 10
13535    /// depending on the store — [`snapshot::version_for`] decides, and a store
13536    /// that has called [`enable_multiplicity`](Self::enable_multiplicity)
13537    /// writes 10 — and it reads 5 through 10. A caller comparing a store's
13538    /// stamp against this value must use `>=`, not `==`, or it will report an
13539    /// opted-in store as needing a migration *down*; `cli::run_migrate` is the
13540    /// worked example.
13541    ///
13542    /// The name is kept for compatibility: it is public API reachable from the
13543    /// CLI and from any embedder, and respelling it would break them for a
13544    /// doc-level clarification.
13545    ///
13546    /// [`snapshot::version_for`]: core_storage::snapshot::version_for
13547    pub fn format_version() -> u16 {
13548        core_storage::snapshot::VERSION
13549    }
13550
13551    /// Test-support: total bytes appended (SimFs only usage).
13552    pub fn fs_total_appended(&self) -> usize
13553    where
13554        F: FsIntrospect,
13555    {
13556        self.fs.total_appended()
13557    }
13558
13559    /// Test-support: successful `Fs::sync` calls (SimFs / counting fs).
13560    pub fn fs_sync_count(&self) -> usize
13561    where
13562        F: FsIntrospect,
13563    {
13564        self.fs.sync_count()
13565    }
13566
13567    /// Consume the db, returning its fs (for crash simulation).
13568    pub fn into_fs(self) -> F {
13569        self.fs
13570    }
13571
13572    pub fn snapshot(&mut self) -> Result<()> {
13573        self.snapshot_with(SnapshotOptions::default())
13574    }
13575
13576    /// Snapshot with explicit options.
13577    ///
13578    /// # `keep_wal`
13579    ///
13580    /// When `keep_wal` is `false` (the default, same as [`snapshot`]):
13581    ///   - The WAL is replaced with a minimal baseline containing one
13582    ///     `EnableFulltext` record per active declaration.  All pre-snapshot
13583    ///     history is discarded; `open_at` can only reach post-snapshot commits.
13584    ///
13585    /// When `keep_wal` is `true`:
13586    ///   - The WAL is left intact.  All pre-snapshot commits remain reachable
13587    ///     via `open_at`.  The existing WAL already contains the original
13588    ///     `EnableFulltext` records, so no baseline re-write is needed; the
13589    ///     recovery guards in `apply()` silently skip any duplicate records on
13590    ///     replay.
13591    ///   - Crash window: a crash after the snapshot write but before the next
13592    ///     WAL write leaves the full pre-snapshot WAL intact.  On reopen the
13593    ///     snapshot is loaded and the WAL replayed idempotently over it — safe
13594    ///     because every `apply()` arm is idempotent when replayed over an
13595    ///     already-current snapshot.
13596    pub fn snapshot_with(&mut self, opts: SnapshotOptions) -> Result<()> {
13597        if self.read_only {
13598            return Err(GraphError::ReadOnly);
13599        }
13600        // A snapshot rewrites `wal.bin` through a tmp+rename, so a peer that is
13601        // appending ends up holding a descriptor on an unlinked inode and loses
13602        // commits it believes durable. Snapshotting therefore requires the
13603        // cross-process write lock, exactly as appending does. Unlike the WAL
13604        // append path this does not go through `log_then_apply_with`, so both
13605        // guards are repeated here.
13606        if self.degraded {
13607            return Err(GraphError::Io(std::io::Error::other(
13608                "database degraded after group-commit fsync failure; reopen required",
13609            )));
13610        }
13611        if self.lock_denied {
13612            return Err(GraphError::Busy { holder: None });
13613        }
13614        // Capture whether snapshot.bin already existed BEFORE this snapshot write.
13615        // Used by the archive path's conservative genesis-chain check: if a prior
13616        // snapshot exists but wal.truncated does not, we cannot distinguish a
13617        // legacy store (may have been truncated in an older code version) from a
13618        // new store that only used keep_wal=true.  Conservative: refuse genesis in
13619        // both cases.  Must be sampled here, before the snapshot write below.
13620        //
13621        // `snapshot_preserved_history` is the one case where the answer is not a
13622        // guess: a snapshot *this handle* took, on a store that had none when it
13623        // opened, and that kept the WAL. The proxy defers to it, because
13624        // otherwise `enable_multiplicity` — whose forced snapshot is exactly
13625        // that — would permanently disqualify the store from a genesis chain it
13626        // is fully entitled to (defect #23).
13627        let had_prior_snapshot = self.fs.snapshot_path().map(|p| p.exists()).unwrap_or(false)
13628            && !self.snapshot_preserved_history;
13629        // Which version this store writes. V9 unless it has opted in to
13630        // multiplicity, in which case V10 — the stamp that makes a reader which
13631        // does not know WAL discriminant 23 refuse the open instead of
13632        // truncating the WAL at the first such frame. The container is
13633        // identical either way; only these two header bytes move.
13634        let snapshot_version = core_storage::snapshot::version_for(self.multiplicity);
13635        self.ensure_v8_base_sections_loaded();
13636        // Ensure provenance is decoded before to_persist() clones it.
13637        self.engine.ensure_provenance_loaded_mut();
13638        let (rule_defs_typed, provenance, rule_tripped, rule_fires) = self.engine.to_persist();
13639        let rule_defs = rule_defs_typed
13640            .iter()
13641            .map(|r| bincode::serialize(r).expect("RuleDef serialize cannot fail"))
13642            .collect();
13643        // Collect HNSW state and IVF state.  When indexes are not yet
13644        // populated (clean open, no mutation since open), pass the retained
13645        // raw bytes through directly so that migrate/snapshot does not
13646        // silently discard fitted approximate-rule indexes.
13647        let hnsw_state = self.engine.export_hnsw_state_passthrough();
13648        let ivf_bytes = if !self.engine.indexes_populated() {
13649            // Pass retained IVF bytes through unchanged (no re-encode).
13650            self.engine.retained_ivf_bytes_clone().unwrap_or_default()
13651        } else {
13652            // Indexes live: encode from current state.
13653            let raw_ivf = self.engine.export_ivf_state();
13654            let ivf_state_map: BTreeMap<String, core_storage::snapshot::PerRuleIvfState> = raw_ivf
13655                .into_iter()
13656                .map(|(name, ((sc, sa, sd), (dc, da, dd)))| {
13657                    (
13658                        name,
13659                        core_storage::snapshot::PerRuleIvfState {
13660                            src: core_storage::snapshot::SideIvfState {
13661                                centroids: sc,
13662                                clusters: sa,
13663                                drift: sd,
13664                            },
13665                            dst: core_storage::snapshot::SideIvfState {
13666                                centroids: dc,
13667                                clusters: da,
13668                                drift: dd,
13669                            },
13670                        },
13671                    )
13672                })
13673                .collect();
13674            if ivf_state_map.is_empty() {
13675                Vec::new()
13676            } else {
13677                bincode::serialize(&ivf_state_map).expect("IVF state serialize cannot fail")
13678            }
13679        };
13680        let view_defs: Vec<Vec<u8>> = self
13681            .view_store
13682            .views()
13683            .map(|v| bincode::serialize(v).expect("ViewDef serialize cannot fail"))
13684            .collect();
13685        if self.base.is_some() {
13686            // V8 merge-snapshot path: encode base+overlay into a new V8 snapshot,
13687            // write it atomically, remap it as the new base, then clear the overlay.
13688            let meta = V8Meta {
13689                labels: (*self.labels).clone(),
13690                edge_props: (*self.edge_props).clone(),
13691                rule_defs,
13692                provenance,
13693                rule_tripped,
13694                rule_fires,
13695                ivf_bytes,
13696                view_defs,
13697                wal_truncated: !opts.keep_wal,
13698                hnsw: hnsw_state,
13699                last_change: self.last_change.clone(),
13700            };
13701            let mut buf: Vec<u8> = Vec::new();
13702            {
13703                // Clone the Arc so the old base stays alive while we encode.
13704                // The borrow of archived_csr (into old_base's mmap) is released
13705                // at the end of this block, before we replace self.base.
13706                let old_base = self.base.clone().expect("is_some checked above");
13707                let archived_csr = old_base.topology().map_err(|e| GraphError::Corrupt {
13708                    detail: format!("v8 snapshot: topology section: {e:?}"),
13709                })?;
13710                let archived_cols = old_base.columns().map_err(|e| GraphError::Corrupt {
13711                    detail: format!("v8 snapshot: columns section: {e:?}"),
13712                })?;
13713                // `None` when the base predates V9 — the migration path: its
13714                // string columns still carry their own tables and this snapshot
13715                // is the rewrite that collapses them into section 12.
13716                let archived_strings =
13717                    old_base
13718                        .string_table()
13719                        .transpose()
13720                        .map_err(|e| GraphError::Corrupt {
13721                            detail: format!("v8 snapshot: strings section: {e:?}"),
13722                        })?;
13723                let archived_edge_props =
13724                    old_base
13725                        .edge_props_section()
13726                        .map_err(|e| GraphError::Corrupt {
13727                            detail: format!("v8 snapshot: edge_props section: {e:?}"),
13728                        })?;
13729                let edge_props_raw =
13730                    old_base
13731                        .edge_props_raw_bytes()
13732                        .map_err(|e| GraphError::Corrupt {
13733                            detail: format!("v8 snapshot: edge_props raw bytes: {e:?}"),
13734                        })?;
13735                let prov_raw =
13736                    old_base
13737                        .provenance_raw_bytes()
13738                        .map_err(|e| GraphError::Corrupt {
13739                            detail: format!("v8 snapshot: provenance raw bytes: {e:?}"),
13740                        })?;
13741                encode_v8(
13742                    Some(archived_csr),
13743                    Some(archived_cols),
13744                    archived_strings,
13745                    Some((archived_edge_props, edge_props_raw)),
13746                    Some(prov_raw),
13747                    &self.topo,
13748                    &self.props,
13749                    &self.ids,
13750                    &self.syms,
13751                    &meta,
13752                    &mut buf,
13753                )?;
13754            }
13755            core_storage::snapshot::stamp_container_version(&mut buf, snapshot_version)?;
13756            self.fs.write_atomic(FileId::Snapshot, &buf)?;
13757            // Remap the freshly-written snapshot as the new base.
13758            // C2: use file mmap on RealFs; fall back to from_bytes on SimFs.
13759            let new_base = if let Some(snap_path) = self.fs.snapshot_path() {
13760                core_storage::v8::MappedBase::map(&snap_path)
13761            } else {
13762                core_storage::v8::MappedBase::from_bytes(buf)
13763            }
13764            .map_err(|e| GraphError::Corrupt {
13765                detail: format!("v8 snapshot: remap new base: {e:?}"),
13766            })?;
13767            self.base = Some(Arc::new(new_base));
13768            // Clear the overlay and prop tombstones — all data is now in the new base.
13769            self.topo = Arc::new(Topology::new());
13770            self.props = Arc::new(core_storage::columns::ColumnStore::new());
13771        } else {
13772            // Legacy path (V5–V7 stores without a V8 base).
13773            //
13774            // Memory-diet path: build V8Meta directly from &self — no SnapshotState
13775            // clone and no encode_v8_from_state intermediate clones.  The big
13776            // structures (self.topo, self.props) are borrowed, not cloned.
13777            // self.edge_props is moved (not cloned) because we immediately clear it
13778            // when we remap the new V8 snapshot as self.base (see below).
13779            //
13780            // Eliminates from peak RSS vs. the old SnapshotState path:
13781            //   • self.topo.clone()      (~topology HashMap footprint)
13782            //   • self.props.clone()     (~column-store footprint)
13783            //   • encode_v8_from_state V8Meta secondary clones (labels, edge_props, …)
13784            let meta = V8Meta {
13785                labels: (*self.labels).clone(),
13786                wal_truncated: !opts.keep_wal,
13787                // Move edge_props out so the large overlay is freed when meta
13788                // drops at end of this block (self.edge_props is now empty; reads
13789                // after base assignment go through the mmap'd base section).
13790                edge_props: std::mem::take(Arc::make_mut(&mut self.edge_props)),
13791                rule_defs,
13792                provenance,
13793                rule_tripped,
13794                rule_fires,
13795                ivf_bytes,
13796                view_defs,
13797                hnsw: hnsw_state,
13798                last_change: self.last_change.clone(),
13799            };
13800            let mut buf = Vec::new();
13801            encode_v8(
13802                None,
13803                None,
13804                None,
13805                None,
13806                None,
13807                &self.topo,
13808                &self.props,
13809                &self.ids,
13810                &self.syms,
13811                &meta,
13812                &mut buf,
13813            )?;
13814            // meta (and the moved edge_props inside it) is no longer needed;
13815            // drop it before the write to keep the peak window narrow.
13816            drop(meta);
13817            core_storage::snapshot::stamp_container_version(&mut buf, snapshot_version)?;
13818            self.fs.write_atomic(FileId::Snapshot, &buf)?;
13819            // Remap the freshly-written V8 snapshot as self.base.
13820            // On RealFs: drop the encode buffer before mmap to recover ~1.9 GiB.
13821            // On SimFs (tests): pass buf to from_bytes.
13822            let new_base = if let Some(snap_path) = self.fs.snapshot_path() {
13823                drop(buf);
13824                core_storage::v8::MappedBase::map(&snap_path)
13825            } else {
13826                core_storage::v8::MappedBase::from_bytes(buf)
13827            }
13828            .map_err(|e| GraphError::Corrupt {
13829                detail: format!("v8 snapshot: remap new base (legacy path): {e:?}"),
13830            })?;
13831            self.base = Some(Arc::new(new_base));
13832            // Free the large heap-allocated decoded state — all data is now in the
13833            // mmap'd base.  Mirrors the V8 merge-snapshot path (see above).
13834            // self.edge_props was already moved into meta and is effectively empty.
13835            self.topo = Arc::new(Topology::new());
13836            self.props = Arc::new(core_storage::columns::ColumnStore::new());
13837        }
13838
13839        if opts.archive_wal {
13840            // History-preserving snapshot (Task 4):
13841            //   1. Snapshot already written above (write_atomic → fsynced).
13842            //   2. Rename WAL → wal.<commit_seq>.archive  (atomic, same fs).
13843            //      Crash window B: crash here leaves archive present, WAL
13844            //      absent.  Reopen: snapshot loaded (full state), no WAL
13845            //      replay.  Archive is NOT replayed into live state — it is
13846            //      pre-snapshot by construction.  Safe.
13847            //   3. Optionally write genesis marker (first archive only, no
13848            //      prior WAL truncation).
13849            //   4. Prune old archives (retention), update horizon floor.
13850            //      Pruning invalidates the genesis chain; delete marker.
13851            //   5. Write new minimal baseline WAL (write_atomic).
13852            //      Crash window C: crash here leaves new archive plus no live
13853            //      WAL.  Same as window B — handled above.
13854            //
13855            // Sample existing archives BEFORE the rename so we can detect
13856            // whether this is the first archive.
13857            let existing_archives = self.fs.list_archives()?;
13858            let is_first_archive = existing_archives.is_empty();
13859
13860            // Compute a globally-monotonic archive name: the name equals the
13861            // cumulative end-frame index of the archive in global commit space.
13862            //
13863            // Using `commit_seq` directly is UNSOUND across sessions: on reopen
13864            // commit_seq is seeded from max(last_change), which underestimates
13865            // the WAL depth when trailing commits (e.g. insert_edge) do not
13866            // update last_change.  A session-2 archive could then receive a name
13867            // ≤ the session-1 archive, causing incorrect sort order or collision.
13868            //
13869            // Instead: read and decode the live WAL here (before the rename) to
13870            // get its exact frame count, then add it to the last known global
13871            // end-frame index (the name of the most recent existing archive, or
13872            // wal_horizon_floor if no archives exist).  This is O(WAL size) but
13873            // snapshot is already serialising the full graph state, so the cost
13874            // is dominated.
13875            let live_wal_bytes_for_name = self.fs.read(FileId::Wal)?;
13876            let (live_frames_for_name, _) = decode_all(&live_wal_bytes_for_name);
13877            let archive_n = existing_archives
13878                .last()
13879                .copied()
13880                .unwrap_or(self.wal_horizon_floor)
13881                + live_frames_for_name.len() as u64;
13882            self.fs.archive_wal(archive_n)?;
13883
13884            // The replacement WAL goes in **immediately**, with no fallible call
13885            // between it and the rename above.
13886            //
13887            // The rename is what removes the store's live declarations — the
13888            // multiplicity opt-in, and every `EnableFulltext` / `EnableIndex` —
13889            // and this write is what puts them back. Every call that used to sit
13890            // in between (the genesis marker, the retention sweep's reads, the
13891            // floor write, the archive deletes) was a `?` that could leave the
13892            // store with neither, so a single transient `Err` was enough to lose
13893            // a declaration that no rebuild can recover (defect #22).
13894            //
13895            // Ordering alone cannot close the crash window between two
13896            // filesystem calls; for the multiplicity declaration the V10 stamp
13897            // does that on the open path. What ordering does close is the much
13898            // wider window in which an ordinary I/O error did it — and that half
13899            // covers all three declarations, not just the one with a stamp.
13900            let mut baseline_wal: Vec<u8> = Vec::new();
13901            // The multiplicity opt-in is a declaration like the two below it,
13902            // and it is re-emitted for the same reason: truncation must not
13903            // silently opt the store back out and stop counting.
13904            if self.multiplicity {
13905                baseline_wal
13906                    .extend_from_slice(&encode_record(&core_storage::wal::MULTIPLICITY_ENABLED));
13907            }
13908            for (label, field) in self.fulltext.enabled_pairs() {
13909                let rec = WalRecord::EnableFulltext {
13910                    label: label.clone(),
13911                    field: field.clone(),
13912                };
13913                baseline_wal.extend_from_slice(&encode_record(&rec));
13914            }
13915            for (label, field) in self.prop_index.enabled_pairs() {
13916                let rec = WalRecord::EnableIndex {
13917                    label: label.clone(),
13918                    field: field.clone(),
13919                };
13920                baseline_wal.extend_from_slice(&encode_record(&rec));
13921            }
13922            self.fs.write_atomic(FileId::Wal, &baseline_wal)?;
13923
13924            // Genesis marker: written once when the first archive is taken
13925            // from a store that has never undergone a WAL-truncating snapshot.
13926            // When present, `open_at` may replay archive-resident commits from
13927            // empty state (the archive chain covers from global index 0).
13928            //
13929            // Two conditions must ALL hold:
13930            //   1. This is the first archive (existing_archives was empty).
13931            //   2. No snapshot.bin existed before this operation (had_prior_snapshot=false).
13932            //      A WAL-truncating snapshot (keep_wal=false) always writes snapshot.bin
13933            //      before truncating the WAL, so if any prior truncating snapshot was taken
13934            //      — even in a previous session — snapshot.bin is present and this condition
13935            //      is false.  This subsumes the cross-session truncation case without
13936            //      requiring a separate wal.truncated sidecar file.
13937            //      For legacy stores (snapshot.bin written by an older code version that
13938            //      may have truncated the WAL), the same conservative refusal applies:
13939            //      we cannot prove the chain is complete, so we refuse genesis (cost =
13940            //      no as-of-through-archives; never silent wrong data).
13941            //      The one exception is a snapshot this handle took itself, on a store
13942            //      that had none when it opened, with the WAL kept: there the answer is
13943            //      known rather than guessed, and `snapshot_preserved_history` says so.
13944            //      Without that exception `enable_multiplicity`'s forced keep_wal
13945            //      snapshot would disqualify the store forever (defect #23).
13946            //      On SimFs (snapshot_path() == None) had_prior_snapshot is always false,
13947            //      so SimFs always passes this check.
13948            if is_first_archive && !had_prior_snapshot {
13949                self.fs.write_genesis_marker()?;
13950                self.archive_genesis_chain = true;
13951            }
13952
13953            // Retention pruning: keep newest `keep` archives; delete oldest.
13954            // Pruning is the ONLY deletion site for archives.
13955            //
13956            // Crash-safety ordering (C1 fix):
13957            //   1. Count frames in surplus archives (reads only — no mutation).
13958            //   2. Advance and PERSIST the horizon floor FIRST via write-then-
13959            //      rename (atomic).  A crash after this point leaves orphaned
13960            //      archives on disk, but the floor is correct.  The opening
13961            //      cleanup sweep (`cleanup_orphaned_archives`) removes them on
13962            //      the next open, so the store is always safe to reopen.
13963            //   3. Delete the genesis marker (floor > 0 already blocks open_at
13964            //      via the conjunctive gate; marker cleanup is belt-and-suspenders).
13965            //   4. Delete surplus archives.  A crash between any two deletes
13966            //      leaves the floor committed and orphaned archives cleaned at
13967            //      next open — never a stale floor with a missing archive prefix.
13968            if let Some(keep) = self.wal_archive_retention {
13969                if keep > 0 {
13970                    let archives = self.fs.list_archives()?;
13971                    // archives is sorted ascending (oldest first)
13972                    if archives.len() as u32 > keep {
13973                        let surplus = archives.len() - keep as usize;
13974                        // Step 1: count pruned frames (reads, no mutation).
13975                        let mut pruned_frames = 0u64;
13976                        for &n in &archives[..surplus] {
13977                            let bytes = self.fs.read_archive(n)?;
13978                            let (frames, _) = decode_all(&bytes);
13979                            pruned_frames += frames.len() as u64;
13980                        }
13981                        // Step 2: advance and persist floor FIRST.
13982                        self.wal_horizon_floor += pruned_frames;
13983                        self.fs.write_horizon_floor(self.wal_horizon_floor)?;
13984                        // The time map must not outlive the commits it
13985                        // describes: an entry below the new floor would resolve
13986                        // a date to a commit the engine can no longer replay,
13987                        // which is worse than having no entry at all.
13988                        self.commit_times.truncate_below(self.wal_horizon_floor);
13989                        self.rewrite_commit_times();
13990                        // Step 3: delete genesis marker (floor > 0 already
13991                        // blocks open_at; this is belt-and-suspenders cleanup).
13992                        if pruned_frames > 0 && self.archive_genesis_chain {
13993                            self.fs.delete_genesis_marker()?;
13994                            self.archive_genesis_chain = false;
13995                        }
13996                        // Step 4: delete surplus archives.  Crash here →
13997                        // orphaned archives; cleaned at next open.
13998                        for &n in &archives[..surplus] {
13999                            self.fs.delete_archive(n)?;
14000                        }
14001                    }
14002                }
14003            }
14004        } else if opts.keep_wal {
14005            // keep_wal=true: WAL is left untouched.  The existing WAL already
14006            // contains the EnableFulltext records from the original enable calls;
14007            // replay is idempotent (guards in apply() skip already-live entries).
14008            // No baseline re-write is needed or safe here — the full WAL history
14009            // must remain intact for open_at to reach pre-snapshot commits.
14010        } else {
14011            // keep_wal=false (default): truncate by replacing the WAL with a
14012            // minimal baseline of one EnableFulltext record per active pair.
14013            //
14014            // Crash-ordering: write_atomic is atomic.
14015            //   • Crash before snapshot write  → WAL unchanged.  Safe.
14016            //   • Crash after snapshot write but before this WAL write → full
14017            //     pre-snapshot WAL still present; open_with replays idempotently.
14018            //   • Crash after both writes → normal post-snapshot state.
14019            //
14020            // Genesis chain: a WAL-truncating snapshot breaks the archive chain
14021            // for any archives taken AFTER this point (their WAL slices would
14022            // not start at genesis).  Delete any existing genesis marker so that
14023            // open_at refuses archive-resident commits.  Future sessions are
14024            // covered by had_prior_snapshot: snapshot.bin written here persists
14025            // across sessions and prevents a later archiving session from
14026            // incorrectly claiming a complete genesis chain.
14027            if self.archive_genesis_chain {
14028                self.fs.delete_genesis_marker()?;
14029                self.archive_genesis_chain = false;
14030            }
14031            // And this handle can no longer prove the WAL is whole: it is about
14032            // to truncate it itself. Same-session archives after this point get
14033            // the conservative answer, exactly as cross-session ones do.
14034            self.snapshot_preserved_history = false;
14035            let mut baseline_wal: Vec<u8> = Vec::new();
14036            // The multiplicity opt-in is a declaration like the two below it,
14037            // and it is re-emitted for the same reason: truncation must not
14038            // silently opt the store back out and stop counting.
14039            if self.multiplicity {
14040                baseline_wal
14041                    .extend_from_slice(&encode_record(&core_storage::wal::MULTIPLICITY_ENABLED));
14042            }
14043            for (label, field) in self.fulltext.enabled_pairs() {
14044                let rec = WalRecord::EnableFulltext {
14045                    label: label.clone(),
14046                    field: field.clone(),
14047                };
14048                baseline_wal.extend_from_slice(&encode_record(&rec));
14049            }
14050            for (label, field) in self.prop_index.enabled_pairs() {
14051                let rec = WalRecord::EnableIndex {
14052                    label: label.clone(),
14053                    field: field.clone(),
14054                };
14055                baseline_wal.extend_from_slice(&encode_record(&rec));
14056            }
14057            self.fs.write_atomic(FileId::Wal, &baseline_wal)?;
14058            // This branch discards history rather than archiving it: the frames
14059            // the map describes are gone and the replacement WAL renumbers from
14060            // the floor, so every surviving entry now names a different frame.
14061            // Keeping them would resolve a date onto an unrelated commit — and
14062            // the stamps written after this point would sit far above the
14063            // store's own frame count, which is how the date surface came to
14064            // refuse commits the index path served perfectly well.
14065            //
14066            // A store that cannot answer a date says so by name
14067            // (`NoRecordedTime`). That is the honest state after discarding the
14068            // history the dates addressed.
14069            self.commit_times = core_storage::commit_times::CommitTimes::default();
14070            self.rewrite_commit_times();
14071        }
14072        // After snapshot the overlay may have changed (V8 merge path clears
14073        // self.topo and self.props). Refresh the MVCC fold so future readers
14074        // see the post-snapshot state rather than stale overlay data.
14075        self.fold_now();
14076        // We wrote the snapshot and (unless keep_wal) replaced the WAL, so both
14077        // markers this handle uses to detect other processes' work must be
14078        // re-taken from disk. Skipping this would make our own snapshot look
14079        // like a peer's on the next staleness check and force a needless
14080        // reload.
14081        self.wal_consumed = self.fs.wal_len().map_err(GraphError::Io)?;
14082        // A snapshot can replace the live WAL with a baseline, which renumbers
14083        // every frame after it. Re-derive the frame cursor from what the store
14084        // now actually holds rather than carrying the pre-snapshot count
14085        // forward — `wal_total_commits` is the same sequence the history
14086        // surfaces index, and the snapshot has already paid a far larger cost
14087        // than one decode.
14088        self.wal_frames_written = self.wal_total_commits()?;
14089        self.snapshot_ident = self.fs.snapshot_ident().map_err(GraphError::Io)?;
14090        Ok(())
14091    }
14092}
14093
14094/// What a batch node insert does when its key is already taken.
14095///
14096/// A mirror rebuild writes a frame onto a store that already has content, so
14097/// "the key exists" is a routine answer rather than a failure. The decision is
14098/// made during the batch's existing validate pass, from one id-map lookup per
14099/// row, so the frame stays atomic and re-ingest stays O(n).
14100#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)]
14101pub enum OnConflict {
14102    /// Refuse the whole frame with [`GraphError::DuplicateKey`]. The default,
14103    /// and the only behaviour before v0.6.10.
14104    #[default]
14105    Error,
14106    /// Leave the stored node exactly as it is — properties, label and edges —
14107    /// and count it in [`BatchOutcome::skipped`].
14108    Skip,
14109    /// Keep the key and make the node's properties **exactly** the supplied
14110    /// props: supplied fields are set, fields absent from the supplied props
14111    /// are removed. A supplied label that differs from the stored one, and a
14112    /// supplied `ns` that would move the node, are row errors — relabelling is
14113    /// [`GraphDb::rename_node`], not a side effect of a rebuild.
14114    ///
14115    /// Two properties are outside "exactly", both because they are not the
14116    /// caller's to supply:
14117    ///
14118    /// - `ns` is immutable, so an omitted `ns` leaves the node where it is
14119    ///   rather than moving it to `default`;
14120    /// - a property a **view** owns is kept, not removed. Supplying one is a
14121    ///   row error, so omitting it cannot be a request to delete it, and
14122    ///   refusing the row instead would make `Replace` impossible for the whole
14123    ///   population a view has written to. Each such field kept is counted in
14124    ///   [`BatchOutcome::kept_view_owned`] — the row is still `replaced` and
14125    ///   still raises no row error, so that count is the only signal a caller
14126    ///   gets that the stored node carries a field its frame did not describe.
14127    Replace,
14128}
14129
14130/// What one [`OnConflict::Replace`] row resolves to.
14131///
14132/// The property writes that make the node exactly the supplied props —
14133/// `Some(value)` is a set, `None` a removal — paired with how many view-owned
14134/// fields the row kept instead of removing, which is the one way the result is
14135/// not exactly the supplied props. See [`MutPreview::plan_replace`].
14136type ReplacePlan = (Vec<(String, Option<Value>)>, usize);
14137
14138/// What one committed batch did.
14139///
14140/// [`BatchBuilder::commit`] returns the first two fields as a tuple; the rest
14141/// exist for [`OnConflict`] and are always zero / empty without it.
14142#[derive(Clone, Debug, Default, PartialEq, Eq)]
14143pub struct BatchOutcome {
14144    /// Node records actually written.
14145    pub nodes_inserted: usize,
14146    /// Edge records actually written. A duplicate edge is a silent no-op under
14147    /// every policy — adjacency is a set — and is not counted.
14148    pub edges_inserted: usize,
14149    /// Rows whose key was taken and whose policy was [`OnConflict::Skip`].
14150    pub skipped: usize,
14151    /// Rows whose key was taken and whose policy was [`OnConflict::Replace`].
14152    pub replaced: usize,
14153    /// View-owned properties an [`OnConflict::Replace`] row **kept** although
14154    /// the caller did not supply them — counted per field, so one row that
14155    /// keeps two contributes two.
14156    ///
14157    /// This is the one respect in which `Replace` does not make a node's props
14158    /// exactly the supplied ones (see [`OnConflict::Replace`]). Those rows
14159    /// still count in `replaced` and still raise no `row_errors`, because
14160    /// nothing went wrong: a view's property is not the caller's to supply or
14161    /// to remove. A mirror rebuild that needs its copy to be byte-exact reads
14162    /// this to learn that the store kept fields its frame did not describe.
14163    pub kept_view_owned: usize,
14164    /// `(row, why)` for rows an [`OnConflict::Replace`] refused. `row` counts
14165    /// node-insert ops in this batch from zero, which for a caller that queues
14166    /// its nodes in order is the index of the offending node. The rest of the
14167    /// frame still commits; the refused row changes nothing.
14168    pub row_errors: Vec<(usize, String)>,
14169}
14170
14171/// The `(nodes_inserted, edges_inserted)` pair every pre-0.6.10 commit entry
14172/// point returns. Keeps those signatures unchanged now that the validate pass
14173/// produces a [`BatchOutcome`].
14174fn inserted_pair(outcome: BatchOutcome) -> (usize, usize) {
14175    (outcome.nodes_inserted, outcome.edges_inserted)
14176}
14177
14178/// One entry of a frame the validate pass has decided on, before
14179/// [`GraphDb::rewrite_wal_dense_planned`] turns it into dense-id records.
14180///
14181/// Almost every entry is already a finished [`WalRecord`]. The exception is a
14182/// duplicate edge insert: its count names a dense triple, and on the batch path
14183/// the endpoints and the edge type may all be created by earlier records in the
14184/// *same* frame, so no id for them exists until the dense rewrite allocates it.
14185/// Carrying the keys this far and resolving them there is what lets the count
14186/// survive the shape a mirror rebuild writes (defect #24).
14187enum PlannedRec {
14188    Rec(WalRecord),
14189    DuplicateCount {
14190        edge_type: String,
14191        src_key: String,
14192        dst_key: String,
14193    },
14194}
14195
14196/// Queued mutation for a [`BatchBuilder`] or [`GraphDb::commit_group`].
14197///
14198/// The `submit_batch` / `commit_group` APIs accept `Vec<BatchOp>` so that
14199/// callers can build a set of mutations without holding `&mut GraphDb` and
14200/// hand them off to the group-committing writer for durable, batched I/O.
14201pub enum BatchOp {
14202    InsertNode {
14203        label: String,
14204        key: String,
14205        props: Vec<(String, Value)>,
14206    },
14207    InsertEdge {
14208        edge_type: String,
14209        src_key: String,
14210        dst_key: String,
14211    },
14212    SetProp {
14213        key: String,
14214        field: String,
14215        value: Value,
14216    },
14217    RemoveProp {
14218        key: String,
14219        field: String,
14220    },
14221    DeleteEdge {
14222        edge_type: String,
14223        src_key: String,
14224        dst_key: String,
14225    },
14226    DeleteNode {
14227        key: String,
14228    },
14229    CreateRule(RuleDef),
14230    DeleteRule {
14231        name: String,
14232    },
14233    /// Rename a node's key. Validated: old must exist, new must not.
14234    RenameNode {
14235        old_key: String,
14236        new_key: String,
14237    },
14238    /// Insert an edge, auto-creating any missing endpoint as a plain node with
14239    /// `placeholder_label` and no props. Rules fire and last-change is updated
14240    /// for each created endpoint (normal InsertNode semantics in the batch frame).
14241    InsertEdgeUpsert {
14242        edge_type: String,
14243        src_key: String,
14244        dst_key: String,
14245        placeholder_label: String,
14246    },
14247    /// Insert `key`, or — when the key is already taken — do what `on_conflict`
14248    /// says. Queued by [`BatchBuilder::insert_node_on_conflict`]; `Error`
14249    /// queues a plain [`BatchOp::InsertNode`] instead, so this variant only
14250    /// ever carries `Skip` or `Replace`.
14251    InsertNodeOnConflict {
14252        label: String,
14253        key: String,
14254        props: Vec<(String, Value)>,
14255        on_conflict: OnConflict,
14256    },
14257}
14258
14259/// Three-way node visibility status used by `check_single_op_authz`.
14260enum NodeAuthzStatus {
14261    /// Node exists in the store and is in the role's read mask.
14262    Visible(String), // carries the node's label
14263    /// Node exists in the store but is NOT in the role's read mask.
14264    Hidden,
14265    /// Node does not exist in the store.
14266    Absent,
14267}
14268
14269/// Overlay of ops already accepted earlier in the same batch. Never written
14270/// back to the database — validation only.
14271#[derive(Default)]
14272struct Overlay {
14273    extra_keys: BTreeSet<String>,
14274    /// Label of each node inserted earlier in this batch. The store does not
14275    /// have these keys yet, so `label_of` cannot answer for them, and
14276    /// `OnConflict::Replace` has to compare labels.
14277    extra_labels: BTreeMap<String, String>,
14278    deleted_keys: BTreeSet<String>,
14279    extra_props: BTreeMap<(String, String), Value>,
14280    removed_props: BTreeSet<(String, String)>,
14281    extra_edges: BTreeSet<(String, String, String)>,
14282    deleted_edges: BTreeSet<(String, String, String)>,
14283    extra_rules: BTreeSet<String>,
14284    deleted_rules: BTreeSet<String>,
14285    /// `rule name → (via_edge, edge_type)` for every via-hop rule accepted
14286    /// earlier in this batch. Feeds the rule-chain cycle check, which otherwise
14287    /// sees only the rules already committed to the engine. Keyed by name so a
14288    /// later `DeleteRule` in the same batch drops the arc with the rule.
14289    extra_rule_arcs: BTreeMap<String, (String, String)>,
14290}
14291
14292/// Read-only view of live db state plus a batch overlay. Shared by single-op
14293/// public methods (empty overlay) and `commit_batch`.
14294struct MutPreview<'a, F: Fs> {
14295    db: &'a GraphDb<F>,
14296    overlay: Overlay,
14297}
14298
14299/// Shortest path from `start` to `target` following `arcs` (`from → to`), or
14300/// `None` if `target` is unreachable.
14301///
14302/// Used for rule-chain cycle detection, where an arc is "a rule hops over
14303/// `from` and writes `to`". Breadth-first over BTree-ordered adjacency, so the
14304/// reported path is stable for a given rule set, and iterative so a pathological
14305/// rule graph cannot overflow the stack.
14306fn find_cycle_through(arcs: &[(String, String)], start: &str, target: &str) -> Option<Vec<String>> {
14307    let mut adj: BTreeMap<&str, BTreeSet<&str>> = BTreeMap::new();
14308    for (from, to) in arcs {
14309        adj.entry(from.as_str()).or_default().insert(to.as_str());
14310    }
14311    let mut parent: BTreeMap<&str, &str> = BTreeMap::new();
14312    let mut visited: BTreeSet<&str> = BTreeSet::new();
14313    let mut queue: std::collections::VecDeque<&str> = std::collections::VecDeque::new();
14314    visited.insert(start);
14315    queue.push_back(start);
14316    while let Some(node) = queue.pop_front() {
14317        if node == target {
14318            let mut path = vec![node.to_string()];
14319            let mut cur = node;
14320            while let Some(&p) = parent.get(cur) {
14321                path.push(p.to_string());
14322                cur = p;
14323            }
14324            path.reverse();
14325            return Some(path);
14326        }
14327        for &next in adj.get(node).into_iter().flatten() {
14328            if visited.insert(next) {
14329                parent.insert(next, node);
14330                queue.push_back(next);
14331            }
14332        }
14333    }
14334    None
14335}
14336
14337impl<'a, F: Fs> MutPreview<'a, F> {
14338    fn new(db: &'a GraphDb<F>) -> Self {
14339        Self {
14340            db,
14341            overlay: Overlay::default(),
14342        }
14343    }
14344
14345    fn has_key(&self, key: &str) -> bool {
14346        if self.overlay.extra_keys.contains(key) {
14347            return true;
14348        }
14349        if self.overlay.deleted_keys.contains(key) {
14350            return false;
14351        }
14352        self.db.ids.get(key).is_some()
14353    }
14354
14355    fn has_prop(&self, key: &str, field: &str) -> bool {
14356        if !self.has_key(key) {
14357            return false;
14358        }
14359        let k = (key.to_string(), field.to_string());
14360        if self.overlay.removed_props.contains(&k) {
14361            return false;
14362        }
14363        if self.overlay.extra_props.contains_key(&k) {
14364            return true;
14365        }
14366        // Fresh identity (first insert in this batch, or delete+reinsert):
14367        // ignore props still sitting on the soon-to-be-tombstoned slot.
14368        if self.overlay.extra_keys.contains(key) {
14369            return false;
14370        }
14371        self.db.get_prop(key, field).is_some()
14372    }
14373
14374    fn has_edge(&self, edge_type: &str, src_key: &str, dst_key: &str) -> bool {
14375        let k = (
14376            edge_type.to_string(),
14377            src_key.to_string(),
14378            dst_key.to_string(),
14379        );
14380        if self.overlay.deleted_edges.contains(&k) {
14381            return false;
14382        }
14383        if self.overlay.extra_edges.contains(&k) {
14384            return true;
14385        }
14386        // A key created in this batch (including reinsert) has no db edges.
14387        if self.overlay.extra_keys.contains(src_key) || self.overlay.extra_keys.contains(dst_key) {
14388            return false;
14389        }
14390        if self.overlay.deleted_keys.contains(src_key)
14391            || self.overlay.deleted_keys.contains(dst_key)
14392        {
14393            return false;
14394        }
14395        let Some(src) = self.db.ids.get(src_key) else {
14396            return false;
14397        };
14398        let Some(dst) = self.db.ids.get(dst_key) else {
14399            return false;
14400        };
14401        let Some(sym) = self.db.syms.get(edge_type) else {
14402            return false;
14403        };
14404        self.db
14405            .topo_view()
14406            .neighbors(sym, Direction::Out, src)
14407            .binary_search(&dst)
14408            .is_ok()
14409    }
14410
14411    fn has_rule(&self, name: &str) -> bool {
14412        if self.overlay.extra_rules.contains(name) {
14413            return true;
14414        }
14415        if self.overlay.deleted_rules.contains(name) {
14416            return false;
14417        }
14418        self.db.engine.rules().any(|r| r.name == name)
14419    }
14420
14421    fn is_rule_owned(&self, edge_type: &str, src_key: &str, dst_key: &str) -> bool {
14422        if self.overlay.extra_keys.contains(src_key) || self.overlay.extra_keys.contains(dst_key) {
14423            return false;
14424        }
14425        if self.overlay.deleted_keys.contains(src_key)
14426            || self.overlay.deleted_keys.contains(dst_key)
14427        {
14428            return false;
14429        }
14430        let Some(src) = self.db.ids.get(src_key) else {
14431            return false;
14432        };
14433        let Some(dst) = self.db.ids.get(dst_key) else {
14434            return false;
14435        };
14436        let Some(et) = self.db.syms.get(edge_type) else {
14437            return false;
14438        };
14439        // extra_rules is deliberately not consulted: a CreateRule earlier in
14440        // this batch has not fired, so it contributes no provenance. That is
14441        // the documented rule-window gap (see GraphDb::batch).
14442        if self.overlay.deleted_rules.is_empty() {
14443            return self.db.engine.is_owned(et, src, dst);
14444        }
14445        for (rule, triples) in self.db.engine.provenance() {
14446            if self.overlay.deleted_rules.contains(rule) {
14447                continue;
14448            }
14449            if triples.contains(&(et, src, dst)) {
14450                return true;
14451            }
14452        }
14453        false
14454    }
14455
14456    /// The refusals a node creation makes, in the order it makes them.
14457    ///
14458    /// A view owns its property, and creating a node that carries one is a
14459    /// write to it exactly as `set_prop` is — so it is refused here, at the one
14460    /// choke-point `GraphDb::insert_node`, `BatchOp::InsertNode` and the
14461    /// no-conflict arm of `BatchOp::InsertNodeOnConflict` all pass through.
14462    ///
14463    /// Leaving creation exempt was not harmless. The value was stored and
14464    /// served: a created node the view has no reason to revisit keeps the
14465    /// caller's number for the life of the handle, and the backfill at the next
14466    /// open overwrites it — so the store answered `deg = 777` before a restart
14467    /// and `deg = 0` after, for a property every other surface calls read-only.
14468    /// It also split one op two ways: supplying a view-owned field under
14469    /// `OnConflict::Replace` was already a row error on a taken key while the
14470    /// same field on a fresh key was accepted.
14471    ///
14472    /// Checked before the key, like [`MutPreview::prepare_remove_prop`], so the
14473    /// answer does not depend on whether the key exists.
14474    fn check_insert_node(&self, key: &str, props: &[(String, Value)]) -> Result<()> {
14475        for (field, _) in props {
14476            if let Some(view_name) = self.db.view_store.view_for_prop(field) {
14477                return Err(GraphError::ViewPropReadOnly {
14478                    view_name: view_name.to_string(),
14479                });
14480            }
14481        }
14482        if self.has_key(key) {
14483            Err(GraphError::DuplicateKey { key: key.into() })
14484        } else {
14485            Ok(())
14486        }
14487    }
14488
14489    fn check_live_key(&self, key: &str) -> Result<()> {
14490        if self.has_key(key) {
14491            Ok(())
14492        } else {
14493            Err(GraphError::KeyNotFound { key: key.into() })
14494        }
14495    }
14496
14497    fn prepare_insert_edge(&self, edge_type: &str, src_key: &str, dst_key: &str) -> Result<bool> {
14498        for k in [src_key, dst_key] {
14499            if !self.has_key(k) {
14500                return Err(GraphError::KeyNotFound { key: k.into() });
14501            }
14502        }
14503        if self.is_rule_owned(edge_type, src_key, dst_key) {
14504            return Err(GraphError::RuleOwned {
14505                detail: format!("edge {edge_type} {src_key}→{dst_key} is rule-owned"),
14506            });
14507        }
14508        // A user-written edge stays inside one namespace. Derived edges do not
14509        // come through here — the engine adds them directly — and the rule
14510        // scoping check is what keeps those pure.
14511        let src_ns = self.namespace_in_batch(src_key);
14512        let dst_ns = self.namespace_in_batch(dst_key);
14513        if src_ns != dst_ns {
14514            return Err(GraphError::CrossNamespace {
14515                src: src_key.to_string(),
14516                src_ns,
14517                dst: dst_key.to_string(),
14518                dst_ns,
14519            });
14520        }
14521        Ok(!self.has_edge(edge_type, src_key, dst_key))
14522    }
14523
14524    fn prepare_remove_prop(&self, key: &str, field: &str) -> Result<bool> {
14525        // A view owns its property, and the refusal has to live here rather
14526        // than on `GraphDb::remove_prop`: `BatchOp::RemoveProp` never meets
14527        // that one, and it is what the HTTP `DELETE /node/{key}/prop/{field}`
14528        // route, `Batch::remove_prop` and the CLI all submit. This is the one
14529        // choke-point every removal passes, exactly as it is for `ns` below.
14530        // Checked before the key, so the answer does not depend on whether the
14531        // key exists — which is also what `GraphDb::remove_prop` answered when
14532        // it carried the only copy of this guard.
14533        if let Some(view_name) = self.db.view_store.view_for_prop(field) {
14534            return Err(GraphError::ViewPropReadOnly {
14535                view_name: view_name.to_string(),
14536            });
14537        }
14538        self.check_live_key(key)?;
14539        // Removing `ns` is changing the namespace — to `default`, the namespace
14540        // an absent property names. It goes through this one choke-point and NOT
14541        // through `rewrite_wal_dense` (a `RemoveProp` needs no dense rewrite), so
14542        // the immutability rule has to be stated here as well. Without it the
14543        // node silently lands in `default` on the next open: the cross-namespace
14544        // edge guard is defeated and a default-bound role reads a tenant's node.
14545        if field == NS_PROP {
14546            let from = self.namespace_in_batch(key);
14547            if from != NS_DEFAULT {
14548                return Err(GraphError::NamespaceImmutable {
14549                    key: key.to_string(),
14550                    from,
14551                    to: NS_DEFAULT.to_string(),
14552                });
14553            }
14554            // Already in `default`: the removal changes no namespace. It is the
14555            // no-op `set_prop` to the current namespace is, not an error.
14556            return Ok(false);
14557        }
14558        Ok(self.has_prop(key, field))
14559    }
14560
14561    fn prepare_delete_edge(&self, edge_type: &str, src_key: &str, dst_key: &str) -> Result<bool> {
14562        for k in [src_key, dst_key] {
14563            if !self.has_key(k) {
14564                return Err(GraphError::KeyNotFound { key: k.into() });
14565            }
14566        }
14567        // Provenance-owned OR a live rule would derive this pair. User-first
14568        // edges that a later rule matches are not in `owned`, but deleting
14569        // them would leave a hole `rebuild_rule` immediately fills.
14570        if self.is_rule_owned(edge_type, src_key, dst_key) {
14571            return Err(GraphError::RuleOwned {
14572                detail: format!(
14573                    "edge {edge_type} {src_key}→{dst_key} is rule-owned; \
14574                     delete or change the owning rule"
14575                ),
14576            });
14577        }
14578        if self.would_derive(edge_type, src_key, dst_key) {
14579            return Err(GraphError::RuleOwned {
14580                detail: format!(
14581                    "edge {edge_type} {src_key}→{dst_key} is rule-owned; \
14582                     delete or change the owning rule, or a live rule would re-derive it"
14583                ),
14584            });
14585        }
14586        Ok(self.has_edge(edge_type, src_key, dst_key))
14587    }
14588
14589    /// True if any live rule (minus overlay-deleted names) would derive
14590    /// `(edge_type, src, dst)` from current overlay-visible props/labels.
14591    /// CreateRule names in `extra_rules` are ignored — same documented
14592    /// same-batch rule-window as [`Self::is_rule_owned`].
14593    ///
14594    /// Mirrored by `guard_matches` in `memory/forget.rs`, which names the
14595    /// rules this refused for: change the two together (ledger row 67).
14596    fn would_derive(&self, edge_type: &str, src_key: &str, dst_key: &str) -> bool {
14597        if src_key == dst_key {
14598            return false;
14599        }
14600        let Some(src_label) = self.label_of(src_key) else {
14601            return false;
14602        };
14603        let Some(dst_label) = self.label_of(dst_key) else {
14604            return false;
14605        };
14606        for rule in self.db.engine.rules() {
14607            if self.overlay.deleted_rules.contains(&rule.name) {
14608                continue;
14609            }
14610            if rule.edge_type != edge_type {
14611                continue;
14612            }
14613            if rule.src_label != src_label || rule.dst_label != dst_label {
14614                continue;
14615            }
14616            let src_props = |f: &str| self.prop_value(src_key, f);
14617            let dst_props = |f: &str| self.prop_value(dst_key, f);
14618            let src_view = NodeView {
14619                key: src_key,
14620                props: &src_props,
14621            };
14622            let dst_view = NodeView {
14623                key: dst_key,
14624                props: &dst_props,
14625            };
14626            if evaluate(&rule.predicate, &src_view, &dst_view).is_some() {
14627                return true;
14628            }
14629        }
14630        false
14631    }
14632
14633    fn label_of(&self, key: &str) -> Option<String> {
14634        if self.overlay.deleted_keys.contains(key) {
14635            return None;
14636        }
14637        // Fresh identities created in this batch have no stored label in the
14638        // overlay; they cannot be provenance-owned yet either.
14639        let id = self.db.ids.get(key)?;
14640        let sym = self.db.labels.get(id as usize).copied()?;
14641        if sym == u32::MAX {
14642            return None;
14643        }
14644        self.db.syms.resolve(sym).map(str::to_string)
14645    }
14646
14647    /// The label `key` carries as this batch sees it — including a node
14648    /// inserted earlier in the same batch, which the store does not have yet.
14649    fn label_in_batch(&self, key: &str) -> Option<String> {
14650        if self.overlay.deleted_keys.contains(key) {
14651            return None;
14652        }
14653        if let Some(label) = self.overlay.extra_labels.get(key) {
14654            return Some(label.clone());
14655        }
14656        self.label_of(key)
14657    }
14658
14659    /// The property writes that make `key`'s props exactly `props`, or why the
14660    /// row is refused.
14661    ///
14662    /// `Some(value)` is a set and `None` is a removal. `store_fields` is every
14663    /// field name the store knows, hoisted by the caller so a frame of N
14664    /// replaces reads the field list once rather than N times.
14665    ///
14666    /// The second half of the pair is how many view-owned fields this row kept
14667    /// rather than removed — the one part of "exactly the supplied props" that
14668    /// does not hold, and the caller's only signal that it did not.
14669    ///
14670    /// The refusals are row errors, not frame errors: a mirror rebuild should
14671    /// learn which of its rows disagree with the store without losing the rows
14672    /// that agree.
14673    fn plan_replace(
14674        &self,
14675        label: &str,
14676        key: &str,
14677        props: &[(String, Value)],
14678        store_fields: &[String],
14679    ) -> std::result::Result<ReplacePlan, String> {
14680        // A different label is a relabel, and a rebuild does not relabel: that
14681        // is `rename_node` or an explicit write, never a side effect here.
14682        let stored = self.label_in_batch(key).unwrap_or_default();
14683        if stored != label {
14684            return Err(format!(
14685                "node {key}: on_conflict=\"replace\" will not relabel {stored:?} to {label:?}; \
14686                 relabelling is rename_node or an explicit write"
14687            ));
14688        }
14689        // `ns` is immutable. Replace removes what the supplied props omit, so
14690        // an omitted `ns` is a move to `default` exactly as a different `ns` is
14691        // a move to that one; both are the same refusal.
14692        let from = self.namespace_in_batch(key);
14693        let to = match props.iter().find(|(field, _)| field == NS_PROP) {
14694            Some((_, Value::Str(ns))) => ns.clone(),
14695            Some((_, value)) => {
14696                return Err(format!(
14697                    "node {key}: {NS_PROP} must be a string naming a namespace, got {value:?}"
14698                ));
14699            }
14700            None => NS_DEFAULT.to_string(),
14701        };
14702        if to != from {
14703            return Err(format!(
14704                "node {key}: {NS_PROP} is immutable; on_conflict=\"replace\" cannot move it \
14705                 from {from:?} to {to:?}"
14706            ));
14707        }
14708
14709        if let Some(why) = self.supplied_view_owned_prop(key, props) {
14710            return Err(why);
14711        }
14712
14713        let supplied: BTreeSet<&str> = props.iter().map(|(field, _)| field.as_str()).collect();
14714        let mut writes = Vec::new();
14715        for (field, value) in props {
14716            // `ns` names the namespace the node is already in, so the write is
14717            // the no-op the dense-rewrite seam would drop anyway.
14718            if field == NS_PROP {
14719                continue;
14720            }
14721            // Already exactly this value: a rebuild of an unchanged row should
14722            // cost no WAL record.
14723            if self.prop_value(key, field).as_ref() == Some(value) {
14724                continue;
14725            }
14726            writes.push((field.clone(), Some(value.clone())));
14727        }
14728        // Everything the node still carries that the supplied props do not.
14729        // `ns` is never removed: it is immutable, and the check above has
14730        // already established the node stays where it is.
14731        let overlay_fields = self
14732            .overlay
14733            .extra_props
14734            .keys()
14735            .filter(|(k, _)| k == key)
14736            .map(|(_, field)| field.as_str());
14737        //
14738        // A view-owned field is filtered out rather than refused. It is not the
14739        // caller's to supply (supplying one is still the row error above) and
14740        // so it is not part of what "exactly the supplied ones" ranges over:
14741        // omitting it is not a request to delete it. Refusing here instead
14742        // would make `replace` impossible for every node a view has written to
14743        // — which on a store carrying a view is the whole population a mirror
14744        // rebuild has to cover.
14745        let omitted: BTreeSet<&str> = store_fields
14746            .iter()
14747            .map(String::as_str)
14748            .chain(overlay_fields)
14749            .filter(|field| {
14750                *field != NS_PROP && !supplied.contains(field) && self.has_prop(key, field)
14751            })
14752            .collect();
14753        // The view-owned half is kept, and counted: the row still commits and
14754        // still reports no error, so without this number a mirror rebuild is
14755        // told it got exactly what it asked for when it did not (defect #18).
14756        let (stale, kept): (Vec<&str>, Vec<&str>) = omitted
14757            .into_iter()
14758            .partition(|field| self.db.view_store.view_for_prop(field).is_none());
14759        writes.extend(stale.into_iter().map(|field| (field.to_string(), None)));
14760        Ok((writes, kept.len()))
14761    }
14762
14763    /// The row error a supplied view-owned field earns, or `None`.
14764    ///
14765    /// Shared by [`MutPreview::plan_replace`] and the no-conflict arm of
14766    /// `BatchOp::InsertNodeOnConflict` so that one op answers a supplied
14767    /// view-owned field the same way whether or not the key was already taken.
14768    fn supplied_view_owned_prop(&self, key: &str, props: &[(String, Value)]) -> Option<String> {
14769        props.iter().find_map(|(field, _)| {
14770            self.db.view_store.view_for_prop(field).map(|view_name| {
14771                format!(
14772                    "node {key}: property {field:?} is owned by view {view_name:?} and is \
14773                     read-only"
14774                )
14775            })
14776        })
14777    }
14778
14779    /// The namespace `key` is in as this batch sees it — including a node
14780    /// inserted earlier in the same batch, which the store does not have yet.
14781    fn namespace_in_batch(&self, key: &str) -> String {
14782        namespace_of_value(self.prop_value(key, NS_PROP).as_ref()).to_string()
14783    }
14784
14785    fn prop_value(&self, key: &str, field: &str) -> Option<Value> {
14786        if !self.has_key(key) {
14787            return None;
14788        }
14789        let k = (key.to_string(), field.to_string());
14790        if self.overlay.removed_props.contains(&k) {
14791            return None;
14792        }
14793        if let Some(v) = self.overlay.extra_props.get(&k) {
14794            return Some(v.clone());
14795        }
14796        if self.overlay.extra_keys.contains(key) {
14797            return None;
14798        }
14799        self.db.get_prop(key, field)
14800    }
14801
14802    fn check_create_rule(&self, def: &RuleDef) -> Result<()> {
14803        def.validate()
14804            .map_err(|e| GraphError::RuleInvalid { detail: e })?;
14805        if self.has_rule(&def.name) {
14806            return Err(GraphError::RuleInvalid {
14807                detail: format!("rule {:?} already exists", def.name),
14808            });
14809        }
14810        // Rule-chain cycle rejection. Derived edges feed via-hop rules, so a
14811        // rule set forms a graph whose arcs are "hops over `via_edge`, writes
14812        // `edge_type`". A cycle in that graph is a rule set that would re-fire
14813        // itself forever; the engine's depth cap would silently truncate it
14814        // instead, leaving an arbitrary partial result. Reject it here, the one
14815        // place that sees the whole rule set.
14816        //
14817        // Rules accepted earlier in the same batch count too: the overlay
14818        // carries their arcs, so a cycle cannot be assembled one op at a time.
14819        if let Some(via) = def.via_edge.as_deref() {
14820            if via == def.edge_type {
14821                return Err(GraphError::RuleInvalid {
14822                    detail: format!("rule chain cycle: {} -> {}", via, def.edge_type),
14823                });
14824            }
14825            let mut arcs: Vec<(String, String)> = self
14826                .db
14827                .engine
14828                .rules()
14829                .filter(|r| !self.overlay.deleted_rules.contains(&r.name))
14830                .filter_map(|r| r.via_edge.clone().map(|v| (v, r.edge_type.clone())))
14831                .collect();
14832            arcs.extend(self.overlay.extra_rule_arcs.values().cloned());
14833            arcs.push((via.to_string(), def.edge_type.clone()));
14834            if let Some(path) = find_cycle_through(&arcs, &def.edge_type, via) {
14835                return Err(GraphError::RuleInvalid {
14836                    detail: format!("rule chain cycle: {} -> {}", via, path.join(" -> ")),
14837                });
14838            }
14839        }
14840        Ok(())
14841    }
14842
14843    fn check_delete_rule(&self, name: &str) -> Result<()> {
14844        if self.has_rule(name) {
14845            Ok(())
14846        } else {
14847            Err(GraphError::RuleNotFound { name: name.into() })
14848        }
14849    }
14850
14851    fn note_insert_node(&mut self, label: &str, key: &str, props: &[(String, Value)]) {
14852        self.overlay.deleted_keys.remove(key);
14853        self.overlay.extra_keys.insert(key.to_string());
14854        self.overlay
14855            .extra_labels
14856            .insert(key.to_string(), label.to_string());
14857        self.overlay.extra_props.retain(|(k, _), _| k != key);
14858        self.overlay.removed_props.retain(|(k, _)| k != key);
14859        for (field, value) in props {
14860            self.overlay
14861                .extra_props
14862                .insert((key.to_string(), field.clone()), value.clone());
14863        }
14864    }
14865
14866    fn note_insert_edge(&mut self, edge_type: &str, src_key: &str, dst_key: &str) {
14867        let k = (
14868            edge_type.to_string(),
14869            src_key.to_string(),
14870            dst_key.to_string(),
14871        );
14872        self.overlay.deleted_edges.remove(&k);
14873        self.overlay.extra_edges.insert(k);
14874    }
14875
14876    fn note_set_prop(&mut self, key: &str, field: &str, value: &Value) {
14877        let k = (key.to_string(), field.to_string());
14878        self.overlay.removed_props.remove(&k);
14879        self.overlay.extra_props.insert(k, value.clone());
14880    }
14881
14882    fn note_remove_prop(&mut self, key: &str, field: &str) {
14883        let k = (key.to_string(), field.to_string());
14884        self.overlay.extra_props.remove(&k);
14885        self.overlay.removed_props.insert(k);
14886    }
14887
14888    fn note_delete_edge(&mut self, edge_type: &str, src_key: &str, dst_key: &str) {
14889        let k = (
14890            edge_type.to_string(),
14891            src_key.to_string(),
14892            dst_key.to_string(),
14893        );
14894        self.overlay.extra_edges.remove(&k);
14895        self.overlay.deleted_edges.insert(k);
14896    }
14897
14898    fn note_delete_node(&mut self, key: &str) {
14899        self.overlay.extra_keys.remove(key);
14900        self.overlay.deleted_keys.insert(key.to_string());
14901        self.overlay.extra_props.retain(|(k, _), _| k != key);
14902        self.overlay.removed_props.retain(|(k, _)| k != key);
14903        self.overlay
14904            .extra_edges
14905            .retain(|(_, s, d)| s != key && d != key);
14906        self.overlay
14907            .deleted_edges
14908            .retain(|(_, s, d)| s != key && d != key);
14909    }
14910
14911    fn note_create_rule(&mut self, def: &RuleDef) {
14912        self.overlay.deleted_rules.remove(&def.name);
14913        self.overlay.extra_rules.insert(def.name.clone());
14914        // Rules accepted earlier in this batch are not in the engine yet, so
14915        // the cycle check would not see their arcs. Keep the arc, not just the
14916        // name, so a batch cannot smuggle in a cycle one op at a time.
14917        if let Some(via) = def.via_edge.clone() {
14918            self.overlay
14919                .extra_rule_arcs
14920                .insert(def.name.clone(), (via, def.edge_type.clone()));
14921        }
14922    }
14923
14924    fn check_rename_node(&self, old: &str, new: &str) -> Result<()> {
14925        if !self.has_key(old) {
14926            return Err(GraphError::KeyNotFound { key: old.into() });
14927        }
14928        if self.has_key(new) {
14929            return Err(GraphError::DuplicateKey { key: new.into() });
14930        }
14931        Ok(())
14932    }
14933
14934    fn note_rename_node(&mut self, old: &str, new: &str) {
14935        // Mark old as deleted so subsequent batch ops cannot reference it.
14936        self.overlay.extra_keys.remove(old);
14937        self.overlay.deleted_keys.insert(old.to_string());
14938        // Mark new as extra so subsequent batch ops can reference it.
14939        self.overlay.deleted_keys.remove(new);
14940        self.overlay.extra_keys.insert(new.to_string());
14941        // Migrate any overlay props from old key to new key.
14942        let new_str = new.to_string();
14943        let transferred: Vec<((String, String), Value)> = self
14944            .overlay
14945            .extra_props
14946            .iter()
14947            .filter(|((k, _), _)| k.as_str() == old)
14948            .map(|((_, f), v)| ((new_str.clone(), f.clone()), v.clone()))
14949            .collect();
14950        self.overlay
14951            .extra_props
14952            .retain(|(k, _), _| k.as_str() != old);
14953        for (k, v) in transferred {
14954            self.overlay.extra_props.insert(k, v);
14955        }
14956        // Migrate removed_props.
14957        let transferred_removed: Vec<(String, String)> = self
14958            .overlay
14959            .removed_props
14960            .iter()
14961            .filter(|(k, _)| k.as_str() == old)
14962            .map(|(_, f)| (new_str.clone(), f.clone()))
14963            .collect();
14964        self.overlay
14965            .removed_props
14966            .retain(|(k, _)| k.as_str() != old);
14967        for k in transferred_removed {
14968            self.overlay.removed_props.insert(k);
14969        }
14970    }
14971
14972    fn note_delete_rule(&mut self, name: &str) {
14973        self.overlay.extra_rules.remove(name);
14974        // Drop its chain arc too: a rule created and then deleted in the same
14975        // batch must not make a later, legal rule look like a cycle.
14976        self.overlay.extra_rule_arcs.remove(name);
14977        self.overlay.deleted_rules.insert(name.to_string());
14978        // Treat the deleted rule's current provenance as gone so a later
14979        // delete_edge of those triples is a no-op (matches sequential).
14980        if let Some(triples) = self.db.engine.provenance().get(name) {
14981            for &(et, s, d) in triples {
14982                let Some(etype) = self.db.syms.resolve(et) else {
14983                    continue;
14984                };
14985                let Some(src) = self.db.ids.key_of(s) else {
14986                    continue;
14987                };
14988                let Some(dst) = self.db.ids.key_of(d) else {
14989                    continue;
14990                };
14991                let k = (etype.to_string(), src.to_string(), dst.to_string());
14992                self.overlay.extra_edges.remove(&k);
14993                self.overlay.deleted_edges.insert(k);
14994            }
14995        }
14996    }
14997}
14998
14999/// Collects mutations and commits them as one WAL `Batch` frame.
15000///
15001/// Holds `&mut GraphDb` for its lifetime. Queue with the same method names
15002/// as [`GraphDb`]; call [`commit`](Self::commit) to validate, log, and apply.
15003/// See [`GraphDb::batch`] for validation and atomicity rules.
15004pub struct BatchBuilder<'a, F: Fs> {
15005    db: &'a mut GraphDb<F>,
15006    ops: Vec<BatchOp>,
15007}
15008
15009impl<'a, F: Fs> BatchBuilder<'a, F> {
15010    pub fn insert_node(
15011        &mut self,
15012        label: &str,
15013        key: &str,
15014        props: Vec<(String, Value)>,
15015    ) -> &mut Self {
15016        self.ops.push(BatchOp::InsertNode {
15017            label: label.into(),
15018            key: key.into(),
15019            props,
15020        });
15021        self
15022    }
15023
15024    /// Queue a node insert whose answer to a taken key is `on_conflict`.
15025    ///
15026    /// [`OnConflict::Error`] queues exactly the op [`insert_node`](Self::insert_node)
15027    /// does, so the default path is unchanged.
15028    pub fn insert_node_on_conflict(
15029        &mut self,
15030        label: &str,
15031        key: &str,
15032        props: Vec<(String, Value)>,
15033        on_conflict: OnConflict,
15034    ) -> &mut Self {
15035        self.ops.push(match on_conflict {
15036            OnConflict::Error => BatchOp::InsertNode {
15037                label: label.into(),
15038                key: key.into(),
15039                props,
15040            },
15041            on_conflict => BatchOp::InsertNodeOnConflict {
15042                label: label.into(),
15043                key: key.into(),
15044                props,
15045                on_conflict,
15046            },
15047        });
15048        self
15049    }
15050
15051    pub fn insert_edge(&mut self, edge_type: &str, src_key: &str, dst_key: &str) -> &mut Self {
15052        self.ops.push(BatchOp::InsertEdge {
15053            edge_type: edge_type.into(),
15054            src_key: src_key.into(),
15055            dst_key: dst_key.into(),
15056        });
15057        self
15058    }
15059
15060    pub fn set_prop(&mut self, key: &str, field: &str, value: Value) -> &mut Self {
15061        self.ops.push(BatchOp::SetProp {
15062            key: key.into(),
15063            field: field.into(),
15064            value,
15065        });
15066        self
15067    }
15068
15069    pub fn remove_prop(&mut self, key: &str, field: &str) -> &mut Self {
15070        self.ops.push(BatchOp::RemoveProp {
15071            key: key.into(),
15072            field: field.into(),
15073        });
15074        self
15075    }
15076
15077    pub fn delete_edge(&mut self, edge_type: &str, src_key: &str, dst_key: &str) -> &mut Self {
15078        self.ops.push(BatchOp::DeleteEdge {
15079            edge_type: edge_type.into(),
15080            src_key: src_key.into(),
15081            dst_key: dst_key.into(),
15082        });
15083        self
15084    }
15085
15086    pub fn delete_node(&mut self, key: &str) -> &mut Self {
15087        self.ops.push(BatchOp::DeleteNode { key: key.into() });
15088        self
15089    }
15090
15091    pub fn create_rule(&mut self, def: RuleDef) -> &mut Self {
15092        self.ops.push(BatchOp::CreateRule(def));
15093        self
15094    }
15095
15096    pub fn delete_rule(&mut self, name: &str) -> &mut Self {
15097        self.ops.push(BatchOp::DeleteRule { name: name.into() });
15098        self
15099    }
15100
15101    /// Queue a node-rename in this batch.
15102    ///
15103    /// Validation (old exists, new not taken) runs at commit time.
15104    pub fn rename_node(&mut self, old_key: &str, new_key: &str) -> &mut Self {
15105        self.ops.push(BatchOp::RenameNode {
15106            old_key: old_key.into(),
15107            new_key: new_key.into(),
15108        });
15109        self
15110    }
15111
15112    /// Queue an edge insert with endpoint auto-creation.
15113    ///
15114    /// Any missing endpoint is created as a plain node `{key, label:
15115    /// placeholder_label, no props}` inside this batch frame. Rules fire and
15116    /// last-change is updated for each auto-created node.
15117    pub fn insert_edge_upsert(
15118        &mut self,
15119        edge_type: &str,
15120        src_key: &str,
15121        dst_key: &str,
15122        placeholder_label: &str,
15123    ) -> &mut Self {
15124        self.ops.push(BatchOp::InsertEdgeUpsert {
15125            edge_type: edge_type.into(),
15126            src_key: src_key.into(),
15127            dst_key: dst_key.into(),
15128            placeholder_label: placeholder_label.into(),
15129        });
15130        self
15131    }
15132
15133    /// Validate every queued op, then log one `Batch` frame and apply.
15134    /// Empty / all-noop batches return `Ok(())` without writing the WAL.
15135    /// A second `commit()` after a successful one is an empty-batch no-op
15136    /// (queued ops were taken).
15137    /// Takes `&mut self` so it chains after the queue methods (`b.insert_node(..).commit()`)
15138    /// and also works as `let mut b = db.batch(); b.insert_node(..); b.commit()`.
15139    ///
15140    /// **Rule-window limitation:** batch validation cannot see edges that a
15141    /// rule created earlier in the *same* batch will derive at apply time, so
15142    /// a `delete_edge` / `insert_edge` in that window is silently no-oped
15143    /// where sequential calls would return `Err(RuleOwned)`. State integrity
15144    /// is unaffected (idempotent apply, provenance intact). Create rules in
15145    /// their own batch, or sequentially, when later ops may touch derived
15146    /// edges.
15147    /// Validate every queued op and commit atomically.
15148    ///
15149    /// Returns `(nodes_inserted, edges_inserted)` — the counts of node and edge
15150    /// WAL records actually written (duplicate edges are silent no-ops and are
15151    /// NOT counted). Both are 0 when the batch is empty or all-noop.
15152    pub fn commit(&mut self) -> Result<(usize, usize)> {
15153        let ops = std::mem::take(&mut self.ops);
15154        self.db.commit_batch(ops)
15155    }
15156
15157    /// [`commit`](Self::commit) with the full [`BatchOutcome`] — the counts a
15158    /// caller needs when its rows carry an [`OnConflict`] policy.
15159    pub fn commit_outcome(&mut self) -> Result<BatchOutcome> {
15160        let ops = std::mem::take(&mut self.ops);
15161        self.db.commit_logged_batch(ops, None, None)
15162    }
15163
15164    /// Same as [`commit`](Self::commit) but tail the inner events with
15165    /// [`MutationEvent::Ingested`] instead of [`MutationEvent::BatchApplied`].
15166    pub(crate) fn commit_ingest(&mut self, label: &str, inserted: usize) -> Result<(usize, usize)> {
15167        let ops = std::mem::take(&mut self.ops);
15168        self.db
15169            .commit_logged_batch(ops, Some((label.to_string(), inserted)), None)
15170            .map(inserted_pair)
15171    }
15172}
15173
15174pub struct NodeRef<'a, F: Fs> {
15175    db: &'a GraphDb<F>,
15176    id: u32,
15177}
15178
15179impl<'a, F: Fs> NodeRef<'a, F> {
15180    pub fn key(&self) -> &str {
15181        self.db.ids.key_of(self.id).expect("dense ids")
15182    }
15183
15184    pub fn label(&self) -> &str {
15185        let sym = self
15186            .db
15187            .labels
15188            .get(self.id as usize)
15189            .copied()
15190            .filter(|&s| s != u32::MAX)
15191            .expect("real nodes always have a label; u32::MAX sentinel cannot occur");
15192        self.db.syms.resolve(sym).expect("interned label symbol")
15193    }
15194
15195    pub fn prop(&self, field: &str) -> Option<Value> {
15196        self.db
15197            .props_view()
15198            .get(self.id, field)
15199            .map(|vr| vr.into_value())
15200    }
15201
15202    /// All stored fields for this node, sorted by field name.
15203    ///
15204    /// Reads from the full base+overlay view so that props stored only in the
15205    /// V8 snapshot base (i.e. before any post-snapshot WAL writes) are visible.
15206    pub fn props(&self) -> BTreeMap<String, Value> {
15207        let mut out = BTreeMap::new();
15208        let pv = self.db.props_view();
15209        for field in pv.field_names() {
15210            if let Some(vr) = pv.get(self.id, &field) {
15211                out.insert(field, vr.into_value());
15212            }
15213        }
15214        out
15215    }
15216
15217    /// depth-N BFS as a ResultSet: columns ["key","label","depth"], BFS order.
15218    pub fn neighborhood(&self, depth: u32, edge_types: Option<&[&str]>, dir: Dir) -> ResultSet {
15219        let view = self.db.view();
15220        let resolved: Option<Vec<u32>> = edge_types.map(|names| {
15221            names
15222                .iter()
15223                .filter_map(|name| view.syms.get(name))
15224                .collect()
15225        });
15226        let nb = neighborhood(&view, self.id, depth, resolved.as_deref(), dir);
15227        let mut rs = ResultSet::new(vec!["key".into(), "label".into(), "depth".into()]);
15228        for (nid, d) in nb.nodes {
15229            let key = view.key_of(nid);
15230            let label = view
15231                .label_of(nid)
15232                .expect("real nodes always have a label; u32::MAX sentinel cannot occur");
15233            rs.push_row(vec![
15234                Some(Value::Str(key.to_string())),
15235                Some(Value::Str(label.to_string())),
15236                Some(Value::Int(d as i64)),
15237            ]);
15238        }
15239        rs
15240    }
15241
15242    /// 1-hop, Both directions: edge-type name → sorted unique neighbor keys.
15243    pub fn grouped_by_edge_type(&self) -> BTreeMap<String, Vec<String>> {
15244        let view = self.db.view();
15245        let mut groups: BTreeMap<String, BTreeSet<String>> = BTreeMap::new();
15246        for e in expand(&view, self.id, None, Dir::Both) {
15247            // Skip edges with unknown etypes (only possible from corrupt large
15248            // TOPOLOGY section; function returns BTreeMap not Result).
15249            let Some(etype) = view.syms.resolve(e.etype) else {
15250                continue;
15251            };
15252            let etype = etype.to_string();
15253            let nbr = if e.src == self.id { e.dst } else { e.src };
15254            groups
15255                .entry(etype)
15256                .or_default()
15257                .insert(view.key_of(nbr).to_string());
15258        }
15259        groups
15260            .into_iter()
15261            .map(|(k, v)| (k, v.into_iter().collect()))
15262            .collect()
15263    }
15264}
15265
15266#[cfg(test)]
15267mod tests {
15268    use super::*;
15269    use core_rules::Predicate;
15270
15271    fn tmp_dir(name: &str) -> std::path::PathBuf {
15272        let d =
15273            std::env::temp_dir().join(format!("graphdb-db-unit-{}-{}", name, std::process::id()));
15274        let _ = std::fs::remove_dir_all(&d);
15275        d
15276    }
15277
15278    fn fk_rule() -> RuleDef {
15279        RuleDef {
15280            name: "works_at".into(),
15281            src_label: "Person".into(),
15282            dst_label: "Org".into(),
15283            predicate: Predicate::KeyMatch {
15284                field: "org_id".into(),
15285            },
15286            edge_type: "WORKS_AT".into(),
15287            weight_prop: None,
15288            max_edges: None,
15289            approximate: false,
15290            via_label: None,
15291            via_edge: None,
15292            via_dir: None,
15293            namespace: None,
15294        }
15295    }
15296
15297    /// Regression guard for the no-views delta-copy fast path.
15298    ///
15299    /// When no views are defined, `pending_deltas_since().to_vec()` must never
15300    /// be called — even during a large CreateRule backfill. The DELTA_COPY_COUNT
15301    /// thread-local is incremented inside every `if !view_store.is_empty()` block;
15302    /// a count of 0 after the entire sequence proves the guard fires correctly.
15303    #[test]
15304    fn no_delta_copy_when_no_views() {
15305        DELTA_COPY_COUNT.with(|c| c.set(0));
15306        let dir = tmp_dir("no-delta-copy");
15307        {
15308            let mut db = GraphDb::open(&dir).unwrap();
15309            // Insert 50 Org + 50 Person nodes with FK links.
15310            for i in 0..50u32 {
15311                db.insert_node("Org", &format!("o{i}"), vec![]).unwrap();
15312            }
15313            for i in 0..50u32 {
15314                db.insert_node(
15315                    "Person",
15316                    &format!("p{i}"),
15317                    vec![("org_id".into(), Value::Str(format!("o{i}")))],
15318                )
15319                .unwrap();
15320            }
15321            // CreateRule backfill should NOT invoke to_vec() when no views are defined.
15322            db.create_rule(fk_rule()).unwrap();
15323
15324            // Counter must stay 0 — no views, no copies.
15325            let copies = DELTA_COPY_COUNT.with(|c| c.get());
15326            assert_eq!(
15327                copies, 0,
15328                "pending_deltas_since().to_vec() called despite no views"
15329            );
15330
15331            // Derived edges must still be correct (the guard skips only the
15332            // empty delta propagation loop, not the rule application itself).
15333            let nbrs = db.neighbors("p0", "WORKS_AT", Direction::Out).unwrap();
15334            assert_eq!(
15335                nbrs,
15336                vec!["o0"],
15337                "rule must derive edges even with no views"
15338            );
15339        }
15340        let _ = std::fs::remove_dir_all(&dir);
15341    }
15342
15343    /// Gating regression: subscribe AFTER a backfill must see no stale events.
15344    /// subscribe BEFORE a backfill must see every edge-fire event.
15345    #[test]
15346    fn subscribe_after_backfill_no_stale_events() {
15347        let dir = tmp_dir("sub-after-backfill");
15348        {
15349            let mut db = GraphDb::open(&dir).unwrap();
15350            for i in 0..10u32 {
15351                db.insert_node("Org", &format!("o{i}"), vec![]).unwrap();
15352                db.insert_node(
15353                    "Person",
15354                    &format!("p{i}"),
15355                    vec![("org_id".into(), Value::Str(format!("o{i}")))],
15356                )
15357                .unwrap();
15358            }
15359            // Create rule BEFORE subscribing — emit_deltas is false during backfill.
15360            db.create_rule(fk_rule()).unwrap();
15361
15362            // Subscribe AFTER the backfill — queue must be empty (no stale events).
15363            let sub = db.subscribe_all_rules().unwrap();
15364            // No events should have queued for the prior backfill.
15365            assert!(
15366                sub.try_recv().is_none(),
15367                "subscribe after backfill must see no stale events"
15368            );
15369
15370            // Inserting a new node now should fire an event (emit_deltas is now true).
15371            db.insert_node("Org", "o_new", vec![]).unwrap();
15372            db.insert_node(
15373                "Person",
15374                "p_new",
15375                vec![("org_id".into(), Value::Str("o_new".into()))],
15376            )
15377            .unwrap();
15378            let ev = sub.recv_timeout(std::time::Duration::from_millis(200));
15379            assert!(
15380                ev.is_some(),
15381                "edge-fire event must arrive after subscribe (emit_deltas=true)"
15382            );
15383        }
15384        let _ = std::fs::remove_dir_all(&dir);
15385    }
15386
15387    /// Gating regression: subscribe BEFORE a backfill → events flow.
15388    #[test]
15389    fn subscribe_before_backfill_events_flow() {
15390        let dir = tmp_dir("sub-before-backfill");
15391        {
15392            let mut db = GraphDb::open(&dir).unwrap();
15393            // Subscribe FIRST — emit_deltas becomes true.
15394            let sub = db.subscribe_all_rules().unwrap();
15395
15396            for i in 0..5u32 {
15397                db.insert_node("Org", &format!("o{i}"), vec![]).unwrap();
15398                db.insert_node(
15399                    "Person",
15400                    &format!("p{i}"),
15401                    vec![("org_id".into(), Value::Str(format!("o{i}")))],
15402                )
15403                .unwrap();
15404            }
15405            // Backfill fires with emit_deltas=true → events queued.
15406            db.create_rule(fk_rule()).unwrap();
15407
15408            // Should receive at least one edge-fired event from the backfill.
15409            let mut received = 0usize;
15410            while sub.try_recv().is_some() {
15411                received += 1;
15412            }
15413            assert!(
15414                received > 0,
15415                "subscribe before backfill must receive edge-fire events (got 0)"
15416            );
15417        }
15418        let _ = std::fs::remove_dir_all(&dir);
15419    }
15420
15421    /// Companion: when a view IS defined, the delta path fires and view values update.
15422    #[test]
15423    fn delta_copy_fires_when_view_exists() {
15424        use core_rules::ViewSource;
15425        DELTA_COPY_COUNT.with(|c| c.set(0));
15426        let dir = tmp_dir("delta-copy-with-view");
15427        {
15428            let mut db = GraphDb::open(&dir).unwrap();
15429            db.insert_node("Org", "o1", vec![]).unwrap();
15430            db.insert_node(
15431                "Person",
15432                "p1",
15433                vec![("org_id".into(), Value::Str("o1".into()))],
15434            )
15435            .unwrap();
15436            // Declare a Degree view so is_empty() returns false.
15437            db.create_view(ViewDef {
15438                name: "degree_out".into(),
15439                label: "Person".into(),
15440                view_prop: "degree_out".into(),
15441                source: ViewSource::Degree {
15442                    edge_type: "WORKS_AT".into(),
15443                    direction: Direction::Out,
15444                },
15445            })
15446            .unwrap();
15447            db.create_rule(fk_rule()).unwrap();
15448
15449            // At least one delta copy should have happened (CreateRule backfill).
15450            let copies = DELTA_COPY_COUNT.with(|c| c.get());
15451            assert!(
15452                copies > 0,
15453                "expected delta copy to fire when a view is defined"
15454            );
15455
15456            // View value should be computed: p1 has one WORKS_AT out-edge.
15457            let info = db.node_info("p1").unwrap();
15458            let degree = info.props.get("degree_out");
15459            assert!(
15460                degree.is_some(),
15461                "view prop should be written to node props"
15462            );
15463        }
15464        let _ = std::fs::remove_dir_all(&dir);
15465    }
15466
15467    /// Regression: `open_at_with` must call `rebuild_all` after WAL replay so
15468    /// derived-edge-driven view values reflect the as-of state rather than just
15469    /// the initial backfill written at `CreateView` time.
15470    ///
15471    /// Base WAL frames (indices 0..=5 before history markers):
15472    ///   0: insert Org "o1"
15473    ///   1: create_view "employee_count" (Degree / WORKS_AT / In) on Org
15474    ///   2: create_rule fk_rule (WORKS_AT, Person→Org via org_id)
15475    ///   3: insert Person "p1" → rule fires WORKS_AT p1→o1 (degree = 1)  ← mid
15476    ///   4: insert Person "p2" → rule fires WORKS_AT p2→o1 (degree = 2)
15477    ///   5: insert Person "p3" → rule fires WORKS_AT p3→o1 (degree = 3)  ← latest
15478    ///
15479    /// Each rule-fire also appends a DerivedEdgeAdded history-marker frame (state
15480    /// no-op), so the total commit count is higher than the base frame count.
15481    /// The "latest" open_at commit is computed dynamically via `wal_commit_count_at`.
15482    ///
15483    /// Without `rebuild_all`, the as-of instance's "emp" view stays at the
15484    /// initial backfill value (0) instead of reflecting the replayed derived edges.
15485    #[test]
15486    fn open_at_derived_edge_view_values_correct() {
15487        use core_rules::ViewSource;
15488        let dir = tmp_dir("open-at-view-rebuild");
15489        {
15490            let mut db = GraphDb::open(&dir).unwrap();
15491            // frame 0
15492            db.insert_node("Org", "o1", vec![]).unwrap();
15493            // frame 1: create view — initial backfill sees 0 derived edges (none fired yet)
15494            db.create_view(ViewDef {
15495                name: "employee_count".into(),
15496                label: "Org".into(),
15497                view_prop: "emp".into(),
15498                source: ViewSource::Degree {
15499                    edge_type: "WORKS_AT".into(),
15500                    direction: Direction::In,
15501                },
15502            })
15503            .unwrap();
15504            // frame 2: create rule — no Persons yet; backfill is a no-op
15505            db.create_rule(fk_rule()).unwrap();
15506            // frame 3: p1 — rule fires WORKS_AT p1→o1; degree = 1
15507            db.insert_node(
15508                "Person",
15509                "p1",
15510                vec![("org_id".into(), Value::Str("o1".into()))],
15511            )
15512            .unwrap();
15513            // frame 4: p2 — degree = 2
15514            db.insert_node(
15515                "Person",
15516                "p2",
15517                vec![("org_id".into(), Value::Str("o1".into()))],
15518            )
15519            .unwrap();
15520            // frame 5: p3 — degree = 3
15521            db.insert_node(
15522                "Person",
15523                "p3",
15524                vec![("org_id".into(), Value::Str("o1".into()))],
15525            )
15526            .unwrap();
15527            // Sanity: normal open sees degree = 3.
15528            assert_eq!(
15529                db.get_view_prop("o1", "emp"),
15530                Some(Value::Int(3)),
15531                "normal db must show degree 3 after 3 derived edges"
15532            );
15533        } // WAL flushed
15534
15535        // Re-open normally to get the authoritative reference value.
15536        let normal_db = GraphDb::open(&dir).unwrap();
15537        let normal_emp = normal_db.get_view_prop("o1", "emp");
15538        assert_eq!(
15539            normal_emp,
15540            Some(Value::Int(3)),
15541            "re-opened normal db must show degree 3"
15542        );
15543
15544        // Latest as-of (last WAL commit): must match the normal open.
15545        // History-marker frames are appended after each rule-fire, so the total
15546        // commit count is computed dynamically rather than hardcoded.
15547        let total = crate::wal_commit_count_at(&dir).unwrap();
15548        let aof_latest = GraphDb::open_at(&dir, total - 1).unwrap();
15549        assert_eq!(
15550            aof_latest.get_view_prop("o1", "emp"),
15551            normal_emp,
15552            "open_at latest: derived-edge view must equal normal open (rebuild_all required)"
15553        );
15554
15555        // Mid-history as-of (commit 3 = p1 insert Batch frame): only p1; degree = 1.
15556        // The DerivedEdgeAdded marker for p1 is at frame 4 (state no-op on replay),
15557        // so replaying 0..=3 correctly re-derives only the p1→o1 edge.
15558        let aof_mid = GraphDb::open_at(&dir, 3).unwrap();
15559        assert_eq!(
15560            aof_mid.get_view_prop("o1", "emp"),
15561            Some(Value::Int(1)),
15562            "open_at mid-history: only p1 exists at frame 3, degree must be 1"
15563        );
15564
15565        let _ = std::fs::remove_dir_all(&dir);
15566    }
15567
15568    /// Pin: subscribe_* on an as-of instance must return Err(ReadOnly) —
15569    /// as-of instances never commit, so distribute_events never runs and any
15570    /// subscription would wait forever.
15571    #[test]
15572    fn subscribe_on_as_of_returns_read_only_error() {
15573        let dir = tmp_dir("sub-as-of-read-only");
15574        {
15575            let mut db = GraphDb::open(&dir).unwrap();
15576            db.insert_node("Org", "o1", vec![]).unwrap();
15577            db.create_rule(fk_rule()).unwrap();
15578        }
15579        let mut aof = GraphDb::open_at(&dir, 0).unwrap();
15580
15581        assert!(
15582            matches!(
15583                aof.subscribe_all_rules(),
15584                Err(core_storage::GraphError::ReadOnly)
15585            ),
15586            "subscribe_all_rules on as-of must return ReadOnly"
15587        );
15588        assert!(
15589            matches!(
15590                aof.subscribe_writes(),
15591                Err(core_storage::GraphError::ReadOnly)
15592            ),
15593            "subscribe_writes on as-of must return ReadOnly"
15594        );
15595        assert!(
15596            matches!(
15597                aof.subscribe_rule("works_at"),
15598                Err(core_storage::GraphError::ReadOnly)
15599            ),
15600            "subscribe_rule on as-of must return ReadOnly"
15601        );
15602        let _ = std::fs::remove_dir_all(&dir);
15603    }
15604
15605    /// Regression: a failed dense WAL rewrite must not leave speculative
15606    /// interns in `syms`. If it does, the next successful mutation logs an
15607    /// `Intern` record with an inflated id; replay (which never saw the
15608    /// orphans) assigns a smaller id and the WAL becomes unreplayable.
15609    #[test]
15610    fn dense_rewrite_error_rolls_back_speculative_interns() {
15611        let dir = tmp_dir("dense-rewrite-rollback");
15612        {
15613            let mut db = GraphDb::open(&dir).unwrap();
15614            db.insert_node("Person", "a", vec![]).unwrap();
15615
15616            // Bypass MutPreview validation to hit the rewrite's own error path
15617            // (same shape as an id-exhaustion failure mid-rewrite). The
15618            // InsertEdge arm interns the edge type before it resolves keys.
15619            let err = db.rewrite_wal_dense(vec![WalRecord::InsertEdge {
15620                edge_type: "ORPHAN_TYPE".into(),
15621                src_key: "missing".into(),
15622                dst_key: "a".into(),
15623            }]);
15624            assert!(err.is_err(), "rewrite of a missing src key must fail");
15625            assert_eq!(
15626                db.syms.get("ORPHAN_TYPE"),
15627                None,
15628                "failed rewrite must roll back speculative interns"
15629            );
15630
15631            // A later successful mutation must produce a replayable WAL.
15632            db.set_prop("a", "later_field", Value::Int(2)).unwrap();
15633        }
15634        let db = GraphDb::open(&dir).expect("WAL must replay after failed rewrite");
15635        assert_eq!(db.get_prop("a", "later_field"), Some(Value::Int(2)));
15636        let _ = std::fs::remove_dir_all(&dir);
15637    }
15638}