Skip to main content

core_api/
db.rs

1use crate::ingest::{IngestOptions, IngestReport};
2use crate::roles::{PropPredicate, RoleDef, RolesFile, WriteScope};
3use crate::subscription::{
4    event_matches, DbEvent, SubEntry, SubFilter, SubInner, Subscription, DEFAULT_SUB_CAPACITY,
5};
6use core_query::cypher::ast::{ret_val_label, ArithOp};
7use core_query::cypher::{
8    execute, execute_union, is_subscribable, is_write_tokens, lex, parse, parse_read, parse_write,
9    plan, Expr, MatchDeleteNodeStmt, NodePat, Operand, Params, Pattern, PlanOp, Query, RetItem,
10    RetVal, WriteStatement,
11};
12use core_query::{eval_cmp, eval_filter, expand, neighborhood, Dir, Filter, GraphView, ResultSet};
13use core_rules::{
14    decode_rule_def, ef_max, evaluate, BuildProgress, EngineEdgeDelta, GraphMut, NodeView,
15    Predicate, RuleDef, RuleEngine, ViewDef, ViewStore,
16};
17use core_storage::fs::{FileId, Fs, FsIntrospect, RealFs};
18use core_storage::fulltext::FulltextIndex;
19use core_storage::property_index::PropertyIndex;
20use core_storage::v8::encode::{
21    archived_hnsw_to_owned, archived_rules_meta_to_owned, archived_to_idmap, archived_to_interner,
22    archived_views_to_owned, decode_last_change_bytes, decode_meta, encode_v8, V8Meta,
23};
24use core_storage::v8::seam::TopologyView;
25use core_storage::wal::{decode_all, encode_record, WalRecord};
26use core_storage::EdgePropsView;
27use core_storage::{
28    namespace_of_value, ColumnStore, Direction, EdgeProps, GraphError, IdMap, Interner, Result,
29    Topology, Value,
30};
31pub use core_storage::{valid_namespace, NS_DEFAULT, NS_MAX_LEN, NS_PROP};
32
33/// Index of [`NS_DEFAULT`] in `GraphDb::ns_names` — always zero, so the
34/// open-time pass over a store with no `ns` column fills `node_ns` with one
35/// constant and allocates no names.
36const NS_DEFAULT_IDX: u32 = 0;
37
38/// The reserved edge property holding a pair's insert count (§5.13).
39///
40/// Absent means 1 — the count is written only from the second insert of a
41/// triple onward, and only on a store that called
42/// [`GraphDb::enable_multiplicity`]. The engine owns the name: Cypher `SET` on
43/// it is refused, as the other reserved names are.
44pub const EDGE_COUNT_PROP: &str = "count";
45use serde::{Deserialize, Serialize};
46use std::collections::{BTreeMap, BTreeSet, HashMap, HashSet};
47use std::sync::Arc;
48
49/// Print a timing checkpoint when MUSHROOMDB_TRACE_OPEN is set.
50/// Zero-cost when the env var is absent (the var check is O(1) after first call).
51macro_rules! trace_open {
52    ($phase:literal, $t:expr) => {
53        if std::env::var("MUSHROOMDB_TRACE_OPEN").is_ok() {
54            eprintln!(
55                "[MUSHROOMDB_TRACE_OPEN] {:40} {:>9.3?}",
56                $phase,
57                $t.elapsed()
58            );
59        }
60    };
61}
62
63/// Print a migration phase checkpoint when MUSHROOMDB_TRACE_MIGRATE is set.
64/// Zero-cost when the env var is absent (the var check is O(1) after first call).
65macro_rules! trace_migrate {
66    ($phase:literal, $t:expr) => {
67        if std::env::var("MUSHROOMDB_TRACE_MIGRATE").is_ok() {
68            eprintln!(
69                "[MUSHROOMDB_TRACE_MIGRATE] {:40} {:>9.3?}",
70                $phase,
71                $t.elapsed()
72            );
73        }
74    };
75}
76
77// Test-only: counts how many times `pending_deltas_since().to_vec()` actually
78// executes (i.e., at least one view is defined). Used to verify the fast-path
79// guard skips the allocation when `view_store.is_empty()`.
80#[cfg(test)]
81thread_local! {
82    static DELTA_COPY_COUNT: std::cell::Cell<usize> = const { std::cell::Cell::new(0) };
83}
84
85// Per-thread count of query-subscription `execute` calls in `distribute_events`.
86//
87// Incremented each time a query subscription actually runs its plan (i.e.,
88// the label-skip fast-path did not fire). Because `distribute_events` is
89// called synchronously on the writer thread, this thread-local correctly
90// isolates each test thread's count even when integration tests run in
91// parallel. Read via [`query_sub_exec_count`].
92thread_local! {
93    static QUERY_SUB_EXECS_TL: std::cell::Cell<usize> = const { std::cell::Cell::new(0) };
94}
95
96/// Return the number of query-subscription re-executions logged on this
97/// thread since the process started (or since last reset via
98/// [`reset_query_sub_exec_count`]).
99///
100/// Primarily for integration tests that verify the label-skip fast-path.
101#[doc(hidden)]
102pub fn query_sub_exec_count() -> usize {
103    QUERY_SUB_EXECS_TL.with(|c| c.get())
104}
105
106/// Reset the per-thread query-subscription execution counter to zero.
107#[doc(hidden)]
108pub fn reset_query_sub_exec_count() {
109    QUERY_SUB_EXECS_TL.with(|c| c.set(0));
110}
111
112// Exact-versus-approximate warnings emitted on this thread. Thread-local for
113// the same reason [`QUERY_SUB_EXECS_TL`] is: integration tests run in parallel
114// and each gets its own thread, so a neighbour's masked search cannot be
115// mistaken for this test's.
116thread_local! {
117    static AMBIGUOUS_EXACTNESS_WARNS: std::cell::Cell<u64> = const { std::cell::Cell::new(0) };
118    static AMBIGUOUS_EXACTNESS_LAST: std::cell::RefCell<Option<String>> =
119        const { std::cell::RefCell::new(None) };
120}
121
122/// How many times a masked, non-exact vector search has explained itself on
123/// this thread since the last [`ambiguous_exactness_warns_reset`].
124///
125/// The line itself is the product; this counter exists so a test can assert it
126/// is printed **once per index** rather than once per call.
127///
128/// **Single-threaded assertions only.** The suppression set this counts is a
129/// `Mutex<HashSet<_>>` on the `GraphDb` — shared by every thread — while the
130/// counter is thread-local. Under a concurrent caller (`serve`, which is the
131/// deployment the warning exists for) the thread that prints the line is not
132/// necessarily the thread that asked, so a zero here does not mean the line was
133/// not printed and a one does not mean it was printed once. It answers
134/// "once per index" only in a test that owns the store.
135#[doc(hidden)]
136pub fn ambiguous_exactness_warns() -> u64 {
137    AMBIGUOUS_EXACTNESS_WARNS.with(|c| c.get())
138}
139
140/// The most recent exactness warning printed on this thread, verbatim.
141///
142/// The advice a caller reads has to be advice that caller can act on, which the
143/// counter alone cannot witness — see
144/// `the_hybrid_path_advises_a_call_the_hybrid_caller_can_make`. Carries the
145/// same single-threaded caveat as [`ambiguous_exactness_warns`].
146#[doc(hidden)]
147pub fn ambiguous_exactness_last_warning() -> Option<String> {
148    AMBIGUOUS_EXACTNESS_LAST.with(|c| c.borrow().clone())
149}
150
151/// Reset this thread's exactness-warning counter and recorded line.
152#[doc(hidden)]
153pub fn ambiguous_exactness_warns_reset() {
154    AMBIGUOUS_EXACTNESS_WARNS.with(|c| c.set(0));
155    AMBIGUOUS_EXACTNESS_LAST.with(|c| *c.borrow_mut() = None);
156}
157
158/// Which signature reached the approximate masked vector leg.
159///
160/// One code path, two entry points, and the advice cannot be the same: a
161/// warning that names an argument the caller's function does not take sends
162/// them looking for a parameter that is not there. `search_hybrid` and
163/// `search_hybrid_scoped` take `(text_field, query_text, vector_field,
164/// query_vec, label, k[, mask])` — no `exact`, no `where`.
165///
166/// The warning is not suppressed on the hybrid path. The approximation is the
167/// same one, and a caller who read `mask=` as a promise of exhaustiveness is
168/// the reader it was written for whichever door they came in by; only the
169/// remedy differs, so only the remedy changes.
170#[derive(Clone, Copy, PartialEq, Eq, Hash, Debug)]
171enum ExactnessCaller {
172    /// `find_similar` / `find_similar_vector_*` — `exact` and `where` are its
173    /// own parameters.
174    Vector,
175    /// `search_hybrid` / `search_hybrid_scoped` — neither argument exists, and
176    /// the leg is one half of a fusion.
177    Hybrid,
178}
179
180impl ExactnessCaller {
181    fn subject(self) -> &'static str {
182        match self {
183            Self::Vector => "a masked vector search",
184            Self::Hybrid => "the vector leg of a masked hybrid search",
185        }
186    }
187
188    fn advice(self) -> &'static str {
189        match self {
190            Self::Vector => "pass exact=True or a where= predicate.",
191            // Names the call that does take the argument, because this one
192            // does not: the caller's own next step, not a parameter hunt.
193            Self::Hybrid => {
194                "run the vector leg on its own with find_similar(field, vector, mask=…, \
195                 exact=True) and fuse it with search() yourself — search_hybrid itself \
196                 takes no exactness argument."
197            }
198        }
199    }
200}
201
202/// Internal state for a single `subscribe_query` subscription.
203///
204/// On every commit, `distribute_events` re-executes `ops` against the current
205/// graph state, diffs the result against `prev_rows`, and pushes
206/// `DbEvent::QueryRowAdded` / `QueryRowRemoved` events to `inner`.
207///
208/// **Full re-run per commit; use LIMIT to bound execution cost.**
209/// (Differential evaluation is roadmap / Phase 5.)
210pub(crate) struct QuerySubEntry {
211    /// Compiled plan for the subscribed Cypher query.
212    ops: Vec<PlanOp>,
213    /// Column names from the first execution (fixed for the subscription lifetime).
214    columns: Vec<String>,
215    /// Serialized (JSON) row key → row data, representing the result set at
216    /// the end of the last commit. Used to diff against the new result.
217    prev_row_map: std::collections::HashMap<String, Vec<Option<Value>>>,
218    /// Weak pointer to the subscriber queue; dead Weak → subscription dropped.
219    inner: std::sync::Weak<SubInner>,
220    /// Interned label sym captured at subscribe time from the plan's leading scan
221    /// (`ScanLabel`, `IndexScan`, or `IndexIntersect` with a concrete label).
222    ///
223    /// `None` means the plan has an `Expand` op (or no recognizable leading scan
224    /// with a concrete label), and this subscription must re-execute on every
225    /// commit without skipping. This is the conservative v0.4.3 boundary: Expand
226    /// queries are never skipped because edges can alter join results regardless
227    /// of which node labels were written.
228    scan_label: Option<u32>,
229}
230
231/// A post-commit mutation notification.
232///
233/// Emitted from `log_then_apply` after the WAL append, fsync, and
234/// in-memory `apply` all succeed. Never emitted for rejected operations
235/// (validation errors, [`GraphError::RuleOwned`], duplicate keys, no-op
236/// deletes/removes). Event payloads carry user keys and rule names, never
237/// internal ids.
238///
239/// **Replay:** [`GraphDb::open`] / [`GraphDb::open_with`] replay the WAL via
240/// `apply` only. Emission lives exclusively in `log_then_apply`, so
241/// recovery is silent even if a sink were installed (it cannot be: the
242/// sink is in-memory and set after open).
243///
244/// **Ordering:** a `Batch` WAL frame emits one event per inner record, then
245/// [`MutationEvent::BatchApplied`]. An ingest commit emits those same inner
246/// events, then [`MutationEvent::Ingested`] (not `BatchApplied`). An empty
247/// or all-noop batch writes no WAL and emits nothing (including no summary).
248///
249/// **Derived edges:** rule-created or retracted edges are not individually
250/// evented — they are recoverable from the triggering mutation plus the live
251/// rule set. Only the triggering record is emitted.
252///
253/// **Wire form:** externally tagged snake_case JSON
254/// (`{"node_inserted":{"label":"A","key":"k"}}`).
255#[derive(Debug, Clone, PartialEq, Eq, Serialize)]
256#[serde(rename_all = "snake_case")]
257pub enum MutationEvent {
258    NodeInserted {
259        label: String,
260        key: String,
261    },
262    PropSet {
263        key: String,
264        field: String,
265    },
266    PropRemoved {
267        key: String,
268        field: String,
269    },
270    EdgeInserted {
271        edge_type: String,
272        src: String,
273        dst: String,
274    },
275    EdgeDeleted {
276        edge_type: String,
277        src: String,
278        dst: String,
279    },
280    NodeDeleted {
281        key: String,
282    },
283    RuleCreated {
284        name: String,
285    },
286    RuleDeleted {
287        name: String,
288    },
289    RuleRebuilt {
290        name: String,
291    },
292    BatchApplied {
293        ops: usize,
294    },
295    Ingested {
296        label: String,
297        inserted: usize,
298    },
299}
300
301fn event_from_record(rec: &WalRecord, intern: &Interner, ids: &IdMap) -> Option<MutationEvent> {
302    match rec {
303        WalRecord::InsertNode { label, key, .. } => Some(MutationEvent::NodeInserted {
304            label: label.clone(),
305            key: key.clone(),
306        }),
307        WalRecord::InsertNodeId { label, key, .. } => Some(MutationEvent::NodeInserted {
308            label: intern.resolve(*label)?.to_string(),
309            key: key.clone(),
310        }),
311        WalRecord::SetProp { key, field, .. } => Some(MutationEvent::PropSet {
312            key: key.clone(),
313            field: field.clone(),
314        }),
315        WalRecord::SetPropId { id, field, .. } => Some(MutationEvent::PropSet {
316            key: ids.key_of(*id)?.to_string(),
317            field: intern.resolve(*field)?.to_string(),
318        }),
319        WalRecord::RemoveProp { key, field } => Some(MutationEvent::PropRemoved {
320            key: key.clone(),
321            field: field.clone(),
322        }),
323        WalRecord::InsertEdge {
324            edge_type,
325            src_key,
326            dst_key,
327        } => Some(MutationEvent::EdgeInserted {
328            edge_type: edge_type.clone(),
329            src: src_key.clone(),
330            dst: dst_key.clone(),
331        }),
332        WalRecord::InsertEdgeId { etype, src, dst } => Some(MutationEvent::EdgeInserted {
333            edge_type: intern.resolve(*etype)?.to_string(),
334            src: ids.key_of(*src)?.to_string(),
335            dst: ids.key_of(*dst)?.to_string(),
336        }),
337        WalRecord::DeleteEdge {
338            edge_type,
339            src_key,
340            dst_key,
341        } => Some(MutationEvent::EdgeDeleted {
342            edge_type: edge_type.clone(),
343            src: src_key.clone(),
344            dst: dst_key.clone(),
345        }),
346        WalRecord::DeleteNode { key } => Some(MutationEvent::NodeDeleted { key: key.clone() }),
347        WalRecord::CreateRule { def_bytes } => {
348            let def: RuleDef = decode_rule_def(def_bytes).ok()?;
349            Some(MutationEvent::RuleCreated { name: def.name })
350        }
351        WalRecord::DeleteRule { name } => Some(MutationEvent::RuleDeleted { name: name.clone() }),
352        WalRecord::RebuildRule { name } => Some(MutationEvent::RuleRebuilt { name: name.clone() }),
353        WalRecord::Batch(_)
354        | WalRecord::CreateView { .. }
355        | WalRecord::DeleteView { .. }
356        | WalRecord::EnableFulltext { .. }
357        | WalRecord::DisableFulltext { .. }
358        | WalRecord::EnableIndex { .. }
359        | WalRecord::DisableIndex { .. }
360        | WalRecord::Intern { .. }
361        // History markers are no-ops for mutation events — they carry no new
362        // state and rules re-derive deterministically on replay.
363        | WalRecord::DerivedEdgeAdded { .. }
364        | WalRecord::DerivedEdgeRetracted { .. }
365        // A count changes neither the node nor the edge population: the pair it
366        // counts was already there, which is why it is written at all.
367        | WalRecord::SetEdgeCount { .. }
368        // RenameNode carries no node/edge count change; no special event.
369        | WalRecord::RenameNode { .. } => None,
370    }
371}
372
373/// Database-wide counters plus per-rule budget/fire stats.
374#[derive(Debug, Clone, PartialEq, Eq, Serialize)]
375pub struct Stats {
376    pub nodes_live: usize,
377    pub nodes_tombstoned: usize,
378    pub edges: u64,
379    pub rules: Vec<RuleStats>,
380    /// How many writes hit the rule-chaining depth cap with work still pending,
381    /// since this handle was opened. Non-zero means some derived edges beyond
382    /// the cap are stale and no single later write will repair them: split the
383    /// rule chain or shorten it. Never persisted, so it resets on reopen.
384    #[serde(default)]
385    pub chain_truncations: u64,
386    /// The oldest commit index history still reaches (the WAL horizon floor).
387    /// `0` means nothing has been pruned and history is complete; a non-zero
388    /// value means events before that commit were pruned and are gone.
389    #[serde(default)]
390    pub history_floor: u64,
391    /// Live node counts per namespace, in name order. Always carries
392    /// `default` — a store is at least its default namespace — so a
393    /// single-tenant store reads `[{"name":"default", …}]` and a reader can
394    /// tell "no namespaces in use" from one entry.
395    #[serde(default)]
396    pub namespaces: Vec<NamespaceStats>,
397}
398
399/// Live node count for one namespace; one entry of [`Stats::namespaces`].
400#[derive(Debug, Clone, PartialEq, Eq, Serialize)]
401pub struct NamespaceStats {
402    pub name: String,
403    pub nodes_live: usize,
404}
405
406/// One rule's provenance size, trip latch, and fire counter.
407///
408/// `tripped` is a one-way latch: once set, the engine adds no new edges for
409/// that rule until [`GraphDb::rebuild_rule`] (and only if the full desired
410/// set then fits). `fires` counts `on_node_changed` evaluations plus
411/// backfill/rebuild participant ticks (rebuild counts even when it is a
412/// provenance no-op).
413#[derive(Debug, Clone, PartialEq, Eq, Serialize)]
414pub struct RuleStats {
415    pub name: String,
416    pub edges: u64,
417    pub tripped: bool,
418    pub fires: u64,
419    /// Whether this rule uses the approximate IVF-Flat candidate path.
420    pub approximate: bool,
421    /// `Some` while this rule's vector index is still being built.
422    ///
423    /// The rule derives **no** edges until it is `None`: the backfill is one
424    /// commit that runs after the index is whole, so a caller never sees a
425    /// partial edge set. Absent from the JSON when the rule is not building,
426    /// which is every rule created over a corpus at or below
427    /// [`core_rules::HNSW_BUILD_BATCH`] vectors.
428    #[serde(default, skip_serializing_if = "Option::is_none")]
429    pub building: Option<BuildProgress>,
430}
431
432/// One entry in the slow-query ring buffer.
433#[derive(Debug, Clone, Serialize)]
434pub struct SlowQueryEntry {
435    /// Execution time in whole milliseconds.
436    pub ms: u64,
437    /// The Cypher query string that was slow.
438    pub query: String,
439    /// The commit sequence number at the time the query ran.
440    pub at_commit: u64,
441}
442
443/// Snapshot of the slow-query log returned by [`GraphDb::slow_query_snapshot`].
444#[derive(Debug, Clone, Serialize)]
445pub struct SlowQuerySnapshot {
446    /// Current threshold in milliseconds (0 = disabled).
447    pub threshold_ms: u64,
448    /// Total number of slow queries ever recorded (not capped by ring size).
449    pub count: u64,
450    /// Most-recent slow queries (up to 16), oldest first.
451    pub last: Vec<SlowQueryEntry>,
452}
453
454/// Internal ring-buffer state protected by a `Mutex` so `query(&self)` can
455/// write to it without a mutable borrow.
456struct SlowQueryLog {
457    entries: std::collections::VecDeque<SlowQueryEntry>,
458    total: u64,
459}
460
461/// Maximum number of entries kept in the slow-query ring buffer.
462const SLOW_QUERY_RING_CAP: usize = 16;
463
464/// Wire summary of a [`Predicate`]. JSON only — `Explanation` is never
465/// bincode-persisted (WAL/snapshots store `RuleDef` bytes, not this type).
466#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
467pub struct PredicateSummary {
468    pub kind: String,
469    pub fields: Vec<String>,
470    pub min: Option<f64>,
471    pub tolerance: Option<f64>,
472    pub km: Option<f64>,
473    pub parts: Option<Vec<PredicateSummary>>,
474    /// True when the owning rule has `approximate=true` (IVF-Flat candidate path).
475    /// Always false for predicates reported without rule context (sub-predicates in `parts`).
476    #[serde(default)]
477    pub approximate: bool,
478}
479
480impl From<&Predicate> for PredicateSummary {
481    fn from(p: &Predicate) -> Self {
482        match p {
483            Predicate::KeyMatch { field } => PredicateSummary {
484                kind: "key_match".into(),
485                fields: vec![field.clone()],
486                min: None,
487                tolerance: None,
488                km: None,
489                parts: None,
490                approximate: false,
491            },
492            Predicate::FieldEqual { field } => PredicateSummary {
493                kind: "field_equal".into(),
494                fields: vec![field.clone()],
495                min: None,
496                tolerance: None,
497                km: None,
498                parts: None,
499                approximate: false,
500            },
501            Predicate::Overlap { field, min } => PredicateSummary {
502                kind: "overlap".into(),
503                fields: vec![field.clone()],
504                min: Some(*min),
505                tolerance: None,
506                km: None,
507                parts: None,
508                approximate: false,
509            },
510            Predicate::NumericWithin { field, tolerance } => PredicateSummary {
511                kind: "numeric_within".into(),
512                fields: vec![field.clone()],
513                min: None,
514                tolerance: Some(*tolerance),
515                km: None,
516                parts: None,
517                approximate: false,
518            },
519            Predicate::GeoRadius { field, km } => PredicateSummary {
520                kind: "geo_radius".into(),
521                fields: vec![field.clone()],
522                min: None,
523                tolerance: None,
524                km: Some(*km),
525                parts: None,
526                approximate: false,
527            },
528            Predicate::VectorSimilar { field, min } => PredicateSummary {
529                kind: "vector_similar".into(),
530                fields: vec![field.clone()],
531                min: Some(*min),
532                tolerance: None,
533                km: None,
534                parts: None,
535                approximate: false,
536            },
537            Predicate::All(inner) => {
538                let parts: Vec<PredicateSummary> = inner.iter().map(Self::from).collect();
539                let mut fields = Vec::new();
540                for part in &parts {
541                    for f in &part.fields {
542                        if !fields.contains(f) {
543                            fields.push(f.clone());
544                        }
545                    }
546                }
547                PredicateSummary {
548                    kind: "all".into(),
549                    fields,
550                    min: None,
551                    tolerance: None,
552                    km: None,
553                    parts: Some(parts),
554                    approximate: false,
555                }
556            }
557            Predicate::Any(inner) => {
558                let parts: Vec<PredicateSummary> = inner.iter().map(Self::from).collect();
559                let mut fields = Vec::new();
560                for part in &parts {
561                    for f in &part.fields {
562                        if !fields.contains(f) {
563                            fields.push(f.clone());
564                        }
565                    }
566                }
567                PredicateSummary {
568                    kind: "any".into(),
569                    fields,
570                    min: None,
571                    tolerance: None,
572                    km: None,
573                    parts: Some(parts),
574                    approximate: false,
575                }
576            }
577        }
578    }
579}
580
581/// Snapshot of a live node's key, label, and columnar properties.
582///
583/// `props` is a [`BTreeMap`] so field order is deterministic (sorted by name)
584/// regardless of insert order or the columnar store's `HashMap` iteration.
585///
586/// Deliberately does not derive `Serialize`: `Value`'s serde form is
587/// internally tagged. Wire JSON is built by `value_to_json` in the server.
588#[derive(Debug, Clone, PartialEq)]
589pub struct NodeInfo {
590    pub key: String,
591    pub label: String,
592    pub props: BTreeMap<String, Value>,
593}
594
595/// Counts returned by [`GraphDb::delete_node`].
596#[derive(Debug, Clone, PartialEq, Eq, Default)]
597pub struct DeleteReport {
598    /// Number of manual (user-inserted) edges removed.
599    pub manual_edges: u64,
600    /// Number of derived (rule-owned) edges retracted.
601    pub derived_edges: u64,
602}
603
604/// One directed edge incident on a node, with provenance membership.
605///
606/// `derived` is true iff `(edge_type, src, dst)` is in the rule engine's
607/// Plan-8 `by_node` provenance index.
608#[derive(Debug, Clone, PartialEq, Eq, Serialize)]
609pub struct EdgeInfo {
610    pub edge_type: String,
611    pub src_key: String,
612    pub dst_key: String,
613    pub derived: bool,
614}
615
616/// One directed edge incident on a node at a point in WAL history, with the
617/// rule that derived it when it is rule-owned.
618///
619/// Returned by [`GraphDb::edges_at`] (sorted by `(edge_type, src_key, dst_key)`)
620/// and by [`GraphDb::what_if_set_prop`].
621#[derive(Debug, Clone, PartialEq, Eq, PartialOrd, Ord, Serialize)]
622pub struct EdgeAt {
623    pub edge_type: String,
624    pub src_key: String,
625    pub dst_key: String,
626    /// `true` when a rule wrote the edge (`DerivedEdgeAdded` in the WAL, or a
627    /// live provenance entry).
628    pub derived: bool,
629    /// The rule that derived the edge. `None` for a manual edge.
630    pub rule: Option<String>,
631}
632
633/// The derived edges a hypothetical property change would retract and derive.
634///
635/// Returned by [`GraphDb::what_if_set_prop`]. Both lists are sorted by
636/// `(edge_type, src_key, dst_key)` and every entry is rule-derived.
637#[derive(Debug, Clone, PartialEq, Eq, Serialize)]
638pub struct WhatIf {
639    /// Derived edges that exist now and would be retracted.
640    pub lost: Vec<EdgeAt>,
641    /// Derived edges that do not exist now and would be derived.
642    pub gained: Vec<EdgeAt>,
643}
644
645/// An edge with mask-aware endpoint visibility.
646///
647/// Returned by [`GraphDb::node_edges_masked`] in [`crate::mask::MaskMode::Stub`]
648/// mode — hidden endpoints carry `*_restricted: true`.
649#[derive(Debug, Clone, PartialEq, Eq)]
650pub struct MaskedEdge {
651    pub edge_type: String,
652    pub src_key: String,
653    /// `true` when `src_key` is in the DB but hidden from the mask.
654    pub src_restricted: bool,
655    pub dst_key: String,
656    /// `true` when `dst_key` is in the DB but hidden from the mask.
657    pub dst_restricted: bool,
658    pub derived: bool,
659}
660
661/// Result of a mask-aware node lookup via [`GraphDb::node_info_masked`].
662///
663/// `None` from that method means the key does not exist (→ 404).
664/// `Some(Restricted)` is only produced when `mask.mode() == MaskMode::Stub`.
665#[derive(Debug, PartialEq)]
666pub enum MaskedNodeResult {
667    Visible(NodeInfo),
668    /// Node exists in the DB but is hidden from this mask.
669    Restricted,
670}
671
672/// One rule-owned edge between two nodes, with the rule name, edge type,
673/// direction (src_key → dst_key), and weight if the rule stores one.
674#[derive(Debug, Clone, PartialEq, Serialize)]
675pub struct Explanation {
676    pub rule: String,
677    pub edge_type: String,
678    pub src_key: String,
679    pub dst_key: String,
680    pub weight: Option<f64>,
681    pub predicate: PredicateSummary,
682    /// For a via-hop rule, the edge type the rule hops over to reach its
683    /// candidates. `None` for a plain two-node rule. A via-hop rule whose
684    /// `via_edge` is itself rule-derived is the chaining case: the hop edge
685    /// was written by another rule in the same commit.
686    #[serde(default)]
687    pub via_edge: Option<String>,
688}
689
690/// Report returned by [`GraphDb::backup_to`].
691#[derive(Debug, Clone)]
692pub struct BackupReport {
693    /// Filenames copied into the destination directory (sorted ascending).
694    pub files: Vec<String>,
695    /// Total bytes written across all copied files.
696    pub bytes: u64,
697    /// `true` when the destination opened cleanly and passed post-copy checks.
698    ///
699    /// For stores that have a `snapshot.bin` this means: all V8 section CRCs
700    /// matched **and** the destination opened without error.
701    ///
702    /// For WAL-only stores (no `snapshot.bin`) there is no snapshot to
703    /// CRC-check; `verified` is `true` when the destination opened and
704    /// replayed the WAL without error (record-level checksums in the WAL
705    /// provide the integrity signal, not section CRCs).
706    pub verified: bool,
707}
708
709/// One directed edge in export form, with optional rule attribution for derived edges.
710///
711/// Returned by [`GraphDb::all_edges_for_export`].
712///
713/// Does not derive `Eq`/`Ord`: `weight` is an `f64` and NaN breaks a total
714/// order. Callers that need a stable edge ordering already sort by
715/// `(edge_type, src, dst)` explicitly (see `all_edges_for_export`).
716#[derive(Debug, Clone, PartialEq, PartialOrd)]
717pub struct ExportEdge {
718    pub edge_type: String,
719    pub src: String,
720    pub dst: String,
721    pub derived: bool,
722    /// Rule name that created this edge, if derived. `None` for manual edges.
723    pub rule: Option<String>,
724    /// The creating rule's declared `weight_prop`, read off this edge, when
725    /// derived and numeric (`Int`/`Float`). `None` for manual edges, derived
726    /// edges whose rule declares no `weight_prop`, or a non-numeric value.
727    pub weight: Option<f64>,
728}
729
730/// One edge type's shape, as [`GraphDb::edge_type_census`] counts it.
731///
732/// Deliberately per *type* and not per edge: everything here is a summary a
733/// caller can print in one line, and none of it costs a record per edge.
734#[derive(Debug, Clone, PartialEq, Eq)]
735pub struct EdgeTypeCensus {
736    pub edge_type: String,
737    /// Directed edges of this type. Counted the way
738    /// [`GraphDb::edge_count`] counts: each edge once, from its source.
739    pub edges: u64,
740    /// Every label seen on a source of this type, sorted.
741    pub src_labels: Vec<String>,
742    /// Every label seen on a destination of this type, sorted.
743    pub dst_labels: Vec<String>,
744    /// The rules that declare this `edge_type`, sorted. Empty for a type
745    /// written by hand.
746    pub rules: Vec<String>,
747    /// `(src key, dst key)` of the first edge of this type in the store's own
748    /// id order — a real pair to quote in an example.
749    pub sample: Option<(String, String)>,
750}
751
752/// Construct the standard write-query result set (columns: created, properties_set, deleted).
753fn write_result_set() -> ResultSet {
754    ResultSet::new(vec![
755        "created".into(),
756        "properties_set".into(),
757        "deleted".into(),
758    ])
759}
760
761fn resolve_merge_set_value(op: &Operand, params: &BTreeMap<String, Value>) -> Result<Value> {
762    match op {
763        Operand::Lit(v) => Ok(v.clone()),
764        Operand::Param(name) => params
765            .get(name)
766            .cloned()
767            .ok_or_else(|| GraphError::QueryError {
768                detail: format!("missing parameter `{name}`"),
769            }),
770        _ => Err(GraphError::QueryError {
771            detail: "ON CREATE/ON MATCH SET value must be a literal or $parameter".into(),
772        }),
773    }
774}
775
776fn operand_node_vars(op: &Operand, out: &mut Vec<String>) {
777    match op {
778        Operand::Prop { var, .. } | Operand::Var(var) => {
779            if !out.contains(var) {
780                out.push(var.clone());
781            }
782        }
783        Operand::FuncCall { args, .. } => {
784            for arg in args {
785                operand_node_vars(arg, out);
786            }
787        }
788        Operand::BinArith { left, right, .. } => {
789            operand_node_vars(left, out);
790            operand_node_vars(right, out);
791        }
792        Operand::Case { branches, default } => {
793            // Branch conditions reference vars already bound (and mask-filtered)
794            // by the MATCH phase, so collecting from the value operands + ELSE
795            // is sufficient for RETURN-projection var discovery.
796            for (_, value) in branches {
797                operand_node_vars(value, out);
798            }
799            if let Some(d) = default {
800                operand_node_vars(d, out);
801            }
802        }
803        Operand::Index { base, index } => {
804            operand_node_vars(base, out);
805            operand_node_vars(index, out);
806        }
807        Operand::Lit(_) | Operand::Param(_) => {}
808    }
809}
810
811fn ret_node_vars(items: &[RetItem]) -> Vec<String> {
812    let mut out = Vec::new();
813    for item in items {
814        match &item.value {
815            RetVal::Var(v) | RetVal::Prop { var: v, .. } => {
816                if !out.contains(v) {
817                    out.push(v.clone());
818                }
819            }
820            RetVal::FuncCall { args, .. } => {
821                for arg in args {
822                    operand_node_vars(arg, &mut out);
823                }
824            }
825            RetVal::ScalarExpr(op) => operand_node_vars(op, &mut out),
826            RetVal::Agg { .. } => {}
827        }
828    }
829    out
830}
831
832fn add_var(out: &mut Vec<String>, v: &str) {
833    if !out.iter().any(|x| x == v) {
834        out.push(v.to_string());
835    }
836}
837
838fn pattern_node_vars(pats: &[Pattern]) -> Vec<String> {
839    let mut out = Vec::new();
840    for p in pats {
841        if let Some(v) = &p.start.var {
842            add_var(&mut out, v);
843        }
844        for (_, dest) in &p.chain {
845            if let Some(v) = &dest.var {
846                add_var(&mut out, v);
847            }
848        }
849    }
850    out
851}
852
853fn pattern_rel_vars(pats: &[Pattern]) -> Vec<String> {
854    let mut out = Vec::new();
855    for p in pats {
856        for (rel, _) in &p.chain {
857            if rel.hops.is_none() {
858                if let Some(v) = &rel.var {
859                    add_var(&mut out, v);
860                }
861            }
862        }
863    }
864    out
865}
866
867fn rel_type_alias(var: &str) -> String {
868    format!("__rt_{var}")
869}
870
871fn ret_column_name(item: &RetItem) -> String {
872    if let Some(alias) = &item.alias {
873        return alias.clone();
874    }
875    // The same naming rule the planner and the executor use, so a
876    // write-statement RETURN names its columns exactly as a read query does.
877    // An aggregate is not legal in a write-statement RETURN; it keeps the
878    // placeholder it always had.
879    ret_val_label(&item.value).unwrap_or_else(|| "<agg>".to_string())
880}
881
882fn eval_set_return_operand<F: Fs>(
883    db: &GraphDb<F>,
884    match_rs: &ResultSet,
885    row: usize,
886    rel_vars: &[String],
887    op: &Operand,
888    params: &BTreeMap<String, Value>,
889) -> Result<Option<Value>> {
890    match op {
891        Operand::Lit(v) => Ok(Some(v.clone())),
892        Operand::Param(name) => params.get(name).cloned().ok_or_else(|| GraphError::QueryError {
893            detail: format!("missing parameter `{name}`"),
894        }).map(Some),
895        Operand::Var(name) if rel_vars.iter().any(|r| r == name) => Err(GraphError::QueryError {
896            detail: format!(
897                "cannot return relationship variable '{name}' bare; return its properties ({name}.field) instead"
898            ),
899        }),
900        Operand::Var(name) => Ok(match_rs.get(row, name).cloned()),
901        Operand::Prop { var, field } => {
902            if rel_vars.iter().any(|r| r == var) {
903                return Ok(None);
904            }
905            let Some(Value::Str(key)) = match_rs.get(row, var) else {
906                return Ok(None);
907            };
908            if let Some(v) = db.get_prop(key, field) {
909                return Ok(Some(v));
910            }
911            // Same stored-wins identity fallback as the read path:
912            // n.key / n.id / n.label, not only get_prop.
913            Ok(match field.as_str() {
914                "key" | "id" => Some(Value::Str(key.clone())),
915                "label" => db
916                    .node_ref(key)
917                    .map(|n| Value::Str(n.label().to_owned())),
918                _ => None,
919            })
920        }
921        Operand::FuncCall { name, args } => {
922            eval_set_return_func(db, match_rs, row, rel_vars, name, args, params)
923        }
924        Operand::BinArith { op, left, right } => {
925            let lv = eval_set_return_operand(db, match_rs, row, rel_vars, left, params)?;
926            let rv = eval_set_return_operand(db, match_rs, row, rel_vars, right, params)?;
927            eval_set_return_arith(op, lv, rv)
928        }
929        Operand::Case { branches, default } => {
930            for (cond, value) in branches {
931                if eval_set_return_expr(db, match_rs, row, rel_vars, cond, params, 0)? {
932                    return eval_set_return_operand(db, match_rs, row, rel_vars, value, params);
933                }
934            }
935            match default {
936                Some(d) => eval_set_return_operand(db, match_rs, row, rel_vars, d, params),
937                None => Ok(None),
938            }
939        }
940        Operand::Index { base, index } => {
941            let base_val = eval_set_return_operand(db, match_rs, row, rel_vars, base, params)?;
942            let idx_val = eval_set_return_operand(db, match_rs, row, rel_vars, index, params)?;
943            Ok(core_query::value_ops::index_list(base_val, idx_val))
944        }
945    }
946}
947
948fn eval_set_return_expr<F: Fs>(
949    db: &GraphDb<F>,
950    match_rs: &ResultSet,
951    row: usize,
952    rel_vars: &[String],
953    expr: &Expr,
954    params: &BTreeMap<String, Value>,
955    depth: u32,
956) -> Result<bool> {
957    if depth > 256 {
958        return Err(GraphError::QueryError {
959            detail: "expression nesting too deep".into(),
960        });
961    }
962    match expr {
963        Expr::And(lhs, rhs) => {
964            let l = eval_set_return_expr(db, match_rs, row, rel_vars, lhs, params, depth + 1)?;
965            let r = eval_set_return_expr(db, match_rs, row, rel_vars, rhs, params, depth + 1)?;
966            Ok(l && r)
967        }
968        Expr::Or(lhs, rhs) => {
969            let l = eval_set_return_expr(db, match_rs, row, rel_vars, lhs, params, depth + 1)?;
970            let r = eval_set_return_expr(db, match_rs, row, rel_vars, rhs, params, depth + 1)?;
971            Ok(l || r)
972        }
973        Expr::Not(inner) => Ok(!eval_set_return_expr(
974            db,
975            match_rs,
976            row,
977            rel_vars,
978            inner,
979            params,
980            depth + 1,
981        )?),
982        Expr::Cmp { lhs, op, rhs } => {
983            let l = eval_set_return_operand(db, match_rs, row, rel_vars, lhs, params)?;
984            let r = eval_set_return_operand(db, match_rs, row, rel_vars, rhs, params)?;
985            match (l, r) {
986                (Some(a), Some(b)) => Ok(eval_cmp(op, &a, &b)),
987                _ => Ok(false),
988            }
989        }
990        Expr::Truthy(op) => {
991            let val = eval_set_return_operand(db, match_rs, row, rel_vars, op, params)?;
992            Ok(match val {
993                None => false,
994                Some(Value::Bool(b)) => b,
995                Some(Value::Int(n)) => n != 0,
996                Some(Value::Float(f)) => f != 0.0,
997                Some(Value::Str(s)) => !s.is_empty(),
998                Some(Value::List(v)) => !v.is_empty(),
999                Some(Value::Map(m)) => !m.is_empty(),
1000            })
1001        }
1002        Expr::IsNull(op) => {
1003            let val = eval_set_return_operand(db, match_rs, row, rel_vars, op, params)?;
1004            Ok(val.is_none())
1005        }
1006        Expr::IsNotNull(op) => {
1007            let val = eval_set_return_operand(db, match_rs, row, rel_vars, op, params)?;
1008            Ok(val.is_some())
1009        }
1010        Expr::In { expr, list } => {
1011            let Some(needle) = eval_set_return_operand(db, match_rs, row, rel_vars, expr, params)?
1012            else {
1013                return Ok(false);
1014            };
1015            for item_op in list {
1016                match eval_set_return_operand(db, match_rs, row, rel_vars, item_op, params)? {
1017                    None => {}
1018                    Some(Value::List(items)) => {
1019                        for item in items {
1020                            if eval_cmp(&core_query::CmpOp::Eq, &needle, &item) {
1021                                return Ok(true);
1022                            }
1023                        }
1024                    }
1025                    Some(item) if eval_cmp(&core_query::CmpOp::Eq, &needle, &item) => {
1026                        return Ok(true);
1027                    }
1028                    Some(_) => {}
1029                }
1030            }
1031            Ok(false)
1032        }
1033    }
1034}
1035
1036fn eval_set_return_arith(
1037    op: &ArithOp,
1038    lv: Option<Value>,
1039    rv: Option<Value>,
1040) -> Result<Option<Value>> {
1041    match (lv, rv) {
1042        (None, _) | (_, None) => Ok(None),
1043        (Some(Value::Int(a)), Some(Value::Int(b))) => {
1044            let result = match op {
1045                ArithOp::Sub => a.saturating_sub(b),
1046                ArithOp::Mul => a.saturating_mul(b),
1047                ArithOp::Add => a.saturating_add(b),
1048                ArithOp::Div => {
1049                    if b == 0 {
1050                        return Err(GraphError::QueryError {
1051                            detail: "division by zero".into(),
1052                        });
1053                    }
1054                    a.checked_div(b).unwrap_or(i64::MAX)
1055                }
1056            };
1057            Ok(Some(Value::Int(result)))
1058        }
1059        (Some(lv), Some(rv)) => {
1060            let a = match &lv {
1061                Value::Float(f) => *f,
1062                Value::Int(i) => *i as f64,
1063                _ => {
1064                    return Err(GraphError::QueryError {
1065                        detail: format!("arithmetic operand must be numeric, got {lv:?}"),
1066                    })
1067                }
1068            };
1069            let b = match &rv {
1070                Value::Float(f) => *f,
1071                Value::Int(i) => *i as f64,
1072                _ => {
1073                    return Err(GraphError::QueryError {
1074                        detail: format!("arithmetic operand must be numeric, got {rv:?}"),
1075                    })
1076                }
1077            };
1078            let result = match op {
1079                ArithOp::Sub => a - b,
1080                ArithOp::Mul => a * b,
1081                ArithOp::Add => a + b,
1082                ArithOp::Div => {
1083                    if b == 0.0 {
1084                        return Err(GraphError::QueryError {
1085                            detail: "division by zero".into(),
1086                        });
1087                    }
1088                    a / b
1089                }
1090            };
1091            Ok(Some(Value::Float(result)))
1092        }
1093    }
1094}
1095
1096fn eval_set_return_func<F: Fs>(
1097    db: &GraphDb<F>,
1098    match_rs: &ResultSet,
1099    row: usize,
1100    rel_vars: &[String],
1101    name: &str,
1102    args: &[Operand],
1103    params: &BTreeMap<String, Value>,
1104) -> Result<Option<Value>> {
1105    let norm = name.to_ascii_lowercase();
1106    if norm == "type" {
1107        if args.len() != 1 {
1108            return Err(GraphError::QueryError {
1109                detail: format!("type() requires exactly 1 argument, got {}", args.len()),
1110            });
1111        }
1112        let Operand::Var(rel) = &args[0] else {
1113            return Err(GraphError::QueryError {
1114                detail: "type() argument must be a relationship variable (e.g. type(r))".into(),
1115            });
1116        };
1117        return Ok(match_rs.get(row, &rel_type_alias(rel)).cloned());
1118    }
1119    if norm == "key" || norm == "id" {
1120        let fname = if norm == "id" { "id" } else { "key" };
1121        if args.len() != 1 {
1122            return Err(GraphError::QueryError {
1123                detail: format!("{fname}() requires exactly 1 argument, got {}", args.len()),
1124            });
1125        }
1126        let Operand::Var(var) = &args[0] else {
1127            return Err(GraphError::QueryError {
1128                detail: format!("{fname}() argument must be a node variable (e.g. {fname}(n))"),
1129            });
1130        };
1131        if rel_vars.iter().any(|r| r == var) {
1132            return Err(GraphError::QueryError {
1133                detail: format!("{fname}() argument `{var}` is a relationship, not a node"),
1134            });
1135        }
1136        // MATCH rows bind node variables to their key string, so the column
1137        // value *is* the key. `id()` aliases `key()`.
1138        return Ok(match_rs.get(row, var).cloned());
1139    }
1140    let mut vals = Vec::with_capacity(args.len());
1141    for arg in args {
1142        vals.push(eval_set_return_operand(
1143            db, match_rs, row, rel_vars, arg, params,
1144        )?);
1145    }
1146    match norm.as_str() {
1147        "tolower" => {
1148            if vals.len() != 1 {
1149                return Err(GraphError::QueryError {
1150                    detail: format!("toLower() requires exactly 1 argument, got {}", vals.len()),
1151                });
1152            }
1153            Ok(vals[0].clone().map(|val| match val {
1154                Value::Str(s) => Value::Str(s.to_ascii_lowercase()),
1155                other => other,
1156            }))
1157        }
1158        "toupper" => {
1159            if vals.len() != 1 {
1160                return Err(GraphError::QueryError {
1161                    detail: format!("toUpper() requires exactly 1 argument, got {}", vals.len()),
1162                });
1163            }
1164            Ok(vals[0].clone().map(|val| match val {
1165                Value::Str(s) => Value::Str(s.to_ascii_uppercase()),
1166                other => other,
1167            }))
1168        }
1169        "size" => match vals.first().cloned().flatten() {
1170            None => Ok(None),
1171            Some(Value::Str(s)) => Ok(Some(Value::Int(s.len() as i64))),
1172            Some(Value::List(items)) => Ok(Some(Value::Int(items.len() as i64))),
1173            Some(_) => Ok(None),
1174        },
1175        "coalesce" => Ok(vals.into_iter().flatten().next()),
1176        "abs" => match vals.first().cloned().flatten() {
1177            None => Ok(None),
1178            Some(Value::Int(n)) => Ok(Some(Value::Int(n.saturating_abs()))),
1179            Some(Value::Float(f)) => Ok(Some(Value::Float(f.abs()))),
1180            Some(_) => Ok(None),
1181        },
1182        "round" => match vals.first().cloned().flatten() {
1183            None => Ok(None),
1184            Some(Value::Float(f)) => Ok(Some(Value::Float(f.round()))),
1185            Some(Value::Int(n)) => Ok(Some(Value::Int(n))),
1186            Some(_) => Ok(None),
1187        },
1188        "decay" => {
1189            if vals.len() != 3 {
1190                return Err(GraphError::QueryError {
1191                    detail: format!("decay() requires exactly 3 arguments, got {}", vals.len()),
1192                });
1193            }
1194            match (vals[0].clone(), vals[1].clone(), vals[2].clone()) {
1195                (None, _, _) | (_, None, _) | (_, _, None) => Ok(None),
1196                (Some(b), Some(a), Some(h)) => {
1197                    let numeric = |v: Value| -> Result<f64> {
1198                        match v {
1199                            Value::Int(n) => Ok(n as f64),
1200                            Value::Float(f) => Ok(f),
1201                            other => Err(GraphError::QueryError {
1202                                detail: format!(
1203                                    "decay() requires numeric arguments, got {other:?}"
1204                                ),
1205                            }),
1206                        }
1207                    };
1208                    let b = numeric(b)?;
1209                    let a = numeric(a)?;
1210                    let h = numeric(h)?;
1211                    if h <= 0.0 {
1212                        return Err(GraphError::QueryError {
1213                            detail: "decay() requires halflife > 0".into(),
1214                        });
1215                    }
1216                    Ok(Some(Value::Float(b * 0.5f64.powf(a / h))))
1217                }
1218            }
1219        }
1220        _ => Err(GraphError::QueryError {
1221            detail: format!(
1222                "unknown function `{name}`; supported: toLower, toUpper, size, coalesce, type, abs, round, decay, key, id"
1223            ),
1224        }),
1225    }
1226}
1227
1228fn eval_set_return_item<F: Fs>(
1229    db: &GraphDb<F>,
1230    match_rs: &ResultSet,
1231    row: usize,
1232    rel_vars: &[String],
1233    item: &RetItem,
1234    params: &BTreeMap<String, Value>,
1235) -> Result<Option<Value>> {
1236    match &item.value {
1237        RetVal::Var(v) => eval_set_return_operand(
1238            db,
1239            match_rs,
1240            row,
1241            rel_vars,
1242            &Operand::Var(v.clone()),
1243            params,
1244        ),
1245        RetVal::Prop { var, field } => eval_set_return_operand(
1246            db,
1247            match_rs,
1248            row,
1249            rel_vars,
1250            &Operand::Prop {
1251                var: var.clone(),
1252                field: field.clone(),
1253            },
1254            params,
1255        ),
1256        RetVal::FuncCall { name, args } => {
1257            eval_set_return_func(db, match_rs, row, rel_vars, name, args, params)
1258        }
1259        RetVal::ScalarExpr(op) => eval_set_return_operand(db, match_rs, row, rel_vars, op, params),
1260        RetVal::Agg { .. } => Err(GraphError::QueryError {
1261            detail: "aggregates are not supported in MATCH … SET … RETURN".into(),
1262        }),
1263    }
1264}
1265
1266/// Project user RETURN from original MATCH rows after SET. No rematch.
1267fn project_set_return_rows<F: Fs>(
1268    db: &GraphDb<F>,
1269    rel_vars: &[String],
1270    match_rs: &ResultSet,
1271    returns: &[RetItem],
1272    params: &BTreeMap<String, Value>,
1273) -> Result<ResultSet> {
1274    let columns: Vec<String> = returns.iter().map(ret_column_name).collect();
1275    let mut out = ResultSet::new(columns);
1276    for row in 0..match_rs.len() {
1277        let mut cells = Vec::with_capacity(returns.len());
1278        for item in returns {
1279            cells.push(eval_set_return_item(
1280                db, match_rs, row, rel_vars, item, params,
1281            )?);
1282        }
1283        out.push_row(cells);
1284    }
1285    Ok(out)
1286}
1287
1288/// Single construction point for a `GraphMut` view over the split-borrowed graph fields.
1289/// Callers use `std::mem::take` on the engine before calling this, then restore it after.
1290/// Extract a `Vec<f64>` from a `Value::List` whose items are all numeric.
1291/// Returns `None` for non-list values or lists with non-numeric elements.
1292/// Extra candidates pulled from an approximate index before re-scoring, over and
1293/// above the `k` asked for.
1294///
1295/// The index orders candidates by `f32` distances, which agree with the exact
1296/// `f64` cosine to about 1e-6. Re-scoring can therefore only reshuffle
1297/// candidates inside a band that narrow — it cannot move a hit past one that is
1298/// further away by more than 1e-6 — so the only way a true top-`k` member can be
1299/// lost is if the index ranked it just outside `k` on the `f32` order. Fetching
1300/// `k + 16` covers any such band up to 16 members wide, which at 1e-6 means 16
1301/// vectors within a millionth of each other in cosine: a duplicate cluster, and
1302/// then the members are interchangeable anyway. `min` is applied to the exact
1303/// score, never to the index's, so a hit sitting on the threshold is decided
1304/// exactly.
1305const VECTOR_RESCORE_MARGIN: usize = 16;
1306
1307/// Cosine similarity between an already-unit query and node `id`'s `field`
1308/// vector, read from the **`f64`** properties. `None` when the node has no
1309/// numeric-list vector there, or its norm is zero.
1310///
1311/// The single definition of the score this API reports. Both the brute-force
1312/// scan and the re-scoring step that follows an index lookup go through it, so
1313/// the two paths cannot disagree — which is the property
1314/// `index_and_brute_force_agree_on_scores` pins.
1315fn exact_vector_similarity(
1316    view: &GraphView<'_>,
1317    id: u32,
1318    field: &str,
1319    q_unit: &[f64],
1320) -> Option<f64> {
1321    let v = view.prop(id, field)?;
1322    let xs = value_as_float_list(&v.into_value())?;
1323    let v_norm: f64 = xs.iter().map(|x| x * x).sum::<f64>().sqrt();
1324    if v_norm == 0.0 {
1325        return None;
1326    }
1327    Some(
1328        q_unit
1329            .iter()
1330            .zip(xs.iter())
1331            .map(|(a, b)| a * (b / v_norm))
1332            .sum(),
1333    )
1334}
1335
1336fn value_as_float_list(v: &Value) -> Option<Vec<f64>> {
1337    match v {
1338        Value::List(items) => items
1339            .iter()
1340            .map(|item| match item {
1341                Value::Float(f) => Some(*f),
1342                Value::Int(i) => Some(*i as f64),
1343                _ => None,
1344            })
1345            .collect(),
1346        _ => None,
1347    }
1348}
1349
1350fn make_graph_mut<'a>(
1351    ids: &'a IdMap,
1352    syms: &'a mut Interner,
1353    labels: &'a [u32],
1354    props: core_storage::v8::seam::ColumnsView<'a>,
1355    topo: &'a mut Topology,
1356    base: &'a Option<std::sync::Arc<core_storage::v8::MappedBase>>,
1357    edge_props: &'a mut EdgeProps,
1358) -> GraphMut<'a> {
1359    GraphMut {
1360        ids,
1361        syms,
1362        labels,
1363        props,
1364        topo,
1365        base_topo: base_csr(base),
1366        edge_props,
1367    }
1368}
1369
1370/// The archived CSR of an open V8 snapshot, for the rule engine's graph reads.
1371///
1372/// A store opened from a snapshot keeps its edges in the mapping and its
1373/// overlay empty, so a rule that reads the graph's shape has to see both.
1374fn base_csr(
1375    base: &Option<std::sync::Arc<core_storage::v8::MappedBase>>,
1376) -> Option<&core_storage::v8::layout::ArchivedCsr> {
1377    base.as_ref().map(|b| {
1378        b.topology()
1379            .expect("base topology section bounds validated at open")
1380    })
1381}
1382
1383/// Build a `ColumnsView` from the disjoint `props` overlay and optional V8 base.
1384///
1385/// Takes explicit field references rather than `&self` so the caller can hold
1386/// simultaneous mutable borrows of other fields (e.g. `syms`, `topo`).
1387fn build_props_view<'a>(
1388    props: &'a ColumnStore,
1389    base: &'a Option<std::sync::Arc<core_storage::v8::MappedBase>>,
1390) -> core_storage::v8::seam::ColumnsView<'a> {
1391    match base {
1392        None => core_storage::v8::seam::ColumnsView::owned(props),
1393        Some(b) => {
1394            let archived = b
1395                .columns()
1396                .expect("base columns section bounds validated at open");
1397            core_storage::v8::seam::ColumnsView::with_base_cached(props, archived, b.mixed_cache())
1398                .with_shared_strings(base_string_table(b))
1399        }
1400    }
1401}
1402
1403/// The base columns section paired with the string table that resolves its
1404/// string ids — what `ViewStore` needs to read a neighbour's string property
1405/// out of a V9 snapshot.
1406fn base_columns(
1407    base: &Option<std::sync::Arc<core_storage::v8::MappedBase>>,
1408) -> Option<core_storage::v8::seam::BaseColumns<'_>> {
1409    base.as_ref().map(|b| core_storage::v8::seam::BaseColumns {
1410        cols: b
1411            .columns()
1412            .expect("base columns section bounds validated at open"),
1413        strings: base_string_table(b),
1414    })
1415}
1416
1417/// The shared string table of a V9 base, or `None` for a pre-V9 one.
1418///
1419/// Every `ColumnsView` built over a base must carry it: without it a V9
1420/// snapshot's string columns, whose own tables are empty, read back as absent.
1421fn base_string_table(
1422    base: &core_storage::v8::MappedBase,
1423) -> Option<&core_storage::v8::layout::ArchivedStringTable> {
1424    base.string_table()
1425        .transpose()
1426        .expect("base strings section bounds validated at open")
1427}
1428
1429fn build_topo_view<'a>(
1430    overlay: &'a Topology,
1431    base: &'a Option<std::sync::Arc<core_storage::v8::MappedBase>>,
1432) -> core_storage::v8::seam::TopologyView<'a> {
1433    match base {
1434        None => core_storage::v8::seam::TopologyView::owned(overlay),
1435        Some(b) => {
1436            let archived_csr = b
1437                .topology()
1438                .expect("base topology section bounds validated at open");
1439            core_storage::v8::seam::TopologyView::with_base(overlay, archived_csr)
1440        }
1441    }
1442}
1443
1444/// When [`GraphDb`] calls `Fs::sync` after a WAL append.
1445///
1446/// Default is [`Strict`](FsyncPolicy::Strict): every `log_then_apply_with`
1447/// fsyncs (single `insert_node` / `set_prop`). Ingest and `write_batch`
1448/// emit one `WalRecord::Batch` and fsync once at that frame (Batched).
1449/// [`Relaxed`](FsyncPolicy::Relaxed) skips WAL sync; [`GraphDb::snapshot`]
1450/// is still durable via `write_atomic`. Crash-recovery DST stays Strict.
1451#[derive(Clone, Copy, Debug, PartialEq, Eq, Default)]
1452pub enum FsyncPolicy {
1453    /// Every WAL commit calls `fs.sync` (today's behavior).
1454    #[default]
1455    Strict,
1456    /// Sync only at a `Batch` frame end. Single-op path stays Strict unless
1457    /// this policy is set on the database.
1458    Batched,
1459    /// Never call `fs.sync`. [`GraphDb::snapshot`] still syncs via `write_atomic`.
1460    Relaxed,
1461}
1462
1463/// A precondition for a compare-and-set batch write.
1464///
1465/// All preconditions in a [`GraphDb::write_batch_cas`] or
1466/// [`crate::SharedDb::submit_batch_cas`] call are checked atomically before
1467/// any operation in the batch is applied.  If any precondition fails, the
1468/// entire batch is rejected with [`GraphError::CasConflict`] and no WAL frame
1469/// is written.
1470///
1471/// # Touch definition
1472///
1473/// A node's last-change commit (`last_changed`) is updated when any of the
1474/// following state-changing WAL records touch it:
1475///
1476/// - `InsertNode` / `InsertNodeId` — the newly-inserted node.
1477/// - `SetProp` / `SetPropId` / `RemoveProp` — the property-bearing node.
1478/// - `InsertEdge` / `InsertEdgeId` / `DeleteEdge` — **both** src and dst
1479///   endpoints (an edge change touches both sides).
1480/// - `DeleteNode` — the node is tombstoned; `last_changed` returns `None`
1481///   for deleted keys so the pre-deletion entry is never observed.
1482///
1483/// History markers (`DerivedEdgeAdded` / `DerivedEdgeRetracted`) are
1484/// state no-ops.  The underlying mutation that triggered rule firing already
1485/// updated the relevant nodes' last-change entries.  Rule-management records
1486/// (`CreateRule`, `DeleteRule`, `RebuildRule`) and view/full-text declarations
1487/// do not touch any node's last-change.
1488#[derive(Debug, Clone, PartialEq, Eq)]
1489pub enum Precondition {
1490    /// The node's last-change commit must equal `expected`.
1491    ///
1492    /// Fails with [`GraphError::CasConflict`] when:
1493    /// - The node does not exist (`last_changed` returns `None`), or
1494    /// - The recorded commit seq does not match `expected`.
1495    NodeUnchangedSince { key: String, expected: u64 },
1496    /// The node must not exist (not inserted, or already deleted).
1497    ///
1498    /// Fails with [`GraphError::CasConflict`] (expected=`u64::MAX`,
1499    /// actual=`last_changed(key).unwrap_or(0)`) when the node is live.
1500    NodeAbsent { key: String },
1501}
1502
1503pub struct GraphDb<F: Fs> {
1504    fs: F,
1505    ids: IdMap,
1506    syms: Interner,
1507    topo: Topology,
1508    props: ColumnStore,
1509    labels: Vec<u32>, // node id -> label symbol
1510    /// Namespace names by index; index [`NS_DEFAULT_IDX`] is always
1511    /// [`NS_DEFAULT`]. Derived beside [`Self::node_ns`], never persisted.
1512    ///
1513    /// A private table rather than the shared [`Interner`]: interning
1514    /// `"default"` at open would add a symbol to the store's symbol table and
1515    /// change the bytes of the next snapshot of a store that has no namespaces
1516    /// at all.
1517    ns_names: Vec<String>,
1518    /// Namespace index per dense node id, into [`Self::ns_names`];
1519    /// [`NS_DEFAULT_IDX`] for a node with no `ns` property.
1520    ///
1521    /// Derived: built by one pass over the `ns` column at open (which reads
1522    /// nothing when the column does not exist) and maintained at every node
1523    /// insert. Never written to a snapshot or the WAL, because the property it
1524    /// mirrors already is. A namespace cannot change, so no other record shape
1525    /// can move a node between namespaces.
1526    node_ns: Vec<u32>,
1527    edge_props: EdgeProps,
1528    engine: RuleEngine,
1529    view_store: ViewStore,
1530    /// Incremental inverted index for full-text-lite search.
1531    /// Rebuild-on-open: populated from WAL replay + rebuild_all at open end.
1532    fulltext: FulltextIndex,
1533    /// Opt-in equality index over scalar node properties.
1534    /// Rebuild-on-open: declarations replay from the WAL, postings rebuild at
1535    /// open end (mirrors `fulltext`).
1536    prop_index: PropertyIndex,
1537    /// Whether this store records insert-count multiplicity (§5.13).
1538    ///
1539    /// Declared like `prop_index`'s enabled pairs — a WAL record replayed at
1540    /// open, re-emitted into the baseline by a truncating snapshot — but it
1541    /// gates a *format* step rather than an index: `WalRecord::SetEdgeCount`
1542    /// (discriminant 23) is written only when this is `true`, so a store that
1543    /// never opts in stays readable by a binary that predates the record.
1544    multiplicity: bool,
1545    event_sink: Option<Box<dyn Fn(MutationEvent) + Send + Sync>>,
1546    /// WAL fsync cadence. Default [`FsyncPolicy::Strict`].
1547    fsync: FsyncPolicy,
1548    /// Monotonically increasing per-commit counter.  A single `log_then_apply_with`
1549    /// call increments this once; all events emitted from that call share the same
1550    /// `commit_seq` value.
1551    commit_seq: u64,
1552    /// RBAC role definitions loaded from `roles.json` at open.
1553    ///
1554    /// `Some(roles)` — loaded successfully (may be empty when no roles are defined).
1555    /// `None` — `roles.json` was present but corrupt; `mask_for_role` returns
1556    /// `Err` for any request (fail-loud, never silently grant empty visibility).
1557    roles: Option<Vec<RoleDef>>,
1558    /// Memo for [`mask_for_role`](GraphDb::mask_for_role), keyed by
1559    /// `(role, commit_seq)` — a scoped reader between two writes resolves once.
1560    ///
1561    /// Shared by `Arc` with every [`ReaderSnapshot`](crate::reader::ReaderSnapshot)
1562    /// taken from this handle. Replaced (not cleared) whenever the role
1563    /// definitions change or the store is reloaded, which `commit_seq` does not
1564    /// record; see [`RoleMaskCache`](crate::mask::RoleMaskCache).
1565    role_masks: Arc<crate::mask::RoleMaskCache>,
1566    /// Which loaded store this handle is, for memos that outlive it.
1567    ///
1568    /// `role_masks` needs no such thing — the handle owns it and replaces it —
1569    /// but a [`Scope`](crate::mask::Scope) is the caller's, so its resolved key
1570    /// leg is stamped with this alongside `commit_seq`. Minted fresh here and
1571    /// again in [`reset_for_reload`](GraphDb::reset_for_reload), at exactly the
1572    /// two points a fresh `RoleMaskCache` is installed; see
1573    /// [`StoreStamp`](crate::mask::StoreStamp) for the invariant.
1574    store_id: crate::mask::StoreId,
1575    /// Live subscriptions.  Entries with a dead `Weak` are pruned on the next
1576    /// distribute_events call.
1577    subscriptions: Vec<SubEntry>,
1578    /// Live query subscriptions. Re-executed on every commit when non-empty.
1579    /// Dead `Weak` entries are pruned inside `distribute_events`.
1580    query_subscriptions: Vec<QuerySubEntry>,
1581    /// Queue capacity for new subscriptions created by this db.  Default is
1582    /// [`DEFAULT_SUB_CAPACITY`]; can be overridden via [`set_sub_capacity`]
1583    /// to test Lagged behaviour with small queues.
1584    sub_capacity: usize,
1585    /// True for as-of instances opened via [`GraphDb::open_at`].
1586    /// Every mutation method and `snapshot()` returns [`GraphError::ReadOnly`]
1587    /// when this flag is set.
1588    read_only: bool,
1589    /// Total WAL commit count at the time [`open_at`] was called.
1590    /// 0 for normal (non-as-of) instances.
1591    total_wal_commits: u64,
1592    /// Immutable mmap-backed base snapshot (V8).  When `Some`, `self.topo` is
1593    /// the WAL-replay overlay (empty at open time, populated by apply()) and
1594    /// reads go through a merged `TopologyView`.  `self.props` is always
1595    /// fully materialized (base + WAL replay) for HNSW/IVF and view compat.
1596    base: Option<Arc<core_storage::v8::MappedBase>>,
1597    // ── MVCC epoch reader state ───────────────────────────────────────────────
1598    /// Most-recent full overlay clone.  Initialized at end of `open_with` /
1599    /// `open_at_with`; refreshed every `FOLD_EVERY_K` commits.
1600    /// `None` only between struct creation and the first fold.
1601    fold_overlay: Option<Arc<crate::reader::FrozenOverlay>>,
1602    /// Per-commit deltas accumulated since the last fold.
1603    delta_tail: Vec<Arc<crate::reader::CommitDelta>>,
1604    /// How many commits have occurred since the last fold.
1605    commits_since_fold: usize,
1606    /// When true, `log_then_apply_with` buffers event notifications instead of
1607    /// firing them immediately.  Used by the group-commit drain thread to defer
1608    /// events until after the group fsync (R2: durability before notification).
1609    /// Cleared to false once the drain thread flushes or discards the buffer.
1610    defer_events: bool,
1611    /// Buffered events accumulated while `defer_events` is true.
1612    deferred_events: Vec<DeferredEvent>,
1613    /// Set to true by the group-commit drain thread when a group fsync fails
1614    /// after WAL truncation.  All subsequent mutation attempts return an IO
1615    /// error until the database is reopened.
1616    degraded: bool,
1617    /// Set to `true` after `ensure_v8_base_sections_loaded` has read provenance,
1618    /// HNSW, and IVF sections from the mmap base into the engine's retained
1619    /// fields.  `false` on all opens until first use; always `true` for non-V8
1620    /// opens (base is None, fast-path sets flag immediately).
1621    v8_sections_loaded: std::sync::atomic::AtomicBool,
1622    /// Serializes the one-time section population in `ensure_v8_base_sections_loaded`.
1623    v8_sections_mutex: std::sync::Mutex<()>,
1624    /// Per-node last-change commit sequence.  `last_change[node_id] = seq` means
1625    /// the node was last modified by commit `seq`.
1626    ///
1627    /// Loaded from V8 section 11 at open; updated on every state-changing commit
1628    /// and WAL replay frame.  V5-V7 stores start with an empty map; pre-WAL-horizon
1629    /// nodes return `None` from `last_changed` until they are next mutated.
1630    ///
1631    /// See [`Precondition`] for the full touch definition.
1632    last_change: HashMap<u32, u64>,
1633    /// WAL archive retention policy set by [`set_wal_archive_retention`].
1634    /// `None` = unlimited (keep all archives); `Some(N)` = keep N newest archives,
1635    /// pruning older ones at snapshot time.  0 is treated as unlimited.
1636    wal_archive_retention: Option<u32>,
1637    /// Global frame index of the first commit that is still reachable through
1638    /// surviving archives.  Persisted to `wal.floor` sidecar when pruning occurs.
1639    /// Default 0 = all history reachable.
1640    wal_horizon_floor: u64,
1641    /// True when the surviving archive chain forms a continuous WAL history
1642    /// starting from the store's first commit (the genesis chain).
1643    ///
1644    /// `open_at` may replay archive-resident commits from empty state only when
1645    /// this flag is true AND `wal_horizon_floor == 0`.  Cleared whenever:
1646    ///   - a WAL-truncating snapshot (`keep_wal=false`) is taken after archives
1647    ///     already exist (breaks the chain for subsequent archives), or
1648    ///   - any archive is pruned (floor advances past zero).
1649    ///
1650    /// Persisted via the `wal.genesis` marker file; loaded from it at open.
1651    archive_genesis_chain: bool,
1652    /// True when this handle can *prove* the live WAL has never been truncated:
1653    /// there was no `snapshot.bin` when it opened the store, and it has taken no
1654    /// truncating snapshot since.
1655    ///
1656    /// The archive path's genesis check asks "did a snapshot exist before this
1657    /// one?" as a proxy for "was the WAL ever truncated". The proxy is sound
1658    /// across sessions — this binary cannot tell a history-preserving snapshot
1659    /// from a truncating one once the handle that took it is gone — but inside
1660    /// one session it is not, and `enable_multiplicity` made that visible: its
1661    /// forced `keep_wal` snapshot left the WAL entirely intact and yet
1662    /// permanently disqualified the store from ever receiving a genesis marker
1663    /// (defect #23). This flag is what the proxy defers to when the answer is
1664    /// actually known.
1665    snapshot_preserved_history: bool,
1666    /// Transient write-authz context set by `write_batch_authz` /
1667    /// `query_write_authz` for the duration of ONE mutation call.
1668    /// Always `None` at rest.  Never serialized, never WAL-replayed.
1669    pending_write_authz: Option<WriteAuthz>,
1670    /// Slow-query threshold in milliseconds.  0 = disabled.
1671    /// Seeded from `MUSHROOMDB_SLOW_QUERY_MS` at open; override via
1672    /// [`GraphDb::set_slow_query_threshold_ms`] (tests must use the setter
1673    /// — env vars are process-global and race parallel test threads).
1674    slow_query_threshold_ms: u64,
1675    /// Ring buffer of recent slow queries (interior-mutable so `query(&self)`
1676    /// can record entries without requiring `&mut self`).
1677    slow_queries: std::sync::Mutex<SlowQueryLog>,
1678    /// `(field, label, caller)` triples whose exact-versus-approximate
1679    /// ambiguity this handle has already explained once. See
1680    /// [`note_ambiguous_exactness`](GraphDb::note_ambiguous_exactness).
1681    /// The caller shape is part of the key because the two shapes give
1682    /// different advice — silencing one with the other would leave a caller
1683    /// reading advice meant for a signature it does not have.
1684    /// Advice bookkeeping, not graph state: a reload keeps it, as the
1685    /// slow-query log does.
1686    warned_ambiguous_exactness: std::sync::Mutex<HashSet<(String, String, ExactnessCaller)>>,
1687    /// Instant at which the database was opened (used by `/metrics` uptime).
1688    started_at: std::time::Instant,
1689    // ── Multi-process state (cross-process lock + WAL tailing) ────────────────
1690    /// Byte offset of the WAL prefix already applied to in-memory state.
1691    ///
1692    /// Advanced by exactly the encoded length of every frame this handle
1693    /// appends, and by the decoded byte count of every tail
1694    /// [`refresh`](GraphDb::refresh) absorbs. Rewound by
1695    /// [`set_wal_consumed`](GraphDb::set_wal_consumed) when the group-commit
1696    /// drain thread truncates a failed group. Compared against the WAL's
1697    /// on-disk length to decide staleness.
1698    wal_consumed: u64,
1699    /// Identity of the snapshot this handle's base state came from, as
1700    /// `(len, mtime_nanos)`. A different value means another process replaced
1701    /// the snapshot and the WAL no longer continues our state: refresh reloads.
1702    snapshot_ident: Option<(u64, u64)>,
1703    /// The options this handle was opened with. Replayed verbatim when
1704    /// `refresh` has to rebuild from disk.
1705    open_opts: OpenOptions,
1706    /// True when this handle holds the cross-process write lock for its whole
1707    /// lifetime (a plain read-write open). Per-write lock acquisition is a
1708    /// no-op on such a handle, and never releases the lock.
1709    holds_lifetime_lock: bool,
1710    /// True between a failed lock acquisition and the end of the write scope
1711    /// that failed. Makes every WAL-appending mutation in that scope return
1712    /// [`GraphError::Busy`] instead of writing.
1713    lock_denied: bool,
1714    /// True for an as-of view opened via [`GraphDb::open_at`]. Such a view is
1715    /// pinned to one commit, so it is never stale and never refreshes — later
1716    /// commits by any process are deliberately invisible to it.
1717    pinned: bool,
1718}
1719
1720/// One group of deferred event notifications, held until the group fsync
1721/// completes.  Replayed by [`GraphDb::flush_deferred_events`].
1722struct DeferredEvent {
1723    rec: core_storage::WalRecord,
1724    engine_deltas: Vec<EngineEdgeDelta>,
1725    seq: u64,
1726    ingest: Option<(String, usize)>,
1727}
1728
1729/// Options for [`GraphDb::open_with_options`].
1730#[derive(Clone, Copy, Debug)]
1731pub struct OpenOptions {
1732    /// Rewrite an old-format snapshot to the current VERSION after a
1733    /// successful load (default `true`). The old snapshot is kept as
1734    /// `snapshot.bin.bak` until the next clean open at the current version,
1735    /// at which point the `.bak` is deleted.
1736    ///
1737    /// Set to `false` to open a store without touching any on-disk files
1738    /// (useful for read-only inspection of a store at an older format).
1739    pub auto_migrate: bool,
1740
1741    /// Write the valid WAL prefix back over a torn tail on open (default
1742    /// `true`). Truncating a genuinely torn tail is correct crash recovery.
1743    ///
1744    /// Set to `false` for an unattended reader. The valid prefix is still
1745    /// decoded and replayed in memory, but nothing is written: a reader that
1746    /// opens while another process is mid-append would otherwise discard a
1747    /// frame that writer believes durable. `mushroomdb recall`, which runs on
1748    /// every prompt, passes `false` for exactly this reason.
1749    pub repair_wal: bool,
1750
1751    /// Open without ever writing to the store (default `false`).
1752    ///
1753    /// A read-only handle:
1754    /// - returns [`GraphError::ReadOnly`] from every mutation and from
1755    ///   `snapshot()`;
1756    /// - performs no disk write at open — no WAL repair write-back and no
1757    ///   auto-migration rewrite, whatever the other two flags say;
1758    /// - never takes the cross-process write lock, so it opens immediately even
1759    ///   while another process is writing, and never makes a writer wait.
1760    ///
1761    /// [`refresh`](GraphDb::refresh) and [`is_stale`](GraphDb::is_stale) work
1762    /// normally, so a read-only handle can follow another process's commits.
1763    pub read_only: bool,
1764}
1765
1766impl Default for OpenOptions {
1767    fn default() -> Self {
1768        Self {
1769            auto_migrate: true,
1770            repair_wal: true,
1771            read_only: false,
1772        }
1773    }
1774}
1775
1776/// How long a writer polls for the cross-process write lock before giving up
1777/// with [`GraphError::Busy`].
1778///
1779/// Long enough to ride out another process's commit (a batch apply plus one
1780/// fsync), short enough that a stuck peer surfaces as an error rather than a
1781/// hang.
1782pub const WRITE_LOCK_WAIT: std::time::Duration = std::time::Duration::from_secs(2);
1783
1784/// Refusal when a `MERGE` create cannot choose a namespace.
1785///
1786/// A role bound to two or more namespaces cannot have its create arm land in
1787/// `default`, and the statement did not name `ns`. The role must name one.
1788pub const MERGE_CREATE_NEEDS_ONE_NAMESPACE: &str =
1789    "role-bound token: MERGE create requires the role to name one namespace";
1790
1791/// Interval between poll attempts while waiting for the cross-process lock.
1792pub(crate) const LOCK_POLL_INTERVAL: std::time::Duration = std::time::Duration::from_millis(10);
1793
1794/// Why `load_from_disk` is running, which decides whether it may repair.
1795#[derive(Clone, Copy, Debug, PartialEq, Eq)]
1796enum LoadOrigin {
1797    /// A fresh open. Crash recovery is this handle's job: a torn WAL tail is
1798    /// the signature of a crash and truncating it is correct, and archives
1799    /// orphaned by an interrupted prune can be swept.
1800    Open,
1801    /// A reload driven by [`GraphDb::refresh`], because another process
1802    /// replaced the snapshot. Nothing here is crash recovery — the store is
1803    /// live and someone else is writing it — so this origin writes nothing.
1804    Reload,
1805}
1806
1807/// Authorization context carried by `write_batch_authz` / `query_write_authz`.
1808///
1809/// `None` at the call site = full authority (today's zero-cost behavior).
1810/// `Some(WriteAuthz)` = role-scoped: the decision table (plan §"authz decision
1811/// table") is evaluated per-op inside `commit_logged_batch` BEFORE any WAL
1812/// record is built.  A denial returns an error with no WAL frame written.
1813///
1814/// The mask is ALWAYS `Omit`-mode: role-token paths must never acknowledge
1815/// hidden-node existence to callers.
1816#[derive(Clone, Debug)]
1817pub struct WriteAuthz {
1818    pub role: String,
1819    pub scope: WriteScope,
1820    /// Resolved by `mask_for_role` under the same write guard as the mutation.
1821    /// Always `Omit`-mode — never `Stub`.
1822    pub mask: crate::mask::NodeMask,
1823}
1824
1825/// The error every role surface gives when `roles.json` did not parse at open.
1826///
1827/// One text, so `mask_for_role` and [`GraphDb::roles_checked`] cannot drift
1828/// apart on the same cause.
1829fn roles_poisoned() -> GraphError {
1830    GraphError::Corrupt {
1831        detail: "roles.json was corrupt at open; fix the file and re-open to restore role access"
1832            .into(),
1833    }
1834}
1835
1836/// Write `bytes` to `snapshot.bin.bak` atomically with full fsync.
1837///
1838/// Uses [`RealFs::write_atomic`] which applies `F_FULLFSYNC` on macOS and
1839/// `sync_all` on other platforms, then renames the `.tmp` file into place and
1840/// syncs the directory entry. This is the only correct path for writing the
1841/// `.bak` — plain `std::fs::write + sync_all` misses both `F_FULLFSYNC` and
1842/// the directory sync.
1843pub fn write_snapshot_bak(dir: &std::path::Path, bytes: &[u8]) -> crate::Result<()> {
1844    use core_storage::fs::{FileId, Fs as _};
1845    RealFs::new(dir)
1846        .map_err(core_storage::GraphError::Io)?
1847        .write_atomic(FileId::SnapshotBak, bytes)
1848        .map_err(core_storage::GraphError::Io)
1849}
1850
1851/// Return the on-disk snapshot format version without decoding the full snapshot.
1852///
1853/// Reads only the 6-byte header (magic + version LE). Returns `None` when no
1854/// snapshot file exists (WAL-only store). Returns an error if the header is
1855/// malformed.
1856pub fn snapshot_version_at(dir: &std::path::Path) -> crate::Result<Option<u16>> {
1857    use std::io::Read as _;
1858    let path = dir.join("snapshot.bin");
1859    let mut header = [0u8; 6];
1860    let n = match std::fs::File::open(&path) {
1861        Ok(mut f) => f.read(&mut header).map_err(core_storage::GraphError::Io)?,
1862        Err(e) if e.kind() == std::io::ErrorKind::NotFound => return Ok(None),
1863        Err(e) => return Err(core_storage::GraphError::Io(e)),
1864    };
1865    core_storage::snapshot::peek_version(&header[..n])
1866}
1867
1868/// Options for [`GraphDb::snapshot_with`].
1869#[derive(Debug, Clone, Default)]
1870pub struct SnapshotOptions {
1871    /// When `true`, the WAL is preserved after the snapshot write.
1872    /// Pre-snapshot commits remain reachable via [`GraphDb::open_at`].
1873    /// When `false` (the default), the WAL is truncated to a minimal
1874    /// baseline so cold-start replay stays fast.
1875    pub keep_wal: bool,
1876    /// When `true`, the current WAL is renamed to `wal.<commit_seq>.archive`
1877    /// before a fresh WAL baseline is written (history-preserving snapshot).
1878    ///
1879    /// This is the feature opt-in: `false` (the default) leaves the existing
1880    /// truncation / keep-wal behaviour byte-identical.  `archive_wal` takes
1881    /// precedence over `keep_wal` when both are set.
1882    ///
1883    /// Archives can be scanned by [`GraphDb::node_history`],
1884    /// [`GraphDb::edge_history`], [`GraphDb::was_linked`], and
1885    /// [`GraphDb::open_at`], extending the reachable history horizon across
1886    /// snapshot boundaries.
1887    pub archive_wal: bool,
1888}
1889
1890/// Derive the scan-label sym for the commit-skip fast-path.
1891///
1892/// Walks `ops` to find the plan's leading scan op (`ScanLabel`, `IndexScan`,
1893/// or `IndexIntersect`) with a concrete label string, then interns it.
1894///
1895/// Returns `None` in all cases where skipping is unsafe:
1896/// - Any `Expand` op is present (edge traversal; edges change results regardless
1897///   of node labels).
1898/// - The leading scan has no label (`ScanLabel { label: None }` — full scan).
1899/// - No recognizable leading scan op is found.
1900///
1901/// This is the conservative v0.4.3 boundary. The caller stores the result in
1902/// [`QuerySubEntry::scan_label`] at subscribe time; `None` means always execute.
1903fn extract_scan_label(ops: &[PlanOp], syms: &mut Interner) -> Option<u32> {
1904    // Any Expand → must always re-execute (edges can change join results).
1905    if ops.iter().any(|op| matches!(op, PlanOp::Expand { .. })) {
1906        return None;
1907    }
1908    for op in ops {
1909        match op {
1910            PlanOp::ScanLabel {
1911                label: Some(label), ..
1912            } => return Some(syms.intern(label)),
1913            PlanOp::IndexScan {
1914                label: Some(label), ..
1915            } => return Some(syms.intern(label)),
1916            PlanOp::IndexIntersect {
1917                label: Some(label), ..
1918            } => return Some(syms.intern(label)),
1919            _ => {}
1920        }
1921    }
1922    None
1923}
1924
1925/// How an as-of read is restricted — the argument to
1926/// [`GraphDb::query_at_scoped`].
1927///
1928/// Every variant is resolved against the graph **as it was at the requested
1929/// commit**, not against the current graph.
1930#[derive(Debug, Clone, Copy)]
1931pub enum AsOfScope<'a> {
1932    /// Everything the named role may see. The role *definition* is the current
1933    /// one — `roles.json` is a sidecar and has no past version — but its
1934    /// `keys` and `labels` are resolved against the as-of graph.
1935    Role(&'a str),
1936    /// An explicit node-key allow-list. Keys that did not exist at that commit
1937    /// resolve to nothing.
1938    Keys(&'a [String]),
1939    /// A role intersected with a client-supplied allow-list. The intersection
1940    /// is the never-widen rule: a client mask can only narrow a role.
1941    RoleAndKeys(&'a str, &'a [String]),
1942    /// Every live node in one namespace, as the graph was at that commit.
1943    ///
1944    /// A namespace cannot change — it is set at insert and immutable — so the
1945    /// answer is simply "the nodes that existed then and are in this
1946    /// namespace". A name no node uses resolves to nothing, never to
1947    /// everything.
1948    Namespace(&'a str),
1949}
1950
1951impl GraphDb<RealFs> {
1952    /// Open the database at `dir` with default options.
1953    ///
1954    /// Equivalent to `open_with_options(dir, OpenOptions::default())`.
1955    /// Old-format snapshots (V5, V6) are automatically migrated to the
1956    /// current version on a successful load (see [`OpenOptions::auto_migrate`]).
1957    pub fn open(dir: &std::path::Path) -> Result<Self> {
1958        Self::open_with_options(dir, OpenOptions::default())
1959    }
1960
1961    /// Open the database at `dir` with explicit options.
1962    ///
1963    /// When `opts.auto_migrate` is `true` (the default) and the on-disk
1964    /// snapshot is an older format version, this function:
1965    ///   1. Copies the current `snapshot.bin` to `snapshot.bin.bak` (atomic
1966    ///      + fsynced) before any modification.
1967    ///   2. Rewrites `snapshot.bin` at the current format version via
1968    ///      [`GraphDb::snapshot_with`] with `keep_wal: true` (WAL preserved).
1969    ///
1970    /// If migration fails the error is returned and the original files are
1971    /// intact (the `.bak` was written before the new snapshot was attempted).
1972    ///
1973    /// A clean open that finds the snapshot already at the current version
1974    /// deletes any leftover `.bak` file.
1975    ///
1976    /// WAL-only stores (no snapshot) are never auto-migrated on open.
1977    ///
1978    /// `opts.repair_wal` controls the other write this function can make; see
1979    /// [`OpenOptions::repair_wal`]. With both flags `false` the open touches
1980    /// no file on disk.
1981    pub fn open_with_options(dir: &std::path::Path, opts: OpenOptions) -> Result<Self> {
1982        Self::open_dir(dir, opts, true)
1983    }
1984
1985    /// Open without taking the cross-process write lock for the handle's
1986    /// lifetime.
1987    ///
1988    /// Only [`SharedDb`](crate::SharedDb) uses this: a long-lived server holds
1989    /// its handle open indefinitely, so it takes the lock per write instead of
1990    /// keeping every other process out of the store for as long as it runs.
1991    pub(crate) fn open_unlocked(dir: &std::path::Path) -> Result<Self> {
1992        Self::open_dir(dir, OpenOptions::default(), false)
1993    }
1994
1995    fn open_dir(dir: &std::path::Path, opts: OpenOptions, hold_lock: bool) -> Result<Self> {
1996        // Header-only peek — 6 bytes, no full decode.
1997        let snap_version = snapshot_version_at(dir)?;
1998
1999        // Full load: decode snapshot + replay WAL + rebuild indexes.
2000        let mut db = Self::open_generic(RealFs::new(dir)?, opts, hold_lock)?;
2001
2002        // A read-only handle writes nothing at open, so it never migrates —
2003        // the old-format snapshot is loaded and left exactly as it is.
2004        if opts.auto_migrate && !opts.read_only {
2005            match snap_version {
2006                Some(ver) if ver < core_storage::snapshot::VERSION => {
2007                    let _tm = std::time::Instant::now();
2008                    // Copy the original snapshot to .bak at OS level — no in-memory
2009                    // buffer required for a 2+ GiB file.
2010                    //
2011                    // Crash-safety: snapshot.bin remains intact (write_atomic inside
2012                    // snapshot_with uses a .tmp+rename) until the V8 write succeeds.
2013                    // A torn .bak on crash is acceptable because the original
2014                    // snapshot.bin is the authoritative source until after the rename.
2015                    std::fs::copy(dir.join("snapshot.bin"), dir.join("snapshot.bin.bak"))
2016                        .map_err(core_storage::GraphError::Io)?;
2017                    trace_migrate!("bak copy done", _tm);
2018                    // Rewrite snapshot at current version; keep WAL intact.
2019                    db.snapshot_with(SnapshotOptions {
2020                        keep_wal: true,
2021                        ..SnapshotOptions::default()
2022                    })?;
2023                    trace_migrate!("snapshot_with done", _tm);
2024                }
2025                Some(_) => {
2026                    // Already current version: remove any leftover .bak.
2027                    let bak = dir.join("snapshot.bin.bak");
2028                    if bak.exists() {
2029                        std::fs::remove_file(&bak).map_err(core_storage::GraphError::Io)?;
2030                    }
2031                }
2032                None => {
2033                    // WAL-only store — nothing to migrate on open.
2034                }
2035            }
2036        }
2037
2038        Ok(db)
2039    }
2040
2041    /// Open a read-only view of the database as it existed after `commit`.
2042    ///
2043    /// Commit indices are 0-based over the current WAL: commit 0 is the state
2044    /// after the first WAL frame, commit N-1 is the state after the N-th (most
2045    /// recent) frame.  Call [`GraphDb::open`] to read the full current state.
2046    ///
2047    /// **Replay base.** [`GraphDb::snapshot`] truncates the WAL when it runs,
2048    /// so as-of can only reach commits recorded in the current WAL (those
2049    /// written after the most recent snapshot, or all commits if no snapshot
2050    /// was ever taken).  Commit 0 in `open_at` always refers to the first
2051    /// frame in the WAL that exists on disk, not the first ever write to the
2052    /// database.  When the on-disk snapshot recorded that it truncated the
2053    /// WAL (V7, default `keep_wal: false`), it is loaded as the base state
2054    /// before frame replay, so the as-of view includes all pre-snapshot data.
2055    /// Snapshots written with `keep_wal: true` (and legacy V5/V6 snapshots)
2056    /// are ignored and replay is WAL-only, as before.
2057    ///
2058    /// **Read-only.** Every mutation method and `snapshot()` on the returned
2059    /// instance returns [`GraphError::ReadOnly`].  Queries, `explain()`, and
2060    /// `stats()` work normally.
2061    ///
2062    /// # Errors
2063    /// - [`GraphError::CommitOutOfRange`] if `commit >= wal_commit_count` (including
2064    ///   when the WAL is empty after a snapshot).
2065    pub fn open_at(dir: &std::path::Path, commit: u64) -> Result<Self> {
2066        Self::open_at_with(RealFs::new(dir)?, commit)
2067    }
2068
2069    /// Run a **read-only** Cypher query against the graph as it existed at
2070    /// `commit` — the "time-travel" / agent-replay query. Opens a temporal view
2071    /// of this store's directory at that commit and executes the read there.
2072    ///
2073    /// The current instance is unaffected. Write statements are rejected (the
2074    /// temporal view is read-only). `commit` is a 0-based WAL commit index;
2075    /// `commit == wal_commit_count` (or `open_at`'s range) yields the newest
2076    /// state. Prefer this over holding many historical instances open.
2077    ///
2078    /// # Errors
2079    /// - [`GraphError::CommitOutOfRange`] if `commit` is past the WAL horizon.
2080    /// - A query error for a malformed or write query.
2081    pub fn query_at(
2082        &self,
2083        commit: u64,
2084        cypher: &str,
2085        params: &std::collections::BTreeMap<String, Value>,
2086    ) -> Result<ResultSet> {
2087        let temporal = self.open_at_for_read(commit, cypher)?;
2088        temporal.query(cypher, params)
2089    }
2090
2091    /// Run a **read-only** Cypher query at `commit`, restricted by `scope`.
2092    ///
2093    /// The **graph** is as of `commit`; the **role definition** is as it is
2094    /// now, because `roles.json` is a sidecar and is never a WAL record — it
2095    /// has no past version to read. A role's `keys` and `labels` are resolved
2096    /// against the commit-`commit` graph, so a role that may see a label sees
2097    /// exactly the nodes that carried it then, and an explicit key that did
2098    /// not exist yet resolves to nothing.
2099    ///
2100    /// [`AsOfScope::RoleAndKeys`] intersects the two: a client allow-list can
2101    /// only narrow what a role may see, never widen it.
2102    ///
2103    /// Write statements are rejected, exactly as [`GraphDb::query_at`] rejects
2104    /// them.
2105    ///
2106    /// # Errors
2107    /// - [`GraphError::CommitOutOfRange`] if `commit` is outside the retained
2108    ///   range; the error carries that range.
2109    /// - [`GraphError::KeyNotFound`] with a `role:` prefix for an unknown role,
2110    ///   or [`GraphError::Corrupt`] when `roles.json` was corrupt at open.
2111    /// - A query error for a malformed or write query.
2112    pub fn query_at_scoped(
2113        &self,
2114        commit: u64,
2115        cypher: &str,
2116        params: &std::collections::BTreeMap<String, Value>,
2117        scope: AsOfScope<'_>,
2118    ) -> Result<ResultSet> {
2119        let temporal = self.open_at_for_read(commit, cypher)?;
2120        let mask = temporal.mask_at_scope(scope)?;
2121        temporal.query_masked(cypher, params, &mask)
2122    }
2123
2124    /// As [`GraphDb::query_at_scoped`], with `namespace` intersected into
2125    /// whatever `scope` resolves to.
2126    ///
2127    /// This is what a surface needs when a caller passes `namespace` beside a
2128    /// `role` or a client mask on a time-travel read: [`AsOfScope`] names one
2129    /// restriction, and the namespace is a second one that composes with it
2130    /// rather than replacing it. The intersection is the never-widen rule — a
2131    /// namespace can only narrow what the scope already allows — and both legs
2132    /// are resolved against the graph as it was at `commit`.
2133    ///
2134    /// `AsOfScope::Namespace(ns)` is still the way to ask for a namespace alone.
2135    pub fn query_at_scoped_in_namespace(
2136        &self,
2137        commit: u64,
2138        cypher: &str,
2139        params: &std::collections::BTreeMap<String, Value>,
2140        scope: AsOfScope<'_>,
2141        namespace: &str,
2142    ) -> Result<ResultSet> {
2143        let temporal = self.open_at_for_read(commit, cypher)?;
2144        let mask = temporal
2145            .mask_at_scope(scope)?
2146            .intersect(&temporal.mask_for_namespace(namespace));
2147        temporal.query_masked(cypher, params, &mask)
2148    }
2149
2150    /// Run a **read-only** Cypher query at `commit`, restricted by a
2151    /// [`Scope`](crate::mask::Scope).
2152    ///
2153    /// [`AsOfScope`] names *one* restriction — a role, a key list, a namespace,
2154    /// or a role-and-keys pair. A `Scope` is the general shape a handle carries,
2155    /// and nesting can give it several role or namespace legs at once, so it
2156    /// cannot be spelled as an `AsOfScope`. This is the entry point a scoped
2157    /// handle uses for time travel; `query_at_scoped` stays the way to ask for
2158    /// one named restriction.
2159    ///
2160    /// Both the graph and the scope's key and namespace legs are resolved
2161    /// against `commit`; a role's *definition* is the current one, because
2162    /// `roles.json` is a sidecar with no past version — the same split
2163    /// [`GraphDb::query_at_scoped`] documents.
2164    ///
2165    /// The scope resolves **cold** here: a temporal handle is its own store, so
2166    /// its ids could never be served to a live read, but filling the scope's
2167    /// one-entry key memo from a handle thrown away at the end of this call
2168    /// would evict the live entry for nothing. See
2169    /// [`Scope::resolve_uncached`](crate::mask::Scope::resolve_uncached).
2170    ///
2171    /// # Errors
2172    /// - [`GraphError::CommitOutOfRange`] if `commit` is outside the retained
2173    ///   range; the error carries that range.
2174    /// - [`GraphError::KeyNotFound`] with a `role:` prefix for an unknown role,
2175    ///   or [`GraphError::Corrupt`] when `roles.json` was corrupt at open.
2176    /// - A query error for a malformed or write query.
2177    pub fn query_at_with_scope(
2178        &self,
2179        commit: u64,
2180        cypher: &str,
2181        params: &std::collections::BTreeMap<String, Value>,
2182        scope: &crate::mask::Scope,
2183    ) -> Result<ResultSet> {
2184        let temporal = self.open_at_for_read(commit, cypher)?;
2185        let mask = scope.resolve_uncached(&temporal)?;
2186        temporal.query_masked(cypher, params, &mask)
2187    }
2188
2189    /// Open the temporal view for a time-travel read and refuse write Cypher.
2190    ///
2191    /// Shared by [`GraphDb::query_at`] and [`GraphDb::query_at_scoped`] so both
2192    /// resolve the commit and reject writes identically.
2193    fn open_at_for_read(&self, commit: u64, cypher: &str) -> Result<Self> {
2194        let dir = self.fs.dir().to_path_buf();
2195        let temporal = Self::open_at(&dir, commit)?;
2196        if is_write_tokens(&lex(cypher).map_err(|e| GraphError::QueryError {
2197            detail: format!("lex: {e}"),
2198        })?) {
2199            return Err(GraphError::QueryError {
2200                detail: "query_at is read-only: write statements are not permitted in a \
2201                         time-travel query"
2202                    .into(),
2203            });
2204        }
2205        Ok(temporal)
2206    }
2207}
2208
2209impl<F: Fs> GraphDb<F> {
2210    /// Open over an arbitrary [`Fs`], repairing a torn WAL tail as usual.
2211    pub fn open_with(fs: F) -> Result<Self> {
2212        Self::open_with_repair(fs, true)
2213    }
2214
2215    /// As [`GraphDb::open_with`], but `repair_wal: false` decodes the valid WAL
2216    /// prefix without writing the truncation back. See
2217    /// [`OpenOptions::repair_wal`].
2218    pub fn open_with_repair(fs: F, repair_wal: bool) -> Result<Self> {
2219        Self::open_generic(
2220            fs,
2221            OpenOptions {
2222                repair_wal,
2223                ..OpenOptions::default()
2224            },
2225            true,
2226        )
2227    }
2228
2229    /// Shared open path.
2230    ///
2231    /// `hold_lock` requests the cross-process write lock for the whole handle
2232    /// lifetime — the right behaviour for a plain read-write `GraphDb`, whose
2233    /// owner writes through it directly. [`SharedDb`](crate::SharedDb) passes
2234    /// `false` and takes the lock per write instead, so that a long-lived
2235    /// server does not keep every other process out of the store.
2236    ///
2237    /// A read-only open never takes the lock regardless of `hold_lock`.
2238    fn open_generic(fs: F, opts: OpenOptions, hold_lock: bool) -> Result<Self> {
2239        let mut db = Self::new_empty(fs, opts);
2240        db.read_only = opts.read_only;
2241        if hold_lock && !opts.read_only {
2242            if !db.poll_lock(WRITE_LOCK_WAIT)? {
2243                return Err(GraphError::Busy { holder: None });
2244            }
2245            db.holds_lifetime_lock = true;
2246        }
2247        db.load_from_disk(LoadOrigin::Open)?;
2248        Ok(db)
2249    }
2250
2251    /// A handle with no state loaded: every field at its empty value, the
2252    /// filesystem and options in place. Only [`load_from_disk`] makes it
2253    /// usable.
2254    fn new_empty(fs: F, opts: OpenOptions) -> Self {
2255        Self {
2256            fs,
2257            ids: IdMap::new(),
2258            syms: Interner::new(),
2259            topo: Topology::new(),
2260            props: ColumnStore::new(),
2261            labels: Vec::new(),
2262            ns_names: vec![NS_DEFAULT.to_string()],
2263            node_ns: Vec::new(),
2264            edge_props: EdgeProps::new(),
2265            engine: RuleEngine::new(),
2266            view_store: ViewStore::new(),
2267            fulltext: FulltextIndex::new(),
2268            prop_index: PropertyIndex::new(),
2269            multiplicity: false,
2270            event_sink: None,
2271            fsync: FsyncPolicy::Strict,
2272            commit_seq: 0,
2273            roles: Some(vec![]),
2274            role_masks: Arc::new(crate::mask::RoleMaskCache::new()),
2275            store_id: crate::mask::StoreId::next(),
2276            subscriptions: Vec::new(),
2277            query_subscriptions: Vec::new(),
2278            sub_capacity: DEFAULT_SUB_CAPACITY,
2279            read_only: false,
2280            total_wal_commits: 0,
2281            base: None,
2282            fold_overlay: None,
2283            delta_tail: Vec::new(),
2284            commits_since_fold: 0,
2285            defer_events: false,
2286            deferred_events: Vec::new(),
2287            degraded: false,
2288            v8_sections_loaded: std::sync::atomic::AtomicBool::new(false),
2289            v8_sections_mutex: std::sync::Mutex::new(()),
2290            last_change: HashMap::new(),
2291            wal_archive_retention: None,
2292            wal_horizon_floor: 0,
2293            archive_genesis_chain: false,
2294            // Nothing is proven until `load_from_disk` has looked at the store.
2295            snapshot_preserved_history: false,
2296            pending_write_authz: None,
2297            slow_query_threshold_ms: std::env::var("MUSHROOMDB_SLOW_QUERY_MS")
2298                .ok()
2299                .and_then(|v| v.parse().ok())
2300                .unwrap_or(100),
2301            slow_queries: std::sync::Mutex::new(SlowQueryLog {
2302                entries: std::collections::VecDeque::new(),
2303                total: 0,
2304            }),
2305            warned_ambiguous_exactness: std::sync::Mutex::new(HashSet::new()),
2306            started_at: std::time::Instant::now(),
2307            wal_consumed: 0,
2308            snapshot_ident: None,
2309            open_opts: opts,
2310            holds_lifetime_lock: false,
2311            lock_denied: false,
2312            pinned: false,
2313        }
2314    }
2315
2316    /// Return every field describing stored graph state to its empty value,
2317    /// leaving this handle's own identity alone.
2318    ///
2319    /// Preserved on purpose: the filesystem, open options, lock ownership, the
2320    /// event sink and subscriptions, fsync policy, degraded flag, and the
2321    /// slow-query configuration and log. A caller that registered a sink or a
2322    /// subscription keeps it across a reload.
2323    fn reset_for_reload(&mut self) {
2324        self.ids = IdMap::new();
2325        self.syms = Interner::new();
2326        self.topo = Topology::new();
2327        self.props = ColumnStore::new();
2328        self.labels = Vec::new();
2329        self.ns_names = vec![NS_DEFAULT.to_string()];
2330        self.node_ns = Vec::new();
2331        self.edge_props = EdgeProps::new();
2332        self.engine = RuleEngine::new();
2333        self.view_store = ViewStore::new();
2334        self.fulltext = FulltextIndex::new();
2335        self.prop_index = PropertyIndex::new();
2336        // Cleared like every other declaration: a reload replays the store's own
2337        // WAL, and the opt-in comes back from it or not at all.
2338        self.multiplicity = false;
2339        self.commit_seq = 0;
2340        self.roles = Some(vec![]);
2341        // A fresh cache, not a cleared one: any reader snapshot still holding
2342        // the old `Arc` keeps it to itself, so nothing it memoised against the
2343        // pre-reload store can be read back through this handle.
2344        self.role_masks = Arc::new(crate::mask::RoleMaskCache::new());
2345        // The same move for memos this handle does not own. `commit_seq` is
2346        // zeroed just above and reseeded from `max(last_change)`, which a
2347        // delete-only commit leaves where it was — so a reload can land back on
2348        // a sequence a caller's `Scope` already cached a mask at. A new id is
2349        // what makes that entry stop matching.
2350        self.store_id = crate::mask::StoreId::next();
2351        self.total_wal_commits = 0;
2352        self.base = None;
2353        self.fold_overlay = None;
2354        self.delta_tail = Vec::new();
2355        self.commits_since_fold = 0;
2356        self.deferred_events = Vec::new();
2357        self.v8_sections_loaded
2358            .store(false, std::sync::atomic::Ordering::Release);
2359        self.last_change = HashMap::new();
2360        self.wal_horizon_floor = 0;
2361        self.archive_genesis_chain = false;
2362        // Re-derived by `load_from_disk` from the store it is about to read.
2363        self.snapshot_preserved_history = false;
2364        self.pending_write_authz = None;
2365        self.wal_consumed = 0;
2366        self.snapshot_ident = None;
2367    }
2368
2369    /// Load the snapshot base and replay the WAL into an empty handle — the
2370    /// whole of what opening a store does after the struct exists.
2371    ///
2372    /// Split out of the open path so that [`refresh`](GraphDb::refresh) can
2373    /// rebuild a handle in place, without ownership of `F`, when another
2374    /// process replaces the snapshot underneath it.
2375    ///
2376    /// `origin` decides whether the two repair writes this function can make
2377    /// are appropriate; see [`LoadOrigin`].
2378    fn load_from_disk(&mut self, origin: LoadOrigin) -> Result<usize> {
2379        // Both writes below are crash recovery, and only an open is entitled to
2380        // perform them. A read-only handle promises to touch nothing, and a
2381        // reload driven by `refresh` is looking at a store another process is
2382        // actively writing: what looks like a torn tail there is a peer
2383        // mid-append, and what looks like an orphaned archive may be one that
2384        // peer is about to reference.
2385        let may_repair = origin == LoadOrigin::Open && !self.open_opts.read_only;
2386        let repair_wal = self.open_opts.repair_wal && may_repair;
2387        let db = self;
2388        db.wal_horizon_floor = db.fs.read_horizon_floor()?;
2389        db.archive_genesis_chain = db.fs.has_genesis_marker();
2390        // Opening cleanup: remove orphaned archives — archives whose frames all
2391        // fall below the horizon floor.  Orphans arise when a crash interrupted
2392        // the retention-prune sequence after the floor was written but before
2393        // all surplus archives were deleted.  Safe to delete: floor already
2394        // accounts for their frames.
2395        if may_repair {
2396            db.cleanup_orphaned_archives()?;
2397        }
2398        let _t0 = std::time::Instant::now();
2399        // Peek 6 bytes to determine snapshot version without reading the full
2400        // file. For RealFs this is a true partial read (O(1)); for SimFs the
2401        // default impl reads all bytes and truncates (still correct).
2402        let snap_header = db.fs.read_prefix(FileId::Snapshot, 6)?;
2403        // V8, V9 and V10 share the mmap-able container; V9 only adds section 12
2404        // and V10 adds nothing but its version stamp. A version outside that set
2405        // falls through to the full-read path below, where `snapshot::decode`
2406        // either handles it (V5–V7) or refuses it by name — which is what stops
2407        // an older binary before it reaches the WAL.
2408        let is_v8 = snap_header.len() >= 6
2409            && &snap_header[0..4] == b"GDB1"
2410            && core_storage::snapshot::is_mmap_container(u16::from_le_bytes([
2411                snap_header[4],
2412                snap_header[5],
2413            ]));
2414        // The version this store is stamped with, or `None` when it has never
2415        // been snapshotted. Read from the same six bytes, with no second read.
2416        let snapshot_version = if snap_header.len() >= 6 && &snap_header[0..4] == b"GDB1" {
2417            Some(u16::from_le_bytes([snap_header[4], snap_header[5]]))
2418        } else {
2419            None
2420        };
2421        // No snapshot means no snapshot has ever truncated the WAL, so this
2422        // handle can prove the history is whole. Once a snapshot exists that
2423        // this handle did not take, it cannot: see `snapshot_preserved_history`.
2424        db.snapshot_preserved_history = snap_header.is_empty();
2425        if is_v8 {
2426            // V8: map the file zero-copy (RealFs) or read full bytes (SimFs).
2427            // No 2.4GB heap Vec is allocated on RealFs.
2428            let mapped = Arc::new(
2429                if let Some(snap_path) = db.fs.snapshot_path() {
2430                    core_storage::v8::MappedBase::map(&snap_path)
2431                } else {
2432                    let snap_bytes = db.fs.read(FileId::Snapshot)?;
2433                    core_storage::v8::MappedBase::from_bytes(snap_bytes)
2434                }
2435                .map_err(|e| GraphError::Corrupt {
2436                    detail: format!("v8: mmap open: {e:?}"),
2437                })?,
2438            );
2439            db.restore_v8_base(Arc::clone(&mapped))?;
2440            trace_open!("restore_v8_base", _t0);
2441            db.base = Some(mapped);
2442            trace_open!("base assigned", _t0);
2443        } else if !snap_header.is_empty() {
2444            // Legacy V5-V7: full read required for decode.
2445            let snap_bytes = db.fs.read(FileId::Snapshot)?;
2446            if let Some(state) = core_storage::snapshot::decode(&snap_bytes)? {
2447                db.restore_snapshot_state(state)?;
2448            }
2449        }
2450        // else: snap_header is empty = no snapshot file, fresh store.
2451        //
2452        // Seed commit_seq from the highest seq persisted in last_change so that
2453        // WAL-replay frames (which start at commit_seq+1) always exceed any seq
2454        // already stored in the snapshot.  Without this, a db with one snapshot
2455        // commit would save last_change["a"]=1, then on reopen the first WAL
2456        // frame would replay at seq=1 again — colliding and making WAL-tail
2457        // mutations indistinguishable from the snapshot baseline.
2458        //
2459        // Safety invariant (seq-recycling):
2460        //   Recycled seqs (those below the seeded baseline) were NEVER stored in
2461        //   last_change because they belonged to a previous db lifetime — a new
2462        //   db starts at commit_seq=0 with an empty last_change.  Therefore no
2463        //   CAS precondition can carry a recycled seq as its `expected` value
2464        //   and accidentally match a live node's last_change entry.
2465        //
2466        // `expected:0` on a deleted-then-reinserted node:
2467        //   After deletion, last_changed() returns None; callers that call
2468        //   last_changed() and then use NodeUnchangedSince get None.unwrap_or(0)
2469        //   = 0.  The reinserted node gets seq > 0, so a subsequent CAS with
2470        //   expected=0 correctly conflicts.  The only way to observe actual=0 in
2471        //   a CasConflict would be a caller that invented expected=0 without ever
2472        //   calling last_changed() — unreachable via the documented API contract.
2473        if let Some(&max_seq) = db.last_change.values().max() {
2474            db.commit_seq = db.commit_seq.max(max_seq);
2475        }
2476        let bytes = db.fs.read(FileId::Wal)?;
2477        let (records, valid_len) = decode_all(&bytes);
2478        // The valid prefix is replayed either way; `repair_wal` only decides
2479        // whether the truncation is written back. A reader that races a live
2480        // appender must not persist a truncation the writer never asked for.
2481        if valid_len < bytes.len() && repair_wal {
2482            db.fs.write_atomic(FileId::Wal, &bytes[..valid_len])?;
2483        }
2484        // WAL-present path: build indexes eagerly BEFORE replay so that the
2485        // first replayed record does not trigger the lazy-init guard (which
2486        // would call reindex_all_load_state on an empty graph, defeating the
2487        // point of restoring IVF/HNSW blobs from the snapshot).
2488        if !records.is_empty() {
2489            db.ensure_v8_base_sections_loaded();
2490            trace_open!("lazy sections loaded (WAL path)", _t0);
2491        }
2492        let replayed = db.apply_frames(records)?;
2493        // ── The multiplicity declaration, recovered from the stamp ───────────
2494        //
2495        // The opt-in is re-emitted into every baseline WAL a snapshot writes, so
2496        // ordinarily the replay above has already found it. But
2497        // `snapshot_with(archive_wal)` renames the live WAL away and writes its
2498        // replacement afterwards, and between those two points the store holds
2499        // no live declaration at all. A crash there — or a single `Err` from any
2500        // call in between — used to opt the store back out on the next open
2501        // (defect #22): it would stop counting and write a **V9** snapshot while
2502        // the archives still carried discriminant 23, which is the exact state
2503        // the V10 stamp exists to prevent.
2504        //
2505        // The V10 stamp is what carries the conclusion. The archive clause is a
2506        // scope restriction, not a second proof — an earlier version of this
2507        // comment, and defect #22, claimed otherwise, and defect #33 corrects
2508        // it. Taking the two in order:
2509        //
2510        // **The stamp.** `snapshot_with` stamps the snapshot from
2511        // `self.multiplicity` *before* it touches the WAL, and nothing rewrites
2512        // a V10 snapshot at V9 while the store believes it is opted in. So a
2513        // V10 stamp says this store reached `enable_multiplicity` far enough to
2514        // write the snapshot — and, decisively, that every older binary already
2515        // refuses this store by name. Opting in here can cost such a reader
2516        // nothing it was not already being told.
2517        //
2518        // **What the archive clause does not prove.** It is *not* evidence that
2519        // the archive was taken while the store was opted in. A store can
2520        // archive at V9 and opt in afterwards, leaving a V10 snapshot standing
2521        // beside an archive whose WAL carries no declaration at all — see
2522        // `a_failed_opt_in_beside_an_archive_comes_back_opted_in`. The inference
2523        // held in the success case by coincidence, not by construction.
2524        //
2525        // **What it does buy: scope.** Without it the recovery would also fire
2526        // on a store that reached the V10 snapshot write and then failed with no
2527        // archive in sight. That store must stay opted out, and can: no WAL was
2528        // renamed away, nothing carries discriminant 23, and its next snapshot
2529        // rewrites at V9, which puts it back within reach of every older reader.
2530        // An archive is the marker for the one state that is not recoverable
2531        // that way — a WAL renamed away that may hold the only copy of the
2532        // declaration. `no_crash_leaves_discriminant_23_unguarded` pins that
2533        // line: it sweeps a workload with no archives at all and refuses a
2534        // V10-implies-enabled rule.
2535        //
2536        // **The invariant, whichever way the clause goes:** the recovery never
2537        // opts in a store whose snapshot is not V10. A V9 store has made no
2538        // promise to an older reader, so opting it in would start writing
2539        // discriminant 23 behind a stamp that does not guard it. Pinned by
2540        // `the_recovery_never_opts_in_a_store_whose_snapshot_is_not_v10` and
2541        // `the_recovery_does_not_opt_a_store_in_by_itself`.
2542        //
2543        // What this recovery cannot do is make the opt-in atomic; it is not,
2544        // and `enable_multiplicity` says so. See defects #32-#34.
2545        if !db.multiplicity
2546            && snapshot_version == Some(core_storage::snapshot::VERSION_10)
2547            && !db.fs.list_archives()?.is_empty()
2548        {
2549            db.multiplicity = true;
2550        }
2551        // The cursor sits at the end of the valid prefix, not the end of the
2552        // file: a torn or still-being-written tail is unconsumed by definition
2553        // and stays visible to `is_stale` until it decodes.
2554        db.wal_consumed = valid_len as u64;
2555        db.snapshot_ident = db.fs.snapshot_ident().map_err(GraphError::Io)?;
2556        trace_open!("wal replay done", _t0);
2557        // Rebuild view values after WAL replay only when there is no V8 base.
2558        // With a V8 base, view values are correct in the snapshot and are updated
2559        // incrementally during WAL replay (on_edge_changed / on_prop_changed).
2560        // A full rebuild would read overlay-only props (empty after restore_v8_base)
2561        // and overwrite correct base values with wrong results (e.g. NeighborAgg
2562        // Sum reads no "score" in overlay → writes 0.0, shadowing the correct
2563        // base value).
2564        if db.base.is_none() {
2565            let topo_view = TopologyView::owned(&db.topo);
2566            db.view_store
2567                .rebuild_all(&mut db.props, &topo_view, &db.ids, &db.syms, &db.labels);
2568        }
2569        // Rebuild full-text index after WAL replay.  Corrects drift from
2570        // per-record incremental apply during replay.
2571        db.fulltext.rebuild_all(
2572            &db.ids,
2573            &db.labels,
2574            &db.syms,
2575            build_props_view(&db.props, &db.base),
2576        );
2577        db.prop_index.rebuild_all(
2578            &db.ids,
2579            &db.labels,
2580            &db.syms,
2581            build_props_view(&db.props, &db.base),
2582        );
2583        // Namespaces: one pass over the `ns` column, after the snapshot is
2584        // restored and the WAL replayed. Replay maintains `node_ns` record by
2585        // record as well; this pass is what makes a snapshot-only open right,
2586        // and it reads nothing on a store with no `ns` column.
2587        db.rebuild_node_ns();
2588        // A mid-build snapshot's HNSW blob carries `complete == false`.
2589        // Register it so `serve`'s ticker sees work without waiting for a write.
2590        db.register_outstanding_index_builds();
2591        // Load roles sidecar. Missing file = no roles (Some(vec![])).
2592        // Corrupt/unparseable = poisoned (None); mask_for_role will fail-loud.
2593        db.roles = Self::load_roles_from_fs(&db.fs)?;
2594        // Capture the initial MVCC fold so reader() is ready immediately.
2595        db.fold_now();
2596        trace_open!("open_with complete", _t0);
2597        Ok(replayed)
2598    }
2599
2600    /// Apply decoded WAL frames to in-memory state, exactly as the open-path
2601    /// replay does — same `apply` calls, same per-frame delta drain, same
2602    /// commit-seq and last-change bookkeeping. Rules therefore fire and derived
2603    /// edges appear identically whether a frame arrives at open, from a local
2604    /// commit, or from another process by way of [`refresh`](GraphDb::refresh).
2605    ///
2606    /// Returns the number of frames applied.
2607    ///
2608    /// Deltas are drained and discarded per frame: replayed frames are already
2609    /// reflected on disk, so they are not news to a subscriber, and draining
2610    /// inside the loop keeps `pending_deltas` O(1) over a large WAL (I-2).
2611    fn apply_frames(&mut self, records: Vec<WalRecord>) -> Result<usize> {
2612        if records.is_empty() {
2613            return Ok(0);
2614        }
2615        // Materialize any state retained in the mmap base before the first
2616        // frame lands, so a replayed record cannot trip the lazy-init guard and
2617        // rebuild indexes from an empty graph. Both calls are idempotent.
2618        self.ensure_v8_base_sections_loaded();
2619        self.engine.consume_retained_state_eager(
2620            &self.ids,
2621            &self.syms,
2622            &self.labels,
2623            build_props_view(&self.props, &self.base),
2624        );
2625        let applied = records.len();
2626        for rec in records {
2627            self.apply(&rec)?;
2628            let _ = self.engine.drain_deltas();
2629            // Track commit_seq during replay so last_change entries are
2630            // consistent with the seqs assigned by log_then_apply_with on
2631            // subsequent live commits.  After N replayed frames, commit_seq=N;
2632            // live commits begin at N+1.
2633            self.commit_seq += 1;
2634            let replay_seq = self.commit_seq;
2635            self.update_last_change_from_rec(&rec, replay_seq);
2636        }
2637        // Enforce I-2: if the per-frame drain above is ever removed or skipped,
2638        // this assert catches the regression in debug builds immediately.
2639        debug_assert_eq!(
2640            self.engine.pending_delta_count(),
2641            0,
2642            "pending_deltas non-empty after replay — \
2643             per-frame drain must run inside the loop to keep memory O(1)"
2644        );
2645        // T2 note: the per-frame drain IS the suppression seam for replay.
2646        // Any future as-of replay path (Plan-15 T2) must drain here to feed
2647        // replaying subscribers; the mechanism is already in place.
2648        let _ = self.engine.drain_deltas(); // belt-and-braces no-op after loop drain
2649        Ok(applied)
2650    }
2651
2652    // ── Multi-process safety: cross-process write lock + WAL tailing ──────────
2653    //
2654    // mushroomdb is many-readers / one-writer across processes. Writers take an
2655    // advisory exclusive lock on the store's `LOCK` file; readers never do.
2656    // Every handle tracks how much of the WAL it has consumed, so it can pick
2657    // up another process's commits by decoding only the new tail rather than
2658    // reopening. See `docs/site/concurrency.md`.
2659
2660    /// Whether the store on disk has moved ahead of (or out from under) this
2661    /// handle's in-memory state.
2662    ///
2663    /// True when the WAL's length differs from this handle's cursor — another
2664    /// process committed, or is mid-append — or when the snapshot file's
2665    /// identity changed. Costs two metadata lookups and reads no file contents,
2666    /// so it is cheap enough for a read path to call.
2667    ///
2668    /// Always false for an as-of view from [`GraphDb::open_at`]: such a view is
2669    /// pinned to one commit and later commits are deliberately invisible to it.
2670    pub fn is_stale(&self) -> Result<bool> {
2671        if self.pinned {
2672            return Ok(false);
2673        }
2674        if self.fs.wal_len().map_err(GraphError::Io)? != self.wal_consumed {
2675            return Ok(true);
2676        }
2677        Ok(self.fs.snapshot_ident().map_err(GraphError::Io)? != self.snapshot_ident)
2678    }
2679
2680    /// Bring this handle up to date with every commit other processes have made,
2681    /// and return how many frames were applied.
2682    ///
2683    /// The WAL tail is decoded from this handle's cursor and applied through the
2684    /// same path the open replay uses, so rules fire and derived edges appear
2685    /// exactly as they would on a fresh open. Interners, id maps and indexes
2686    /// stay valid for the same reason.
2687    ///
2688    /// A frame another process is still writing is left alone: a trailing
2689    /// partial frame is a wait, not a corruption, and the handle stays stale
2690    /// until that frame is complete. Nothing is written to disk, so a read-only
2691    /// handle can refresh freely.
2692    ///
2693    /// When the snapshot file's identity changed, or the WAL is shorter than
2694    /// this handle's cursor, the WAL no longer continues our state — another
2695    /// process snapshotted or archived. The handle is then rebuilt from disk
2696    /// with the options it was opened with, and the return value is the number
2697    /// of frames in the new WAL.
2698    ///
2699    /// Returns 0 for an as-of view, which never follows later commits.
2700    ///
2701    /// # Errors
2702    ///
2703    /// An error here leaves the handle **degraded**: it got partway through
2704    /// applying the tail, or partway through a reload, so its in-memory state
2705    /// no longer matches any point on disk. Further mutations are refused and
2706    /// the handle must be reopened. Nothing on disk was damaged — the store
2707    /// itself is fine, and a fresh open recovers it.
2708    pub fn refresh(&mut self) -> Result<u64> {
2709        if self.pinned {
2710            return Ok(0);
2711        }
2712        let disk_ident = self.fs.snapshot_ident().map_err(GraphError::Io)?;
2713        let wal_len = self.fs.wal_len().map_err(GraphError::Io)?;
2714        if disk_ident != self.snapshot_ident || wal_len < self.wal_consumed {
2715            // The WAL no longer continues our state: rebuild from disk. State
2716            // is cleared first, so a failed load leaves an empty handle — mark
2717            // it degraded rather than let a caller read an empty graph as if
2718            // it were the store's contents.
2719            self.reset_for_reload();
2720            return match self.load_from_disk(LoadOrigin::Reload) {
2721                Ok(frames) => Ok(frames as u64),
2722                Err(e) => {
2723                    self.degraded = true;
2724                    Err(e)
2725                }
2726            };
2727        }
2728        if wal_len == self.wal_consumed {
2729            return Ok(0);
2730        }
2731        let tail = self
2732            .fs
2733            .read_range(FileId::Wal, self.wal_consumed)
2734            .map_err(GraphError::Io)?;
2735        let (records, valid_len) = decode_all(&tail);
2736        let applied = match self.apply_frames(records) {
2737            Ok(n) => n,
2738            Err(e) => {
2739                // Some frames landed and some did not, and the cursor cannot
2740                // say how many. Advancing it would skip the rest; leaving it
2741                // would replay what already applied. Neither is recoverable in
2742                // place, so refuse further writes and require a reopen.
2743                self.degraded = true;
2744                return Err(e);
2745            }
2746        };
2747        // Advance by the bytes actually decoded, never by the file length: an
2748        // incomplete trailing frame stays unconsumed for the next refresh.
2749        self.wal_consumed += valid_len as u64;
2750        if applied > 0 {
2751            // Peer commits must reach `reader()` snapshots taken from here on.
2752            // A full fold is what open does; refresh does not build per-commit
2753            // deltas, so there is nothing cheaper that stays correct.
2754            self.fold_now();
2755        }
2756        Ok(applied as u64)
2757    }
2758
2759    /// Byte offset of the WAL prefix this handle has applied.
2760    ///
2761    /// Exposed for tests that assert the cursor tracks appended bytes exactly.
2762    #[doc(hidden)]
2763    pub fn wal_consumed(&self) -> u64 {
2764        self.wal_consumed
2765    }
2766
2767    /// Rewind the WAL cursor after the group-commit drain thread truncated a
2768    /// failed group off the tail, so the cursor still describes the file.
2769    pub(crate) fn set_wal_consumed(&mut self, len: u64) {
2770        self.wal_consumed = len;
2771    }
2772
2773    /// One non-blocking attempt at the cross-process write lock.
2774    ///
2775    /// Takes `&self` so a caller can poll for the lock *before* it acquires the
2776    /// in-process write guard. That ordering is what keeps a busy peer in
2777    /// another process from stalling this process's readers.
2778    ///
2779    /// A handle that owns the lock for its lifetime always succeeds.
2780    pub(crate) fn try_cross_process_lock(&self) -> Result<bool> {
2781        if self.holds_lifetime_lock {
2782            return Ok(true);
2783        }
2784        self.fs.try_lock_exclusive().map_err(GraphError::Io)
2785    }
2786
2787    /// Poll for the cross-process write lock until `wait` elapses.
2788    ///
2789    /// One attempt is always made, so a zero wait is a single try. Returns
2790    /// `false` when the lock is still held elsewhere at the deadline; nothing
2791    /// has been written and retrying later is safe.
2792    ///
2793    /// Only the plain-`GraphDb` open path uses this, where the caller owns the
2794    /// handle outright. [`SharedDb`](crate::SharedDb) polls
2795    /// [`try_cross_process_lock`](GraphDb::try_cross_process_lock) itself so
2796    /// that it holds no in-process guard while it waits.
2797    fn poll_lock(&self, wait: std::time::Duration) -> Result<bool> {
2798        let deadline = std::time::Instant::now() + wait;
2799        loop {
2800            if self.try_cross_process_lock()? {
2801                return Ok(true);
2802            }
2803            let now = std::time::Instant::now();
2804            if now >= deadline {
2805                return Ok(false);
2806            }
2807            std::thread::sleep(LOCK_POLL_INTERVAL.min(deadline.saturating_duration_since(now)));
2808        }
2809    }
2810
2811    /// Open a cross-process write scope, given the outcome of an already-made
2812    /// lock attempt.
2813    ///
2814    /// The caller polls for the lock first — outside any in-process guard — and
2815    /// passes what it got. On success this refreshes, so the writes about to
2816    /// happen land on top of every other process's commits. On failure the
2817    /// handle refuses WAL-appending mutations and `snapshot()` with
2818    /// [`GraphError::Busy`] until [`end_write_lock`](GraphDb::end_write_lock)
2819    /// closes the scope, so a caller holding a guard cannot write behind
2820    /// another process's back.
2821    ///
2822    /// A handle that already owns the lock for its lifetime skips the refresh:
2823    /// no other process can have written, so there is nothing to pick up.
2824    pub(crate) fn enter_write_scope(&mut self, acquired: bool) -> Result<()> {
2825        self.lock_denied = !acquired;
2826        if !acquired || self.holds_lifetime_lock {
2827            return Ok(());
2828        }
2829        if let Err(e) = self.refresh() {
2830            // Do not hold a lock we cannot use: release it and let the caller
2831            // see the underlying failure.
2832            let _ = self.fs.unlock();
2833            self.lock_denied = true;
2834            return Err(e);
2835        }
2836        Ok(())
2837    }
2838
2839    /// Close a cross-process write scope opened by
2840    /// [`enter_write_scope`](GraphDb::enter_write_scope): release the lock and
2841    /// clear the Busy latch. Safe to call when the lock was never taken.
2842    pub(crate) fn end_write_lock(&mut self) {
2843        self.lock_denied = false;
2844        if !self.holds_lifetime_lock {
2845            // Releasing a lock we do not hold is a no-op; a failure to release
2846            // is reported by the OS closing the descriptor at handle drop.
2847            let _ = self.fs.unlock();
2848        }
2849    }
2850
2851    /// As-of replay for [`GraphDb::open_at`]: snapshot base (only when the
2852    /// snapshot truncated the WAL) plus the first `commit + 1` WAL frames;
2853    /// see [`GraphDb::open_at`] for the semantics.  The per-frame drain
2854    /// mirrors `open_with` exactly so pending_delta_count is 0 on exit.
2855    /// Restore all persisted state from a decoded snapshot. Shared by
2856    /// `open_with` and (when the snapshot truncated the WAL) `open_at_with`.
2857    fn restore_snapshot_state(
2858        &mut self,
2859        state: core_storage::snapshot::SnapshotState,
2860    ) -> Result<()> {
2861        self.ids = state.ids;
2862        self.syms = state.syms;
2863        self.topo = state.topo;
2864        self.props = state.props;
2865        self.labels = state.labels;
2866        self.edge_props = state.edge_props;
2867        // Cross-section label integrity for V5/V7 snapshots: same invariants as
2868        // restore_v8_base.  A crafted bincode snapshot with a short `labels` vec,
2869        // out-of-range sym ids, or a sentinel label on a live node would otherwise
2870        // open successfully and panic later in `NodeRef::label()` or
2871        // `neighborhood_masked()`.  Catching it here turns those into typed
2872        // `GraphError::Corrupt` at open time.
2873        {
2874            let ids_len = self.ids.len();
2875            if self.labels.len() != ids_len {
2876                return Err(GraphError::Corrupt {
2877                    detail: format!(
2878                        "snapshot: labels vec has {} entries but id table has {} total slots",
2879                        self.labels.len(),
2880                        ids_len,
2881                    ),
2882                });
2883            }
2884            let syms_len = self.syms.len() as u32;
2885            for (i, &sym) in self.labels.iter().enumerate() {
2886                let is_tombstoned = self.ids.is_tombstoned(i as u32);
2887                if sym == u32::MAX {
2888                    if !is_tombstoned {
2889                        return Err(GraphError::Corrupt {
2890                            detail: format!(
2891                                "snapshot: live node at id slot {i} has sentinel label (u32::MAX)"
2892                            ),
2893                        });
2894                    }
2895                } else if sym >= syms_len {
2896                    return Err(GraphError::Corrupt {
2897                        detail: format!(
2898                            "snapshot: label at id slot {i} references sym {sym} \
2899                             which is out of interner range ({syms_len})"
2900                        ),
2901                    });
2902                }
2903            }
2904        }
2905        let defs: Vec<RuleDef> = state
2906            .rule_defs
2907            .iter()
2908            .map(|b| {
2909                decode_rule_def(b).map_err(|e| GraphError::Corrupt {
2910                    detail: format!("snapshot rule_def deserialize: {e}"),
2911                })
2912            })
2913            .collect::<Result<Vec<_>>>()?;
2914        self.engine =
2915            RuleEngine::from_persist(defs, state.provenance, state.rule_tripped, state.rule_fires);
2916        // Candidate indexes are rebuilt lazily on the first mutation (see
2917        // RuleEngine::on_node_changed).  HNSW blobs and IVF centroids from the
2918        // snapshot are retained without deserializing so that:
2919        //   - clean-open (empty WAL): indexes stay empty; blobs load on first
2920        //     ANN query via ensure_hnsw_loaded, or on first mutation via the
2921        //     lazy-init guard which calls reindex_all_load_state (the scan
2922        //     skips the HNSW build for every side the blob supplies).
2923        //   - WAL-present: open_with calls consume_retained_state_eager before
2924        //     replay so HNSW/IVF are live before any record fires the hooks.
2925        let ivf_bytes = if state.ivf_state.is_empty() {
2926            Vec::new()
2927        } else {
2928            bincode::serialize(&state.ivf_state).expect("IVF state serialize cannot fail")
2929        };
2930        // Store blobs without eagerly deserializing them.
2931        // `self.ids` is the snapshot's id table at this point — WAL replay has
2932        // not run — so its length is the line an interrupted build is detected
2933        // against.
2934        let snapshot_ids = self.ids.len() as u32;
2935        self.engine
2936            .store_snapshot_state(state.hnsw_state, ivf_bytes, snapshot_ids);
2937        // Restore view defs from snapshot (V5).
2938        // The ColumnStore already contains view values from the snapshot;
2939        // use restore_view (no collision check, no backfill) so the store
2940        // is aware of the definitions.  rebuild_all runs after WAL replay.
2941        for def_bytes in &state.view_defs {
2942            let def: ViewDef =
2943                bincode::deserialize(def_bytes).map_err(|e| GraphError::Corrupt {
2944                    detail: format!("snapshot view_def deserialize: {e}"),
2945                })?;
2946            self.view_store
2947                .restore_view(def)
2948                .map_err(|e| GraphError::Corrupt {
2949                    detail: format!("snapshot view restore: {e}"),
2950                })?;
2951        }
2952        Ok(())
2953    }
2954
2955    /// Restore all persisted state from a V8 `MappedBase` snapshot, **except**
2956    /// topology (`self.topo` stays empty and serves as the WAL-replay overlay).
2957    ///
2958    /// `self.props` IS fully materialised from the base so that HNSW/IVF blob
2959    /// deserialization and view rebuild have access to all column data.
2960    fn restore_v8_base(&mut self, mapped: Arc<core_storage::v8::MappedBase>) -> Result<()> {
2961        self.ids = archived_to_idmap(mapped.ids().map_err(|e| GraphError::Corrupt {
2962            detail: format!("v8: ids section: {e:?}"),
2963        })?);
2964        self.syms = archived_to_interner(mapped.syms().map_err(|e| GraphError::Corrupt {
2965            detail: format!("v8: syms section: {e:?}"),
2966        })?);
2967
2968        // C1: self.props is left as an empty overlay. Column reads go through
2969        // props_view() (ColumnsView::with_base), which consults the archived base
2970        // section zero-copy. This avoids the O(columns) heap copy at every open.
2971
2972        // self.topo deliberately left as Topology::new() — overlay path.
2973
2974        let meta = decode_meta(mapped.meta_bytes().map_err(|e| GraphError::Corrupt {
2975            detail: format!("v8: meta section: {e:?}"),
2976        })?)
2977        .map_err(|e| GraphError::Corrupt {
2978            detail: format!("v8: meta decode: {e:?}"),
2979        })?;
2980        self.labels = meta.labels;
2981        // Cross-section label integrity: labels must cover every id slot (live
2982        // and tombstoned), every non-sentinel sym must be within the interner's
2983        // bound, and no live (non-tombstoned) node may carry the u32::MAX
2984        // sentinel label.  Without this check, a crafted snapshot where the META
2985        // section (small, CRC-validated) holds a short `labels` vec, out-of-range
2986        // sym ids, or a sentinel label on a live node, would open successfully
2987        // and then panic in `NodeRef::label()`, `neighborhood_masked()`, and
2988        // related read paths.  Catching the inconsistency here converts those
2989        // panics into typed `GraphError::Corrupt` at open time.
2990        {
2991            let ids_len = self.ids.len();
2992            if self.labels.len() != ids_len {
2993                return Err(GraphError::Corrupt {
2994                    detail: format!(
2995                        "v8: labels section has {} entries but id table has {} total slots",
2996                        self.labels.len(),
2997                        ids_len,
2998                    ),
2999                });
3000            }
3001            let syms_len = self.syms.len() as u32;
3002            for (i, &sym) in self.labels.iter().enumerate() {
3003                let is_tombstoned = self.ids.is_tombstoned(i as u32);
3004                if sym == u32::MAX {
3005                    // Sentinel is only valid for tombstoned slots.
3006                    if !is_tombstoned {
3007                        return Err(GraphError::Corrupt {
3008                            detail: format!(
3009                                "v8: live node at id slot {i} has sentinel label (u32::MAX)"
3010                            ),
3011                        });
3012                    }
3013                } else if sym >= syms_len {
3014                    return Err(GraphError::Corrupt {
3015                        detail: format!(
3016                            "v8: label at id slot {i} references sym {sym} \
3017                             which is out of interner range ({syms_len})"
3018                        ),
3019                    });
3020                }
3021            }
3022        }
3023        // C3: self.edge_props stays as an empty overlay.  Reads go through
3024        // edge_props_view() which consults the mmap'd base section zero-copy
3025        // via EdgePropsView::with_base.  No heap decode at open time.
3026
3027        // Restore rule engine.
3028        let (rule_def_bytes, rule_tripped, rule_fires) =
3029            archived_rules_meta_to_owned(mapped.rules_meta_section().map_err(|e| {
3030                GraphError::Corrupt {
3031                    detail: format!("v8: rules_meta section: {e:?}"),
3032                }
3033            })?);
3034        let defs: Vec<RuleDef> = rule_def_bytes
3035            .iter()
3036            .map(|b| {
3037                decode_rule_def(b).map_err(|e| GraphError::Corrupt {
3038                    detail: format!("v8: rule_def deserialize: {e}"),
3039                })
3040            })
3041            .collect::<Result<Vec<_>>>()?;
3042        self.engine = RuleEngine::from_persist(defs, BTreeMap::new(), rule_tripped, rule_fires);
3043        // C4+C5: provenance, HNSW, and IVF sections are NOT read here.
3044        // `ensure_v8_base_sections_loaded` reads them on first use from
3045        // `self.base` (set by the caller immediately after this returns).
3046        // A clean open touches only: header + IDS + SYMS + META + RULES_META.
3047
3048        // Restore view definitions.
3049        let view_defs =
3050            archived_views_to_owned(mapped.views_section().map_err(|e| GraphError::Corrupt {
3051                detail: format!("v8: views section: {e:?}"),
3052            })?);
3053        for def_bytes in &view_defs {
3054            let def: ViewDef =
3055                bincode::deserialize(def_bytes).map_err(|e| GraphError::Corrupt {
3056                    detail: format!("v8: view_def deserialize: {e}"),
3057                })?;
3058            self.view_store
3059                .restore_view(def)
3060                .map_err(|e| GraphError::Corrupt {
3061                    detail: format!("v8: view restore: {e}"),
3062                })?;
3063        }
3064        // Load the last-change map from section 11 (small section; load eagerly).
3065        // Pre-Task-3 snapshots lack this section; `last_change_bytes` returns &[]
3066        // in that case and `decode_last_change_bytes` returns an empty map.
3067        let last_change_raw = mapped
3068            .last_change_bytes()
3069            .map_err(|e| GraphError::Corrupt {
3070                detail: format!("v8: last_change section: {e:?}"),
3071            })?;
3072        self.last_change = decode_last_change_bytes(last_change_raw);
3073
3074        // Validate that all deferred sections (provenance, HNSW, IVF) fit within
3075        // the file.  Pure bounds check — no bytes read, no page faults triggered.
3076        // Catches truncated snapshots at open time before the lazy deferred reads.
3077        mapped.validate_section_bounds().map_err(|e| match e {
3078            GraphError::Corrupt { detail } => GraphError::Corrupt {
3079                detail: format!("v8: section bounds: {detail}"),
3080            },
3081            other => other,
3082        })?;
3083        Ok(())
3084    }
3085
3086    /// Read provenance, HNSW, and IVF sections from the mmap base into the
3087    /// engine's retained fields on first call.  Subsequent calls are a no-op
3088    /// (AtomicBool fast-path).
3089    ///
3090    /// Must be called before any code path that reads or mutates engine
3091    /// provenance, HNSW, or IVF state:
3092    /// - WAL replay (before `consume_retained_state_eager`)
3093    /// - First mutation (`log_then_apply_with`)
3094    /// - Read-only paths (`stats`, `explain`, `node_edges`)
3095    /// - Snapshot (`snapshot_with`)
3096    ///
3097    /// No-op for fresh stores and V5-V7 opens (`self.base` is `None`).
3098    fn ensure_v8_base_sections_loaded(&self) {
3099        use std::sync::atomic::Ordering;
3100        if self.v8_sections_loaded.load(Ordering::Acquire) {
3101            return;
3102        }
3103        let _guard = self
3104            .v8_sections_mutex
3105            .lock()
3106            .expect("v8 sections mutex poisoned");
3107        if self.v8_sections_loaded.load(Ordering::Acquire) {
3108            return; // another caller populated while we waited
3109        }
3110        let _t = std::time::Instant::now();
3111        if let Some(base) = &self.base {
3112            // Provenance: raw rkyv bytes; CRC validated inside section_bytes.
3113            // Bounds are already validated at open time (restore_v8_base →
3114            // validate_section_bounds) — unreachable post-validate_section_bounds;
3115            // unwrap_or_default is a safety belt against impossible errors.
3116            let prov_bytes = base
3117                .provenance_raw_bytes()
3118                .map(|b| b.to_vec())
3119                .unwrap_or_default();
3120            self.engine.store_provenance_bytes(prov_bytes);
3121            // HNSW: decode rkyv blobs into owned map.
3122            let hnsw_state = base
3123                .hnsw_section()
3124                .map(archived_hnsw_to_owned)
3125                .unwrap_or_default();
3126            // IVF: raw bincode bytes; deserialized on first mutation/query.
3127            let ivf_bytes = base.ivf_bytes().map(|b| b.to_vec()).unwrap_or_default();
3128            // Called before WAL replay on a WAL-present open (`open_with`) and
3129            // before any write on a clean one, so this is the snapshot's count.
3130            let snapshot_ids = self.ids.len() as u32;
3131            self.engine
3132                .store_snapshot_state(hnsw_state, ivf_bytes, snapshot_ids);
3133        }
3134        self.v8_sections_loaded.store(true, Ordering::Release);
3135        if std::env::var("MUSHROOMDB_TRACE_OPEN").is_ok() {
3136            eprintln!(
3137                "[MUSHROOMDB_TRACE_OPEN] ensure_v8_base_sections_loaded: {:>9.3?}",
3138                _t.elapsed()
3139            );
3140        }
3141    }
3142
3143    /// Return a `TopologyView` that merges the mmap'd base (when present) with
3144    /// the in-memory WAL overlay.  Used by all read paths in db.rs that need
3145    /// the full merged topology without going through `self.view()`.
3146    fn topo_view(&self) -> TopologyView<'_> {
3147        match self.base {
3148            None => TopologyView::owned(&self.topo),
3149            Some(ref base) => {
3150                // SAFETY: base lives as long as self; section bounds validated at open.
3151                // topology() uses access_unchecked; all field reads are bounds-checked in seam.rs.
3152                let archived = base
3153                    .topology()
3154                    .expect("base topology section bounds validated at open");
3155                TopologyView::with_base(&self.topo, archived)
3156            }
3157        }
3158    }
3159
3160    /// Return a `ColumnsView` that merges the mmap'd base columns (when a V8
3161    /// snapshot is open) with the in-memory WAL overlay.  Reads consult the
3162    /// overlay first, then fall through to the archived base section zero-copy.
3163    fn props_view(&self) -> core_storage::v8::seam::ColumnsView<'_> {
3164        match self.base {
3165            None => core_storage::v8::seam::ColumnsView::owned(&self.props),
3166            Some(ref base) => {
3167                // columns() uses access_unchecked; field reads are bounds-checked in seam.rs.
3168                let archived = base
3169                    .columns()
3170                    .expect("base columns section bounds validated at open");
3171                core_storage::v8::seam::ColumnsView::with_base_cached(
3172                    &self.props,
3173                    archived,
3174                    base.mixed_cache(),
3175                )
3176                .with_shared_strings(base_string_table(base))
3177            }
3178        }
3179    }
3180
3181    /// Return an `EdgePropsView` that merges the mmap'd base edge-props section
3182    /// (when a V8 snapshot is open) with the in-memory WAL overlay.
3183    ///
3184    /// Reads consult the overlay first (for post-snapshot mutations), then fall
3185    /// through to the archived base section zero-copy.  Tombstones in the
3186    /// overlay mask deleted-from-base entries.
3187    fn edge_props_view(&self) -> EdgePropsView<'_> {
3188        match self.base {
3189            None => EdgePropsView::owned(&self.edge_props),
3190            Some(ref base) => {
3191                // edge_props_section() uses access_unchecked; field reads bounds-checked in seam.rs.
3192                let archived = base
3193                    .edge_props_section()
3194                    .expect("base edge_props section bounds validated at open");
3195                EdgePropsView::with_base(&self.edge_props, archived)
3196            }
3197        }
3198    }
3199
3200    fn open_at_with(fs: F, commit: u64) -> Result<Self> {
3201        // An as-of view never writes and is pinned to one commit: it takes no
3202        // cross-process lock and does not follow later commits.
3203        let mut db = Self::new_empty(
3204            fs,
3205            OpenOptions {
3206                repair_wal: false,
3207                auto_migrate: false,
3208                read_only: true,
3209            },
3210        );
3211        db.pinned = true; // read_only is set after replay, but pinning is immediate
3212        db.wal_horizon_floor = db.fs.read_horizon_floor()?;
3213        db.archive_genesis_chain = db.fs.has_genesis_marker();
3214        // Same orphaned-archive cleanup as open_with: floor was written first
3215        // during pruning, so a crash may have left stale archives below floor.
3216        db.cleanup_orphaned_archives()?;
3217        // Collect archive frames (oldest-first) and live WAL frames.
3218        // Archives represent pre-snapshot history; the snapshot captures the
3219        // cumulative state at the time of archiving.  Crash-window guarantee:
3220        //   A: crash before rename → WAL intact, no archive. Reopen: normal.
3221        //   B: crash after rename, before new WAL → archive present, WAL
3222        //      absent. Reopen: snapshot loaded (full state), no WAL replay.
3223        //   C: crash after new baseline WAL written → normal post-archive.
3224        let archive_ns = db.fs.list_archives()?;
3225        let mut archive_frames_all: Vec<WalRecord> = Vec::new();
3226        for n in &archive_ns {
3227            let arc_bytes = db.fs.read_archive(*n)?;
3228            let (arc_frames, _) = decode_all(&arc_bytes);
3229            archive_frames_all.extend(arc_frames);
3230        }
3231        let total_archive_frames = archive_frames_all.len() as u64;
3232
3233        let live_bytes = db.fs.read(FileId::Wal)?;
3234        let (live_records, _valid_len) = decode_all(&live_bytes);
3235        let total_surviving = total_archive_frames + live_records.len() as u64;
3236        // Global total including any pruned history below the horizon floor.
3237        let total = db.wal_horizon_floor + total_surviving;
3238
3239        // Horizon and range check.
3240        if commit < db.wal_horizon_floor {
3241            return Err(GraphError::CommitOutOfRange {
3242                commit,
3243                total,
3244                floor: db.wal_horizon_floor,
3245            });
3246        }
3247        if commit >= total {
3248            return Err(GraphError::CommitOutOfRange {
3249                commit,
3250                total,
3251                floor: db.wal_horizon_floor,
3252            });
3253        }
3254
3255        // Local index into surviving frames (0 = first frame of oldest archive).
3256        let local = commit - db.wal_horizon_floor;
3257
3258        if local < total_archive_frames {
3259            // Target commit is in an archive.  Correct replay from empty state
3260            // is only possible when the archive chain is an uninterrupted
3261            // genesis chain (first archive taken from a fresh store, no prior
3262            // WAL truncation) and no archives have been pruned (floor == 0).
3263            //
3264            // If either condition is violated the prefix needed to reconstruct
3265            // the requested state is gone; refuse rather than return wrong data.
3266            if db.wal_horizon_floor > 0 || !db.archive_genesis_chain {
3267                return Err(GraphError::CommitOutOfRange {
3268                    commit,
3269                    total,
3270                    floor: db.wal_horizon_floor,
3271                });
3272            }
3273            // Replay all archive frames up to and including the target commit
3274            // from an empty database state.  Archives must be replayed in order
3275            // so that dense-id intern tables are built up correctly.
3276            for rec in archive_frames_all.into_iter().take((local + 1) as usize) {
3277                db.apply(&rec)?;
3278                let _ = db.engine.drain_deltas();
3279            }
3280        } else {
3281            // Target commit is in the live WAL: load snapshot as base, then
3282            // replay the needed live WAL prefix.
3283            //
3284            // Base state: a truncating snapshot (wal_truncated=true) compacts
3285            // all pre-truncation / pre-archive commits.  Dense-id records in
3286            // the live WAL reference ids/interns that the snapshot provides.
3287            // Peek 6 bytes (same pattern as open_with).
3288            let snap_header = db.fs.read_prefix(FileId::Snapshot, 6)?;
3289            let is_v8 = snap_header.len() >= 6
3290                && &snap_header[0..4] == b"GDB1"
3291                && core_storage::snapshot::is_mmap_container(u16::from_le_bytes([
3292                    snap_header[4],
3293                    snap_header[5],
3294                ]));
3295            if is_v8 {
3296                let state = if let Some(snap_path) = db.fs.snapshot_path() {
3297                    let mapped = core_storage::v8::MappedBase::map(&snap_path).map_err(|e| {
3298                        GraphError::Corrupt {
3299                            detail: format!("v8: open_at mmap: {e:?}"),
3300                        }
3301                    })?;
3302                    core_storage::snapshot::decode_v8_from_mapped(&mapped)?
3303                } else {
3304                    let snap_bytes = db.fs.read(FileId::Snapshot)?;
3305                    core_storage::snapshot::decode(&snap_bytes)?
3306                };
3307                if let Some(state) = state {
3308                    if state.wal_truncated {
3309                        db.restore_snapshot_state(state)?;
3310                    }
3311                }
3312            } else if !snap_header.is_empty() {
3313                let snap_bytes = db.fs.read(FileId::Snapshot)?;
3314                if let Some(state) = core_storage::snapshot::decode(&snap_bytes)? {
3315                    if state.wal_truncated {
3316                        db.restore_snapshot_state(state)?;
3317                    }
3318                }
3319            }
3320            // else: snap_header empty = no snapshot file.
3321            let live_local = local - total_archive_frames;
3322            for rec in live_records.into_iter().take((live_local + 1) as usize) {
3323                db.apply(&rec)?;
3324                let _ = db.engine.drain_deltas();
3325            }
3326        }
3327        // Pin: pending_delta_count must be 0 after as-of replay, mirroring T1's
3328        // post-loop assert in open_with.
3329        debug_assert_eq!(
3330            db.engine.pending_delta_count(),
3331            0,
3332            "pending_deltas non-empty after open_at replay — \
3333             per-frame drain must run inside the loop to keep memory O(1)"
3334        );
3335        let _ = db.engine.drain_deltas(); // belt-and-braces no-op
3336                                          // Rebuild view values after WAL replay so derived-edge-driven views
3337                                          // reflect the as-of state.  open_at always uses the legacy path (no V8
3338                                          // base), so topo_view is always owned.
3339        {
3340            let topo_view = TopologyView::owned(&db.topo);
3341            db.view_store
3342                .rebuild_all(&mut db.props, &topo_view, &db.ids, &db.syms, &db.labels);
3343        }
3344        // Rebuild full-text index for as-of view (mirrors open_with pattern).
3345        db.fulltext.rebuild_all(
3346            &db.ids,
3347            &db.labels,
3348            &db.syms,
3349            build_props_view(&db.props, &db.base),
3350        );
3351        db.prop_index.rebuild_all(
3352            &db.ids,
3353            &db.labels,
3354            &db.syms,
3355            build_props_view(&db.props, &db.base),
3356        );
3357        // Namespaces on the temporal handle, built by the same pass the live
3358        // open uses, so an as-of mask narrows by the namespaces of that commit.
3359        db.rebuild_node_ns();
3360        // Load roles sidecar (current roles, not point-in-time).
3361        db.roles = Self::load_roles_from_fs(&db.fs)?;
3362        db.read_only = true;
3363        db.total_wal_commits = total;
3364        // Capture initial fold so reader() is immediately usable.
3365        db.fold_now();
3366        Ok(db)
3367    }
3368
3369    /// Whether this instance is a read-only as-of view.
3370    pub fn is_read_only(&self) -> bool {
3371        self.read_only
3372    }
3373
3374    // ── MVCC epoch reader ─────────────────────────────────────────────────────
3375
3376    /// Clone the current overlay state into a new `FrozenOverlay` and reset
3377    /// the delta tail. Called automatically every `FOLD_EVERY_K` commits and at
3378    /// the end of `open_with` / `open_at_with` to prime the reader.
3379    fn fold_now(&mut self) {
3380        let frozen = crate::reader::FrozenOverlay {
3381            ids: self.ids.clone(),
3382            syms: self.syms.clone(),
3383            topo: self.topo.clone(),
3384            props: self.props.clone(),
3385            labels: self.labels.clone(),
3386            edge_props: self.edge_props.clone(),
3387            roles: self.roles.clone(),
3388            fulltext: self.fulltext.clone(),
3389        };
3390        self.fold_overlay = Some(Arc::new(frozen));
3391        self.delta_tail.clear();
3392        self.commits_since_fold = 0;
3393    }
3394
3395    /// Capture a lock-free reader snapshot of the current db state.
3396    ///
3397    /// The read lock is held only for the duration of this call (to clone a
3398    /// handful of `Arc` handles). Subsequent query operations run without any
3399    /// lock.
3400    pub fn reader(&self) -> crate::reader::ReaderSnapshot {
3401        crate::reader::ReaderSnapshot::new(
3402            self.fold_overlay
3403                .clone()
3404                .expect("fold_overlay is always Some after open_with; call reader() after open"),
3405            self.base.clone(),
3406            self.delta_tail.clone(),
3407            // The snapshot's effective state is exactly this handle's state at
3408            // this commit, so it shares the memo and its version key.
3409            self.commit_seq,
3410            Arc::clone(&self.role_masks),
3411        )
3412    }
3413
3414    /// Append a delta the reader cannot apply, so that a corrupt overlay is
3415    /// reachable from a test.
3416    ///
3417    /// Compiled only under `test-hooks`, which the server's dev-dependency on
3418    /// this crate turns on. One call permanently corrupts every
3419    /// [`ReaderSnapshot`](crate::reader::ReaderSnapshot) taken from the handle,
3420    /// so it must not be in the published surface: `#[doc(hidden)]` hides it
3421    /// from rustdoc and from nothing else. The feature gate — not
3422    /// `#[cfg(test)]` — because its only callers are in `crates/server/tests`,
3423    /// a different crate, exactly as `core_rules`'s index counters are.
3424    ///
3425    /// [`ReaderSnapshot::effective`](crate::reader::ReaderSnapshot) folds the
3426    /// delta tail into a clone of the frozen overlay and answers
3427    /// [`GraphError::Corrupt`] when a record will not apply. Nothing a caller
3428    /// can do produces that state — `apply_one`'s failures are disagreements
3429    /// between the tail and the fold it is applied to, which the write path
3430    /// cannot create — so the `Corrupt` arm of every scoped reader method was
3431    /// reachable only by inspection until this hook existed. An `Intern` record
3432    /// claiming an id the frozen interner will not hand back is the smallest
3433    /// such disagreement.
3434    ///
3435    /// Only the tail is touched. This handle's own state is untouched and
3436    /// `commit_seq` does not move, so a role mask already memoised at this
3437    /// version stays memoised — which is exactly the state in which the HTTP
3438    /// role branches reach a scoped read with a corrupt overlay under them.
3439    #[cfg(any(test, feature = "test-hooks"))]
3440    #[doc(hidden)]
3441    pub fn push_unapplyable_delta_for_test(&mut self) {
3442        self.delta_tail.push(Arc::new(crate::reader::CommitDelta {
3443            records: vec![WalRecord::Intern {
3444                id: u32::MAX,
3445                text: "delta-tail-corruption".into(),
3446            }],
3447            derived_inserts: Vec::new(),
3448            derived_deletes: Vec::new(),
3449        }));
3450    }
3451
3452    /// Total number of WAL commits at the time [`open_at`] was called.
3453    /// Returns 0 for normal (non-as-of) instances.
3454    pub fn total_wal_commits(&self) -> u64 {
3455        self.total_wal_commits
3456    }
3457
3458    /// Apply a record to in-memory state. Used by both live writes and replay,
3459    /// so replay is definitionally identical to the original execution.
3460    fn apply(&mut self, rec: &WalRecord) -> Result<()> {
3461        // Before the record mutates anything: a store restored from a snapshot
3462        // defers building its candidate indexes until the first write, and that
3463        // build is a full node scan. Left where it used to fire — inside the
3464        // engine hook, after `props.set` and the label assignment — the scan
3465        // read the half-applied record and took the in-flight node's vector for
3466        // one the snapshot should have carried, which read as an interrupted
3467        // vector-index build and cost a full `RebuildRule` on the first
3468        // embedded write after every reopen. Hoisted here the scan sees exactly
3469        // the persisted state; the record's own hook then files its vector
3470        // through the ordinary insert path a line later.
3471        self.populate_indexes_before_write();
3472        match rec {
3473            WalRecord::InsertNode { label, key, props } => {
3474                let id = self.ids.try_insert(key)?;
3475                let sym = self.syms.intern(label);
3476                if self.labels.len() <= id as usize {
3477                    // gap slots are sentinels, never valid label symbols
3478                    self.labels.resize(id as usize + 1, u32::MAX);
3479                }
3480                self.labels[id as usize] = sym;
3481                let mut ns_name = NS_DEFAULT.to_string();
3482                for (field, value) in props {
3483                    if field == NS_PROP {
3484                        ns_name = namespace_of_value(Some(value)).to_string();
3485                    }
3486                    self.props.set(id, field, value.clone());
3487                }
3488                self.set_node_ns(id, &ns_name);
3489                // Initialize view values for the new node before the engine runs so
3490                // delta-based increments start from a known zero baseline.
3491                self.view_store
3492                    .init_node_views(id, &mut self.props, &self.syms, &self.labels);
3493                // Fire rules for the newly inserted node.
3494                let cursor = self.engine.pending_delta_count();
3495                let mut eng = std::mem::take(&mut self.engine);
3496                {
3497                    let mut gm = make_graph_mut(
3498                        &self.ids,
3499                        &mut self.syms,
3500                        &self.labels,
3501                        build_props_view(&self.props, &self.base),
3502                        &mut self.topo,
3503                        &self.base,
3504                        &mut self.edge_props,
3505                    );
3506                    eng.on_node_changed(id, None, &mut gm);
3507                }
3508                self.engine = eng;
3509                // Process derived-edge deltas for view maintenance.
3510                // Fast path: skip the O(delta_count) allocation when no views exist.
3511                if !self.view_store.is_empty() {
3512                    #[cfg(test)]
3513                    DELTA_COPY_COUNT.with(|c| c.set(c.get() + 1));
3514                    let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
3515                    for d in &new_deltas {
3516                        self.view_store.on_edge_changed(
3517                            d.etype_sym,
3518                            d.src_id,
3519                            d.dst_id,
3520                            d.fired,
3521                            &mut self.props,
3522                            &build_topo_view(&self.topo, &self.base),
3523                            &self.ids,
3524                            &self.syms,
3525                            &self.labels,
3526                            base_columns(&self.base),
3527                        );
3528                    }
3529                }
3530                // Full-text index maintenance: index enabled fields for this label.
3531                if self.fulltext.has_label(label) {
3532                    for (field, value) in props {
3533                        if self.fulltext.is_enabled(label, field) {
3534                            self.fulltext.add_tokens(id, field, value);
3535                        }
3536                    }
3537                }
3538                // Property (equality) index maintenance.
3539                if self.prop_index.has_label(label) {
3540                    for (field, value) in props {
3541                        self.prop_index.set(label, field, id, value);
3542                    }
3543                }
3544            }
3545            WalRecord::InsertEdge {
3546                edge_type,
3547                src_key,
3548                dst_key,
3549            } => {
3550                let src = self.ids.get(src_key).ok_or_else(|| GraphError::Corrupt {
3551                    detail: format!("wal replay references unknown key {src_key}"),
3552                })?;
3553                let dst = self.ids.get(dst_key).ok_or_else(|| GraphError::Corrupt {
3554                    detail: format!("wal replay references unknown key {dst_key}"),
3555                })?;
3556                let etype = self.syms.intern(edge_type);
3557                // Skip if the edge is already visible in the merged base+overlay
3558                // view.  This keeps WAL replay idempotent when the WAL contains
3559                // pre-snapshot records that are already encoded in a V8 base
3560                // (keep_wal=true opens and crash-before-truncation scenarios).
3561                if self.base.is_some()
3562                    && self
3563                        .topo_view()
3564                        .neighbors(etype, Direction::Out, src)
3565                        .contains(&dst)
3566                {
3567                    return Ok(());
3568                }
3569                self.topo.add_edge(etype, src, dst);
3570                // View maintenance for manual edge insert.
3571                self.view_store.on_edge_changed(
3572                    etype,
3573                    src,
3574                    dst,
3575                    true,
3576                    &mut self.props,
3577                    &build_topo_view(&self.topo, &self.base),
3578                    &self.ids,
3579                    &self.syms,
3580                    &self.labels,
3581                    base_columns(&self.base),
3582                );
3583                // Rule engine: via-hop rules must update when user edges change.
3584                let cursor = self.engine.pending_delta_count();
3585                let mut eng = std::mem::take(&mut self.engine);
3586                {
3587                    let mut gm = make_graph_mut(
3588                        &self.ids,
3589                        &mut self.syms,
3590                        &self.labels,
3591                        build_props_view(&self.props, &self.base),
3592                        &mut self.topo,
3593                        &self.base,
3594                        &mut self.edge_props,
3595                    );
3596                    eng.on_edge_changed(edge_type, src, dst, &mut gm);
3597                }
3598                self.engine = eng;
3599                if !self.view_store.is_empty() {
3600                    let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
3601                    for d in &new_deltas {
3602                        self.view_store.on_edge_changed(
3603                            d.etype_sym,
3604                            d.src_id,
3605                            d.dst_id,
3606                            d.fired,
3607                            &mut self.props,
3608                            &build_topo_view(&self.topo, &self.base),
3609                            &self.ids,
3610                            &self.syms,
3611                            &self.labels,
3612                            base_columns(&self.base),
3613                        );
3614                    }
3615                }
3616            }
3617            WalRecord::SetProp { key, field, value } => {
3618                let id = self.ids.get(key).ok_or_else(|| GraphError::Corrupt {
3619                    detail: format!("wal replay references unknown key {key}"),
3620                })?;
3621                let old_value = build_props_view(&self.props, &self.base)
3622                    .get(id, field)
3623                    .map(|vr| vr.into_value());
3624                self.props.set(id, field, value.clone());
3625                // Fire rules for the changed field.
3626                let cursor = self.engine.pending_delta_count();
3627                let mut eng = std::mem::take(&mut self.engine);
3628                {
3629                    let mut gm = make_graph_mut(
3630                        &self.ids,
3631                        &mut self.syms,
3632                        &self.labels,
3633                        build_props_view(&self.props, &self.base),
3634                        &mut self.topo,
3635                        &self.base,
3636                        &mut self.edge_props,
3637                    );
3638                    eng.on_node_changed(id, Some((field, old_value)), &mut gm);
3639                }
3640                self.engine = eng;
3641                // Derived-edge deltas → view updates.
3642                if !self.view_store.is_empty() {
3643                    #[cfg(test)]
3644                    DELTA_COPY_COUNT.with(|c| c.set(c.get() + 1));
3645                    let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
3646                    for d in &new_deltas {
3647                        self.view_store.on_edge_changed(
3648                            d.etype_sym,
3649                            d.src_id,
3650                            d.dst_id,
3651                            d.fired,
3652                            &mut self.props,
3653                            &build_topo_view(&self.topo, &self.base),
3654                            &self.ids,
3655                            &self.syms,
3656                            &self.labels,
3657                            base_columns(&self.base),
3658                        );
3659                    }
3660                }
3661                // Neighbor-aggregate views that read `field` must also update.
3662                self.view_store.on_prop_changed(
3663                    id,
3664                    field,
3665                    &mut self.props,
3666                    &build_topo_view(&self.topo, &self.base),
3667                    &self.ids,
3668                    &self.syms,
3669                    &self.labels,
3670                    base_columns(&self.base),
3671                );
3672                // Full-text index maintenance: update tokens for this field if indexed.
3673                if self.fulltext.field_indexed(field) {
3674                    let label_opt = self.labels.get(id as usize).and_then(|&sym| {
3675                        if sym == u32::MAX {
3676                            None
3677                        } else {
3678                            self.syms.resolve(sym)
3679                        }
3680                    });
3681                    if let Some(label) = label_opt {
3682                        if self.fulltext.is_enabled(label, field) {
3683                            self.fulltext.remove_node_field(id, field);
3684                            self.fulltext.add_tokens(id, field, value);
3685                        }
3686                    }
3687                }
3688                // Property (equality) index maintenance: re-key this node's value.
3689                if self.prop_index.field_indexed(field) {
3690                    let label_opt = self.labels.get(id as usize).and_then(|&sym| {
3691                        if sym == u32::MAX {
3692                            None
3693                        } else {
3694                            self.syms.resolve(sym)
3695                        }
3696                    });
3697                    if let Some(label) = label_opt {
3698                        self.prop_index.set(label, field, id, value);
3699                    }
3700                }
3701            }
3702            WalRecord::Intern { id, text } => {
3703                if let Some(existing) = self.syms.get(text) {
3704                    if existing != *id {
3705                        return Err(GraphError::Corrupt {
3706                            detail: format!(
3707                                "wal intern mismatch for {text:?}: have {existing}, record {id}"
3708                            ),
3709                        });
3710                    }
3711                } else {
3712                    let got = self.syms.intern(text);
3713                    if got != *id {
3714                        return Err(GraphError::Corrupt {
3715                            detail: format!(
3716                                "wal intern assigned {got} for {text:?}, record wanted {id}"
3717                            ),
3718                        });
3719                    }
3720                }
3721            }
3722            WalRecord::InsertNodeId { label, key, props } => {
3723                let id = self.ids.try_insert(key)?;
3724                if self.labels.len() <= id as usize {
3725                    self.labels.resize(id as usize + 1, u32::MAX);
3726                }
3727                self.labels[id as usize] = *label;
3728                let label_str = self
3729                    .syms
3730                    .resolve(*label)
3731                    .ok_or_else(|| GraphError::Corrupt {
3732                        detail: format!("wal InsertNodeId unknown label intern {label}"),
3733                    })?
3734                    .to_string();
3735                let mut ns_name = NS_DEFAULT.to_string();
3736                for (field_sym, value) in props {
3737                    let field =
3738                        self.syms
3739                            .resolve(*field_sym)
3740                            .ok_or_else(|| GraphError::Corrupt {
3741                                detail: format!(
3742                                    "wal InsertNodeId unknown field intern {field_sym}"
3743                                ),
3744                            })?;
3745                    if field == NS_PROP {
3746                        ns_name = namespace_of_value(Some(value)).to_string();
3747                    }
3748                    self.props.set(id, field, value.clone());
3749                }
3750                self.set_node_ns(id, &ns_name);
3751                self.view_store
3752                    .init_node_views(id, &mut self.props, &self.syms, &self.labels);
3753                let cursor = self.engine.pending_delta_count();
3754                let mut eng = std::mem::take(&mut self.engine);
3755                {
3756                    let mut gm = make_graph_mut(
3757                        &self.ids,
3758                        &mut self.syms,
3759                        &self.labels,
3760                        build_props_view(&self.props, &self.base),
3761                        &mut self.topo,
3762                        &self.base,
3763                        &mut self.edge_props,
3764                    );
3765                    eng.on_node_changed(id, None, &mut gm);
3766                }
3767                self.engine = eng;
3768                if !self.view_store.is_empty() {
3769                    #[cfg(test)]
3770                    DELTA_COPY_COUNT.with(|c| c.set(c.get() + 1));
3771                    let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
3772                    for d in &new_deltas {
3773                        self.view_store.on_edge_changed(
3774                            d.etype_sym,
3775                            d.src_id,
3776                            d.dst_id,
3777                            d.fired,
3778                            &mut self.props,
3779                            &build_topo_view(&self.topo, &self.base),
3780                            &self.ids,
3781                            &self.syms,
3782                            &self.labels,
3783                            base_columns(&self.base),
3784                        );
3785                    }
3786                }
3787                if self.fulltext.has_label(&label_str) {
3788                    for (field_sym, value) in props {
3789                        let Some(field) = self.syms.resolve(*field_sym) else {
3790                            continue;
3791                        };
3792                        if self.fulltext.is_enabled(&label_str, field) {
3793                            self.fulltext.add_tokens(id, field, value);
3794                        }
3795                    }
3796                }
3797                if self.prop_index.has_label(&label_str) {
3798                    for (field_sym, value) in props {
3799                        let Some(field) = self.syms.resolve(*field_sym) else {
3800                            continue;
3801                        };
3802                        self.prop_index.set(&label_str, field, id, value);
3803                    }
3804                }
3805            }
3806            WalRecord::InsertEdgeId { etype, src, dst } => {
3807                // Replay-over-snapshot: dense ids in the pre-snapshot WAL may
3808                // already be tombstoned. Skip rather than attaching edges to
3809                // dead ids (DeleteNode keys the live re-insert, not the old id).
3810                if self.ids.is_tombstoned(*src)
3811                    || self.ids.is_tombstoned(*dst)
3812                    || self.ids.key_of(*src).is_none()
3813                    || self.ids.key_of(*dst).is_none()
3814                {
3815                    return Ok(());
3816                }
3817                // Skip if already visible in the merged view (same idempotency
3818                // guard as InsertEdge above: prevents double-counting when
3819                // pre-snapshot WAL records are replayed over a V8 base).
3820                if self.base.is_some()
3821                    && self
3822                        .topo_view()
3823                        .neighbors(*etype, Direction::Out, *src)
3824                        .contains(dst)
3825                {
3826                    return Ok(());
3827                }
3828                self.topo.add_edge(*etype, *src, *dst);
3829                self.view_store.on_edge_changed(
3830                    *etype,
3831                    *src,
3832                    *dst,
3833                    true,
3834                    &mut self.props,
3835                    &build_topo_view(&self.topo, &self.base),
3836                    &self.ids,
3837                    &self.syms,
3838                    &self.labels,
3839                    base_columns(&self.base),
3840                );
3841                // Rule engine: via-hop rules fire when user via-edges are inserted.
3842                // Resolve etype back to string so on_edge_changed can match rules by name.
3843                if let Some(etype_str) = self.syms.resolve(*etype).map(|s| s.to_string()) {
3844                    let cursor = self.engine.pending_delta_count();
3845                    let mut eng = std::mem::take(&mut self.engine);
3846                    {
3847                        let mut gm = make_graph_mut(
3848                            &self.ids,
3849                            &mut self.syms,
3850                            &self.labels,
3851                            build_props_view(&self.props, &self.base),
3852                            &mut self.topo,
3853                            &self.base,
3854                            &mut self.edge_props,
3855                        );
3856                        eng.on_edge_changed(&etype_str, *src, *dst, &mut gm);
3857                    }
3858                    self.engine = eng;
3859                    if !self.view_store.is_empty() {
3860                        let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
3861                        for d in &new_deltas {
3862                            self.view_store.on_edge_changed(
3863                                d.etype_sym,
3864                                d.src_id,
3865                                d.dst_id,
3866                                d.fired,
3867                                &mut self.props,
3868                                &build_topo_view(&self.topo, &self.base),
3869                                &self.ids,
3870                                &self.syms,
3871                                &self.labels,
3872                                base_columns(&self.base),
3873                            );
3874                        }
3875                    }
3876                }
3877            }
3878            WalRecord::SetPropId { id, field, value } => {
3879                if self.ids.is_tombstoned(*id) || self.ids.key_of(*id).is_none() {
3880                    return Ok(());
3881                }
3882                let field_str = self
3883                    .syms
3884                    .resolve(*field)
3885                    .ok_or_else(|| GraphError::Corrupt {
3886                        detail: format!("wal SetPropId unknown field intern {field}"),
3887                    })?
3888                    .to_string();
3889                let old_value = build_props_view(&self.props, &self.base)
3890                    .get(*id, &field_str)
3891                    .map(|vr| vr.into_value());
3892                self.props.set(*id, &field_str, value.clone());
3893                let cursor = self.engine.pending_delta_count();
3894                let mut eng = std::mem::take(&mut self.engine);
3895                {
3896                    let mut gm = make_graph_mut(
3897                        &self.ids,
3898                        &mut self.syms,
3899                        &self.labels,
3900                        build_props_view(&self.props, &self.base),
3901                        &mut self.topo,
3902                        &self.base,
3903                        &mut self.edge_props,
3904                    );
3905                    eng.on_node_changed(*id, Some((field_str.as_str(), old_value)), &mut gm);
3906                }
3907                self.engine = eng;
3908                if !self.view_store.is_empty() {
3909                    #[cfg(test)]
3910                    DELTA_COPY_COUNT.with(|c| c.set(c.get() + 1));
3911                    let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
3912                    for d in &new_deltas {
3913                        self.view_store.on_edge_changed(
3914                            d.etype_sym,
3915                            d.src_id,
3916                            d.dst_id,
3917                            d.fired,
3918                            &mut self.props,
3919                            &build_topo_view(&self.topo, &self.base),
3920                            &self.ids,
3921                            &self.syms,
3922                            &self.labels,
3923                            base_columns(&self.base),
3924                        );
3925                    }
3926                }
3927                self.view_store.on_prop_changed(
3928                    *id,
3929                    &field_str,
3930                    &mut self.props,
3931                    &build_topo_view(&self.topo, &self.base),
3932                    &self.ids,
3933                    &self.syms,
3934                    &self.labels,
3935                    base_columns(&self.base),
3936                );
3937                if self.fulltext.field_indexed(&field_str) {
3938                    let label_opt = self.labels.get(*id as usize).and_then(|&sym| {
3939                        if sym == u32::MAX {
3940                            None
3941                        } else {
3942                            self.syms.resolve(sym)
3943                        }
3944                    });
3945                    if let Some(label) = label_opt {
3946                        if self.fulltext.is_enabled(label, &field_str) {
3947                            self.fulltext.remove_node_field(*id, &field_str);
3948                            self.fulltext.add_tokens(*id, &field_str, value);
3949                        }
3950                    }
3951                }
3952                if self.prop_index.field_indexed(&field_str) {
3953                    let label_opt = self.labels.get(*id as usize).and_then(|&sym| {
3954                        if sym == u32::MAX {
3955                            None
3956                        } else {
3957                            self.syms.resolve(sym)
3958                        }
3959                    });
3960                    if let Some(label) = label_opt {
3961                        self.prop_index.set(label, &field_str, *id, value);
3962                    }
3963                }
3964            }
3965            WalRecord::CreateRule { def_bytes } => {
3966                let def: RuleDef = decode_rule_def(def_bytes).map_err(|e| GraphError::Corrupt {
3967                    detail: format!("CreateRule def_bytes deserialize failed: {e}"),
3968                })?;
3969                // Replay-over-snapshot idempotency: the rule was captured in the snapshot
3970                // so the engine already has it; silently skip to avoid a spurious
3971                // RuleInvalid error in the crash window between snapshot write and WAL
3972                // truncation.
3973                if self.engine.rules().any(|r| r.name == def.name) {
3974                    return Ok(());
3975                }
3976                let cursor = self.engine.pending_delta_count();
3977                let mut eng = std::mem::take(&mut self.engine);
3978                let result = {
3979                    let mut gm = make_graph_mut(
3980                        &self.ids,
3981                        &mut self.syms,
3982                        &self.labels,
3983                        build_props_view(&self.props, &self.base),
3984                        &mut self.topo,
3985                        &self.base,
3986                        &mut self.edge_props,
3987                    );
3988                    eng.create_rule(def, &mut gm)
3989                };
3990                self.engine = eng;
3991                result.map_err(|e| GraphError::RuleInvalid { detail: e })?;
3992                // Derived-edge fires from backfill → view updates.
3993                // Fast path: skip O(edge_count) allocation when no views exist.
3994                if !self.view_store.is_empty() {
3995                    #[cfg(test)]
3996                    DELTA_COPY_COUNT.with(|c| c.set(c.get() + 1));
3997                    let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
3998                    for d in &new_deltas {
3999                        self.view_store.on_edge_changed(
4000                            d.etype_sym,
4001                            d.src_id,
4002                            d.dst_id,
4003                            d.fired,
4004                            &mut self.props,
4005                            &build_topo_view(&self.topo, &self.base),
4006                            &self.ids,
4007                            &self.syms,
4008                            &self.labels,
4009                            base_columns(&self.base),
4010                        );
4011                    }
4012                }
4013            }
4014            WalRecord::DeleteRule { name } => {
4015                // Replay-over-snapshot idempotency: the snapshot already captured the
4016                // post-delete state so the rule is absent; silently skip to avoid a
4017                // spurious RuleNotFound error in the crash window between snapshot write
4018                // and WAL truncation.
4019                if !self.engine.rules().any(|r| r.name == *name) {
4020                    return Ok(());
4021                }
4022                let cursor = self.engine.pending_delta_count();
4023                let mut eng = std::mem::take(&mut self.engine);
4024                let result = {
4025                    let mut gm = make_graph_mut(
4026                        &self.ids,
4027                        &mut self.syms,
4028                        &self.labels,
4029                        build_props_view(&self.props, &self.base),
4030                        &mut self.topo,
4031                        &self.base,
4032                        &mut self.edge_props,
4033                    );
4034                    eng.delete_rule(name, &mut gm)
4035                };
4036                self.engine = eng;
4037                result.map_err(|_| GraphError::RuleNotFound { name: name.clone() })?;
4038                // Derived-edge retractions → view updates.
4039                if !self.view_store.is_empty() {
4040                    #[cfg(test)]
4041                    DELTA_COPY_COUNT.with(|c| c.set(c.get() + 1));
4042                    let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
4043                    for d in &new_deltas {
4044                        self.view_store.on_edge_changed(
4045                            d.etype_sym,
4046                            d.src_id,
4047                            d.dst_id,
4048                            d.fired,
4049                            &mut self.props,
4050                            &build_topo_view(&self.topo, &self.base),
4051                            &self.ids,
4052                            &self.syms,
4053                            &self.labels,
4054                            base_columns(&self.base),
4055                        );
4056                    }
4057                }
4058            }
4059            WalRecord::RemoveProp { key, field } => {
4060                // Recovery-safe: unknown key or already-absent field is a
4061                // clean no-op. Crash-window replay over a snapshot that
4062                // already applied this record must not Err.
4063                let Some(id) = self.ids.get(key) else {
4064                    return Ok(());
4065                };
4066                // Read old value through the seam for rule retraction.
4067                let old = build_props_view(&self.props, &self.base)
4068                    .get(id, field)
4069                    .map(|vr| vr.into_value());
4070                self.props.remove(id, field);
4071                // If the base still supplies the value after the overlay removal,
4072                // record a tombstone so ColumnsView::get does not resurrect it.
4073                // This covers both the base-only case AND the both-resident case:
4074                //   base-only (in_overlay=false): old prop was only in base, remove
4075                //     is a no-op on overlay, base still visible → tombstone needed.
4076                //   both-resident (in_overlay=true): overlay had v2, base has v1;
4077                //     removing overlay uncovers v1 → tombstone needed.
4078                // Idempotent on double-replay: second pass sees the tombstone →
4079                // get() returns None → condition is false → no duplicate tombstone.
4080                if build_props_view(&self.props, &self.base)
4081                    .get(id, field)
4082                    .is_some()
4083                {
4084                    self.props.record_prop_tombstone(id, field);
4085                }
4086                let cursor = self.engine.pending_delta_count();
4087                let mut eng = std::mem::take(&mut self.engine);
4088                {
4089                    let mut gm = make_graph_mut(
4090                        &self.ids,
4091                        &mut self.syms,
4092                        &self.labels,
4093                        build_props_view(&self.props, &self.base),
4094                        &mut self.topo,
4095                        &self.base,
4096                        &mut self.edge_props,
4097                    );
4098                    eng.on_node_changed(id, Some((field, old)), &mut gm);
4099                }
4100                self.engine = eng;
4101                // Derived-edge deltas → view updates.
4102                if !self.view_store.is_empty() {
4103                    #[cfg(test)]
4104                    DELTA_COPY_COUNT.with(|c| c.set(c.get() + 1));
4105                    let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
4106                    for d in &new_deltas {
4107                        self.view_store.on_edge_changed(
4108                            d.etype_sym,
4109                            d.src_id,
4110                            d.dst_id,
4111                            d.fired,
4112                            &mut self.props,
4113                            &build_topo_view(&self.topo, &self.base),
4114                            &self.ids,
4115                            &self.syms,
4116                            &self.labels,
4117                            base_columns(&self.base),
4118                        );
4119                    }
4120                }
4121                // Neighbor-aggregate views that read `field` must also update.
4122                self.view_store.on_prop_changed(
4123                    id,
4124                    field,
4125                    &mut self.props,
4126                    &build_topo_view(&self.topo, &self.base),
4127                    &self.ids,
4128                    &self.syms,
4129                    &self.labels,
4130                    base_columns(&self.base),
4131                );
4132                // Full-text index maintenance: remove tokens for this field.
4133                if self.fulltext.field_indexed(field) {
4134                    self.fulltext.remove_node_field(id, field);
4135                }
4136                // Property (equality) index maintenance: drop this node's entry.
4137                if self.prop_index.field_indexed(field) {
4138                    if let Some(label) = self.labels.get(id as usize).and_then(|&sym| {
4139                        (sym != u32::MAX).then(|| self.syms.resolve(sym)).flatten()
4140                    }) {
4141                        self.prop_index.remove_node(label, field, id);
4142                    }
4143                }
4144            }
4145            WalRecord::DeleteEdge {
4146                edge_type,
4147                src_key,
4148                dst_key,
4149            } => {
4150                // Recovery-safe: unknown keys, unknown etype, or already-
4151                // absent edge is a clean no-op (remove_edge returns false).
4152                let Some(src) = self.ids.get(src_key) else {
4153                    return Ok(());
4154                };
4155                let Some(dst) = self.ids.get(dst_key) else {
4156                    return Ok(());
4157                };
4158                let Some(etype) = self.syms.get(edge_type) else {
4159                    return Ok(());
4160                };
4161                // I3: phantom-tombstone guard.  When a V8 base is present, a
4162                // DeleteEdge WAL record for an edge that was already absorbed into
4163                // the new base (i.e. neither in overlay nor in base) must be skipped.
4164                // Without this guard, remove_edge records a tombstone for an edge
4165                // that no longer exists, incorrectly understating edge_count.
4166                if self.base.is_some()
4167                    && !self
4168                        .topo_view()
4169                        .neighbors(etype, core_storage::topology::Direction::Out, src)
4170                        .contains(&dst)
4171                {
4172                    return Ok(());
4173                }
4174                self.topo.remove_edge(etype, src, dst);
4175                self.edge_props.remove_edge(etype, src, dst);
4176                // View maintenance for manual edge delete (topo already updated above).
4177                self.view_store.on_edge_changed(
4178                    etype,
4179                    src,
4180                    dst,
4181                    false,
4182                    &mut self.props,
4183                    &build_topo_view(&self.topo, &self.base),
4184                    &self.ids,
4185                    &self.syms,
4186                    &self.labels,
4187                    base_columns(&self.base),
4188                );
4189                // Rule engine: via-hop rules must retract when user via-edges are deleted.
4190                let cursor = self.engine.pending_delta_count();
4191                let mut eng = std::mem::take(&mut self.engine);
4192                {
4193                    let mut gm = make_graph_mut(
4194                        &self.ids,
4195                        &mut self.syms,
4196                        &self.labels,
4197                        build_props_view(&self.props, &self.base),
4198                        &mut self.topo,
4199                        &self.base,
4200                        &mut self.edge_props,
4201                    );
4202                    eng.on_edge_changed(edge_type, src, dst, &mut gm);
4203                }
4204                self.engine = eng;
4205                if !self.view_store.is_empty() {
4206                    let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
4207                    for d in &new_deltas {
4208                        self.view_store.on_edge_changed(
4209                            d.etype_sym,
4210                            d.src_id,
4211                            d.dst_id,
4212                            d.fired,
4213                            &mut self.props,
4214                            &build_topo_view(&self.topo, &self.base),
4215                            &self.ids,
4216                            &self.syms,
4217                            &self.labels,
4218                            base_columns(&self.base),
4219                        );
4220                    }
4221                }
4222            }
4223            WalRecord::DeleteNode { key } => {
4224                // Recovery-safe: already-tombstoned / unknown key is a clean
4225                // no-op. Crash-window replay over a snapshot that already
4226                // applied this record cannot recover the retired id from the
4227                // key (`IdMap::get` is None), so every subsequent step is
4228                // skipped. Each step is independently idempotent if invoked
4229                // twice on a still-live id: retraction is a no-op on empty
4230                // provenance, `remove_edge` returns false, `remove_all` is a
4231                // no-op, `ids.delete` returns None, label sentinel is sticky.
4232                let Some(n) = self.ids.get(key) else {
4233                    return Ok(());
4234                };
4235
4236                // (1) Retract derived edges + de-index while props/labels live.
4237                let cursor = self.engine.pending_delta_count();
4238                let mut eng = std::mem::take(&mut self.engine);
4239                {
4240                    let mut gm = make_graph_mut(
4241                        &self.ids,
4242                        &mut self.syms,
4243                        &self.labels,
4244                        build_props_view(&self.props, &self.base),
4245                        &mut self.topo,
4246                        &self.base,
4247                        &mut self.edge_props,
4248                    );
4249                    eng.on_node_removed(n, &mut gm);
4250                }
4251                self.engine = eng;
4252                // Derived-edge retractions → view updates for neighbors.
4253                if !self.view_store.is_empty() {
4254                    #[cfg(test)]
4255                    DELTA_COPY_COUNT.with(|c| c.set(c.get() + 1));
4256                    let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
4257                    for d in &new_deltas {
4258                        self.view_store.on_edge_changed(
4259                            d.etype_sym,
4260                            d.src_id,
4261                            d.dst_id,
4262                            d.fired,
4263                            &mut self.props,
4264                            &build_topo_view(&self.topo, &self.base),
4265                            &self.ids,
4266                            &self.syms,
4267                            &self.labels,
4268                            base_columns(&self.base),
4269                        );
4270                    }
4271                }
4272
4273                // (2) Sweep ALL remaining edges incident to n, both directions,
4274                // every etype. This cascade is intentionally mask-independent:
4275                // topology integrity requires removing every edge touching the
4276                // deleted node regardless of the caller's visibility scope.
4277                // (The mask limits which nodes a role's read phase can return;
4278                // the WAL delete always executes with full storage authority.)
4279                // Collect then remove so neighbor slices stay valid during
4280                // iteration. Remove from topo first, then call view maintenance
4281                // so Avg/Min/Max recompute sees the correct (reduced) neighbor set.
4282                let etypes: Vec<u32> = self.topo.etypes().collect();
4283                let mut doomed = Vec::new();
4284                for et in &etypes {
4285                    for &dst in self.topo.neighbors(*et, Direction::Out, n).as_ref() {
4286                        doomed.push((*et, n, dst));
4287                    }
4288                    for &src in self.topo.neighbors(*et, Direction::In, n).as_ref() {
4289                        doomed.push((*et, src, n));
4290                    }
4291                }
4292                for (et, s, d) in doomed {
4293                    self.topo.remove_edge(et, s, d);
4294                    self.edge_props.remove_edge(et, s, d);
4295                    // View maintenance: n's own view values will be cleared by
4296                    // remove_all below; only update surviving neighbors.
4297                    self.view_store.on_edge_changed(
4298                        et,
4299                        s,
4300                        d,
4301                        false,
4302                        &mut self.props,
4303                        &build_topo_view(&self.topo, &self.base),
4304                        &self.ids,
4305                        &self.syms,
4306                        &self.labels,
4307                        base_columns(&self.base),
4308                    );
4309                }
4310
4311                // (3) Drop every remaining prop (`ColumnStore::remove_all`).
4312                self.props.remove_all(n);
4313                // Full-text index maintenance: remove all tokens for this node.
4314                self.fulltext.remove_node(n);
4315                // Property (equality) index maintenance: drop all entries for n.
4316                self.prop_index.remove_node_all(n);
4317
4318                // (4) Retire the dense id and stamp the label sentinel.
4319                self.ids.delete(key);
4320                if let Some(slot) = self.labels.get_mut(n as usize) {
4321                    *slot = u32::MAX;
4322                }
4323            }
4324            WalRecord::Batch(inner) => {
4325                // Apply each inner record in order through the same apply path.
4326                // Inner records are validated free of nested Batch by encode_record.
4327                for rec in inner {
4328                    self.apply(rec)?;
4329                }
4330            }
4331            WalRecord::RebuildRule { name } => {
4332                // Replay-over-snapshot idempotency: the snapshot may already
4333                // reflect a later delete_rule, so the rule is absent; skip.
4334                if !self.engine.rules().any(|r| r.name == *name) {
4335                    return Ok(());
4336                }
4337                let cursor = self.engine.pending_delta_count();
4338                let mut eng = std::mem::take(&mut self.engine);
4339                let result = {
4340                    let mut gm = make_graph_mut(
4341                        &self.ids,
4342                        &mut self.syms,
4343                        &self.labels,
4344                        build_props_view(&self.props, &self.base),
4345                        &mut self.topo,
4346                        &self.base,
4347                        &mut self.edge_props,
4348                    );
4349                    eng.rebuild(name, &mut gm)
4350                };
4351                self.engine = eng;
4352                result.map_err(|_| GraphError::RuleNotFound { name: name.clone() })?;
4353                // Derived-edge delta changes → view updates.
4354                if !self.view_store.is_empty() {
4355                    #[cfg(test)]
4356                    DELTA_COPY_COUNT.with(|c| c.set(c.get() + 1));
4357                    let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
4358                    for d in &new_deltas {
4359                        self.view_store.on_edge_changed(
4360                            d.etype_sym,
4361                            d.src_id,
4362                            d.dst_id,
4363                            d.fired,
4364                            &mut self.props,
4365                            &build_topo_view(&self.topo, &self.base),
4366                            &self.ids,
4367                            &self.syms,
4368                            &self.labels,
4369                            base_columns(&self.base),
4370                        );
4371                    }
4372                }
4373            }
4374            WalRecord::CreateView { def_bytes } => {
4375                let def: ViewDef =
4376                    bincode::deserialize(def_bytes).map_err(|e| GraphError::Corrupt {
4377                        detail: format!("CreateView def_bytes deserialize failed: {e}"),
4378                    })?;
4379                // Replay-over-snapshot idempotency: view already present → skip.
4380                if self.view_store.has_view(&def.name) {
4381                    return Ok(());
4382                }
4383                self.view_store
4384                    .create_view(
4385                        def,
4386                        &mut self.props,
4387                        &build_topo_view(&self.topo, &self.base),
4388                        &self.ids,
4389                        &self.syms,
4390                        &self.labels,
4391                    )
4392                    .map_err(|e| GraphError::RuleInvalid { detail: e })?;
4393            }
4394            WalRecord::DeleteView { name } => {
4395                // Replay-over-snapshot idempotency: view already absent → skip.
4396                if !self.view_store.has_view(name) {
4397                    return Ok(());
4398                }
4399                self.view_store
4400                    .delete_view(name, &mut self.props, &self.ids, &self.labels, &self.syms)
4401                    .map_err(|_| GraphError::RuleNotFound { name: name.clone() })?;
4402            }
4403            WalRecord::EnableFulltext { label, field } => {
4404                // Replay-over-snapshot idempotency: already enabled → skip.
4405                if self.fulltext.is_enabled(label, field) {
4406                    return Ok(());
4407                }
4408                self.fulltext.enable(label, field);
4409                // Backfill: index all live nodes of this label that have the field.
4410                let n = self.ids.len() as u32;
4411                for id in 0..n {
4412                    let Some(&sym) = self.labels.get(id as usize) else {
4413                        continue;
4414                    };
4415                    if sym == u32::MAX {
4416                        continue; // tombstoned
4417                    }
4418                    let Some(lbl) = self.syms.resolve(sym) else {
4419                        continue;
4420                    };
4421                    if lbl != label {
4422                        continue;
4423                    }
4424                    if let Some(value) = build_props_view(&self.props, &self.base)
4425                        .get(id, field)
4426                        .map(|vr| vr.into_value())
4427                    {
4428                        self.fulltext.add_tokens(id, field, &value);
4429                    }
4430                }
4431            }
4432            WalRecord::DisableFulltext { label, field } => {
4433                // Replay-over-snapshot idempotency: already disabled → skip.
4434                if !self.fulltext.is_enabled(label, field) {
4435                    return Ok(());
4436                }
4437                // If another label still indexes this field, the postings column
4438                // is kept — but it must not contain node_ids from the now-disabled
4439                // label.  Remove them before calling disable() so the field_indexed
4440                // guard inside disable() sees the correct post-removal state.
4441                if self.fulltext.field_indexed_by_other(label, field) {
4442                    if let Some(label_sym) = self.syms.get(label) {
4443                        for (node_id, &lsym) in self.labels.iter().enumerate() {
4444                            if lsym == label_sym {
4445                                self.fulltext.remove_node_field(node_id as u32, field);
4446                            }
4447                        }
4448                    }
4449                }
4450                self.fulltext.disable(label, field);
4451            }
4452            WalRecord::EnableIndex { label, field } => {
4453                // Replay-over-snapshot idempotency: already enabled → skip.
4454                if self.prop_index.is_enabled(label, field) {
4455                    return Ok(());
4456                }
4457                self.prop_index.enable(label, field);
4458                // Backfill: index all live nodes of this label that have the field.
4459                let n = self.ids.len() as u32;
4460                for id in 0..n {
4461                    let Some(&sym) = self.labels.get(id as usize) else {
4462                        continue;
4463                    };
4464                    if sym == u32::MAX {
4465                        continue; // tombstoned
4466                    }
4467                    let Some(lbl) = self.syms.resolve(sym) else {
4468                        continue;
4469                    };
4470                    if lbl != label {
4471                        continue;
4472                    }
4473                    if let Some(value) = build_props_view(&self.props, &self.base)
4474                        .get(id, field)
4475                        .map(|vr| vr.into_value())
4476                    {
4477                        self.prop_index.set(label, field, id, &value);
4478                    }
4479                }
4480            }
4481            WalRecord::DisableIndex { label, field } => {
4482                self.prop_index.disable(label, field);
4483            }
4484            // ── insert-count multiplicity (§5.13) ────────────────────────────
4485            //
4486            // Two shapes, told apart by `count`: the opt-in declaration, and an
4487            // absolute count for one triple. Absolute is what makes this
4488            // idempotent over a snapshot base — a pre-snapshot frame replayed
4489            // over a base that already folded it in lands on the same number
4490            // rather than adding to it, which is the failure a delta (or a count
4491            // derived from `InsertEdgeId` records) would have.
4492            WalRecord::SetEdgeCount {
4493                etype,
4494                src,
4495                dst,
4496                count,
4497            } => {
4498                if rec.is_multiplicity_decl() {
4499                    self.multiplicity = true;
4500                } else {
4501                    self.edge_props.set(
4502                        *etype,
4503                        *src,
4504                        *dst,
4505                        EDGE_COUNT_PROP,
4506                        Value::Int(*count as i64),
4507                    );
4508                }
4509            }
4510            // History markers carry no replay state — rules re-derive edges
4511            // deterministically on open/replay. Skip unconditionally.
4512            WalRecord::DerivedEdgeAdded { .. } | WalRecord::DerivedEdgeRetracted { .. } => {}
4513            // ── rename_node ──────────────────────────────────────────────────
4514            WalRecord::RenameNode { old_key, new_key } => {
4515                // Recovery-safe: if old_key is already gone (key was renamed
4516                // by a snapshot or a prior replay frame), skip cleanly.
4517                if self.ids.get(old_key).is_none() {
4518                    return Ok(());
4519                }
4520                // The rename only updates the key-table; the dense id, all
4521                // topo edges, props, labels, and rule state are id-indexed and
4522                // require no change.
4523                self.ids
4524                    .rename(old_key, new_key)
4525                    .map_err(|e| GraphError::Corrupt {
4526                        detail: format!("wal replay RenameNode {old_key}→{new_key}: {e}"),
4527                    })?;
4528            }
4529        }
4530        Ok(())
4531    }
4532
4533    /// Intern `s` in `syms` and emit a WAL `Intern` record so `*Id` records
4534    /// replay on WAL-only `open_at` (no snapshot intern table). Apply is
4535    /// idempotent when the string is already bound. Always emit: after
4536    /// `snapshot()` the WAL is truncated and live intern is not on disk.
4537    fn intern_wal(&mut self, s: &str) -> (u32, WalRecord) {
4538        let id = if let Some(id) = self.syms.get(s) {
4539            id
4540        } else {
4541            self.syms.intern(s)
4542        };
4543        (
4544            id,
4545            WalRecord::Intern {
4546                id,
4547                text: s.to_string(),
4548            },
4549        )
4550    }
4551
4552    /// Rewrite user-facing records into dense-id records. On `Err`, no live
4553    /// state is left mutated: speculative interns made while building the
4554    /// output are rolled back, so a later successful mutation cannot log an
4555    /// `Intern` record whose id replay would never reproduce.
4556    fn rewrite_wal_dense(&mut self, recs: Vec<WalRecord>) -> Result<Vec<WalRecord>> {
4557        self.rewrite_wal_dense_planned(recs.into_iter().map(PlannedRec::Rec).collect())
4558    }
4559
4560    /// [`rewrite_wal_dense`](Self::rewrite_wal_dense) for a frame that still
4561    /// carries [`PlannedRec::DuplicateCount`] entries — the shape a batch
4562    /// produces, where a duplicate's count can only be named once this pass has
4563    /// assigned the frame's own ids.
4564    fn rewrite_wal_dense_planned(&mut self, recs: Vec<PlannedRec>) -> Result<Vec<WalRecord>> {
4565        let syms_checkpoint = self.syms.len();
4566        let result = self.rewrite_wal_dense_inner(recs);
4567        if result.is_err() {
4568            self.syms.truncate(syms_checkpoint);
4569        }
4570        result
4571    }
4572
4573    fn rewrite_wal_dense_inner(&mut self, recs: Vec<PlannedRec>) -> Result<Vec<WalRecord>> {
4574        let mut out = Vec::with_capacity(recs.len());
4575        // Node ids allocated by later apply(InsertNodeId) in this same batch.
4576        let mut pending: std::collections::HashMap<String, u32> = std::collections::HashMap::new();
4577        // Namespace of each node inserted earlier in this same frame, so a SET
4578        // on a node this frame created is measured against the namespace it was
4579        // created in rather than against the store, where it does not exist yet.
4580        let mut pending_ns: std::collections::HashMap<String, String> =
4581            std::collections::HashMap::new();
4582        let mut interned = std::collections::HashSet::<u32>::new();
4583        let mut next = u32::try_from(self.ids.len()).map_err(|_| GraphError::Corrupt {
4584            detail: "id space exhausted".into(),
4585        })?;
4586        // Insert counts this frame has already raised. `edge_insert_count`
4587        // reads committed state, which cannot see a count queued earlier in
4588        // this same frame, so N duplicates of one pair would otherwise all
4589        // compute `committed + 1` and the last would win.
4590        let mut pending_counts: HashMap<(u32, u32, u32), u64> = HashMap::new();
4591        let lookup = |ids: &IdMap,
4592                      pending: &std::collections::HashMap<String, u32>,
4593                      key: &str|
4594         -> Option<u32> { ids.get(key).or_else(|| pending.get(key).copied()) };
4595        for rec in recs {
4596            // A duplicate insert's count, resolved here and nowhere else.
4597            //
4598            // This is the only pass that knows the frame's own ids: a node
4599            // created earlier in the same frame has no dense id until the
4600            // `InsertNodeId` above allocates one, and an edge type first used in
4601            // this frame is not in `syms` until `intern_wal` puts it there.
4602            // Resolving the count in the batch's validate pass instead — where
4603            // it used to live — meant that a duplicate whose endpoints or type
4604            // were created in the same frame silently produced no count at all,
4605            // which is exactly the shape a mirror rebuild writes (defect #24).
4606            let rec = match rec {
4607                PlannedRec::Rec(rec) => rec,
4608                PlannedRec::DuplicateCount {
4609                    edge_type,
4610                    src_key,
4611                    dst_key,
4612                } => {
4613                    let (etype, intern) = self.intern_wal(&edge_type);
4614                    if interned.insert(etype) {
4615                        out.push(intern);
4616                    }
4617                    let src = lookup(&self.ids, &pending, &src_key).ok_or_else(|| {
4618                        GraphError::Corrupt {
4619                            detail: format!("dense WAL rewrite missing src {src_key}"),
4620                        }
4621                    })?;
4622                    let dst = lookup(&self.ids, &pending, &dst_key).ok_or_else(|| {
4623                        GraphError::Corrupt {
4624                            detail: format!("dense WAL rewrite missing dst {dst_key}"),
4625                        }
4626                    })?;
4627                    let count = pending_counts
4628                        .get(&(etype, src, dst))
4629                        .copied()
4630                        .unwrap_or_else(|| self.edge_insert_count(etype, src, dst))
4631                        .saturating_add(1);
4632                    pending_counts.insert((etype, src, dst), count);
4633                    out.push(WalRecord::SetEdgeCount {
4634                        etype,
4635                        src,
4636                        dst,
4637                        count,
4638                    });
4639                    continue;
4640                }
4641            };
4642            match rec {
4643                WalRecord::InsertNode { label, key, props } => {
4644                    // Namespace validation and normalisation, on the one seam
4645                    // every user-visible node insert passes through: insert_node,
4646                    // a batch, ingest, Cypher CREATE and MERGE all arrive here
4647                    // before the WAL append, and replay never does.
4648                    let (props, ns_name) = Self::normalise_insert_ns(&key, props)?;
4649                    pending_ns.insert(key.clone(), ns_name);
4650                    let (label_id, intern) = self.intern_wal(&label);
4651                    if interned.insert(label_id) {
4652                        out.push(intern);
4653                    }
4654                    let mut props_id = Vec::with_capacity(props.len());
4655                    for (field, value) in props {
4656                        let (field_id, intern) = self.intern_wal(&field);
4657                        if interned.insert(field_id) {
4658                            out.push(intern);
4659                        }
4660                        props_id.push((field_id, value));
4661                    }
4662                    if lookup(&self.ids, &pending, &key).is_none() {
4663                        pending.insert(key.clone(), next);
4664                        next = next.checked_add(1).ok_or_else(|| GraphError::Corrupt {
4665                            detail: "id space exhausted".into(),
4666                        })?;
4667                    }
4668                    out.push(WalRecord::InsertNodeId {
4669                        label: label_id,
4670                        key,
4671                        props: props_id,
4672                    });
4673                }
4674                WalRecord::SetProp { key, field, value } => {
4675                    // A namespace is set at insert and fixed after: the write is
4676                    // refused when it would move the node, and dropped when it
4677                    // names the namespace the node is already in. Checked here
4678                    // so set_prop, a batch, Cypher SET/MERGE and every upsert
4679                    // that merges props get the same answer.
4680                    if field == NS_PROP {
4681                        let Value::Str(ref to) = value else {
4682                            return Err(GraphError::RuleInvalid {
4683                                detail: format!(
4684                                    "node {key}: {NS_PROP} must be a string naming a namespace, \
4685                                     got {value:?}"
4686                                ),
4687                            });
4688                        };
4689                        let from = pending_ns
4690                            .get(&key)
4691                            .cloned()
4692                            .or_else(|| self.namespace_of(&key))
4693                            .unwrap_or_else(|| NS_DEFAULT.to_string());
4694                        let to = to.clone();
4695                        if to != from {
4696                            return Err(GraphError::NamespaceImmutable {
4697                                key: key.clone(),
4698                                from,
4699                                to,
4700                            });
4701                        }
4702                        continue;
4703                    }
4704                    let id =
4705                        lookup(&self.ids, &pending, &key).ok_or_else(|| GraphError::Corrupt {
4706                            detail: format!("dense WAL rewrite missing key {key}"),
4707                        })?;
4708                    let (field_id, intern) = self.intern_wal(&field);
4709                    if interned.insert(field_id) {
4710                        out.push(intern);
4711                    }
4712                    out.push(WalRecord::SetPropId {
4713                        id,
4714                        field: field_id,
4715                        value,
4716                    });
4717                }
4718                WalRecord::InsertEdge {
4719                    edge_type,
4720                    src_key,
4721                    dst_key,
4722                } => {
4723                    let (etype, intern) = self.intern_wal(&edge_type);
4724                    if interned.insert(etype) {
4725                        out.push(intern);
4726                    }
4727                    let src = lookup(&self.ids, &pending, &src_key).ok_or_else(|| {
4728                        GraphError::Corrupt {
4729                            detail: format!("dense WAL rewrite missing src {src_key}"),
4730                        }
4731                    })?;
4732                    let dst = lookup(&self.ids, &pending, &dst_key).ok_or_else(|| {
4733                        GraphError::Corrupt {
4734                            detail: format!("dense WAL rewrite missing dst {dst_key}"),
4735                        }
4736                    })?;
4737                    out.push(WalRecord::InsertEdgeId { etype, src, dst });
4738                }
4739                WalRecord::RenameNode {
4740                    ref old_key,
4741                    ref new_key,
4742                } => {
4743                    // Track the rename in `pending` so subsequent InsertEdge /
4744                    // SetProp records in this batch can resolve the new key.
4745                    let id = lookup(&self.ids, &pending, old_key).ok_or_else(|| {
4746                        GraphError::Corrupt {
4747                            detail: format!(
4748                                "dense WAL rewrite: RenameNode old key {old_key} not found"
4749                            ),
4750                        }
4751                    })?;
4752                    pending.remove(old_key.as_str());
4753                    pending.insert(new_key.clone(), id);
4754                    out.push(rec);
4755                }
4756                // # Symbol-order invariant (load-bearing)
4757                //
4758                // Write-time and replay-time symbol assignment must agree: every
4759                // symbol in a `Batch` frame has to receive the same dense id when
4760                // the frame's records are replayed in order as it received when
4761                // the frame was written.
4762                //
4763                // A rule's backfill interns its `edge_type` lazily
4764                // (`core_rules::engine`, every `g.syms.intern(&def.edge_type)`
4765                // site), and that backfill runs from `apply` — during the
4766                // `CreateRule` record itself, and again from any later
4767                // `InsertNodeId` in the same frame that makes the rule fire. At
4768                // write time the whole batch is rewritten before any of it is
4769                // applied, so a later `InsertEdge` in the same batch would win the
4770                // lower id for its edge type; on replay the rule's lazy intern
4771                // gets there first and steals it, and the `Intern` record fails at
4772                // the `wal intern assigned …` check in `apply`.
4773                //
4774                // Pre-interning the rule's `edge_type` here, and emitting its
4775                // `Intern` record ahead of the `CreateRule` record, makes both
4776                // orders identical. `weight_prop` needs no pre-intern:
4777                // `EdgeProps::set` keys props by `String`, never through the
4778                // interner. `via_edge` needs none either: via-hop rules resolve it
4779                // with `syms.get` and skip when it is absent.
4780                //
4781                // `RebuildRule` and `DeleteRule` need no such handling here:
4782                // `RebuildRule` has no `BatchOp` variant, so it never appears
4783                // inside a `Batch` today — it is only ever issued as its own
4784                // standalone commit (`rebuild_rule`, or the auto-rebuild path
4785                // that logs it as a second commit after the triggering op).
4786                // `DeleteRule` does have a `BatchOp` variant and can appear
4787                // inside a `Batch`, but it carries only a rule `name` — no
4788                // `edge_type` or other symbol that needs pre-interning — so
4789                // only `CreateRule` needs this arm.
4790                WalRecord::CreateRule { ref def_bytes } => {
4791                    let def = decode_rule_def(def_bytes).map_err(|e| GraphError::Corrupt {
4792                        detail: format!("CreateRule def_bytes deserialize failed: {e}"),
4793                    })?;
4794                    let (etype, intern) = self.intern_wal(&def.edge_type);
4795                    if interned.insert(etype) {
4796                        out.push(intern);
4797                    }
4798                    out.push(rec);
4799                }
4800                other => out.push(other),
4801            }
4802        }
4803        Ok(out)
4804    }
4805
4806    fn log_dense(&mut self, recs: Vec<WalRecord>) -> Result<()> {
4807        let recs = self.rewrite_wal_dense(recs)?;
4808        match recs.len() {
4809            0 => Ok(()),
4810            1 => self.log_then_apply(recs.into_iter().next().unwrap()),
4811            _ => self.log_then_apply(WalRecord::Batch(recs)),
4812        }
4813    }
4814
4815    /// Durable write, then notify the event sink. Replay (`apply` during
4816    /// `open`) never enters this function, so it is the replay-silent seam.
4817    fn log_then_apply(&mut self, rec: WalRecord) -> Result<()> {
4818        self.log_then_apply_with(rec, None, self.fsync)
4819    }
4820
4821    /// Whether this frame must fsync under `policy`.
4822    ///
4823    /// Batched contract: user-visible batches (>1 mutation) fsync; single
4824    /// mutations do not. The dense rewrite wraps a single mutation in a
4825    /// `Batch([Intern.., <one *Id record>])`, so `Intern` records are excluded
4826    /// from the count — removing that filter would make every single-op write
4827    /// fsync under Batched (or, if the threshold were raised instead, skip a
4828    /// needed fsync for real two-op batches).
4829    fn wal_needs_sync(policy: FsyncPolicy, rec: &WalRecord) -> bool {
4830        match policy {
4831            FsyncPolicy::Relaxed => false,
4832            FsyncPolicy::Strict => true,
4833            FsyncPolicy::Batched => match rec {
4834                // Intern + one mutation is the single-op rewrite, not a user batch.
4835                WalRecord::Batch(inner) => {
4836                    inner
4837                        .iter()
4838                        .filter(|r| !matches!(r, WalRecord::Intern { .. }))
4839                        .count()
4840                        > 1
4841                }
4842                _ => false,
4843            },
4844        }
4845    }
4846
4847    /// # Apply-infallibility invariant (load-bearing)
4848    ///
4849    /// The ordering is: WAL append → fsync → apply. If `apply` returned `Err`
4850    /// for a `Batch` frame after a successful WAL write, the WAL would contain
4851    /// the full frame while in-memory state would reflect only the ops before
4852    /// the failure. On reopen, WAL replay would then apply the entire batch —
4853    /// diverging permanently from what the pre-crash process had in memory.
4854    ///
4855    /// For `Batch` frames this situation cannot arise because:
4856    /// - All validation runs via `commit_logged_batch`/`MutPreview` **before**
4857    ///   the WAL write. `MutPreview` uses the same `&mut self` that apply will
4858    ///   use, with no concurrent mutation between validation exit and apply entry.
4859    /// - Every `apply` arm for a validated op is either infallible by construction
4860    ///   (`InsertNode`, `RemoveProp`, `DeleteEdge`, `DeleteNode`), has idempotency
4861    ///   guards that return `Ok(())` (`CreateRule`, `DeleteRule`), or is
4862    ///   guaranteed-present by validation (`InsertEdge`/`SetProp` key lookups).
4863    /// - `on_node_changed` and `on_node_removed` return `()` — never `Err`.
4864    ///
4865    /// A `debug_assert!` below fires in debug builds if `apply` ever returns
4866    /// `Err` for a `Batch` frame, making any future regression immediately visible
4867    /// in tests rather than silently diverging crash-recovery behaviour.
4868    fn log_then_apply_with(
4869        &mut self,
4870        rec: WalRecord,
4871        ingest: Option<(String, usize)>,
4872        policy: FsyncPolicy,
4873    ) -> Result<()> {
4874        // Read-only guard: as-of instances must never write the WAL.
4875        if self.read_only {
4876            return Err(GraphError::ReadOnly);
4877        }
4878        // Degraded guard: fsync failure left WAL truncated, or a refresh failed
4879        // partway; in-memory state is ahead of (or out of step with) the
4880        // on-disk WAL, so further mutations would deepen the divergence.
4881        // Reopen the database to recover.  Checked before the lock guard: this
4882        // is the more serious condition and the more useful error.
4883        if self.degraded {
4884            return Err(GraphError::Io(std::io::Error::other(
4885                "database degraded after group-commit fsync failure; reopen required",
4886            )));
4887        }
4888        // Cross-process guard: this write scope asked for the store's write
4889        // lock and did not get it. Writing anyway would append frames on top of
4890        // a WAL another process is extending, so refuse instead.
4891        if self.lock_denied {
4892            return Err(GraphError::Busy { holder: None });
4893        }
4894        // Ensure retained provenance bytes are decoded into the live mutable
4895        // fields before any mutation touches self.engine.provenance.  This is a
4896        // no-op if provenance was never stored (fresh store) or has already been
4897        // consumed (subsequent mutations).  WAL replay calls apply() directly
4898        // and is covered by consume_retained_state_eager before replay.
4899        self.ensure_v8_base_sections_loaded();
4900        self.engine.ensure_provenance_loaded_mut();
4901        // Invariant (I-1): no stale deltas may enter from a previous apply.
4902        // If any engine method ever accumulates deltas before erroring, they would
4903        // contaminate the *next* commit's event stream. This assert fires in debug
4904        // builds, making any future regression visible at the earliest point.
4905        debug_assert_eq!(
4906            self.engine.pending_delta_count(),
4907            0,
4908            "stale engine deltas at log_then_apply_with entry — \
4909             a previous apply arm may have accumulated deltas before erroring; \
4910             the caller must drain_deltas() on any error path before returning"
4911        );
4912        let frame = encode_record(&rec);
4913        self.fs.append(FileId::Wal, &frame)?;
4914        // The cursor advances by exactly the bytes appended: these frames are
4915        // ours and already applied, so a later refresh must not replay them.
4916        self.wal_consumed += frame.len() as u64;
4917        if Self::wal_needs_sync(policy, &rec) {
4918            self.fs.sync(FileId::Wal)?;
4919        }
4920        // Marker writing always needs the engine deltas, but the engine only
4921        // accumulates them when emit_deltas is true (normally gated on subscribers
4922        // or views being present).  Enable emission for this apply if it is
4923        // currently off, then restore the original state unconditionally via an
4924        // RAII guard — this prevents a panic in apply() from leaking the flag.
4925        // The same guard resets the engine's transient chaining state. A panic
4926        // unwinding out of a rule hook would otherwise leave `chain_depth`
4927        // non-zero, which makes every later `begin_chain` decide chaining is
4928        // already running and silently switch it off for good.
4929        struct RestoreEmitDeltas(*mut RuleEngine, bool);
4930        impl Drop for RestoreEmitDeltas {
4931            fn drop(&mut self) {
4932                // SAFETY: pointer into self (GraphDb); guard is dropped within
4933                // this frame before log_then_apply_with returns.
4934                unsafe {
4935                    (*self.0).set_emit_deltas(self.1);
4936                    (*self.0).reset_chain_state();
4937                }
4938            }
4939        }
4940        let original_emit = self.engine.emit_deltas();
4941        if !original_emit {
4942            self.engine.set_emit_deltas(true);
4943        }
4944        // SAFETY: raw pointer into self; guard dropped within this frame.
4945        let _emit_guard = RestoreEmitDeltas(&mut self.engine as *mut _, original_emit);
4946
4947        let apply_result = self.apply(&rec);
4948        // For Batch frames, post-validation apply must be infallible (see above).
4949        // A debug_assert here catches any future change that makes apply fallible
4950        // before the caller notices via silent WAL/memory divergence.
4951        if matches!(&rec, WalRecord::Batch(_)) {
4952            debug_assert!(
4953                apply_result.is_ok(),
4954                "Batch apply returned Err after successful WAL write — \
4955                 the validate-then-apply invariant has been violated; \
4956                 see log_then_apply_with invariant doc"
4957            );
4958        }
4959        if apply_result.is_err() {
4960            // Discard any partial deltas accumulated by the failed apply.
4961            // They must not ride the next commit's event stream (I-1).
4962            // _emit_guard restores emit_deltas on drop automatically.
4963            let _ = self.engine.drain_deltas();
4964            let _ = self.engine.take_rebuild_needed();
4965            apply_result?;
4966        }
4967        self.commit_seq += 1;
4968        let seq = self.commit_seq;
4969        // Update per-node last-change map for the committed record.
4970        // Must happen after commit_seq is incremented so the seq is correct.
4971        self.update_last_change_from_rec(&rec, seq);
4972        // Drain engine deltas and distribute to subscribers before the existing
4973        // MutationEvent sink fires — both happen post-fsync, post-apply.
4974        // _emit_guard restores emit_deltas after this line when it drops.
4975        let engine_deltas = self.engine.drain_deltas();
4976
4977        // Append history-marker WAL records for any derived-edge changes so
4978        // that `edge_history` and `was_linked` can surface rule-attributed
4979        // events. Markers are STATE NO-OPS during replay; they are written
4980        // without an additional fsync (the triggering commit's sync already
4981        // happened; the next commit's sync covers these lazily).
4982        if !engine_deltas.is_empty() {
4983            let markers: Vec<WalRecord> = engine_deltas
4984                .iter()
4985                .map(|d| {
4986                    if d.fired {
4987                        WalRecord::DerivedEdgeAdded {
4988                            rule: d.rule.clone(),
4989                            edge_type: d.edge_type.clone(),
4990                            src_key: d.src_key.clone(),
4991                            dst_key: d.dst_key.clone(),
4992                        }
4993                    } else {
4994                        WalRecord::DerivedEdgeRetracted {
4995                            rule: d.rule.clone(),
4996                            edge_type: d.edge_type.clone(),
4997                            src_key: d.src_key.clone(),
4998                            dst_key: d.dst_key.clone(),
4999                        }
5000                    }
5001                })
5002                .collect();
5003            let marker_frame = if markers.len() == 1 {
5004                markers.into_iter().next().unwrap()
5005            } else {
5006                WalRecord::Batch(markers)
5007            };
5008            // Ignore append errors: markers are best-effort history
5009            // annotations. Losing them does not affect state correctness.
5010            // The cursor only advances when the bytes actually landed.
5011            let marker_bytes = encode_record(&marker_frame);
5012            if self.fs.append(FileId::Wal, &marker_bytes).is_ok() {
5013                self.wal_consumed += marker_bytes.len() as u64;
5014            }
5015        }
5016
5017        // Record MVCC CommitDelta for the epoch reader.  The WAL record is
5018        // stored as-is (including any nested Batch / Intern records); the
5019        // ReaderSnapshot's apply_one function handles all variants.
5020        {
5021            let derived_inserts = engine_deltas
5022                .iter()
5023                .filter(|d| d.fired)
5024                .map(|d| (d.etype_sym, d.src_id, d.dst_id))
5025                .collect();
5026            let derived_deletes = engine_deltas
5027                .iter()
5028                .filter(|d| !d.fired)
5029                .map(|d| (d.etype_sym, d.src_id, d.dst_id))
5030                .collect();
5031            let delta = Arc::new(crate::reader::CommitDelta {
5032                records: vec![rec.clone()],
5033                derived_inserts,
5034                derived_deletes,
5035            });
5036            self.delta_tail.push(delta);
5037            self.commits_since_fold += 1;
5038            if self.commits_since_fold >= crate::reader::FOLD_EVERY_K {
5039                self.fold_now();
5040            }
5041        }
5042
5043        if self.defer_events {
5044            // Group-commit drain thread: hold events until after the group
5045            // fsync so subscribers only observe durable data (R2).
5046            self.deferred_events.push(DeferredEvent {
5047                rec: rec.clone(),
5048                engine_deltas,
5049                seq,
5050                ingest,
5051            });
5052        } else {
5053            self.distribute_events(&rec, &engine_deltas, seq);
5054            self.emit_committed(&rec, ingest);
5055        }
5056        // Drift is only known after apply, so auto-rebuild cannot join the
5057        // triggering op's WAL frame. Issue RebuildRule as a second commit.
5058        // Skip when `rec` is itself RebuildRule: rebuild resets drift, so a
5059        // retrigger loop is impossible if the fit succeeded, but we still
5060        // drain the flag so a leftover cannot re-enter.
5061        // One slice of any outstanding vector-index build rides here too, so a
5062        // store that is being written to finishes its build without anyone
5063        // calling `pump_index_build`. A rule that becomes whole joins the same
5064        // RebuildRule loop below.
5065        let mut rebuilds = self.engine.take_rebuild_needed();
5066        if !matches!(&rec, WalRecord::RebuildRule { .. }) {
5067            // Not after `CreateRule`: that record's own apply already did the
5068            // rule's first slice, and pumping again here would make one
5069            // `create_rule` call do two slices' work under one lock.
5070            // Nothing pending is the overwhelmingly common case and must cost
5071            // a map lookup, not an engine swap: a store being written to has
5072            // long since populated its indexes, so the `pump_index_build`
5073            // entry point owns the not-yet-populated case on its own.
5074            if !matches!(&rec, WalRecord::CreateRule { .. })
5075                && !self.engine.builds_in_progress().is_empty()
5076            {
5077                rebuilds.extend(self.pump_one_slice().into_iter().map(|b| b.rule));
5078            }
5079            let mut failed = Vec::new();
5080            for name in rebuilds {
5081                if self.engine.rules().any(|r| r.name == name) {
5082                    // User op is already durable. A failed second commit must
5083                    // not surface as the caller's error.
5084                    if let Err(e) =
5085                        self.log_then_apply(WalRecord::RebuildRule { name: name.clone() })
5086                    {
5087                        eprintln!(
5088                            "auto-rebuild of rule {name:?} failed after durable user commit: {e}"
5089                        );
5090                        failed.push(name);
5091                    }
5092                }
5093            }
5094            for name in failed {
5095                self.engine.queue_rebuild_needed(name);
5096            }
5097        }
5098        Ok(())
5099    }
5100
5101    /// Install a post-commit hook. Replaces any previous sink.
5102    ///
5103    /// The sink runs inside `log_then_apply` after a successful
5104    /// durable commit, while the caller still holds `&mut self`. When this
5105    /// database is behind a [`crate::SharedDb`], that means the **write
5106    /// guard is held**. The sink must never call `read` / `write` (or any
5107    /// other method) on the same `SharedDb` — the `RwLock` is not
5108    /// re-entrant and doing so deadlocks. The sink is `Send + Sync`;
5109    /// `std::sync::mpsc::Sender` is not `Sync` and will not type-check.
5110    /// Intended examples: `std::sync::mpsc::SyncSender`,
5111    /// `tokio::sync::mpsc::Sender`, `tokio::sync::broadcast::Sender`
5112    /// (non-blocking `send`), or `Arc<Mutex<Vec<MutationEvent>>>`.
5113    pub fn set_event_sink(&mut self, sink: Box<dyn Fn(MutationEvent) + Send + Sync>) {
5114        self.event_sink = Some(sink);
5115    }
5116
5117    /// Whether a post-commit event sink is currently installed.
5118    pub fn has_event_sink(&self) -> bool {
5119        self.event_sink.is_some()
5120    }
5121
5122    /// Set WAL fsync cadence. Default [`FsyncPolicy::Strict`].
5123    pub fn set_fsync_policy(&mut self, p: FsyncPolicy) {
5124        self.fsync = p;
5125    }
5126
5127    /// Return the current WAL fsync cadence.
5128    pub fn fsync_policy(&self) -> FsyncPolicy {
5129        self.fsync
5130    }
5131
5132    // ── Group-commit event deferral ───────────────────────────────────────────
5133
5134    /// Enable or disable deferred event mode.
5135    ///
5136    /// When `true`, event notifications (subscription `DbEvent`s and legacy
5137    /// `MutationEvent` sink calls) are buffered rather than fired immediately.
5138    /// Call [`flush_deferred_events`] after the group fsync to deliver them,
5139    /// or [`discard_deferred_events`] if the fsync failed and the group must
5140    /// be treated as lost.
5141    pub fn set_deferred_events_mode(&mut self, defer: bool) {
5142        self.defer_events = defer;
5143    }
5144
5145    /// Fire all buffered events accumulated since [`set_deferred_events_mode`]
5146    /// was set to true.  Clears the buffer.
5147    ///
5148    /// Called by the drain thread AFTER a successful group fsync, so
5149    /// subscribers observe only data that is durably on disk.
5150    pub fn flush_deferred_events(&mut self) {
5151        let events = std::mem::take(&mut self.deferred_events);
5152        for de in events {
5153            self.distribute_events(&de.rec, &de.engine_deltas, de.seq);
5154            self.emit_committed(&de.rec, de.ingest);
5155        }
5156    }
5157
5158    /// Discard all buffered events without firing them.
5159    ///
5160    /// Called by the drain thread when a group fsync fails: the WAL has been
5161    /// truncated back to the pre-group offset, so the committed-but-unsynced
5162    /// ops must not be observable to subscribers.
5163    pub fn discard_deferred_events(&mut self) {
5164        self.deferred_events.clear();
5165    }
5166
5167    // ── Degraded state ────────────────────────────────────────────────────────
5168
5169    /// Mark this database as degraded.
5170    ///
5171    /// Called by the group-commit drain thread after a group fsync failure and
5172    /// WAL truncation: the in-memory state is now ahead of the on-disk WAL, so
5173    /// further mutations would deepen the divergence.  All subsequent calls to
5174    /// [`log_then_apply_with`] return `Err` until the database is reopened.
5175    pub fn set_degraded(&mut self) {
5176        self.degraded = true;
5177    }
5178
5179    fn emit(&self, ev: MutationEvent) {
5180        if let Some(sink) = &self.event_sink {
5181            sink(ev);
5182        }
5183    }
5184
5185    fn emit_committed(&self, rec: &WalRecord, ingest: Option<(String, usize)>) {
5186        match rec {
5187            WalRecord::Batch(inner) => {
5188                for r in inner {
5189                    if let Some(ev) = event_from_record(r, &self.syms, &self.ids) {
5190                        self.emit(ev);
5191                    }
5192                }
5193                match ingest {
5194                    Some((label, inserted)) => {
5195                        self.emit(MutationEvent::Ingested { label, inserted })
5196                    }
5197                    None => {
5198                        let ops = inner
5199                            .iter()
5200                            .filter(|r| !matches!(r, WalRecord::Intern { .. }))
5201                            .count();
5202                        if ops > 1 {
5203                            self.emit(MutationEvent::BatchApplied { ops });
5204                        }
5205                    }
5206                }
5207            }
5208            other => {
5209                if let Some(ev) = event_from_record(other, &self.syms, &self.ids) {
5210                    self.emit(ev);
5211                }
5212            }
5213        }
5214    }
5215
5216    // -----------------------------------------------------------------------
5217    // Subscription API
5218    // -----------------------------------------------------------------------
5219
5220    /// Distribute post-commit events to all live subscribers.
5221    ///
5222    /// Build a row-key → row-data map from a [`ResultSet`].
5223    ///
5224    /// Each row is serialized to JSON to form its key; a debug fallback is used
5225    /// if serialization fails. Used by both the initial-seed path in
5226    /// [`Self::subscribe_query`] and the per-commit diff path in
5227    /// [`Self::distribute_events`] to keep the two in sync.
5228    fn result_to_row_map(
5229        result: &core_query::ResultSet,
5230    ) -> std::collections::HashMap<String, Vec<Option<Value>>> {
5231        (0..result.len())
5232            .map(|i| {
5233                let row = result.row(i).to_vec();
5234                let key = serde_json::to_string(&row).unwrap_or_else(|_| format!("{row:?}"));
5235                (key, row)
5236            })
5237            .collect()
5238    }
5239
5240    /// Collect the set of label syms touched by a WAL record.
5241    ///
5242    /// Returns `Some(set)` when every record in this commit can be attributed to
5243    /// a known label sym. Returns `None` when the commit must not be skipped:
5244    /// edge records, unresolvable key→label lookups, or any record type not in
5245    /// the explicit handled set.
5246    ///
5247    /// Handled record types and their actions:
5248    /// - `InsertNode`   → look up label in interner (fails → None)
5249    /// - `InsertNodeId` → label sym is carried directly
5250    /// - `SetProp`      → resolve key→id→label (fails → None)
5251    /// - `DeleteNode`   → resolve key→id→label (fails → None)
5252    /// - `Batch`        → recurse into every inner record
5253    /// - `InsertEdge`, `DeleteEdge`, `InsertEdgeId` → always None (edge records)
5254    /// - everything else → None (conservative)
5255    fn commit_touched_labels(
5256        rec: &WalRecord,
5257        syms: &Interner,
5258        ids: &IdMap,
5259        labels: &[u32],
5260    ) -> Option<BTreeSet<u32>> {
5261        let mut out = BTreeSet::new();
5262        if Self::collect_touched_labels(rec, syms, ids, labels, &mut out) {
5263            Some(out)
5264        } else {
5265            None
5266        }
5267    }
5268
5269    fn collect_touched_labels(
5270        rec: &WalRecord,
5271        syms: &Interner,
5272        ids: &IdMap,
5273        labels: &[u32],
5274        out: &mut BTreeSet<u32>,
5275    ) -> bool {
5276        match rec {
5277            // String-key insert: the dense rewrite converts this to
5278            // [Intern, InsertNodeId], so this arm fires only for legacy WAL
5279            // records written before the dense path was added.
5280            WalRecord::InsertNode { label, .. } => {
5281                if let Some(sym) = syms.get(label) {
5282                    out.insert(sym);
5283                    true
5284                } else {
5285                    false
5286                }
5287            }
5288            // Dense-id insert (produced by rewrite_wal_dense for every
5289            // insert_node call in the current codebase).
5290            WalRecord::InsertNodeId { label, .. } => {
5291                out.insert(*label);
5292                true
5293            }
5294            // String-key prop set: dense path converts to [Intern, SetPropId].
5295            WalRecord::SetProp { key, .. } => {
5296                if let Some(sym) = Self::resolve_key_label_sym(key, ids, labels) {
5297                    out.insert(sym);
5298                    true
5299                } else {
5300                    false
5301                }
5302            }
5303            // Dense-id prop set (produced by rewrite_wal_dense for set_prop).
5304            WalRecord::SetPropId { id, .. } => {
5305                if let Some(sym) = labels.get(*id as usize).copied().filter(|&s| s != u32::MAX) {
5306                    out.insert(sym);
5307                    true
5308                } else {
5309                    false
5310                }
5311            }
5312            WalRecord::DeleteNode { key } => {
5313                if let Some(sym) = Self::resolve_key_label_sym(key, ids, labels) {
5314                    out.insert(sym);
5315                    true
5316                } else {
5317                    false
5318                }
5319            }
5320            WalRecord::Batch(inner) => inner
5321                .iter()
5322                .all(|r| Self::collect_touched_labels(r, syms, ids, labels, out)),
5323            // Intern is a pure metadata record — it does not touch any node's
5324            // label and is safe to skip for the label-skip predicate.
5325            WalRecord::Intern { .. } => true,
5326            // Edge records: always re-execute (edges can change join results).
5327            WalRecord::InsertEdge { .. }
5328            | WalRecord::DeleteEdge { .. }
5329            | WalRecord::InsertEdgeId { .. } => false,
5330            _ => false,
5331        }
5332    }
5333
5334    /// Resolve a node key to its label sym via the dense id table.
5335    /// Returns `None` if the key is unknown or the label is a tombstone sentinel.
5336    fn resolve_key_label_sym(key: &str, ids: &IdMap, labels: &[u32]) -> Option<u32> {
5337        let id = ids.get(key)?;
5338        let sym = labels.get(id as usize).copied()?;
5339        (sym != u32::MAX).then_some(sym)
5340    }
5341
5342    /// Distribute post-commit events to all live subscribers.
5343    ///
5344    /// Called from `log_then_apply_with` after apply + fsync, before the
5345    /// legacy MutationEvent sink. Prunes dead `Weak` entries in-place.
5346    ///
5347    /// Query subscriptions (subscribe_query) re-execute their plan on every
5348    /// call and diff the result against the previous run. Zero overhead when
5349    /// no query subscriptions are active.
5350    fn distribute_events(&mut self, rec: &WalRecord, engine_deltas: &[EngineEdgeDelta], seq: u64) {
5351        if self.subscriptions.is_empty() && self.query_subscriptions.is_empty() {
5352            return;
5353        }
5354
5355        if !self.subscriptions.is_empty() {
5356            // Build write events from the WAL record.
5357            let write_events: Vec<DbEvent> =
5358                Self::write_events_from_record(rec, seq, &self.syms, &self.ids);
5359
5360            // Build edge events from engine deltas.  Weight is looked up from
5361            // edge_props at distribution time (after apply), so it's always fresh.
5362            let edge_events: Vec<DbEvent> = engine_deltas
5363                .iter()
5364                .map(|d| {
5365                    if d.fired {
5366                        // The score lives under the rule's declared weight_prop,
5367                        // which is not always the literal "weight".
5368                        let prop = self
5369                            .engine
5370                            .rules()
5371                            .find(|r| r.name == d.rule)
5372                            .and_then(|r| r.weight_prop.as_deref());
5373                        let weight = prop.and_then(|p| {
5374                            self.edge_props
5375                                .get(d.etype_sym, d.src_id, d.dst_id, p)
5376                                .and_then(|v| {
5377                                    if let core_storage::Value::Float(f) = v {
5378                                        Some(*f)
5379                                    } else {
5380                                        None
5381                                    }
5382                                })
5383                        });
5384                        DbEvent::EdgeFired {
5385                            rule: d.rule.clone(),
5386                            src_key: d.src_key.clone(),
5387                            dst_key: d.dst_key.clone(),
5388                            edge_type: d.edge_type.clone(),
5389                            weight,
5390                            commit_seq: seq,
5391                        }
5392                    } else {
5393                        DbEvent::EdgeRetracted {
5394                            rule: d.rule.clone(),
5395                            src_key: d.src_key.clone(),
5396                            dst_key: d.dst_key.clone(),
5397                            edge_type: d.edge_type.clone(),
5398                            commit_seq: seq,
5399                        }
5400                    }
5401                })
5402                .collect();
5403
5404            // Prune dead entries; push matching events to live ones.
5405            self.subscriptions.retain(|entry| {
5406                let Some(inner) = entry.inner.upgrade() else {
5407                    return false;
5408                };
5409                for ev in &write_events {
5410                    if event_matches(ev, &entry.filter) {
5411                        inner.push(ev.clone());
5412                    }
5413                }
5414                for ev in &edge_events {
5415                    if event_matches(ev, &entry.filter) {
5416                        inner.push(ev.clone());
5417                    }
5418                }
5419                true
5420            });
5421
5422            // Turn off delta accumulation if all subscribers dropped and no views remain.
5423            if self.subscriptions.is_empty() && self.view_store.is_empty() {
5424                self.engine.set_emit_deltas(false);
5425            }
5426        }
5427
5428        // Query subscriptions: full re-run per commit, then diff rows.
5429        // IMPORTANT: full re-execution on every commit — use LIMIT to bound cost.
5430        // Differential evaluation is roadmap / Phase 5.
5431        if !self.query_subscriptions.is_empty() {
5432            // Take the list out so we can call self.view() without borrow conflict.
5433            let mut query_subs = std::mem::take(&mut self.query_subscriptions);
5434            let empty_params = BTreeMap::new();
5435            query_subs.retain_mut(|entry| {
5436                let Some(inner) = entry.inner.upgrade() else {
5437                    return false; // subscriber dropped — prune
5438                };
5439                // Label-skip: if the plan has a known scan label and this commit
5440                // can be proven to touch only different labels (and no rule-derived
5441                // edge deltas fired), the result set cannot have changed — skip.
5442                if let Some(scan_sym) = entry.scan_label {
5443                    if engine_deltas.is_empty() {
5444                        let touched =
5445                            Self::commit_touched_labels(rec, &self.syms, &self.ids, &self.labels);
5446                        if touched.map(|t| !t.contains(&scan_sym)).unwrap_or(false) {
5447                            return true; // safe to skip — result set unchanged
5448                        }
5449                    }
5450                }
5451                QUERY_SUB_EXECS_TL.with(|c| c.set(c.get() + 1));
5452                let result = match execute(&self.view(), &entry.ops, &Params(&empty_params)) {
5453                    Ok(r) => r,
5454                    Err(e) => {
5455                        // Keep the subscription alive; skip the diff for this commit.
5456                        // Re-run errors are transient (e.g., planner change) and
5457                        // self-heal when the next commit succeeds.
5458                        eprintln!("[mushroomdb] subscribe_query re-run failed: {e}");
5459                        return true;
5460                    }
5461                };
5462                // Build new row map: serialized-key → row data.
5463                let new_row_map = Self::result_to_row_map(&result);
5464                // Removed rows: in prev but not in new.
5465                for (key, row) in &entry.prev_row_map {
5466                    if !new_row_map.contains_key(key) {
5467                        inner.push(DbEvent::QueryRowRemoved {
5468                            columns: entry.columns.clone(),
5469                            row: row.clone(),
5470                        });
5471                    }
5472                }
5473                // Added rows: in new but not in prev.
5474                for (key, row) in &new_row_map {
5475                    if !entry.prev_row_map.contains_key(key) {
5476                        inner.push(DbEvent::QueryRowAdded {
5477                            columns: entry.columns.clone(),
5478                            row: row.clone(),
5479                        });
5480                    }
5481                }
5482                entry.prev_row_map = new_row_map;
5483                true
5484            });
5485            self.query_subscriptions = query_subs;
5486        }
5487    }
5488
5489    /// Returns `true` if any live subscriber or view definition requires delta
5490    /// accumulation. Used to set `engine.emit_deltas` on subscribe/view DDL.
5491    fn needs_emit_deltas(&self) -> bool {
5492        !self.view_store.is_empty()
5493            || self
5494                .subscriptions
5495                .iter()
5496                .any(|e| e.inner.upgrade().is_some())
5497    }
5498
5499    /// Convert a WAL record into `DbEvent` write events with the given seq.
5500    fn write_events_from_record(
5501        rec: &WalRecord,
5502        seq: u64,
5503        intern: &Interner,
5504        ids: &IdMap,
5505    ) -> Vec<DbEvent> {
5506        match rec {
5507            WalRecord::InsertNode { label, key, .. } => vec![DbEvent::NodeInserted {
5508                label: label.clone(),
5509                key: key.clone(),
5510                commit_seq: seq,
5511            }],
5512            // *Id arms run after a successful apply, so resolution can only
5513            // fail on a programming error. Skip the event rather than emit a
5514            // fabricated "" that clients can't tell from a real empty value
5515            // (mirrors event_from_record returning None).
5516            WalRecord::InsertNodeId { label, key, .. } => intern
5517                .resolve(*label)
5518                .map(|label| DbEvent::NodeInserted {
5519                    label: label.to_string(),
5520                    key: key.clone(),
5521                    commit_seq: seq,
5522                })
5523                .into_iter()
5524                .collect(),
5525            WalRecord::SetProp { key, field, .. } => vec![DbEvent::PropSet {
5526                key: key.clone(),
5527                field: field.clone(),
5528                commit_seq: seq,
5529            }],
5530            WalRecord::SetPropId { id, field, .. } => ids
5531                .key_of(*id)
5532                .zip(intern.resolve(*field))
5533                .map(|(key, field)| DbEvent::PropSet {
5534                    key: key.to_string(),
5535                    field: field.to_string(),
5536                    commit_seq: seq,
5537                })
5538                .into_iter()
5539                .collect(),
5540            WalRecord::RemoveProp { key, field } => vec![DbEvent::PropRemoved {
5541                key: key.clone(),
5542                field: field.clone(),
5543                commit_seq: seq,
5544            }],
5545            WalRecord::InsertEdge {
5546                edge_type,
5547                src_key,
5548                dst_key,
5549            } => vec![DbEvent::EdgeInserted {
5550                edge_type: edge_type.clone(),
5551                src: src_key.clone(),
5552                dst: dst_key.clone(),
5553                commit_seq: seq,
5554            }],
5555            WalRecord::InsertEdgeId { etype, src, dst } => (|| {
5556                Some(DbEvent::EdgeInserted {
5557                    edge_type: intern.resolve(*etype)?.to_string(),
5558                    src: ids.key_of(*src)?.to_string(),
5559                    dst: ids.key_of(*dst)?.to_string(),
5560                    commit_seq: seq,
5561                })
5562            })()
5563            .into_iter()
5564            .collect(),
5565            WalRecord::DeleteEdge {
5566                edge_type,
5567                src_key,
5568                dst_key,
5569            } => vec![DbEvent::EdgeDeleted {
5570                edge_type: edge_type.clone(),
5571                src: src_key.clone(),
5572                dst: dst_key.clone(),
5573                commit_seq: seq,
5574            }],
5575            WalRecord::DeleteNode { key } => vec![DbEvent::NodeDeleted {
5576                key: key.clone(),
5577                commit_seq: seq,
5578            }],
5579            WalRecord::Batch(inner) => inner
5580                .iter()
5581                .flat_map(|r| Self::write_events_from_record(r, seq, intern, ids))
5582                .collect(),
5583            WalRecord::CreateRule { .. }
5584            | WalRecord::DeleteRule { .. }
5585            | WalRecord::RebuildRule { .. }
5586            | WalRecord::CreateView { .. }
5587            | WalRecord::DeleteView { .. }
5588            | WalRecord::EnableFulltext { .. }
5589            | WalRecord::DisableFulltext { .. }
5590            | WalRecord::EnableIndex { .. }
5591            | WalRecord::DisableIndex { .. }
5592            | WalRecord::Intern { .. }
5593            // History markers produce no DbEvent — the engine delta already
5594            // fired the EdgeFired/EdgeRetracted subscription events.
5595            | WalRecord::DerivedEdgeAdded { .. }
5596            | WalRecord::DerivedEdgeRetracted { .. }
5597            // A count is not an edge event: the pair it counts already fired one
5598            // when it was first inserted.
5599            | WalRecord::SetEdgeCount { .. }
5600            | WalRecord::RenameNode { .. } => vec![],
5601        }
5602    }
5603
5604    /// Subscribe to edge-fire and edge-retract events for one named rule.
5605    ///
5606    /// Returns `Err(GraphError::RuleNotFound)` if `rule_name` is not
5607    /// currently registered. Dropping the returned [`Subscription`] handle
5608    /// unregisters the subscriber — no further events are queued, no
5609    /// resources leak.
5610    pub fn subscribe_rule(&mut self, rule_name: &str) -> core_storage::Result<Subscription> {
5611        if self.read_only {
5612            return Err(core_storage::GraphError::ReadOnly);
5613        }
5614        if !self.engine.rules().any(|r| r.name == rule_name) {
5615            return Err(core_storage::GraphError::RuleNotFound {
5616                name: rule_name.to_string(),
5617            });
5618        }
5619        let inner = SubInner::new(self.sub_capacity());
5620        self.subscriptions.push(SubEntry {
5621            filter: SubFilter::Rule(rule_name.to_string()),
5622            inner: std::sync::Arc::downgrade(&inner),
5623        });
5624        self.engine.set_emit_deltas(true);
5625        Ok(Subscription(inner))
5626    }
5627
5628    /// Subscribe to edge-fire and edge-retract events for **all** rules.
5629    ///
5630    /// Returns `Err(GraphError::ReadOnly)` if called on an as-of instance —
5631    /// as-of instances never commit, so `distribute_events` never runs and the
5632    /// subscription would never deliver events.
5633    pub fn subscribe_all_rules(&mut self) -> core_storage::Result<Subscription> {
5634        if self.read_only {
5635            return Err(core_storage::GraphError::ReadOnly);
5636        }
5637        let inner = SubInner::new(self.sub_capacity());
5638        self.subscriptions.push(SubEntry {
5639            filter: SubFilter::AllRules,
5640            inner: std::sync::Arc::downgrade(&inner),
5641        });
5642        self.engine.set_emit_deltas(true);
5643        Ok(Subscription(inner))
5644    }
5645
5646    /// Subscribe to write events: node insert/delete, prop set/remove.
5647    ///
5648    /// Does not include edge-fire / edge-retract (rule-derived edge events).
5649    ///
5650    /// Returns `Err(GraphError::ReadOnly)` if called on an as-of instance —
5651    /// as-of instances never commit, so `distribute_events` never runs and the
5652    /// subscription would never deliver events.
5653    pub fn subscribe_writes(&mut self) -> core_storage::Result<Subscription> {
5654        if self.read_only {
5655            return Err(core_storage::GraphError::ReadOnly);
5656        }
5657        let inner = SubInner::new(self.sub_capacity());
5658        self.subscriptions.push(SubEntry {
5659            filter: SubFilter::Writes,
5660            inner: std::sync::Arc::downgrade(&inner),
5661        });
5662        self.engine.set_emit_deltas(true);
5663        Ok(Subscription(inner))
5664    }
5665
5666    /// Subscribe to incremental Cypher query results.
5667    ///
5668    /// Parses and plans `cypher`; rejects the query if the plan is not in the
5669    /// allowlisted subset (see [`core_query::cypher::is_subscribable`]):
5670    ///   - `MATCH (n:Label) WHERE … RETURN … [LIMIT n]`
5671    ///   - `MATCH (a)-[r:TYPE]->(b) RETURN … [LIMIT n]`  (exactly one hop)
5672    ///
5673    /// SKIP is not supported — it shifts the result window on every commit,
5674    /// causing spurious Added/Removed churn for rows whose data never changed.
5675    /// Multi-hop Expand chains are not supported; each additional MATCH clause
5676    /// widens scope beyond the documented single-scan / single-hop subset.
5677    ///
5678    /// After each successful commit, the plan is **fully re-executed** and the
5679    /// result is diffed against the previous run. Added rows produce
5680    /// [`DbEvent::QueryRowAdded`]; removed rows produce
5681    /// [`DbEvent::QueryRowRemoved`].
5682    ///
5683    /// **Full re-run per commit; use LIMIT to bound execution cost.**
5684    /// The existing 1 M intermediate-row cap applies. Differential evaluation
5685    /// is roadmap / Phase 5.
5686    ///
5687    /// Returns `Err(GraphError::ReadOnly)` if called on an as-of instance —
5688    /// as-of instances never commit, so `distribute_events` never runs and the
5689    /// subscription would never deliver events.
5690    ///
5691    /// Returns `Err(GraphError::QueryError)` if the query fails to parse, plan,
5692    /// or if the plan shape is not in the allowlist.
5693    pub fn subscribe_query(&mut self, cypher: &str) -> Result<Subscription> {
5694        if self.read_only {
5695            return Err(GraphError::ReadOnly);
5696        }
5697        let tokens = lex(cypher).map_err(|e| GraphError::QueryError {
5698            detail: format!("lex: {e}"),
5699        })?;
5700        let ast = parse(&tokens).map_err(|e| GraphError::QueryError {
5701            detail: format!("parse: {e}"),
5702        })?;
5703        let ops = plan(&ast).map_err(|e| GraphError::QueryError {
5704            detail: format!("plan: {e}"),
5705        })?;
5706        if !is_subscribable(&ops) {
5707            return Err(GraphError::QueryError {
5708                detail: "subscribe_query only supports allowlisted plan shapes: \
5709                         MATCH (n:Label) WHERE … RETURN … [LIMIT n] or \
5710                         MATCH (a)-[r:TYPE]->(b) RETURN … [LIMIT n] (exactly one hop). \
5711                         Not supported: multi-hop Expand chains, SKIP (creates \
5712                         unstable offset windows), ORDER BY, DISTINCT, aggregates, \
5713                         variable-length paths, OPTIONAL MATCH, WITH, UNWIND. \
5714                         Use LIMIT to bound re-execution cost."
5715                    .to_string(),
5716            });
5717        }
5718        // Execute once to capture initial state (initial rows are not emitted as
5719        // events — the subscriber learns the baseline via the first query call).
5720        let empty_params = BTreeMap::new();
5721        let initial = execute(&self.view(), &ops, &Params(&empty_params)).map_err(|e| {
5722            GraphError::QueryError {
5723                detail: format!("execute: {e}"),
5724            }
5725        })?;
5726        let columns = initial.columns().to_vec();
5727        let prev_row_map = Self::result_to_row_map(&initial);
5728        let inner = SubInner::new(self.sub_capacity());
5729        // Derive the scan-label sym for the commit-skip fast-path.  Any Expand op
5730        // or unrecognized leading scan → None (always re-execute).
5731        let scan_label = extract_scan_label(&ops, &mut self.syms);
5732        self.query_subscriptions.push(QuerySubEntry {
5733            ops,
5734            columns,
5735            prev_row_map,
5736            inner: std::sync::Arc::downgrade(&inner),
5737            scan_label,
5738        });
5739        Ok(Subscription(inner))
5740    }
5741
5742    /// Queue capacity used for new subscriptions.
5743    fn sub_capacity(&self) -> usize {
5744        self.sub_capacity
5745    }
5746
5747    /// Override per-subscriber queue capacity for subsequently created
5748    /// subscriptions on this db instance.
5749    ///
5750    /// Default is [`DEFAULT_SUB_CAPACITY`] (65,536 events). Use a smaller
5751    /// value in tests to exercise the [`DbEvent::Lagged`] path without
5752    /// generating tens of thousands of events.
5753    ///
5754    /// This is a test-support escape hatch. Calling it in production reduces
5755    /// subscriber reliability (more Lagged events). It is hidden from rustdoc
5756    /// to discourage accidental production use.
5757    #[doc(hidden)]
5758    pub fn set_sub_capacity(&mut self, capacity: usize) {
5759        self.sub_capacity = capacity;
5760    }
5761
5762    // -----------------------------------------------------------------------
5763
5764    /// Start an atomic batch.
5765    ///
5766    /// The returned [`BatchBuilder`] borrows `self` mutably until
5767    /// [`BatchBuilder::commit`]. Builder methods queue ops only — no
5768    /// validation, no WAL I/O. `commit` validates every queued op against
5769    /// live state plus preceding ops in this batch (duplicate key inside
5770    /// the batch is `Err`; an edge between two nodes created earlier in
5771    /// the batch is valid; `delete_node` then insert of the same key is a
5772    /// fresh identity). Validation never mutates the database. Any failure
5773    /// leaves WAL bytes and in-memory state identical to before `commit`.
5774    /// On success, one `WalRecord::Batch` frame is appended (one fsync)
5775    /// and each inner record is applied in order so rules fire per record.
5776    /// An empty batch, or a batch of only no-ops, writes zero WAL bytes.
5777    ///
5778    /// **Rule-window limitation:** batch validation cannot see edges that a
5779    /// rule created earlier in the *same* batch will derive at apply time, so
5780    /// a `delete_edge` / `insert_edge` in that window is silently no-oped
5781    /// where sequential calls would return `Err(RuleOwned)`. State integrity
5782    /// is unaffected (idempotent apply, provenance intact). Create rules in
5783    /// their own batch, or sequentially, when later ops may touch derived
5784    /// edges.
5785    pub fn batch(&mut self) -> BatchBuilder<'_, F> {
5786        BatchBuilder {
5787            db: self,
5788            ops: Vec::new(),
5789        }
5790    }
5791
5792    /// Closure-style atomic write batch.
5793    ///
5794    /// Equivalent to calling [`GraphDb::batch`], invoking `build` to queue ops,
5795    /// then committing. All ops queued inside `build` are validated in order and
5796    /// committed as a single `WalRecord::Batch` frame (one fsync). Rules fire
5797    /// once per inner record, in order, after commit — semantically identical to
5798    /// sequential single-op writes.
5799    ///
5800    /// **Error semantics — validate-then-apply.** `build` queues ops without
5801    /// touching the database. [`BatchBuilder::commit`] validates every op against
5802    /// live state plus earlier ops in this batch before writing anything. If op N
5803    /// fails validation (duplicate key, unknown key, rule-owned edge, …) the
5804    /// entire batch is rejected: no WAL bytes are written and no in-memory state
5805    /// changes. The database is identical to its state before `write_batch` was
5806    /// called.
5807    ///
5808    /// **Atomicity is crash-level, NOT isolation-level.** On replay after a crash,
5809    /// a partial (torn) `Batch` frame applies NONE of its ops — the frame is
5810    /// either fully applied or not at all. However, while applying a committed
5811    /// batch, concurrent readers may observe intermediate states as ops are applied
5812    /// sequentially in memory. There is no interactive transaction isolation in v1.
5813    /// This is documented as "crash-atomic write batches; no interactive
5814    /// transactions or read isolation."
5815    ///
5816    /// **Returns** `(nodes_inserted, edges_inserted)`. An empty or all-noop batch
5817    /// writes zero WAL bytes and returns `(0, 0)`.
5818    ///
5819    /// # Example
5820    ///
5821    /// ```rust,ignore
5822    /// let (nodes, edges) = db.write_batch(|b| {
5823    ///     b.insert_node("Person", "alice", vec![("age".into(), Value::Int(30))]);
5824    ///     b.insert_node("Person", "bob", vec![]);
5825    ///     b.insert_edge("KNOWS", "alice", "bob");
5826    ///     b.set_prop("alice", "role", Value::Str("admin".into()));
5827    ///     b.delete_node("old_key");
5828    /// })?;
5829    /// // One fsync; on crash replay: all five ops land or none do.
5830    /// ```
5831    pub fn write_batch<C>(&mut self, build: C) -> Result<(usize, usize)>
5832    where
5833        C: FnOnce(&mut BatchBuilder<'_, F>),
5834    {
5835        let mut b = self.batch();
5836        build(&mut b);
5837        b.commit()
5838    }
5839
5840    /// Insert `rows` as nodes of `label`. One call is one atomic batch:
5841    /// auto-declared KeyMatch rules (if any) first, then the accepted node
5842    /// inserts, so incremental fire sees the new rules. Per-row key problems
5843    /// are collected in [`IngestReport::row_errors`] and skipped; a commit
5844    /// `Err` means nothing was applied.
5845    ///
5846    /// Auto-FK rule names are `auto_fk_<src_label_lowercase>_<field>` so
5847    /// distinct source labels sharing an FK field each get their own rule.
5848    pub fn ingest(
5849        &mut self,
5850        label: &str,
5851        rows: Vec<BTreeMap<String, Value>>,
5852        opts: &IngestOptions,
5853    ) -> Result<IngestReport> {
5854        self.ingest_with_edges(label, rows, opts, &[])
5855    }
5856
5857    /// [`ingest`] plus user edges in the **same** previewed WAL batch.
5858    /// A failing edge rejects the whole request; nothing is applied.
5859    pub fn ingest_with_edges(
5860        &mut self,
5861        label: &str,
5862        rows: Vec<BTreeMap<String, Value>>,
5863        opts: &IngestOptions,
5864        edges: &[(String, String, String)],
5865    ) -> Result<IngestReport> {
5866        crate::ingest::run(self, label, rows, opts, edges)
5867    }
5868
5869    /// Parse `json` as an array of objects and ingest via [`GraphDb::ingest`].
5870    ///
5871    /// JSON `null` fields are silently omitted (not stored, not a row error).
5872    /// Nested objects and arrays-of-objects are a per-row error (row skipped).
5873    /// Parse failures and a top-level value that is not an array of objects
5874    /// return [`GraphError::IngestError`].
5875    pub fn ingest_json(
5876        &mut self,
5877        label: &str,
5878        json: &str,
5879        opts: &IngestOptions,
5880    ) -> Result<IngestReport> {
5881        crate::ingest::run_json(self, label, json, opts)
5882    }
5883
5884    fn commit_logged_batch(
5885        &mut self,
5886        ops: Vec<BatchOp>,
5887        ingest: Option<(String, usize)>,
5888        // Two-source rule: write_batch_authz threads authz here directly (never
5889        // touches pending_write_authz); query_write_authz sets the field instead
5890        // and passes None.  Only one source is non-None per call.
5891        param_authz: Option<WriteAuthz>,
5892    ) -> Result<BatchOutcome> {
5893        // Read-only guard: catches empty-batch calls before the early-return
5894        // that skips log_then_apply_with, ensuring all mutation entry points fail.
5895        if self.read_only {
5896            return Err(GraphError::ReadOnly);
5897        }
5898        // Ensure provenance is decoded before MutPreview accesses it
5899        // (note_delete_rule / is_rule_owned may call engine.provenance()).
5900        self.engine.ensure_provenance_loaded_mut();
5901
5902        // ── Authz pre-check ──────────────────────────────────────────────────
5903        // Evaluate the decision table per-op BEFORE MutPreview so that a denial
5904        // produces no WAL frame (all-or-nothing at the authz boundary extends
5905        // the existing validate-then-apply contract to role-scope checks).
5906        //
5907        // `batch_created` tracks key→label for nodes created by earlier ops in
5908        // THIS batch, so InsertEdgeUpsert can count same-batch placeholder nodes
5909        // as visible without needing to call `self.ids.get` on not-yet-committed
5910        // keys (they won't be there yet).
5911        //
5912        // Two-source rule: param_authz (write_batch_authz path) takes precedence;
5913        // fall back to self.pending_write_authz (query_write_authz/Cypher path).
5914        // Cloning the field copy avoids a simultaneous borrow of self.ids below.
5915        let authz_opt = param_authz.or_else(|| self.pending_write_authz.clone());
5916        if let Some(ref authz) = authz_opt {
5917            let mut batch_created: BTreeMap<String, String> = BTreeMap::new();
5918            for op in &ops {
5919                self.check_single_op_authz(authz, op, &batch_created)?;
5920                // Update batch_created after a passing authz check so that
5921                // subsequent ops in this batch see the nodes as "about to exist".
5922                match op {
5923                    BatchOp::InsertNode { label, key, .. } => {
5924                        // Only track genuinely new nodes (absent from the
5925                        // snapshot at authz-check time). A pre-existing visible
5926                        // key would be a DuplicateKey — not a real creation —
5927                        // so MutPreview handles it. Letting it into batch_created
5928                        // would allow a later SetProp to bypass update_labels
5929                        // via the "batch-created → always updatable" ruling
5930                        // (delete+recreate exploit, fix for I1 review round 2).
5931                        //
5932                        // Accepted edge: for a delete+recreate-with-different-
5933                        // label batch, node_status resolves the pre-delete
5934                        // (store) label for any subsequent update checks. This
5935                        // grants no net-new capability — a role that can delete+
5936                        // create can already place arbitrary props via
5937                        // InsertNode's own props field.
5938                        if self.ids.get(key.as_str()).is_none() {
5939                            batch_created.insert(key.clone(), label.clone());
5940                        }
5941                    }
5942                    BatchOp::InsertEdgeUpsert {
5943                        placeholder_label,
5944                        src_key,
5945                        dst_key,
5946                        ..
5947                    } => {
5948                        // Both endpoints will be created if not already in store.
5949                        for ep_key in [src_key, dst_key] {
5950                            if self.ids.get(ep_key.as_str()).is_none()
5951                                && !batch_created.contains_key(ep_key.as_str())
5952                            {
5953                                batch_created.insert(ep_key.clone(), placeholder_label.clone());
5954                            }
5955                        }
5956                    }
5957                    _ => {}
5958                }
5959            }
5960        }
5961
5962        let mut outcome = BatchOutcome::default();
5963        let recs = {
5964            let mut preview = MutPreview::new(self);
5965            let mut recs = Vec::with_capacity(ops.len());
5966            // Which node row we are on, counted over the node-insert ops only.
5967            // A caller that queues its rows in order reads this straight back
5968            // as the index into its own list.
5969            let mut node_row = 0usize;
5970            // Every field name the store knows, which a `Replace` needs to work
5971            // out what it removes. Resolved on the first `Replace` in the frame
5972            // and reused, so N replaces read the field list once, not N times.
5973            let mut store_fields: Option<Vec<String>> = None;
5974            // Duplicate inserts this frame has to count, each paired with the
5975            // position in `recs` it belongs at. The count itself is named in the
5976            // dense rewrite and not here: a duplicate's endpoints and edge type
5977            // may all be created by earlier ops in this same frame, and nothing
5978            // in the frame has a dense id yet. See [`PlannedRec`].
5979            let mut deferred_counts: Vec<(usize, String, String, String)> = Vec::new();
5980            for op in ops {
5981                match op {
5982                    BatchOp::InsertNode { label, key, props } => {
5983                        node_row += 1;
5984                        preview.check_insert_node(&key, &props)?;
5985                        preview.note_insert_node(&label, &key, &props);
5986                        recs.push(WalRecord::InsertNode { label, key, props });
5987                    }
5988                    BatchOp::InsertNodeOnConflict {
5989                        label,
5990                        key,
5991                        props,
5992                        on_conflict,
5993                    } => {
5994                        let row = node_row;
5995                        node_row += 1;
5996                        if !preview.has_key(&key) {
5997                            // No conflict: an ordinary insert on any policy —
5998                            // except that a supplied view-owned field is the
5999                            // same mistake here as on a taken key, and gets the
6000                            // same row error rather than a frame error. Without
6001                            // this, one op answered one request two ways
6002                            // depending on whether the store already had the
6003                            // key (defect #19).
6004                            if let Some(why) = preview.supplied_view_owned_prop(&key, &props) {
6005                                outcome.row_errors.push((row, why));
6006                                continue;
6007                            }
6008                            preview.note_insert_node(&label, &key, &props);
6009                            recs.push(WalRecord::InsertNode { label, key, props });
6010                            continue;
6011                        }
6012                        match on_conflict {
6013                            OnConflict::Error => {
6014                                return Err(GraphError::DuplicateKey { key });
6015                            }
6016                            OnConflict::Skip => outcome.skipped += 1,
6017                            OnConflict::Replace => {
6018                                if store_fields.is_none() {
6019                                    store_fields = Some(preview.db.props_view().field_names());
6020                                }
6021                                let fields = store_fields.as_deref().unwrap_or_default();
6022                                match preview.plan_replace(&label, &key, &props, fields) {
6023                                    Ok((writes, kept_view_owned)) => {
6024                                        outcome.kept_view_owned += kept_view_owned;
6025                                        for (field, value) in writes {
6026                                            match value {
6027                                                Some(value) => {
6028                                                    preview.note_set_prop(&key, &field, &value);
6029                                                    recs.push(WalRecord::SetProp {
6030                                                        key: key.clone(),
6031                                                        field,
6032                                                        value,
6033                                                    });
6034                                                }
6035                                                None => {
6036                                                    preview.note_remove_prop(&key, &field);
6037                                                    recs.push(WalRecord::RemoveProp {
6038                                                        key: key.clone(),
6039                                                        field,
6040                                                    });
6041                                                }
6042                                            }
6043                                        }
6044                                        outcome.replaced += 1;
6045                                    }
6046                                    Err(why) => outcome.row_errors.push((row, why)),
6047                                }
6048                            }
6049                        }
6050                    }
6051                    BatchOp::InsertEdge {
6052                        edge_type,
6053                        src_key,
6054                        dst_key,
6055                    } => {
6056                        if preview.prepare_insert_edge(&edge_type, &src_key, &dst_key)? {
6057                            preview.note_insert_edge(&edge_type, &src_key, &dst_key);
6058                            recs.push(WalRecord::InsertEdge {
6059                                edge_type,
6060                                src_key,
6061                                dst_key,
6062                            });
6063                        } else if preview.db.multiplicity {
6064                            // A duplicate inside a batch counts the way a
6065                            // duplicate through `insert_edge` does: `ingest` and
6066                            // Cypher `CREATE` reach this choke-point and not
6067                            // that one, and a count only one entry point keeps
6068                            // would be worse than no count at all.
6069                            //
6070                            // This is the one gate on discriminant 23 from the
6071                            // batch path: a store that never opted in queues
6072                            // nothing here and so writes no such record.
6073                            deferred_counts.push((recs.len(), edge_type, src_key, dst_key));
6074                        }
6075                    }
6076                    BatchOp::SetProp { key, field, value } => {
6077                        if let Some(view_name) = preview.db.view_store.view_for_prop(&field) {
6078                            return Err(GraphError::ViewPropReadOnly {
6079                                view_name: view_name.to_string(),
6080                            });
6081                        }
6082                        preview.check_live_key(&key)?;
6083                        preview.note_set_prop(&key, &field, &value);
6084                        recs.push(WalRecord::SetProp { key, field, value });
6085                    }
6086                    BatchOp::RemoveProp { key, field } => {
6087                        if preview.prepare_remove_prop(&key, &field)? {
6088                            preview.note_remove_prop(&key, &field);
6089                            recs.push(WalRecord::RemoveProp { key, field });
6090                        }
6091                    }
6092                    BatchOp::DeleteEdge {
6093                        edge_type,
6094                        src_key,
6095                        dst_key,
6096                    } => {
6097                        if preview.prepare_delete_edge(&edge_type, &src_key, &dst_key)? {
6098                            preview.note_delete_edge(&edge_type, &src_key, &dst_key);
6099                            recs.push(WalRecord::DeleteEdge {
6100                                edge_type,
6101                                src_key,
6102                                dst_key,
6103                            });
6104                        }
6105                    }
6106                    BatchOp::DeleteNode { key } => {
6107                        preview.check_live_key(&key)?;
6108                        preview.note_delete_node(&key);
6109                        recs.push(WalRecord::DeleteNode { key });
6110                    }
6111                    BatchOp::CreateRule(def) => {
6112                        preview.check_create_rule(&def)?;
6113                        let def_bytes =
6114                            bincode::serialize(&def).map_err(|e| GraphError::Corrupt {
6115                                detail: format!("serialize rule: {e}"),
6116                            })?;
6117                        preview.note_create_rule(&def);
6118                        recs.push(WalRecord::CreateRule { def_bytes });
6119                    }
6120                    BatchOp::DeleteRule { name } => {
6121                        preview.check_delete_rule(&name)?;
6122                        preview.note_delete_rule(&name);
6123                        recs.push(WalRecord::DeleteRule { name });
6124                    }
6125                    BatchOp::RenameNode { old_key, new_key } => {
6126                        preview.check_rename_node(&old_key, &new_key)?;
6127                        preview.note_rename_node(&old_key, &new_key);
6128                        recs.push(WalRecord::RenameNode { old_key, new_key });
6129                    }
6130                    BatchOp::InsertEdgeUpsert {
6131                        edge_type,
6132                        src_key,
6133                        dst_key,
6134                        placeholder_label,
6135                    } => {
6136                        // Auto-create any missing endpoints as plain InsertNode ops.
6137                        // Rules fire and last-change is updated for each created node.
6138                        for key in [&src_key, &dst_key] {
6139                            if !preview.has_key(key) {
6140                                // A placeholder endpoint carries no props, so
6141                                // the view-owned check has nothing to refuse.
6142                                preview.check_insert_node(key, &[])?;
6143                                preview.note_insert_node(&placeholder_label, key, &[]);
6144                                recs.push(WalRecord::InsertNode {
6145                                    label: placeholder_label.clone(),
6146                                    key: key.clone(),
6147                                    props: vec![],
6148                                });
6149                            }
6150                        }
6151                        if preview.prepare_insert_edge(&edge_type, &src_key, &dst_key)? {
6152                            preview.note_insert_edge(&edge_type, &src_key, &dst_key);
6153                            recs.push(WalRecord::InsertEdge {
6154                                edge_type,
6155                                src_key,
6156                                dst_key,
6157                            });
6158                        } else if preview.db.multiplicity {
6159                            // Same choke-point, same gate as `BatchOp::InsertEdge`
6160                            // above: an upsert that finds the pair already there
6161                            // is a duplicate insert and counts as one.
6162                            deferred_counts.push((recs.len(), edge_type, src_key, dst_key));
6163                        }
6164                    }
6165                }
6166            }
6167            // Splice the deferred counts back into the positions they were
6168            // raised at, so a count still sits exactly where the duplicate did
6169            // — before any later op in the frame that deletes the pair.
6170            let mut planned: Vec<PlannedRec> =
6171                Vec::with_capacity(recs.len() + deferred_counts.len());
6172            let mut deferred = deferred_counts.into_iter().peekable();
6173            for (i, rec) in recs.into_iter().enumerate() {
6174                while deferred.peek().is_some_and(|(at, ..)| *at == i) {
6175                    let (_, edge_type, src_key, dst_key) = deferred.next().expect("just peeked");
6176                    planned.push(PlannedRec::DuplicateCount {
6177                        edge_type,
6178                        src_key,
6179                        dst_key,
6180                    });
6181                }
6182                planned.push(PlannedRec::Rec(rec));
6183            }
6184            for (_, edge_type, src_key, dst_key) in deferred {
6185                planned.push(PlannedRec::DuplicateCount {
6186                    edge_type,
6187                    src_key,
6188                    dst_key,
6189                });
6190            }
6191            planned
6192        };
6193        // A frame that is nothing but skips or refused rows writes no WAL, but
6194        // it still has counts to report, so the early returns carry `outcome`
6195        // rather than zeros.
6196        if recs.is_empty() {
6197            return Ok(outcome);
6198        }
6199        // rewrite_wal_dense converts every InsertNode/InsertEdge into its
6200        // *Id form, so only the dense variants can appear in `recs` here.
6201        let recs = self.rewrite_wal_dense_planned(recs)?;
6202        // The rewrite can empty a non-empty batch: a `SET n.ns` naming the
6203        // namespace the node is already in is a no-op and is dropped there. An
6204        // empty `Batch` frame would still take a commit sequence and a WAL
6205        // record, so a batch that turns out to be nothing writes nothing.
6206        if recs.is_empty() {
6207            return Ok(outcome);
6208        }
6209        outcome.nodes_inserted = recs
6210            .iter()
6211            .filter(|r| matches!(r, WalRecord::InsertNodeId { .. }))
6212            .count();
6213        outcome.edges_inserted = recs
6214            .iter()
6215            .filter(|r| matches!(r, WalRecord::InsertEdgeId { .. }))
6216            .count();
6217        // Ingest / write_batch / query_write: one Batch frame, one fsync per call
6218        // under Strict.  Pass self.fsync directly so Strict stays Strict —
6219        // wal_needs_sync(Strict, _) always returns true regardless of op count.
6220        // Mapping Strict → Batched (the prior bug) caused wal_needs_sync to
6221        // short-circuit on single-op batches and silently skip the fsync.
6222        // Batched fsyncs only for multi-op batches; Relaxed always skips.
6223        self.log_then_apply_with(WalRecord::Batch(recs), ingest, self.fsync)?;
6224        Ok(outcome)
6225    }
6226
6227    fn commit_batch(&mut self, ops: Vec<BatchOp>) -> Result<(usize, usize)> {
6228        self.commit_logged_batch(ops, None, None).map(inserted_pair)
6229    }
6230
6231    /// Commit one submission WITHOUT an fsync — for use inside `commit_group`
6232    /// and the group-commit drain thread, which do a single group fsync later.
6233    fn commit_batch_nosync(&mut self, ops: Vec<BatchOp>) -> Result<(usize, usize)> {
6234        // Restore fsync policy even on panic via a raw-pointer drop guard.
6235        // A panic here would poison the RwLock anyway, but the correct policy
6236        // must be in place if the guard is ever unwrapped.
6237        struct RestoreFsync(*mut FsyncPolicy, FsyncPolicy);
6238        impl Drop for RestoreFsync {
6239            fn drop(&mut self) {
6240                // SAFETY: the pointer is valid for the full duration of
6241                // commit_batch_nosync; the guard is dropped before the frame
6242                // returns, and GraphDb outlives this frame.
6243                unsafe {
6244                    *self.0 = self.1;
6245                }
6246            }
6247        }
6248        let saved = self.fsync;
6249        // SAFETY: raw pointer into self; guard dropped within this frame.
6250        let _g = RestoreFsync(&mut self.fsync as *mut FsyncPolicy, saved);
6251        self.fsync = FsyncPolicy::Relaxed;
6252        self.commit_logged_batch(ops, None, None).map(inserted_pair)
6253    }
6254
6255    /// Commit multiple op-batches as a **group**: each submission gets its own
6256    /// WAL `Batch` frame, but there is exactly **one** `Fs::sync` for the whole
6257    /// group (under `Strict` / `Batched` policy; `Relaxed` skips all syncs).
6258    ///
6259    /// # Durability semantics
6260    ///
6261    /// A crash before the group fsync may lose **all** submissions in the group.
6262    /// A crash after the group fsync preserves all of them.  No submission is
6263    /// ever torn: each WAL frame is either fully applied on replay or dropped
6264    /// in its entirety (CRC-protected frame boundaries).
6265    ///
6266    /// Events and subscription notifications fire per-submission immediately
6267    /// after apply, which may be before the group fsync.  From a subscriber's
6268    /// perspective this is equivalent to the `Relaxed` durability window.
6269    /// Submitters using [`SharedDb::submit_batch`] only unblock after the group
6270    /// fsync, so from their perspective durability is fully guaranteed.
6271    ///
6272    /// # MVCC interplay
6273    ///
6274    /// Each submission records its own `CommitDelta`; the fold-every-K counter
6275    /// increments per submission (not per group), preserving existing reader
6276    /// snapshot semantics.
6277    ///
6278    /// # Returns
6279    ///
6280    /// One `Result<(nodes_inserted, edges_inserted)>` per input group element,
6281    /// in order.  Failures are per-submission (validation errors); the group
6282    /// fsync error (if any) is returned as the second tuple element.
6283    pub fn commit_group(
6284        &mut self,
6285        groups: Vec<Vec<BatchOp>>,
6286    ) -> (Vec<Result<(usize, usize)>>, Option<GraphError>) {
6287        let mut results = Vec::with_capacity(groups.len());
6288        for ops in groups {
6289            results.push(self.commit_batch_nosync(ops));
6290        }
6291        let any_ok = results.iter().any(|r| r.is_ok());
6292        let sync_err = if self.fsync != FsyncPolicy::Relaxed && any_ok {
6293            self.fs
6294                .sync(core_storage::fs::FileId::Wal)
6295                .map_err(GraphError::Io)
6296                .err()
6297        } else {
6298            None
6299        };
6300        (results, sync_err)
6301    }
6302
6303    /// Like [`commit_group`] but skips the group fsync entirely.
6304    ///
6305    /// Used by the drain thread to apply submissions under the write lock and
6306    /// then perform the single fsync OUTSIDE the lock (via
6307    /// `core_storage::sync_wal_at`), reducing the write-lock hold time visible
6308    /// to concurrent readers.
6309    pub fn commit_group_nosync(
6310        &mut self,
6311        groups: Vec<Vec<BatchOp>>,
6312    ) -> Vec<Result<(usize, usize)>> {
6313        let mut results = Vec::with_capacity(groups.len());
6314        for ops in groups {
6315            results.push(self.commit_batch_nosync(ops));
6316        }
6317        results
6318    }
6319
6320    pub fn insert_node(
6321        &mut self,
6322        label: &str,
6323        key: &str,
6324        props: Vec<(String, Value)>,
6325    ) -> Result<()> {
6326        if self.read_only {
6327            return Err(GraphError::ReadOnly);
6328        }
6329        MutPreview::new(self).check_insert_node(key, &props)?;
6330        self.log_dense(vec![WalRecord::InsertNode {
6331            label: label.into(),
6332            key: key.into(),
6333            props,
6334        }])
6335    }
6336
6337    /// Insert a user edge. `Ok(true)` when the pair was new, `Ok(false)` when it
6338    /// was already there — the question is "was this pair new", and a duplicate
6339    /// does not make it so.
6340    ///
6341    /// On a store that called [`enable_multiplicity`](Self::enable_multiplicity)
6342    /// a duplicate is no longer a total no-op: it raises the pair's insert count
6343    /// (§5.13). Adjacency is still a set, so [`degree`](Self::degree) is
6344    /// unchanged and the return value is still `Ok(false)`; the count is visible
6345    /// only through [`degree_multiplicity`](Self::degree_multiplicity) and the
6346    /// reserved [`EDGE_COUNT_PROP`]. On every other store a duplicate writes
6347    /// nothing at all, as it always has.
6348    pub fn insert_edge(&mut self, edge_type: &str, src_key: &str, dst_key: &str) -> Result<bool> {
6349        if self.read_only {
6350            return Err(GraphError::ReadOnly);
6351        }
6352        if !MutPreview::new(self).prepare_insert_edge(edge_type, src_key, dst_key)? {
6353            // The pair exists. The only thing left to record is that it was
6354            // asked for again, and only where the store asked to be told.
6355            if let Some(rec) = self.edge_count_record(edge_type, src_key, dst_key) {
6356                self.log_then_apply(rec)?;
6357            }
6358            return Ok(false);
6359        }
6360        self.log_dense(vec![WalRecord::InsertEdge {
6361            edge_type: edge_type.into(),
6362            src_key: src_key.into(),
6363            dst_key: dst_key.into(),
6364        }])?;
6365        Ok(true)
6366    }
6367
6368    pub fn set_prop(&mut self, key: &str, field: &str, value: Value) -> Result<()> {
6369        if self.read_only {
6370            return Err(GraphError::ReadOnly);
6371        }
6372        if let Some(view_name) = self.view_store.view_for_prop(field) {
6373            return Err(GraphError::ViewPropReadOnly {
6374                view_name: view_name.to_string(),
6375            });
6376        }
6377        MutPreview::new(self).check_live_key(key)?;
6378        self.log_dense(vec![WalRecord::SetProp {
6379            key: key.into(),
6380            field: field.into(),
6381            value,
6382        }])
6383    }
6384
6385    /// Set several properties on one live node in a single WAL commit.
6386    ///
6387    /// Every per-property check [`set_prop`](Self::set_prop) runs — view-owned
6388    /// names, live key, the `ns` immutability rule and its type — is evaluated
6389    /// for the whole list before any record is logged. The first refusal
6390    /// returns and the node is unchanged. An empty list writes nothing.
6391    pub fn set_props(&mut self, key: &str, props: Vec<(String, Value)>) -> Result<()> {
6392        if self.read_only {
6393            return Err(GraphError::ReadOnly);
6394        }
6395        MutPreview::new(self).check_live_key(key)?;
6396        for (field, _) in &props {
6397            if let Some(view_name) = self.view_store.view_for_prop(field) {
6398                return Err(GraphError::ViewPropReadOnly {
6399                    view_name: view_name.to_string(),
6400                });
6401            }
6402        }
6403        if props.is_empty() {
6404            return Ok(());
6405        }
6406        self.write_batch(|b| {
6407            for (field, value) in props {
6408                b.set_prop(key, &field, value);
6409            }
6410        })
6411        .map(|_| ())
6412    }
6413
6414    /// Remove a property. Returns `Ok(false)` (and does not log) if the field
6415    /// is already absent. Unknown or tombstoned keys are `Err(KeyNotFound)`.
6416    /// A field a view owns is `Err(ViewPropReadOnly)` — stated once, in
6417    /// [`MutPreview::prepare_remove_prop`], so that the batch ops reaching that
6418    /// same choke-point cannot miss it.
6419    pub fn remove_prop(&mut self, key: &str, field: &str) -> Result<bool> {
6420        if self.read_only {
6421            return Err(GraphError::ReadOnly);
6422        }
6423        if !MutPreview::new(self).prepare_remove_prop(key, field)? {
6424            return Ok(false);
6425        }
6426        self.log_then_apply(WalRecord::RemoveProp {
6427            key: key.into(),
6428            field: field.into(),
6429        })?;
6430        Ok(true)
6431    }
6432
6433    /// Delete a user edge. Returns `Ok(false)` (and does not log) if the edge
6434    /// is absent. Unknown keys are `Err(KeyNotFound)`. Rule-owned edges — in
6435    /// provenance, or a pair a live rule would derive — are `Err(RuleOwned)`
6436    /// (the rule would just put the edge back; delete or change the rule).
6437    pub fn delete_edge(&mut self, edge_type: &str, src_key: &str, dst_key: &str) -> Result<bool> {
6438        if self.read_only {
6439            return Err(GraphError::ReadOnly);
6440        }
6441        if !MutPreview::new(self).prepare_delete_edge(edge_type, src_key, dst_key)? {
6442            return Ok(false);
6443        }
6444        self.log_then_apply(WalRecord::DeleteEdge {
6445            edge_type: edge_type.into(),
6446            src_key: src_key.into(),
6447            dst_key: dst_key.into(),
6448        })?;
6449        Ok(true)
6450    }
6451
6452    /// Delete a live node. Unknown or already-tombstoned keys are
6453    /// `Err(KeyNotFound)` and are not logged. Validation runs before the WAL
6454    /// write; `apply` of a logged `DeleteNode` for an already-tombstoned key
6455    /// (crash window) is a clean no-op.
6456    ///
6457    /// Returns a [`DeleteReport`] with counts of manual and derived edges
6458    /// removed (computed from live state before the deletion is applied).
6459    pub fn delete_node(&mut self, key: &str) -> Result<DeleteReport> {
6460        if self.read_only {
6461            return Err(GraphError::ReadOnly);
6462        }
6463        // Provenance must be loaded before we query provenance_touching.
6464        self.engine.ensure_provenance_loaded_mut();
6465        let id = self
6466            .ids
6467            .get(key)
6468            .ok_or_else(|| GraphError::KeyNotFound { key: key.into() })?;
6469
6470        // Count edges before the delete is applied so we can report counts.
6471        let derived_set: BTreeSet<(u32, u32, u32)> = self
6472            .engine
6473            .provenance_touching(id)
6474            .map(|(_, etype, src, dst)| (etype, src, dst))
6475            .collect();
6476        let derived_edges = derived_set.len() as u64;
6477
6478        let mut total_topo = 0u64;
6479        let tv = self.topo_view();
6480        for et in tv.etypes() {
6481            total_topo += tv.neighbors(et, Direction::Out, id).len() as u64
6482                + tv.neighbors(et, Direction::In, id).len() as u64;
6483        }
6484        // For symmetric rules (e.g. Overlap), a→b and b→a are two separate directed
6485        // triples in both the topo scan (Out and In from id) and in provenance_touching.
6486        // The subtraction remains correct because both counts include both directions.
6487        let manual_edges = total_topo.saturating_sub(derived_edges);
6488
6489        self.log_then_apply(WalRecord::DeleteNode { key: key.into() })?;
6490        Ok(DeleteReport {
6491            manual_edges,
6492            derived_edges,
6493        })
6494    }
6495
6496    /// Rename a live node's key.  The dense id (and therefore all edges,
6497    /// props, history, and last-change tracking) is unaffected.
6498    ///
6499    /// Returns `Err(KeyNotFound)` if `old` is not a live key.
6500    /// Returns `Err(DuplicateKey)` if `new` is already live.
6501    pub fn rename_node(&mut self, old: &str, new: &str) -> Result<()> {
6502        if self.read_only {
6503            return Err(GraphError::ReadOnly);
6504        }
6505        MutPreview::new(self).check_rename_node(old, new)?;
6506        self.log_then_apply(WalRecord::RenameNode {
6507            old_key: old.into(),
6508            new_key: new.into(),
6509        })
6510    }
6511
6512    /// Return the IVF drift counter for the dst-side candidate index of `rule`.
6513    /// `None` if the rule does not exist or is not approximate.
6514    ///
6515    /// The drift counter increments on IVF insert/remove after the last fit.
6516    /// When dst-side drift exceeds [`core_rules::IVF_DRIFT_REBUILD`], apply
6517    /// WAL-logs `RebuildRule` as a second commit (rebuild resets the counter).
6518    pub fn ivf_dst_drift(&self, rule: &str) -> Option<u64> {
6519        // SideIvfExport = (centroids, node→cluster, drift)
6520        self.engine
6521            .export_ivf_state()
6522            .remove(rule)
6523            .map(|(_src, dst)| dst.2)
6524    }
6525
6526    /// Validate and WAL-log a new rule, then backfill derived edges inside apply.
6527    /// Validation and duplicate-name check run before logging so invalid rules
6528    /// never enter the WAL.
6529    pub fn create_rule(&mut self, def: RuleDef) -> Result<()> {
6530        if self.read_only {
6531            return Err(GraphError::ReadOnly);
6532        }
6533        MutPreview::new(self).check_create_rule(&def)?;
6534        let def_bytes = bincode::serialize(&def).map_err(|e| GraphError::Corrupt {
6535            detail: format!("serialize rule: {e}"),
6536        })?;
6537        self.log_then_apply(WalRecord::CreateRule { def_bytes })
6538    }
6539
6540    /// Override this handle's HNSW build-slice size, or `None` to restore
6541    /// [`core_rules::HNSW_BUILD_BATCH`].
6542    ///
6543    /// Exposed for tests that need a small slice without a large corpus; not
6544    /// part of the stable surface.
6545    #[doc(hidden)]
6546    pub fn set_hnsw_build_batch(&mut self, batch: Option<usize>) {
6547        self.engine.set_hnsw_build_batch(batch);
6548    }
6549
6550    /// Rules whose vector index is still being built, in name order.
6551    ///
6552    /// The same list [`GraphDb::stats`] reports per rule in `building`.
6553    /// After a clean open this includes a build a snapshot cut short, so
6554    /// `serve`'s ticker can pump it without a write.
6555    pub fn builds_in_progress(&self) -> Vec<BuildProgress> {
6556        self.engine.builds_in_progress()
6557    }
6558
6559    /// Advance any vector index still building and backfill each rule that
6560    /// finishes. Returns what is still outstanding.
6561    ///
6562    /// A map lookup when nothing is pending, so it is cheap to call on a timer.
6563    /// One write lock and at most [`core_rules::HNSW_BUILD_BATCH`] vector
6564    /// inserts per pending rule per call, so a caller can drive a large build
6565    /// to completion without ever holding the lock for more than a slice.
6566    ///
6567    /// A rule that finishes here is backfilled through the same
6568    /// `WalRecord::RebuildRule` second commit that IVF drift already uses, so
6569    /// its derived edges are produced by [`GraphDb::rebuild_rule`]'s code path
6570    /// and appear all at once.
6571    ///
6572    /// Every ordinary write pumps one slice on its own (see the post-commit
6573    /// hook in `log_then_apply_with`), so this is for quiescent stores and for
6574    /// operators who want the build finished before traffic arrives.
6575    pub fn pump_index_build(&mut self) -> Result<Vec<BuildProgress>> {
6576        Ok(self.pump_index_build_reporting()?.1)
6577    }
6578
6579    /// [`GraphDb::pump_index_build`], also reporting the builds that **this**
6580    /// call finished, so a progress display can say so.
6581    ///
6582    /// A build can be registered and completed inside a single call — that is
6583    /// what a mid-build snapshot looks like on reopen, where the index scan
6584    /// finishes the graph and only the backfill is outstanding — and the
6585    /// outstanding list alone cannot show that anything happened.
6586    pub fn pump_index_build_reporting(
6587        &mut self,
6588    ) -> Result<(Vec<BuildProgress>, Vec<BuildProgress>)> {
6589        // A read-only handle cannot issue the `RebuildRule` a finished build
6590        // needs, so it would advance the index and then silently fail to
6591        // produce the edges. Refusing is the honest answer.
6592        if self.read_only {
6593            return Err(GraphError::ReadOnly);
6594        }
6595        let finished = self.pump_one_slice();
6596        for done in &finished {
6597            // The index is whole but the rule still owns no edges. A failed
6598            // second commit must leave the rule re-pumpable rather than
6599            // silently edge-less, so the error is surfaced here — unlike the
6600            // post-commit hook, this call is not riding someone else's commit.
6601            self.log_then_apply(WalRecord::RebuildRule {
6602                name: done.rule.clone(),
6603            })?;
6604        }
6605        Ok((finished, self.engine.builds_in_progress()))
6606    }
6607
6608    /// Run the deferred candidate-index build, if it is still owed, against the
6609    /// graph as it stands *now* — before the caller applies anything.
6610    ///
6611    /// A no-op bool test once the indexes are populated, which is after the
6612    /// first write of the handle's life, and for a store with no rules at all.
6613    fn populate_indexes_before_write(&mut self) {
6614        if !self.engine.needs_index_population() {
6615            return;
6616        }
6617        // The retained snapshot blobs arrive with the V8 base sections; without
6618        // them the scan would rebuild every graph the snapshot already holds.
6619        self.ensure_v8_base_sections_loaded();
6620        if !self.engine.needs_index_population() {
6621            return;
6622        }
6623        let mut eng = std::mem::take(&mut self.engine);
6624        {
6625            let gm = make_graph_mut(
6626                &self.ids,
6627                &mut self.syms,
6628                &self.labels,
6629                build_props_view(&self.props, &self.base),
6630                &mut self.topo,
6631                &self.base,
6632                &mut self.edge_props,
6633            );
6634            eng.populate_indexes(&gm);
6635        }
6636        self.engine = eng;
6637    }
6638
6639    /// One slice of build work for every pending rule. Returns the rules whose
6640    /// index just became whole, which the caller must `RebuildRule`.
6641    ///
6642    /// Goes through the engine even with nothing pending when the indexes have
6643    /// not been populated yet: that call adopts the persisted graphs and, for
6644    /// an incomplete blob already registered at open, leaves the remainder to
6645    /// this slice rather than inserting it inline.
6646    fn pump_one_slice(&mut self) -> Vec<BuildProgress> {
6647        // The retained snapshot blobs — and the id count an interrupted build
6648        // is recognised against — arrive with the V8 base sections, which a
6649        // clean open reads lazily. Without this a freshly opened handle pumps
6650        // against empty retained state and concludes there is nothing to do,
6651        // which is precisely the store `build-index` exists for.
6652        self.ensure_v8_base_sections_loaded();
6653        let mut eng = std::mem::take(&mut self.engine);
6654        let finished = {
6655            let mut gm = make_graph_mut(
6656                &self.ids,
6657                &mut self.syms,
6658                &self.labels,
6659                build_props_view(&self.props, &self.base),
6660                &mut self.topo,
6661                &self.base,
6662                &mut self.edge_props,
6663            );
6664            eng.pump_index_build(&mut gm)
6665        };
6666        self.engine = eng;
6667        finished
6668    }
6669
6670    /// Register a sliced build a snapshot cut short, from blobs with
6671    /// `complete == false`.
6672    ///
6673    /// Peeks the V8 mmap for incomplete entries without copying complete
6674    /// graphs. V5–V7 already hold the blobs in the engine from restore.
6675    fn register_outstanding_index_builds(&mut self) {
6676        if self.engine.indexes_populated() {
6677            return;
6678        }
6679        let extra = self.collect_incomplete_hnsw_blobs();
6680        let mut eng = std::mem::take(&mut self.engine);
6681        {
6682            let gm = make_graph_mut(
6683                &self.ids,
6684                &mut self.syms,
6685                &self.labels,
6686                build_props_view(&self.props, &self.base),
6687                &mut self.topo,
6688                &self.base,
6689                &mut self.edge_props,
6690            );
6691            eng.register_incomplete_hnsw_builds(&extra, &gm);
6692        }
6693        self.engine = eng;
6694    }
6695
6696    /// Incomplete `(src, dst)` HNSW blobs from the V8 mmap, copied only when
6697    /// `complete` is false. Empty when there is no mmap base (V5–V7 uses the
6698    /// engine's retained map instead).
6699    fn collect_incomplete_hnsw_blobs(&self) -> BTreeMap<String, (Vec<u8>, Vec<u8>)> {
6700        let Some(base) = &self.base else {
6701            return BTreeMap::new();
6702        };
6703        let Ok(archived) = base.hnsw_section() else {
6704            return BTreeMap::new();
6705        };
6706        archived
6707            .rules
6708            .iter()
6709            .filter_map(|e| {
6710                let src = e.src_blob.as_slice();
6711                let dst = e.dst_blob.as_slice();
6712                if core_rules::hnsw::hnsw_blob_complete(src) == Some(false)
6713                    || core_rules::hnsw::hnsw_blob_complete(dst) == Some(false)
6714                {
6715                    Some((e.name.as_str().to_string(), (src.to_vec(), dst.to_vec())))
6716                } else {
6717                    None
6718                }
6719            })
6720            .collect()
6721    }
6722
6723    /// WAL-log rule deletion. Returns RuleNotFound if the rule does not exist.
6724    pub fn delete_rule(&mut self, name: &str) -> Result<()> {
6725        if self.read_only {
6726            return Err(GraphError::ReadOnly);
6727        }
6728        MutPreview::new(self).check_delete_rule(name)?;
6729        self.log_then_apply(WalRecord::DeleteRule { name: name.into() })
6730    }
6731
6732    /// Return a snapshot of all registered rules.
6733    pub fn rules(&self) -> Vec<RuleDef> {
6734        self.engine.rules().cloned().collect()
6735    }
6736
6737    // -----------------------------------------------------------------------
6738    // Rule suggestion API
6739    // -----------------------------------------------------------------------
6740
6741    /// Profile the database and suggest linking rules with previewed edge counts.
6742    ///
6743    /// Uses the default seed ([`core_rules::SUGGEST_DEFAULT_SEED`]) for deterministic
6744    /// sampling. Suggestions are sorted by estimated edge count (descending).
6745    /// **NO auto-accept** — call [`GraphDb::create_rule`] explicitly to apply.
6746    pub fn suggest_rules(&self) -> Vec<core_rules::RuleSuggestion> {
6747        self.suggest_rules_seeded(core_rules::SUGGEST_DEFAULT_SEED)
6748    }
6749
6750    /// Like [`suggest_rules`] but with a caller-supplied RNG seed for
6751    /// reproducibility. Same seed + same data = identical output.
6752    pub fn suggest_rules_seeded(&self, seed: u64) -> Vec<core_rules::RuleSuggestion> {
6753        self.suggest_rules_with_config(&core_rules::suggest::SuggestConfig::default(), seed)
6754            .suggestions
6755    }
6756
6757    /// [`suggest_rules_seeded`] with a fully custom [`SuggestConfig`].
6758    ///
6759    /// Returns a [`core_rules::SuggestReport`] that includes both the candidate list
6760    /// and a `truncated` flag indicating whether the global budget fired before all
6761    /// candidates were evaluated.
6762    pub fn suggest_rules_with_config(
6763        &self,
6764        config: &core_rules::suggest::SuggestConfig,
6765        seed: u64,
6766    ) -> core_rules::SuggestReport {
6767        use std::collections::BTreeMap;
6768
6769        // Collect (node_id, key) pairs per label, skipping tombstoned nodes.
6770        let mut label_nodes: BTreeMap<String, Vec<(u32, String)>> = BTreeMap::new();
6771        for id in 0..self.ids.len() as u32 {
6772            let Some(key) = self.ids.key_of(id) else {
6773                continue;
6774            };
6775            let Some(&sym) = self.labels.get(id as usize) else {
6776                continue;
6777            };
6778            if sym == u32::MAX {
6779                continue; // tombstoned
6780            }
6781            let Some(label) = self.syms.resolve(sym) else {
6782                continue;
6783            };
6784            label_nodes
6785                .entry(label.to_string())
6786                .or_default()
6787                .push((id, key.to_string()));
6788        }
6789
6790        let existing = self.rules();
6791        let pv = build_props_view(&self.props, &self.base);
6792        let all_fields: Vec<String> = pv.field_names();
6793
6794        core_rules::suggest::suggest_rules(
6795            &label_nodes,
6796            &|id, field| pv.get(id, field).map(|vr| vr.into_value()),
6797            &all_fields,
6798            &existing,
6799            config,
6800            seed,
6801        )
6802    }
6803
6804    /// Recompute a rule's derived edges from scratch. WAL-logged so un-trip
6805    /// plus later mutations replay identically (rebuild is a pure function
6806    /// of state).
6807    ///
6808    /// Only exit from the tripped latch: if the full desired set fits the
6809    /// budget, it is applied completely and `tripped` clears; if it still
6810    /// exceeds the budget, provenance is left untouched and `tripped` stays
6811    /// true. Counts as a fire evaluation (see [`RuleStats::fires`]).
6812    /// Unknown rule → `RuleNotFound`, nothing logged.
6813    pub fn rebuild_rule(&mut self, name: &str) -> Result<()> {
6814        if self.read_only {
6815            return Err(GraphError::ReadOnly);
6816        }
6817        if !self.engine.rules().any(|r| r.name == name) {
6818            return Err(GraphError::RuleNotFound { name: name.into() });
6819        }
6820        self.log_then_apply(WalRecord::RebuildRule { name: name.into() })
6821    }
6822
6823    // -----------------------------------------------------------------------
6824    // Materialized view API
6825    // -----------------------------------------------------------------------
6826
6827    /// Register a new materialized property view, backfill its values for all
6828    /// existing nodes, and WAL-log the definition.
6829    ///
6830    /// # Errors
6831    /// - `ReadOnly`: called on an as-of instance.
6832    /// - `RuleInvalid`: name collision, view_prop collision, or invalid def.
6833    pub fn create_view(&mut self, def: ViewDef) -> Result<()> {
6834        if self.read_only {
6835            return Err(GraphError::ReadOnly);
6836        }
6837        // Pre-validate before WAL write.
6838        def.validate()
6839            .map_err(|e| GraphError::RuleInvalid { detail: e })?;
6840        if self.view_store.has_view(&def.name) {
6841            return Err(GraphError::RuleInvalid {
6842                detail: format!("view {:?} already exists", def.name),
6843            });
6844        }
6845        if let Some(existing) = self.view_store.view_for_prop(&def.view_prop) {
6846            return Err(GraphError::RuleInvalid {
6847                detail: format!(
6848                    "view_prop {:?} is already used by view {:?}",
6849                    def.view_prop, existing
6850                ),
6851            });
6852        }
6853        let def_bytes = bincode::serialize(&def).map_err(|e| GraphError::Corrupt {
6854            detail: format!("serialize view: {e}"),
6855        })?;
6856        // Enable delta accumulation before the view is registered so subsequent
6857        // incremental edge events reach view maintenance from this point onward.
6858        // (The backfill inside create_view reads topo directly; it does not rely
6859        // on pending deltas.)
6860        self.engine.set_emit_deltas(true);
6861        self.log_then_apply(WalRecord::CreateView { def_bytes })
6862    }
6863
6864    /// Remove a named view and delete its values from every node.
6865    ///
6866    /// # Errors
6867    /// - `ReadOnly`: called on an as-of instance.
6868    /// - `RuleNotFound`: view does not exist.
6869    pub fn delete_view(&mut self, name: &str) -> Result<()> {
6870        if self.read_only {
6871            return Err(GraphError::ReadOnly);
6872        }
6873        if !self.view_store.has_view(name) {
6874            return Err(GraphError::RuleNotFound { name: name.into() });
6875        }
6876        let result = self.log_then_apply(WalRecord::DeleteView { name: name.into() });
6877        // After deletion, disable accumulation if no listeners remain.
6878        if !self.needs_emit_deltas() {
6879            self.engine.set_emit_deltas(false);
6880        }
6881        result
6882    }
6883
6884    /// Snapshot of all registered view definitions.
6885    pub fn views(&self) -> Vec<ViewDef> {
6886        self.view_store.views().cloned().collect()
6887    }
6888
6889    // -----------------------------------------------------------------------
6890    // Full-text-lite API
6891    // -----------------------------------------------------------------------
6892
6893    /// Enable full-text indexing for all nodes of `label` on property `field`.
6894    ///
6895    /// After this call, every subsequent write to `(label, field)` is reflected
6896    /// in the index incrementally.  Existing nodes are backfilled immediately.
6897    /// The declaration is persisted as a WAL record; the index itself is rebuilt
6898    /// from scratch on re-open (no snapshot format changes).
6899    ///
6900    /// # Errors
6901    /// - [`GraphError::ReadOnly`]: called on an as-of instance.
6902    /// - [`GraphError::RuleInvalid`]: `(label, field)` is already indexed.
6903    pub fn enable_fulltext(&mut self, label: &str, field: &str) -> Result<()> {
6904        if self.read_only {
6905            return Err(GraphError::ReadOnly);
6906        }
6907        if self.fulltext.is_enabled(label, field) {
6908            return Err(GraphError::RuleInvalid {
6909                detail: format!("full-text index for ({label:?}, {field:?}) already enabled"),
6910            });
6911        }
6912        self.log_then_apply(WalRecord::EnableFulltext {
6913            label: label.into(),
6914            field: field.into(),
6915        })
6916    }
6917
6918    /// Disable full-text indexing for `(label, field)` and drop its postings.
6919    ///
6920    /// # Errors
6921    /// - [`GraphError::ReadOnly`]: called on an as-of instance.
6922    /// - [`GraphError::RuleNotFound`]: `(label, field)` is not currently indexed.
6923    pub fn disable_fulltext(&mut self, label: &str, field: &str) -> Result<()> {
6924        if self.read_only {
6925            return Err(GraphError::ReadOnly);
6926        }
6927        if !self.fulltext.is_enabled(label, field) {
6928            return Err(GraphError::RuleNotFound {
6929                name: format!("fulltext({label},{field})"),
6930            });
6931        }
6932        self.log_then_apply(WalRecord::DisableFulltext {
6933            label: label.into(),
6934            field: field.into(),
6935        })
6936    }
6937
6938    /// Whether `(label, field)` is currently indexed for full-text search.
6939    pub fn is_fulltext_enabled(&self, label: &str, field: &str) -> bool {
6940        self.fulltext.is_enabled(label, field)
6941    }
6942
6943    /// Every `(label, field)` pair with a live full-text index, sorted.
6944    ///
6945    /// Note that [`GraphDb::search`] is keyed by field alone — a pair only
6946    /// declares which nodes are *indexed*, so callers that want to search
6947    /// everything indexed should query each distinct field once.
6948    pub fn fulltext_pairs(&self) -> Vec<(String, String)> {
6949        let mut v: Vec<(String, String)> = self.fulltext.enabled_pairs().cloned().collect();
6950        v.sort();
6951        v
6952    }
6953
6954    /// Enable an equality index for all nodes of `label` on scalar property
6955    /// `field`. Subsequent `WHERE n.field = value` lookups become O(matches)
6956    /// instead of an O(N_label) scan. Existing nodes are backfilled; the
6957    /// declaration persists via WAL and the postings rebuild on re-open.
6958    ///
6959    /// # Errors
6960    /// - [`GraphError::ReadOnly`]: called on an as-of instance.
6961    /// - [`GraphError::RuleInvalid`]: `(label, field)` is already indexed.
6962    pub fn enable_index(&mut self, label: &str, field: &str) -> Result<()> {
6963        if self.read_only {
6964            return Err(GraphError::ReadOnly);
6965        }
6966        if self.prop_index.is_enabled(label, field) {
6967            return Err(GraphError::RuleInvalid {
6968                detail: format!("property index for ({label:?}, {field:?}) already enabled"),
6969            });
6970        }
6971        self.log_then_apply(WalRecord::EnableIndex {
6972            label: label.into(),
6973            field: field.into(),
6974        })
6975    }
6976
6977    /// Disable the equality index for `(label, field)` and drop its postings.
6978    ///
6979    /// # Errors
6980    /// - [`GraphError::ReadOnly`]: called on an as-of instance.
6981    /// - [`GraphError::RuleNotFound`]: `(label, field)` is not currently indexed.
6982    pub fn disable_index(&mut self, label: &str, field: &str) -> Result<()> {
6983        if self.read_only {
6984            return Err(GraphError::ReadOnly);
6985        }
6986        if !self.prop_index.is_enabled(label, field) {
6987            return Err(GraphError::RuleNotFound {
6988                name: format!("index({label},{field})"),
6989            });
6990        }
6991        self.log_then_apply(WalRecord::DisableIndex {
6992            label: label.into(),
6993            field: field.into(),
6994        })
6995    }
6996
6997    /// Whether `(label, field)` currently has an equality index.
6998    pub fn is_index_enabled(&self, label: &str, field: &str) -> bool {
6999        self.prop_index.is_enabled(label, field)
7000    }
7001
7002    /// Start recording insert-count multiplicity on this store (§5.13).
7003    ///
7004    /// Adjacency stays a set and nothing about an existing read changes: a
7005    /// duplicate [`insert_edge`](Self::insert_edge) still returns `Ok(false)`
7006    /// and still leaves [`degree`](Self::degree) alone. What it gains is that
7007    /// the duplicate is *counted*, as the reserved edge property
7008    /// [`EDGE_COUNT_PROP`], readable through
7009    /// [`degree_multiplicity`](Self::degree_multiplicity).
7010    ///
7011    /// # This is a one-way step, and that is why it is a call
7012    ///
7013    /// The count is durable, so it is written to the WAL — as discriminant 23,
7014    /// which no release before v0.6.10 knows. A reader meeting an unknown WAL
7015    /// discriminant cannot know what the record would have changed, so it
7016    /// cannot degrade the way an unreadable index blob can. **After this call
7017    /// the store can no longer be read by an older binary, and there is no call
7018    /// that undoes it.** Gating the record behind this method is what keeps
7019    /// that step a decision an operator makes when they want the feature,
7020    /// rather than one everybody takes by upgrading.
7021    ///
7022    /// # It fails loudly, and that costs a snapshot
7023    ///
7024    /// An older binary does not refuse discriminant 23 — it truncates the WAL
7025    /// at it and, with `repair_wal`, persists the truncation. So this call also
7026    /// writes a **V10 snapshot**, a version no earlier release knows, and it
7027    /// writes it *first*: the snapshot is read before the WAL, so an older
7028    /// binary stops at `snapshot: unsupported version 10` with the WAL
7029    /// untouched. Taking the snapshot before appending the record is what makes
7030    /// the guard unconditional — the store is never, at any interruption point,
7031    /// carrying the record without the stamp that announces it.
7032    ///
7033    /// The snapshot keeps the WAL (`keep_wal: true`): opting in is not a
7034    /// compaction, and history reachable by [`open_at`](Self::open_at) stays
7035    /// reachable. On a large store the call therefore costs one full snapshot
7036    /// write.
7037    ///
7038    /// # What it costs a store that archives
7039    ///
7040    /// Writing `snapshot.bin` is also how the archive path decides whether the
7041    /// store may have a *genesis chain* — whether `open_at` can replay
7042    /// archive-resident commits from empty state. The rule is conservative: a
7043    /// snapshot that was already on disk might have been a truncating one, and
7044    /// once the handle that took it is gone this binary cannot tell. A
7045    /// `keep_wal` snapshot taken by **this** handle is the case where it can, so
7046    /// opting in and then archiving **in the same session** keeps the chain.
7047    ///
7048    /// Opting in, closing the store, and archiving in a *later* session does
7049    /// not — but that is the answer any store with a prior snapshot gets, not
7050    /// something this call causes. A store that wants the chain should take its
7051    /// first archive in the session that opted in.
7052    ///
7053    /// Calling it on a store that has already opted in writes nothing and
7054    /// returns `Ok(())`: an operator should not have to ask first.
7055    ///
7056    /// # This call is not atomic, and an `Err` does not undo it
7057    ///
7058    /// There is no rollback here, and there never was one. An `Err` means this
7059    /// handle stopped believing the store is opted in — `self.multiplicity` is
7060    /// reset, so this handle reports `false` from then on — and nothing more. It
7061    /// says nothing about what reached disk. Two reachable failures leave the
7062    /// opt-in standing:
7063    ///
7064    /// * **The declaration landed and only its fsync failed.** `log_then_apply`
7065    ///   appends, then syncs; a failed barrier leaves `MULTIPLICITY_ENABLED`
7066    ///   already in `wal.bin`. The next open replays it and the store is opted
7067    ///   in. No archive is involved — this one predates the recovery below.
7068    /// * **The declaration never landed, but the V10 snapshot did, on a store
7069    ///   that already had an archive.** The open-time recovery in
7070    ///   `load_from_disk` reads V10-beside-an-archive as an interrupted archive
7071    ///   sequence and opts the store in.
7072    ///
7073    /// So a failed call may leave the opt-in on disk immediately (the first
7074    /// case) or conjure it at the next open (the second), and nothing puts the
7075    /// store back out. Treat `Err` as "the outcome is unknown", not as "nothing
7076    /// happened".
7077    ///
7078    /// **This is safe, and the ordering is the reason.** The V10 stamp is
7079    /// written *before* the declaration, so every one of these intermediate
7080    /// states is one an older binary refuses by name rather than truncates at.
7081    /// The failure direction costs a refusal, never a commit. That ordering is
7082    /// the property worth protecting, not the atomicity this call never had.
7083    ///
7084    /// **To know where the store stands, ask the store.** Reopen it and call
7085    /// [`is_multiplicity_enabled`](Self::is_multiplicity_enabled); that is the
7086    /// only answer that accounts for what reached disk.
7087    ///
7088    /// The one case that really does leave the store opted out is a failure with
7089    /// no archive present and no record written: a stray V10 snapshot remains,
7090    /// costing an older reader a refusal it did not strictly need, and *that*
7091    /// store's next snapshot rewrites at V9.
7092    ///
7093    /// # Errors
7094    /// - [`GraphError::ReadOnly`]: called on an as-of instance.
7095    /// - Anything [`snapshot_with`](Self::snapshot_with) can return, and
7096    ///   anything the WAL append or its fsync can return. See the atomicity
7097    ///   section above for what the store is left holding.
7098    pub fn enable_multiplicity(&mut self) -> Result<()> {
7099        if self.read_only {
7100            return Err(GraphError::ReadOnly);
7101        }
7102        if self.multiplicity {
7103            return Ok(());
7104        }
7105        // The snapshot goes first, and the order is the guard.
7106        //
7107        // An older reader refuses a V10 snapshot by name and stops; it does not
7108        // refuse discriminant 23, it truncates the WAL at it. So the store must
7109        // never hold the record without the snapshot that announces it — not
7110        // even for the width of one fsync. Writing the snapshot before the
7111        // record makes the only reachable intermediate state "V10 snapshot, no
7112        // record", which is merely conservative: this binary reads it as a
7113        // store that has not opted in, and an older one refuses it.
7114        //
7115        // `keep_wal: true` because opting in is not a compaction: an operator
7116        // asking for multiplicity has not asked to lose the history `open_at`
7117        // can reach.
7118        self.multiplicity = true;
7119        let forced = self
7120            .snapshot_with(SnapshotOptions {
7121                keep_wal: true,
7122                ..SnapshotOptions::default()
7123            })
7124            .and_then(|()| self.log_then_apply(core_storage::wal::MULTIPLICITY_ENABLED));
7125        if forced.is_err() {
7126            // This handle stops believing it is opted in. That is all this line
7127            // does — it is not a rollback, and cannot be one: the declaration
7128            // may already be in `wal.bin` (the append succeeded and only the
7129            // fsync failed), and even when it is not, the V10 snapshot beside an
7130            // existing archive is enough for the open-time recovery to opt the
7131            // store in. See the "not atomic" section on this method.
7132            //
7133            // It fails in the safe direction either way: the V10 stamp reached
7134            // disk before anything a v0.6.9 reader would truncate at, so the
7135            // worst an interruption costs that reader is a refusal by name.
7136            self.multiplicity = false;
7137        }
7138        forced
7139    }
7140
7141    /// Whether this store records insert-count multiplicity.
7142    ///
7143    /// `false` on every store that has not called
7144    /// [`enable_multiplicity`](Self::enable_multiplicity) — which is every
7145    /// store that did not ask for it, including one upgraded from an earlier
7146    /// release.
7147    pub fn is_multiplicity_enabled(&self) -> bool {
7148        self.multiplicity
7149    }
7150
7151    /// How many times `(etype, src, dst)` has been inserted: the reserved
7152    /// `count` edge property, or 1 when it is absent.
7153    ///
7154    /// Answers 1 for a pair on a store that never opted in, which is the truth
7155    /// available there — the pair was inserted at least once, and the store
7156    /// kept no record of any second insert.
7157    fn edge_insert_count(&self, etype: u32, src: u32, dst: u32) -> u64 {
7158        match self.edge_props_view().get(etype, src, dst, EDGE_COUNT_PROP) {
7159            Some(Value::Int(n)) if n > 0 => n as u64,
7160            _ => 1,
7161        }
7162    }
7163
7164    /// The `SetEdgeCount` record a duplicate insert of `(edge_type, src_key,
7165    /// dst_key)` should log, or `None` when nothing should be written.
7166    ///
7167    /// `None` when the store has not opted in, so **no discriminant-23 record
7168    /// is written at all** — the gate the whole feature rests on.
7169    ///
7170    /// The other two `None`s are unreachable from the one caller. This is the
7171    /// single-mutation path, where `prepare_insert_edge` has already refused a
7172    /// missing endpoint and an existing pair's edge type is necessarily
7173    /// interned. A batch is the case where a pair's endpoints and type can all
7174    /// be created by the same frame, and it does not come through here: it
7175    /// queues a [`PlannedRec::DuplicateCount`] and names the count in the dense
7176    /// rewrite, which is the only pass that knows the frame's own ids.
7177    fn edge_count_record(
7178        &self,
7179        edge_type: &str,
7180        src_key: &str,
7181        dst_key: &str,
7182    ) -> Option<WalRecord> {
7183        if !self.multiplicity {
7184            return None;
7185        }
7186        let etype = self.syms.get(edge_type)?;
7187        let src = self.ids.get(src_key)?;
7188        let dst = self.ids.get(dst_key)?;
7189        Some(WalRecord::SetEdgeCount {
7190            etype,
7191            src,
7192            dst,
7193            count: self.edge_insert_count(etype, src, dst).saturating_add(1),
7194        })
7195    }
7196
7197    /// Search a full-text-indexed field.
7198    ///
7199    /// Returns `(node_key, match_count)` pairs sorted by match_count descending,
7200    /// ties broken by key (lexicographic).  Tombstoned nodes are excluded.
7201    ///
7202    /// **Query syntax:**
7203    /// - Space-separated terms are AND'd: `"foo bar"` requires both.
7204    /// - `OR` between terms forms disjunction: `"foo OR bar"` matches either.
7205    /// - Trailing `*` on a term is a prefix match: `"rust*"` matches `rustlang`, `rusty`.
7206    /// - `AND` keyword is accepted explicitly and is the default.
7207    /// - Tokenization is unicode-alphanumeric (same as index time); case-insensitive.
7208    ///
7209    /// **Unindexed field:** returns `Ok(vec![])` if `field` is not indexed.
7210    /// Pin: this is the documented, tested, stable behavior for v1.
7211    ///
7212    /// **Memory / performance:** O(postings) lookup; no scan.  The index is
7213    /// in-memory and proportional to total indexed text across all enabled fields.
7214    ///
7215    /// **v2 grammar:** supports `"phrase"`, `-negation`, `prefix*`, `OR`, `AND`.
7216    /// Results are BM25-scored (k1=1.2, b=0.75) and sorted by score descending,
7217    /// key ascending for deterministic tiebreaking.
7218    pub fn search(&self, field: &str, query: &str) -> Vec<(String, f64)> {
7219        // Resolve node_ids to keys (excluding tombstones) then re-sort by
7220        // (score DESC, key ASC) to give a deterministic, key-lexicographic
7221        // tiebreak.  FulltextIndex::search sorts by (score DESC, node_id ASC)
7222        // which diverges from key order when nodes were not inserted in key-lex order.
7223        let mut results: Vec<(String, f64)> = self
7224            .fulltext
7225            .search(field, query, 0)
7226            .into_iter()
7227            .filter_map(|(id, score)| self.ids.key_of(id).map(|key| (key.to_string(), score)))
7228            .collect();
7229        results.sort_by(|a, b| {
7230            b.1.partial_cmp(&a.1)
7231                .unwrap_or(std::cmp::Ordering::Equal)
7232                .then(a.0.cmp(&b.0))
7233        });
7234        results
7235    }
7236
7237    /// [`search`](Self::search), stopping at the `k` best hits.
7238    ///
7239    /// Same ranking and the same deterministic tiebreak, but the index drops
7240    /// everything past `k` before any key is resolved, so a caller that wants
7241    /// the top few out of a field that matched thousands does not pay to
7242    /// materialise and re-sort the tail. `k == 0` means no limit, exactly as
7243    /// [`search`](Self::search) behaves.
7244    ///
7245    /// The BM25 scoring itself is not bounded by `k` — every candidate is
7246    /// scored either way — so this trims the resolve and the sort, not the
7247    /// search.
7248    pub fn search_top(&self, field: &str, query: &str, k: usize) -> Vec<(String, f64)> {
7249        // A tombstoned id resolves to nothing, so asking the index for exactly
7250        // `k` could return fewer. Over-fetching a little and truncating after
7251        // the filter keeps the count right without unbounding the call.
7252        let want = if k == 0 { 0 } else { k.saturating_mul(2) };
7253        let mut results: Vec<(String, f64)> = self
7254            .fulltext
7255            .search(field, query, want)
7256            .into_iter()
7257            .filter_map(|(id, score)| self.ids.key_of(id).map(|key| (key.to_string(), score)))
7258            .collect();
7259        results.sort_by(|a, b| {
7260            b.1.partial_cmp(&a.1)
7261                .unwrap_or(std::cmp::Ordering::Equal)
7262                .then(a.0.cmp(&b.0))
7263        });
7264        if k > 0 {
7265            results.truncate(k);
7266        }
7267        results
7268    }
7269
7270    /// Hybrid search: Reciprocal Rank Fusion (RRF) over fulltext + vector results.
7271    ///
7272    /// Takes up to `4*k` fulltext hits for `(text_field, query_text)` and up to
7273    /// `4*k` vector hits for `(vector_field, query_vec, min=0.0)`, then fuses
7274    /// them with RRF using a fixed constant of 60.
7275    ///
7276    /// ```text
7277    /// score(d) = Σ  1 / (60 + rank_i(d))    (rank 1-based per list)
7278    /// ```
7279    ///
7280    /// Returns the top `k` nodes by fused score, ties broken by node key
7281    /// ascending (deterministic).
7282    ///
7283    /// # Vector leg fallback
7284    ///
7285    /// When `query_vec` is empty the vector leg is skipped entirely and
7286    /// results are ranked by the text list alone through the same RRF path
7287    /// (each text result scores `1/(60 + rank)` from that single list).
7288    ///
7289    /// When `label` is `None`, the vector leg **always** returns empty results.
7290    /// Internally `label` is mapped to `""`, which does not match any rule-created
7291    /// HNSW index (all such indexes are keyed to a specific non-empty label), and
7292    /// the brute-force fallback finds no nodes with an empty label.  The fused
7293    /// ranking is therefore text-only in this case.
7294    pub fn search_hybrid(
7295        &self,
7296        text_field: &str,
7297        query_text: &str,
7298        vector_field: &str,
7299        query_vec: &[f64],
7300        label: Option<&str>,
7301        k: usize,
7302    ) -> Vec<(String, f64)> {
7303        self.search_hybrid_inner(
7304            text_field,
7305            query_text,
7306            vector_field,
7307            query_vec,
7308            label,
7309            k,
7310            None,
7311        )
7312    }
7313
7314    /// [`search_hybrid`](Self::search_hybrid) with **each leg** filtered to the
7315    /// mask before the fusion.
7316    ///
7317    /// Filtering the fused list afterwards would quietly return fewer than `k`.
7318    /// Each leg over-fetches `4*k` candidates, so when the visible nodes rank
7319    /// below `4*k` hidden ones neither leg carries them into the fusion at all
7320    /// and the post-filter has nothing left to keep. Filtering first spends the
7321    /// `4*k` on **visible** hits, so a scoped call is as long as the corpus it
7322    /// can see allows.
7323    ///
7324    /// The ranks that enter RRF are therefore the ranks of the visible corpus,
7325    /// not the visible entries of the store-wide ranking. The constant stays 60
7326    /// and the tiebreak stays key-ascending.
7327    #[allow(clippy::too_many_arguments)]
7328    pub fn search_hybrid_scoped(
7329        &self,
7330        text_field: &str,
7331        query_text: &str,
7332        vector_field: &str,
7333        query_vec: &[f64],
7334        label: Option<&str>,
7335        k: usize,
7336        mask: &crate::mask::NodeMask,
7337    ) -> Vec<(String, f64)> {
7338        self.search_hybrid_inner(
7339            text_field,
7340            query_text,
7341            vector_field,
7342            query_vec,
7343            label,
7344            k,
7345            Some(mask),
7346        )
7347    }
7348
7349    /// The body shared by [`search_hybrid`](Self::search_hybrid) and
7350    /// [`search_hybrid_scoped`](Self::search_hybrid_scoped). `mask = None` is
7351    /// the unscoped contract unchanged: the filter below is then a no-op and
7352    /// the vector leg is the same unmasked call it has always been.
7353    #[allow(clippy::too_many_arguments)]
7354    fn search_hybrid_inner(
7355        &self,
7356        text_field: &str,
7357        query_text: &str,
7358        vector_field: &str,
7359        query_vec: &[f64],
7360        label: Option<&str>,
7361        k: usize,
7362        mask: Option<&crate::mask::NodeMask>,
7363    ) -> Vec<(String, f64)> {
7364        use std::collections::HashMap;
7365
7366        const RRF_K: f64 = 60.0;
7367        let pool = 4 * k;
7368
7369        // Accumulate per-node RRF scores.
7370        let mut scores: HashMap<String, f64> = HashMap::new();
7371
7372        // Text leg. The mask bites on the candidates, before `take(pool)`, so
7373        // the over-fetch is a budget of visible hits rather than one a hidden
7374        // prefix can exhaust.
7375        let text_hits = self.search(text_field, query_text);
7376        let visible_text = text_hits
7377            .into_iter()
7378            .filter(|(key, _count)| mask.is_none_or(|m| m.contains_node(self, key)));
7379        for (rank0, (key, _count)) in visible_text.take(pool).enumerate() {
7380            let rank = (rank0 + 1) as f64;
7381            *scores.entry(key).or_insert(0.0) += 1.0 / (RRF_K + rank);
7382        }
7383
7384        // Vector leg (skipped when query_vec is empty). The masked variant
7385        // applies the mask before its own k-truncation, for the same reason.
7386        if !query_vec.is_empty() {
7387            // `ExactnessCaller::Hybrid`: the leg is the same one
7388            // `find_similar_vector_masked` runs, but the advice its warning
7389            // gives has to fit *this* signature, which has no `exact`.
7390            let vec_hits = self
7391                .find_similar_vector_as(
7392                    vector_field,
7393                    label,
7394                    query_vec,
7395                    pool,
7396                    0.0,
7397                    mask,
7398                    None,
7399                    false,
7400                    ExactnessCaller::Hybrid,
7401                )
7402                .expect("find_similar_vector_as is infallible without where_");
7403            for (rank0, (key, _sim)) in vec_hits.into_iter().enumerate() {
7404                let rank = (rank0 + 1) as f64;
7405                *scores.entry(key).or_insert(0.0) += 1.0 / (RRF_K + rank);
7406            }
7407        }
7408
7409        // Sort: score DESC, then key ASC for deterministic tie-breaking.
7410        let mut ranked: Vec<(String, f64)> = scores.into_iter().collect();
7411        ranked.sort_by(|a, b| {
7412            b.1.partial_cmp(&a.1)
7413                .unwrap_or(std::cmp::Ordering::Equal)
7414                .then(a.0.cmp(&b.0))
7415        });
7416        ranked.truncate(k);
7417        ranked
7418    }
7419
7420    /// For DST/testing: scratch BM25 search over live nodes without the index.
7421    /// Walks every live node, re-stems field tokens, computes corpus stats, and
7422    /// returns BM25-ranked results.
7423    ///
7424    /// The oracle: the ordered key list of `search(field, q)` must equal that of
7425    /// `scratch_search(field, q)` at every quiescent state.
7426    #[doc(hidden)]
7427    pub fn scratch_search(&self, field: &str, query: &str) -> Vec<(String, f64)> {
7428        use core_storage::fulltext::{parse_query, value_tokens_stemmed_with_positions};
7429        use std::collections::BTreeMap;
7430
7431        let groups = parse_query(query);
7432        if groups.is_empty() {
7433            return vec![];
7434        }
7435
7436        // --- Pass 1: collect all live indexed nodes with stemmed token data ---
7437        struct NodeData {
7438            key: String,
7439            /// stemmed_token → positions (sorted)
7440            tokens: BTreeMap<String, Vec<u32>>,
7441            dl: u32,
7442        }
7443
7444        let mut nodes: Vec<NodeData> = Vec::new();
7445        for id in 0..self.ids.len() as u32 {
7446            let Some(key) = self.ids.key_of(id) else {
7447                continue;
7448            };
7449            let Some(&sym) = self.labels.get(id as usize) else {
7450                continue;
7451            };
7452            if sym == u32::MAX {
7453                continue;
7454            }
7455            let label = match self.syms.resolve(sym) {
7456                Some(l) => l,
7457                None => continue,
7458            };
7459            if !self.fulltext.is_enabled(label, field) {
7460                continue;
7461            }
7462            let Some(value) = self.props_view().get(id, field).map(|vr| vr.into_value()) else {
7463                continue;
7464            };
7465            // Use value_tokens_stemmed_with_positions so list elements are
7466            // separated by POSITION_GAP — identical to the index path, which
7467            // prevents phrase queries from matching across element boundaries.
7468            let stemmed_with_pos = match &value {
7469                Value::Str(_) | Value::List(_) => value_tokens_stemmed_with_positions(&value),
7470                _ => continue,
7471            };
7472            let dl = stemmed_with_pos.len() as u32;
7473            let mut tok_map: BTreeMap<String, Vec<u32>> = BTreeMap::new();
7474            for (tok, pos) in stemmed_with_pos {
7475                tok_map.entry(tok).or_default().push(pos);
7476            }
7477            nodes.push(NodeData {
7478                key: key.to_string(),
7479                tokens: tok_map,
7480                dl,
7481            });
7482        }
7483
7484        if nodes.is_empty() {
7485            return vec![];
7486        }
7487
7488        // --- BM25 corpus stats ---
7489        let n = nodes.len() as f64;
7490        let avg_dl: f64 = nodes.iter().map(|nd| nd.dl as f64).sum::<f64>() / n;
7491        // df per stemmed token across all live indexed nodes.
7492        let mut df_map: BTreeMap<&str, f64> = BTreeMap::new();
7493        for nd in &nodes {
7494            for tok in nd.tokens.keys() {
7495                *df_map.entry(tok.as_str()).or_insert(0.0) += 1.0;
7496            }
7497        }
7498
7499        const K1: f64 = 1.2;
7500        const B: f64 = 0.75;
7501
7502        // --- Pass 2: score each node against each OR-group ---
7503        let mut results: Vec<(String, f64)> = Vec::new();
7504        for nd in &nodes {
7505            let dl = nd.dl as f64;
7506            let mut total_score = 0.0f64;
7507
7508            'group: for group in &groups {
7509                let mut group_score = 0.0f64;
7510
7511                for term in group {
7512                    if term.negated {
7513                        // Negated: if doc has this stemmed token → group fails.
7514                        let present = if term.prefix {
7515                            nd.tokens.keys().any(|t| t.starts_with(term.token.as_str()))
7516                        } else {
7517                            nd.tokens.contains_key(term.token.as_str())
7518                        };
7519                        if present {
7520                            continue 'group;
7521                        }
7522                        continue;
7523                    }
7524                    if term.prefix {
7525                        // Prefix: sum BM25 for all matching stemmed tokens.
7526                        let mut prefix_matched = false;
7527                        for (tok, positions) in &nd.tokens {
7528                            if tok.starts_with(term.token.as_str()) {
7529                                let tf = positions.len() as f64;
7530                                let df = df_map.get(tok.as_str()).copied().unwrap_or(1.0);
7531                                let idf = ((n - df + 0.5) / (df + 0.5) + 1.0).ln();
7532                                let tf_norm =
7533                                    tf * (K1 + 1.0) / (tf + K1 * (1.0 - B + B * dl / avg_dl));
7534                                group_score += idf * tf_norm;
7535                                prefix_matched = true;
7536                            }
7537                        }
7538                        if !prefix_matched {
7539                            continue 'group;
7540                        }
7541                    } else {
7542                        // term.token is already stemmed by parse_query; use directly.
7543                        match nd.tokens.get(term.token.as_str()) {
7544                            None => continue 'group,
7545                            Some(positions) => {
7546                                let tf = positions.len() as f64;
7547                                let df = df_map.get(term.token.as_str()).copied().unwrap_or(1.0);
7548                                let idf = ((n - df + 0.5) / (df + 0.5) + 1.0).ln();
7549                                let tf_norm =
7550                                    tf * (K1 + 1.0) / (tf + K1 * (1.0 - B + B * dl / avg_dl));
7551                                group_score += idf * tf_norm;
7552                            }
7553                        }
7554                    }
7555                }
7556
7557                if group_score > 0.0 {
7558                    total_score += group_score;
7559                }
7560            }
7561
7562            if total_score > 0.0 {
7563                results.push((nd.key.clone(), total_score));
7564            }
7565        }
7566
7567        results.sort_by(|a, b| {
7568            b.1.partial_cmp(&a.1)
7569                .unwrap_or(std::cmp::Ordering::Equal)
7570                .then(a.0.cmp(&b.0))
7571        });
7572        results
7573    }
7574
7575    /// Return the current view-maintained value of `view_prop` for node `key`.
7576    /// Equivalent to `get_prop` but documents that it reads a view-managed column.
7577    pub fn get_view_prop(&self, key: &str, view_prop: &str) -> Option<Value> {
7578        let id = self.ids.get(key)?;
7579        self.props_view()
7580            .get(id, view_prop)
7581            .map(|vr| vr.into_value())
7582    }
7583
7584    /// For testing / DST oracle: scratch recompute of a view value for one node.
7585    ///
7586    /// Returns `None` if the node does not exist, the view does not exist, or
7587    /// the view has no result for the node (e.g. Avg with no qualifying neighbors).
7588    #[doc(hidden)]
7589    pub fn scratch_view_value(&self, key: &str, view_name: &str) -> Option<Value> {
7590        let node = self.ids.get(key)?;
7591        let def = self.view_store.views().find(|v| v.name == view_name)?;
7592        // Use TopologyView so that NeighborAgg sees base + overlay edges
7593        // without materialising a temporary Topology (I1).
7594        let topo_view = self.topo_view();
7595        core_rules::views::compute_view_value(
7596            def,
7597            node,
7598            self.props_view(),
7599            &topo_view,
7600            &self.ids,
7601            &self.syms,
7602            &self.labels,
7603        )
7604    }
7605
7606    // -----------------------------------------------------------------------
7607    // Graph algorithm API
7608    // -----------------------------------------------------------------------
7609
7610    /// Run PageRank over the unified topology (manual + derived edges).
7611    ///
7612    /// Returns a [`PageRankReport`] with scores sorted descending (ties: key
7613    /// ascending).  Set `config.edge_type` to restrict to one edge type.
7614    /// `config.converged` is `true` only when the power iteration converged
7615    /// within `config.max_iters` and within any time budget.
7616    pub fn pagerank(&self, config: &crate::algo::PageRankConfig) -> crate::algo::PageRankReport {
7617        let topo = build_topo_view(&self.topo, &self.base);
7618        let edge_props = self.edge_props_view();
7619        crate::algo::pagerank(
7620            &topo,
7621            &self.ids,
7622            &self.syms,
7623            &self.labels,
7624            &edge_props,
7625            config,
7626        )
7627    }
7628
7629    /// Weakly-connected components over the unified topology (treated as
7630    /// undirected regardless of how edges were inserted).
7631    ///
7632    /// Component IDs are the key of the smallest member in the component
7633    /// (deterministic).  Result sorted by (component_id, key).
7634    pub fn connected_components(&self, config: &crate::algo::WccConfig) -> crate::algo::WccReport {
7635        let topo = build_topo_view(&self.topo, &self.base);
7636        let edge_props = self.edge_props_view();
7637        crate::algo::wcc(
7638            &topo,
7639            &self.ids,
7640            &self.syms,
7641            &self.labels,
7642            &edge_props,
7643            config,
7644        )
7645    }
7646
7647    /// Degree centrality for every live node.
7648    ///
7649    /// `direction`: `AlgoDir::Out` = out-degree, `AlgoDir::In` = in-degree,
7650    /// `AlgoDir::Both` = out + in (total directed degree).
7651    ///
7652    /// For one-shot ranking use this; for a live property updated on every
7653    /// write, create a Degree materialized view instead (see `docs/site/algorithms.md`).
7654    pub fn degree_centrality(
7655        &self,
7656        config: &crate::algo::DegreeConfig,
7657    ) -> crate::algo::DegreeReport {
7658        let topo = build_topo_view(&self.topo, &self.base);
7659        let edge_props = self.edge_props_view();
7660        crate::algo::degree_centrality(
7661            &topo,
7662            &self.ids,
7663            &self.syms,
7664            &self.labels,
7665            &edge_props,
7666            config,
7667        )
7668    }
7669
7670    /// Louvain community detection over the unified topology (undirected).
7671    ///
7672    /// See [`crate::algo::LouvainConfig`] for edge-type/weight/label
7673    /// restriction and [`crate::algo::CommunityReport`] for the shape of the
7674    /// result (communities sorted size-desc, then smallest member key asc).
7675    pub fn communities(&self, config: &crate::algo::LouvainConfig) -> crate::algo::CommunityReport {
7676        let topo = build_topo_view(&self.topo, &self.base);
7677        let edge_props = self.edge_props_view();
7678        crate::algo::louvain(
7679            &topo,
7680            &self.ids,
7681            &self.syms,
7682            &self.labels,
7683            &edge_props,
7684            config,
7685        )
7686    }
7687
7688    /// Write a vector of `(node_key, score)` pairs as `prop_name` on each node,
7689    /// atomically via a single write-batch (one WAL frame, one fsync).
7690    ///
7691    /// # Errors
7692    /// - [`GraphError::ReadOnly`]: called on an as-of instance.
7693    /// - [`GraphError::RuleInvalid`]: `prop_name` is managed by an existing view
7694    ///   (collision check mirrors `create_view`).
7695    /// - [`GraphError::KeyNotFound`]: a key in `scores` does not exist as a live node.
7696    pub fn write_scores(&mut self, prop_name: &str, scores: &[(String, f64)]) -> Result<()> {
7697        if self.read_only {
7698            return Err(GraphError::ReadOnly);
7699        }
7700        // Collision check: refuse if prop_name is view-managed.
7701        if let Some(view_name) = self.view_store.view_for_prop(prop_name) {
7702            return Err(GraphError::RuleInvalid {
7703                detail: format!(
7704                    "prop {:?} is managed by view {:?} and cannot be written as scores",
7705                    prop_name, view_name
7706                ),
7707            });
7708        }
7709        // Refuse if prop_name is a view name itself (confusing namespace collision).
7710        if self.view_store.has_view(prop_name) {
7711            return Err(GraphError::RuleInvalid {
7712                detail: format!(
7713                    "prop_name {:?} collides with an existing view name",
7714                    prop_name
7715                ),
7716            });
7717        }
7718        // Write all scores in a single crash-atomic batch.
7719        self.write_batch(|b| {
7720            for (key, score) in scores {
7721                b.set_prop(key, prop_name, Value::Float(*score));
7722            }
7723        })?;
7724        Ok(())
7725    }
7726
7727    /// Return the value of `field` for the node with key `key`, or `None` if
7728    /// the node or field is absent.  Reads through the overlay-over-base
7729    /// `ColumnsView`, materialising base values on demand (zero heap cost for
7730    /// overlay hits; one clone per base hit).
7731    pub fn get_prop(&self, key: &str, field: &str) -> Option<Value> {
7732        let id = self.ids.get(key)?;
7733        self.props_view().get(id, field).map(|vr| vr.into_value())
7734    }
7735
7736    pub fn has_node(&self, key: &str) -> bool {
7737        self.ids.get(key).is_some()
7738    }
7739
7740    /// Borrow the raw id map. Used by `NodeMask::from_keys` to resolve keys.
7741    pub(crate) fn ids(&self) -> &IdMap {
7742        &self.ids
7743    }
7744
7745    // -----------------------------------------------------------------------
7746    // Namespaces
7747    // -----------------------------------------------------------------------
7748
7749    /// The index `name` already has in `ns_names`, if any.
7750    fn ns_index_of(&self, name: &str) -> Option<u32> {
7751        self.ns_names
7752            .iter()
7753            .position(|n| n == name)
7754            .map(|i| i as u32)
7755    }
7756
7757    /// The index for `name`, appending it to `ns_names` when it is new.
7758    ///
7759    /// The table holds one entry per distinct namespace in the store — a
7760    /// tenant count, not a node count — so the linear scan is cheaper than a
7761    /// map and keeps `namespaces()` allocation-free of a second index.
7762    fn ns_index_for(&mut self, name: &str) -> u32 {
7763        match self.ns_index_of(name) {
7764            Some(i) => i,
7765            None => {
7766                self.ns_names.push(name.to_string());
7767                (self.ns_names.len() - 1) as u32
7768            }
7769        }
7770    }
7771
7772    /// The namespace name at `idx`, or [`NS_DEFAULT`] for an index this handle
7773    /// does not know (unreachable; the default is the narrowing answer).
7774    fn ns_name(&self, idx: u32) -> &str {
7775        self.ns_names
7776            .get(idx as usize)
7777            .map(String::as_str)
7778            .unwrap_or(NS_DEFAULT)
7779    }
7780
7781    /// The namespace index of dense node `id`, defaulting for an id with no
7782    /// entry (a node inserted before this handle rebuilt the array cannot
7783    /// exist: every insert path maintains it).
7784    fn node_ns_idx(&self, id: u32) -> u32 {
7785        self.node_ns
7786            .get(id as usize)
7787            .copied()
7788            .unwrap_or(NS_DEFAULT_IDX)
7789    }
7790
7791    /// File node `id` under namespace `name`, growing `node_ns` as `labels`
7792    /// grows. Called from `apply` for every node insert, live and replayed.
7793    fn set_node_ns(&mut self, id: u32, name: &str) {
7794        let idx = if name == NS_DEFAULT {
7795            NS_DEFAULT_IDX
7796        } else {
7797            self.ns_index_for(name)
7798        };
7799        if self.node_ns.len() <= id as usize {
7800            self.node_ns.resize(id as usize + 1, NS_DEFAULT_IDX);
7801        }
7802        self.node_ns[id as usize] = idx;
7803    }
7804
7805    /// Rebuild `node_ns` from the `ns` column — one pass, at the end of an
7806    /// open or a reload, after the snapshot is restored and the WAL replayed.
7807    ///
7808    /// A store with no `ns` column reads nothing: the column-name check fails
7809    /// and the vector is filled with one constant.
7810    fn rebuild_node_ns(&mut self) {
7811        let total = self.ids.len();
7812        self.ns_names.truncate(1);
7813        self.node_ns.clear();
7814        self.node_ns.resize(total, NS_DEFAULT_IDX);
7815        let has_ns_column = {
7816            let cv = self.props_view();
7817            cv.field_names().iter().any(|f| f == NS_PROP)
7818        };
7819        if !has_ns_column {
7820            return;
7821        }
7822        // Collected first so the props view is released before `ns_index_for`
7823        // takes `&mut self`.
7824        let named: Vec<(u32, String)> = {
7825            let cv = self.props_view();
7826            (0..total as u32)
7827                .filter_map(|id| match cv.get(id, NS_PROP).map(|vr| vr.into_value()) {
7828                    Some(Value::Str(s)) if s != NS_DEFAULT => Some((id, s)),
7829                    _ => None,
7830                })
7831                .collect()
7832        };
7833        for (id, name) in named {
7834            let idx = self.ns_index_for(&name);
7835            self.node_ns[id as usize] = idx;
7836        }
7837    }
7838
7839    /// Every namespace with at least one live node, in name order.
7840    ///
7841    /// `["default"]` on any store that has never named a namespace, including
7842    /// an empty one: a store is always at least its default namespace.
7843    pub fn namespaces(&self) -> Vec<String> {
7844        let mut out: BTreeSet<&str> = BTreeSet::new();
7845        out.insert(NS_DEFAULT);
7846        for (id, &idx) in self.node_ns.iter().enumerate() {
7847            if idx == NS_DEFAULT_IDX || !self.is_live_node(id as u32) {
7848                continue;
7849            }
7850            out.insert(self.ns_name(idx));
7851        }
7852        out.into_iter().map(str::to_string).collect()
7853    }
7854
7855    /// The namespace of `key`, or `None` when the key names no live node.
7856    pub fn namespace_of(&self, key: &str) -> Option<String> {
7857        let id = self.ids.get(key)?;
7858        if !self.is_live_node(id) {
7859            return None;
7860        }
7861        Some(self.ns_name(self.node_ns_idx(id)).to_string())
7862    }
7863
7864    /// Every live node in `namespace`, as a visibility mask.
7865    ///
7866    /// Built off `node_ns` on whichever handle this is, so on a temporal handle
7867    /// it is the namespace's membership at that commit. A name no node uses
7868    /// gives an empty mask — a namespace scope never widens.
7869    pub fn mask_for_namespace(&self, namespace: &str) -> crate::mask::NodeMask {
7870        let Some(idx) = self.ns_index_of(namespace) else {
7871            return crate::mask::NodeMask::from_ids(std::collections::HashSet::new());
7872        };
7873        let visible: std::collections::HashSet<u32> = (0..self.ids.len() as u32)
7874            .filter(|&id| self.node_ns_idx(id) == idx && self.is_live_node(id))
7875            .collect();
7876        crate::mask::NodeMask::from_ids(visible)
7877    }
7878
7879    /// Live-node test used by the namespace accessors: a deleted node keeps its
7880    /// dense id and its `node_ns` slot, and the label sentinel is what marks it
7881    /// gone — the same test `mask_for_role`'s label leg applies implicitly.
7882    fn is_live_node(&self, id: u32) -> bool {
7883        self.labels
7884            .get(id as usize)
7885            .is_some_and(|&sym| sym != u32::MAX)
7886            && self.ids.key_of(id).is_some()
7887    }
7888
7889    /// Per-namespace live node counts for [`Stats`], in name order.
7890    fn namespace_stats(&self) -> Vec<NamespaceStats> {
7891        let mut counts: BTreeMap<&str, usize> = BTreeMap::new();
7892        counts.insert(NS_DEFAULT, 0);
7893        for id in 0..self.ids.len() as u32 {
7894            if !self.is_live_node(id) {
7895                continue;
7896            }
7897            *counts
7898                .entry(self.ns_name(self.node_ns_idx(id)))
7899                .or_insert(0) += 1;
7900        }
7901        counts
7902            .into_iter()
7903            .filter(|&(name, n)| n > 0 || name == NS_DEFAULT)
7904            .map(|(name, nodes_live)| NamespaceStats {
7905                name: name.to_string(),
7906                nodes_live,
7907            })
7908            .collect()
7909    }
7910
7911    /// The namespace a create-class op would put its node in: the `ns` entry of
7912    /// the props it carries, normalised, with absent meaning [`NS_DEFAULT`].
7913    fn created_namespace<'a>(key: &str, props: &'a [(String, Value)]) -> Result<&'a str> {
7914        Ok(namespace_of_value(Self::sole_ns_entry(key, props)?))
7915    }
7916
7917    /// The one `ns` entry in a node's props, or `None` when it carries none.
7918    ///
7919    /// A props list naming `ns` twice is refused. Without that refusal the
7920    /// write path and the authorisation path can read the same list
7921    /// differently — one taking the first entry, the other the last — and
7922    /// `CREATE (n:L {ns: 'mine', ns: 'theirs'})` lands a node in a namespace
7923    /// the role was checked against the other of. One entry is the only shape
7924    /// where "the node's namespace" is a single fact, so it is the only shape
7925    /// accepted, and every reader of it agrees by construction.
7926    fn sole_ns_entry<'a>(key: &str, props: &'a [(String, Value)]) -> Result<Option<&'a Value>> {
7927        let mut found: Option<&'a Value> = None;
7928        for (field, value) in props {
7929            if field != NS_PROP {
7930                continue;
7931            }
7932            if found.is_some() {
7933                return Err(GraphError::RuleInvalid {
7934                    detail: format!(
7935                        "node {key}: {NS_PROP} is given more than once; a node has exactly \
7936                         one namespace"
7937                    ),
7938                });
7939            }
7940            found = Some(value);
7941        }
7942        Ok(found)
7943    }
7944
7945    /// The definition of the role a write authorisation names.
7946    ///
7947    /// `None` when `roles.json` was corrupt at open or the role has since been
7948    /// removed — neither can reach a write, because the authorisation carries a
7949    /// mask `mask_for_role` already resolved for that name.
7950    fn role_def_for(&self, role: &str) -> Option<&RoleDef> {
7951        self.roles.as_ref()?.iter().find(|r| r.name == role)
7952    }
7953
7954    /// Validate the `ns` entry of a node's props and drop an explicit default.
7955    ///
7956    /// Runs on the write path only (see `rewrite_wal_dense`), never on replay:
7957    /// a record that reached the WAL was already accepted here.
7958    fn normalise_insert_ns(
7959        key: &str,
7960        props: Vec<(String, Value)>,
7961    ) -> Result<(Vec<(String, Value)>, String)> {
7962        // One `ns` or none: this is where that is enforced, so every later
7963        // reader of the list — the authorisation gate, the two `apply` arms,
7964        // `node_ns` — is looking at a single entry and cannot disagree about
7965        // which one counts.
7966        Self::sole_ns_entry(key, &props)?;
7967        let mut name = NS_DEFAULT.to_string();
7968        let mut out = Vec::with_capacity(props.len());
7969        for (field, value) in props {
7970            if field != NS_PROP {
7971                out.push((field, value));
7972                continue;
7973            }
7974            let Value::Str(ref s) = value else {
7975                return Err(GraphError::RuleInvalid {
7976                    detail: format!(
7977                        "node {key}: {NS_PROP} must be a string naming a namespace, \
7978                         got {value:?}"
7979                    ),
7980                });
7981            };
7982            if !valid_namespace(s) {
7983                return Err(GraphError::RuleInvalid {
7984                    detail: format!(
7985                        "node {key}: {s:?} is not a valid namespace name — 1 to {NS_MAX_LEN} \
7986                         characters of [A-Za-z0-9_.-]"
7987                    ),
7988                });
7989            }
7990            name = s.clone();
7991            // An explicit default stores nothing, so a single-tenant store
7992            // never grows an `ns` column.
7993            if name != NS_DEFAULT {
7994                out.push((field, value));
7995            }
7996        }
7997        Ok((out, name))
7998    }
7999
8000    // -----------------------------------------------------------------------
8001    // RBAC role resolution
8002    // -----------------------------------------------------------------------
8003
8004    /// Parse `roles.json` bytes from `fs`.
8005    ///
8006    /// Return values:
8007    ///   `Ok(Some(roles))` — file absent (returns `vec![]`) **or** file present
8008    ///                       and valid; in both cases `mask_for_role` uses the
8009    ///                       list normally (an absent file means no roles defined).
8010    ///   `Ok(None)`        — file present but corrupt or unrecognised version
8011    ///                       → poisoned state; `mask_for_role` returns `Err` for
8012    ///                       any role name until the file is fixed and the DB
8013    ///                       re-opened (or `apply_schema` is called to repair it).
8014    ///
8015    /// Note: `None` signals corruption, not absence — the opposite of what an
8016    /// optional "file missing" convention would suggest.  The open path stores
8017    /// this result on `db.roles` directly.
8018    fn load_roles_from_fs(fs: &F) -> Result<Option<Vec<RoleDef>>> {
8019        let bytes = fs.read(FileId::Roles).map_err(GraphError::Io)?;
8020        if bytes.is_empty() {
8021            // Empty bytes means either the file is absent or zero-byte — both
8022            // are treated identically as "no roles defined".  A zero-byte
8023            // roles.json does NOT widen access: an absent file and a zero-byte
8024            // file both resolve to an empty role list (sees nothing by default).
8025            return Ok(Some(vec![]));
8026        }
8027        match serde_json::from_slice::<RolesFile>(&bytes) {
8028            Ok(f) if matches!(f.version, 1..=4) => Ok(Some(f.roles)),
8029            // Corrupt or unrecognised version (>4): poison the roles state.
8030            // Never widen: a version this binary does not know may carry a
8031            // narrowing this binary would not apply.
8032            _ => Ok(None),
8033        }
8034    }
8035
8036    /// Resolve a role to a node-visibility mask against the current graph state.
8037    ///
8038    /// Returns `Err` when:
8039    /// - `roles.json` was present but corrupt at open (poisoned state), or
8040    /// - `role` does not match any defined role name.
8041    ///
8042    /// The mask union is: explicit `keys` (unknown keys silently ignored) plus
8043    /// all live nodes carrying any label in `labels` that also pass the role's
8044    /// [`visible_where`](crate::roles::RoleDef::visible_where) predicate, if it
8045    /// has one.  Label resolution is live — new nodes of an allowed label are
8046    /// visible without re-applying the schema, and a property edited out of the
8047    /// predicate takes its node out of the mask on the next read.  An empty
8048    /// union = empty mask = sees nothing.
8049    ///
8050    /// This is the one resolver every read path calls, live and as-of alike, so
8051    /// the predicate applies everywhere at once.  On an as-of handle the role
8052    /// *definition* is the current one and the graph is the historical one: the
8053    /// predicate is evaluated against the property values at the commit being
8054    /// read.
8055    ///
8056    /// The result is memoised per `(role, commit_seq)`, so a scoped reader
8057    /// between two writes resolves the role once.  See
8058    /// [`RoleMaskCache`](crate::mask::RoleMaskCache) for why that cannot go
8059    /// stale.
8060    pub fn mask_for_role(&self, role: &str) -> Result<crate::mask::NodeMask> {
8061        self.role_masks
8062            .get_or_build(role, self.commit_seq, || self.build_mask_for_role(role))
8063            .map(|m| (*m).clone())
8064    }
8065
8066    /// The mask an [`AsOfScope`] names, resolved against this handle.
8067    ///
8068    /// Shared by [`GraphDb::query_at_scoped`] and
8069    /// [`GraphDb::query_at_scoped_in_namespace`] so one scope resolves one way
8070    /// however the namespace leg is added.
8071    fn mask_at_scope(&self, scope: AsOfScope<'_>) -> Result<crate::mask::NodeMask> {
8072        // One resolver answers "what may this role see" — `mask_for_role` — and
8073        // it runs against this handle, so on a temporal one the answer is the
8074        // as-of one.
8075        Ok(match scope {
8076            AsOfScope::Role(role) => self.mask_for_role(role)?,
8077            AsOfScope::Keys(keys) => {
8078                crate::mask::NodeMask::from_keys(self, keys.iter().map(String::as_str))
8079            }
8080            AsOfScope::RoleAndKeys(role, keys) => {
8081                self.mask_for_role(role)?
8082                    .intersect(&crate::mask::NodeMask::from_keys(
8083                        self,
8084                        keys.iter().map(String::as_str),
8085                    ))
8086            }
8087            AsOfScope::Namespace(namespace) => self.mask_for_namespace(namespace),
8088        })
8089    }
8090
8091    /// Resolve `role` against the current graph, ignoring the memo.
8092    fn build_mask_for_role(&self, role: &str) -> Result<crate::mask::NodeMask> {
8093        let roles = self.roles.as_ref().ok_or_else(roles_poisoned)?;
8094        let def = roles
8095            .iter()
8096            .find(|r| r.name == role)
8097            .ok_or_else(|| GraphError::KeyNotFound {
8098                key: format!("role:{role}"),
8099            })?;
8100
8101        let mut visible = std::collections::HashSet::new();
8102
8103        // Key leg: resolve explicit keys to dense ids (unknown keys ignored).
8104        // An administrative grant, never narrowed by the predicate.
8105        for key in &def.keys {
8106            if let Some(id) = self.ids.get(key) {
8107                visible.insert(id);
8108            }
8109        }
8110
8111        // Label leg: live scan — iterate labels vec for matching symbol, and
8112        // when the role carries a predicate, test the property as well.  The
8113        // property comes from the store's own merged view (overlay over the
8114        // mmap'd base), so an as-of handle reads the values of its own commit.
8115        let props = def.visible_where.as_ref().map(|_| self.props_view());
8116        for label_name in &def.labels {
8117            if let Some(sym) = self.syms.get(label_name) {
8118                for (i, &s) in self.labels.iter().enumerate() {
8119                    if s != sym {
8120                        continue;
8121                    }
8122                    let id = i as u32;
8123                    match (&def.visible_where, &props) {
8124                        (Some(pred), Some(view)) => {
8125                            let value = view.get(id, &pred.field).map(|vr| vr.into_value());
8126                            if pred.holds(value.as_ref()) {
8127                                visible.insert(id);
8128                            }
8129                        }
8130                        _ => {
8131                            visible.insert(id);
8132                        }
8133                    }
8134                }
8135            }
8136        }
8137
8138        // Namespace leg: an intersection over the whole union, the key leg
8139        // included. A namespace is a tenancy boundary, so a key naming a node in
8140        // another tenant's namespace is not an administrative grant — and
8141        // `apply_schema` has already refused that role, so this only has to be
8142        // right about the node that moved into existence afterwards.
8143        if def.namespaces.is_some() {
8144            visible.retain(|&id| def.sees_namespace(self.ns_name(self.node_ns_idx(id))));
8145        }
8146
8147        Ok(crate::mask::NodeMask::from_ids(visible))
8148    }
8149
8150    /// Return the current list of role definitions.
8151    ///
8152    /// Returns an empty list when no roles are defined or when `roles.json`
8153    /// was corrupt at open (check [`mask_for_role`](Self::mask_for_role) for
8154    /// the fail-loud error in that case, or call
8155    /// [`roles_checked`](Self::roles_checked), which is this readout with that
8156    /// error in it).
8157    pub fn roles(&self) -> Vec<RoleDef> {
8158        self.roles.as_deref().unwrap_or(&[]).to_vec()
8159    }
8160
8161    /// The role definitions, or the poison error when `roles.json` was corrupt
8162    /// at open.
8163    ///
8164    /// [`roles`](Self::roles) answers `[]` both for a store that defines no
8165    /// roles and for one whose sidecar did not parse, and a caller validating a
8166    /// role name at boot cannot tell those apart. The wrong reading of the pair
8167    /// is the dangerous one: a store with no roles at all is an unrestricted
8168    /// store, so a poisoned file would read as "nothing is restricted here".
8169    ///
8170    /// This is the same answer, for the same cause, that
8171    /// [`mask_for_role`](Self::mask_for_role) gives on the first read.
8172    pub fn roles_checked(&self) -> Result<Vec<RoleDef>> {
8173        match self.roles.as_deref() {
8174            Some(roles) => Ok(roles.to_vec()),
8175            None => Err(roles_poisoned()),
8176        }
8177    }
8178
8179    // ── Role-scoped write authz ───────────────────────────────────────────────
8180
8181    /// Execute `ops` with optional role-scoped write authorization.
8182    ///
8183    /// - `None` → full authority, identical to [`write_batch`](Self::write_batch)
8184    ///   (zero-cost bypass of all authz checks).
8185    /// - `Some(authz)` → the decision table is evaluated per-op BEFORE any WAL
8186    ///   record is built.  A denial returns an error with no WAL frame written
8187    ///   (all-or-nothing at the authz boundary, then at the MutPreview boundary).
8188    ///
8189    /// See the plan's "authz decision table" section for the full semantics.
8190    pub fn write_batch_authz(
8191        &mut self,
8192        authz: Option<&WriteAuthz>,
8193        ops: Vec<BatchOp>,
8194    ) -> Result<(usize, usize)> {
8195        // Thread authz as a direct parameter — never touches pending_write_authz.
8196        self.commit_logged_batch(ops, None, authz.cloned())
8197            .map(inserted_pair)
8198    }
8199
8200    /// Execute a Cypher write statement with role-scoped write authorization.
8201    ///
8202    /// Resolves scope + mask from `self.roles` inside the call (same write-guard
8203    /// lifetime as execution, satisfying §5 lock discipline).  The resolved
8204    /// `WriteAuthz` is stored as `pending_write_authz` for the duration of the
8205    /// call so that all inner `batch.commit()` calls are authz-checked.
8206    ///
8207    /// MERGE is handled specially: the MERGE scope precondition (§3.3) is
8208    /// checked in `exec_merge` BEFORE `has_node` to close the §6.2
8209    /// timing-oracle item (hidden ≡ absent for unscoped roles).
8210    ///
8211    /// Roles with `write: None` (v1 behavior) → `RoleWriteDenied` with
8212    /// "this endpoint is not permitted".
8213    pub fn query_write_authz(
8214        &mut self,
8215        role: &str,
8216        cypher: &str,
8217        params: &BTreeMap<String, Value>,
8218    ) -> Result<ResultSet> {
8219        // Resolve scope (fails fast if role has no write scope).
8220        // write:None → byte-identical v1 blanket-403 body (plan §v1-sidecar mandate).
8221        let scope =
8222            {
8223                let roles = self.roles.as_deref().ok_or_else(|| GraphError::Corrupt {
8224                    detail: "roles.json was corrupt at open; re-open to restore role access".into(),
8225                })?;
8226                let def = roles.iter().find(|r| r.name == role).ok_or_else(|| {
8227                    GraphError::KeyNotFound {
8228                        key: format!("role:{role}"),
8229                    }
8230                })?;
8231                def.write
8232                    .clone()
8233                    .ok_or_else(|| GraphError::RoleWriteDenied {
8234                        reason: "role-bound token: writes are not permitted".into(),
8235                    })?
8236            };
8237        // Resolve mask inside the call (same guard, §5 coherence).
8238        let mask = self.mask_for_role(role)?;
8239        self.pending_write_authz = Some(WriteAuthz {
8240            role: role.into(),
8241            scope,
8242            mask,
8243        });
8244        // RAII guard: always clears pending_write_authz on scope exit, including
8245        // on panic or early-return, mirroring the RestoreEmitDeltas precedent.
8246        struct ClearPendingAuthzOnDrop(*mut Option<WriteAuthz>);
8247        impl Drop for ClearPendingAuthzOnDrop {
8248            fn drop(&mut self) {
8249                // SAFETY: pointer into the owning GraphDb; guard is dropped
8250                // within this function's frame before it returns.
8251                unsafe { *self.0 = None };
8252            }
8253        }
8254        // SAFETY: raw pointer into self; guard dropped before this fn returns.
8255        let _authz_guard = ClearPendingAuthzOnDrop(&mut self.pending_write_authz as *mut _);
8256        let tokens = lex(cypher).map_err(|e| GraphError::QueryError {
8257            detail: format!("lex: {e}"),
8258        })?;
8259        let stmt = parse_write(&tokens).map_err(|e| GraphError::QueryError {
8260            detail: format!("parse: {e}"),
8261        })?;
8262        self.exec_write_stmt(stmt, params)
8263    }
8264
8265    /// Execute `ops` with optional role-scoped write authorization, suppressing
8266    /// fsync (for use inside the group-commit drain thread, which performs one
8267    /// group fsync after releasing the write lock).
8268    ///
8269    /// Identical to [`write_batch_authz`] except the fsync policy is temporarily
8270    /// forced to `Relaxed` for the duration of the call, matching the drain-thread
8271    /// contract established by [`commit_batch_nosync`].
8272    pub(crate) fn write_batch_authz_nosync(
8273        &mut self,
8274        authz: Option<&WriteAuthz>,
8275        ops: Vec<BatchOp>,
8276    ) -> Result<(usize, usize)> {
8277        let saved = self.fsync;
8278        struct RestoreFsync(*mut FsyncPolicy, FsyncPolicy);
8279        impl Drop for RestoreFsync {
8280            fn drop(&mut self) {
8281                // SAFETY: pointer into the owning GraphDb; guard is dropped
8282                // within the enclosing function's frame before it returns.
8283                unsafe { *self.0 = self.1 };
8284            }
8285        }
8286        // SAFETY: raw pointer into self; guard dropped before this fn returns.
8287        let _g = RestoreFsync(&mut self.fsync as *mut FsyncPolicy, saved);
8288        self.fsync = FsyncPolicy::Relaxed;
8289        self.commit_logged_batch(ops, None, authz.cloned())
8290            .map(inserted_pair)
8291    }
8292
8293    /// Execute a `/ingest` request with role-scoped write authorization.
8294    ///
8295    /// Resolves the role's `WriteScope` and `NodeMask` inside this call (same
8296    /// write-guard lifetime as the mutation, satisfying §5 lock discipline).
8297    /// Sets `pending_write_authz` for the duration of the call so that the
8298    /// `commit_ingest` → `commit_logged_batch` path picks up the authz context
8299    /// and evaluates the decision table per-op before any WAL write.
8300    ///
8301    /// §7.3: roles with empty `create_labels` will see every `InsertNode` op
8302    /// denied by the decision table with the appropriate §4.3 scope reason;
8303    /// no special HTTP-layer check is needed.
8304    ///
8305    /// Roles with `write: None` return `RoleWriteDenied` with
8306    /// "writes are not permitted" (byte-identical to v1 blanket 403).
8307    pub fn ingest_with_edges_authz(
8308        &mut self,
8309        role: &str,
8310        label: &str,
8311        rows: Vec<std::collections::BTreeMap<String, Value>>,
8312        opts: &crate::ingest::IngestOptions,
8313        edges: &[(String, String, String)],
8314    ) -> Result<crate::ingest::IngestReport> {
8315        // Resolve scope (fails fast if role has no write scope).
8316        // write:None → byte-identical v1 blanket-403 body (plan §v1-sidecar mandate).
8317        let scope =
8318            {
8319                let roles = self.roles.as_deref().ok_or_else(|| GraphError::Corrupt {
8320                    detail: "roles.json was corrupt at open; re-open to restore role access".into(),
8321                })?;
8322                let def = roles.iter().find(|r| r.name == role).ok_or_else(|| {
8323                    GraphError::KeyNotFound {
8324                        key: format!("role:{role}"),
8325                    }
8326                })?;
8327                def.write
8328                    .clone()
8329                    .ok_or_else(|| GraphError::RoleWriteDenied {
8330                        reason: "role-bound token: writes are not permitted".into(),
8331                    })?
8332            };
8333        let mask = self.mask_for_role(role)?;
8334        self.pending_write_authz = Some(WriteAuthz {
8335            role: role.into(),
8336            scope,
8337            mask,
8338        });
8339        // RAII guard: always clears pending_write_authz on scope exit, including
8340        // on panic or early-return, mirroring the RestoreEmitDeltas precedent.
8341        struct ClearPendingAuthzOnDrop(*mut Option<WriteAuthz>);
8342        impl Drop for ClearPendingAuthzOnDrop {
8343            fn drop(&mut self) {
8344                // SAFETY: pointer into the owning GraphDb; guard is dropped
8345                // within this function's frame before it returns.
8346                unsafe { *self.0 = None };
8347            }
8348        }
8349        // SAFETY: raw pointer into self; guard dropped before this fn returns.
8350        let _authz_guard = ClearPendingAuthzOnDrop(&mut self.pending_write_authz as *mut _);
8351        self.ingest_with_edges(label, rows, opts, edges)
8352    }
8353
8354    /// Evaluate the write-authz decision table for one `BatchOp`.
8355    ///
8356    /// Called by `commit_logged_batch` for each op when `pending_write_authz`
8357    /// is `Some`, BEFORE MutPreview.  A denial returns an error immediately;
8358    /// the remaining ops are not evaluated and no WAL frame is written.
8359    ///
8360    /// `batch_created` carries the key→label pairs of nodes that earlier ops in
8361    /// THIS batch will create.  Used by `InsertEdgeUpsert` to count same-batch
8362    /// placeholder nodes as visible (spec: "a placeholder endpoint the SAME
8363    /// batch creates counts as visible if its label passed the create-class gate").
8364    fn check_single_op_authz(
8365        &self,
8366        authz: &WriteAuthz,
8367        op: &BatchOp,
8368        batch_created: &BTreeMap<String, String>,
8369    ) -> Result<()> {
8370        // Helper: 3-way node status under the authz mask.
8371        //
8372        // Batch-created nodes (from earlier InsertNode in THIS batch) are treated
8373        // as Visible with their recorded label — their create gate already passed
8374        // and they are not yet in self.ids (not committed).  This fixes the
8375        // MERGE+ON CREATE SET case where InsertNode + SetProp arrive together:
8376        // the SetProp must not see the node as Absent.
8377        let node_status = |key: &str| -> NodeAuthzStatus {
8378            if let Some(label) = batch_created.get(key) {
8379                return NodeAuthzStatus::Visible(label.clone());
8380            }
8381            match self.ids.get(key) {
8382                None => NodeAuthzStatus::Absent,
8383                Some(id) if !authz.mask.contains_id(id) => NodeAuthzStatus::Hidden,
8384                Some(id) => {
8385                    let label = self
8386                        .labels
8387                        .get(id as usize)
8388                        .and_then(|&sym| {
8389                            if sym == u32::MAX {
8390                                None
8391                            } else {
8392                                self.syms.resolve(sym).map(str::to_string)
8393                            }
8394                        })
8395                        .unwrap_or_default();
8396                    NodeAuthzStatus::Visible(label)
8397                }
8398            }
8399        };
8400
8401        // Helper: is an InsertEdgeUpsert endpoint visible?
8402        // A same-batch placeholder counts as visible if its label passed
8403        // the create-class gate (spec "upsert placeholder-counts-as-visible").
8404        let upsert_ep_visible = |ep_key: &str, placeholder_label: &str| -> bool {
8405            // In store and visible?
8406            if let Some(id) = self.ids.get(ep_key) {
8407                return authz.mask.contains_id(id);
8408            }
8409            // Created by an earlier op in this batch?
8410            if let Some(created_label) = batch_created.get(ep_key) {
8411                return authz.scope.create_labels.contains(created_label);
8412            }
8413            // Will be created by THIS InsertEdgeUpsert: placeholder_label
8414            // must pass the create-class gate.
8415            authz
8416                .scope
8417                .create_labels
8418                .contains(&placeholder_label.to_string())
8419        };
8420
8421        match op {
8422            // RenameNode / CreateRule / DeleteRule: defense-in-depth gate.
8423            // These ops are never routed to role-scoped paths by the HTTP layer,
8424            // but we 403 them here to close any future bypass route.
8425            //
8426            // InsertNodeOnConflict joins them: it is reachable only from the
8427            // embedded Python binding, which has no role token, and `Replace`
8428            // is a create and an update at once. Rather than split the decision
8429            // table for an op no role-scoped path constructs, refuse it — a
8430            // role-scoped caller writes through the ops that are already in the
8431            // table.
8432            BatchOp::RenameNode { .. }
8433            | BatchOp::CreateRule(_)
8434            | BatchOp::DeleteRule { .. }
8435            | BatchOp::InsertNodeOnConflict { .. } => {
8436                return Err(GraphError::RoleWriteDenied {
8437                    reason: "role-bound token: this endpoint is not permitted".into(),
8438                });
8439            }
8440
8441            // ── CREATE-class: InsertNode ─────────────────────────────────────
8442            //
8443            // Decision table row 1 (scope-before-lookup): check label in
8444            // create_labels BEFORE any key lookup.  This is the structural
8445            // closure of the §6.2 timing-oracle item — the denial fires even
8446            // when the store is EMPTY (see test_create_scope_denied_empty_store).
8447            BatchOp::InsertNode { label, key, props } => {
8448                if !authz.scope.create_labels.contains(label) {
8449                    return Err(GraphError::RoleWriteDenied {
8450                        reason: format!(
8451                            "role-bound token: label '{}' not in write scope (create_labels)",
8452                            label
8453                        ),
8454                    });
8455                }
8456                // A role bound to namespaces may only create inside them. The
8457                // never-widen rule is about what a write makes visible to *any*
8458                // party, not only to the writer: a node this role could never
8459                // read back is a write into somebody else's tenancy. Also a
8460                // scope check, so it runs before the key lookup — it discloses
8461                // nothing about the store. Covers Cypher `CREATE` and the node
8462                // `MERGE` creates, both of which arrive as this op.
8463                // Resolved before the role lookup so a props list naming `ns`
8464                // twice is refused for every role, scoped or not: it is the same
8465                // malformed write the seam refuses, and leaving it to the seam
8466                // would mean the gate had already read one of the two.
8467                let target = Self::created_namespace(key, props)?;
8468                if let Some(def) = self.role_def_for(&authz.role) {
8469                    if !def.sees_namespace(target) {
8470                        return Err(GraphError::RoleWriteDenied {
8471                            reason: format!(
8472                                "role-bound token: namespace '{target}' not in the role's \
8473                                 namespaces"
8474                            ),
8475                        });
8476                    }
8477                }
8478                // Row 2/3: key lookup.
8479                match self.ids.get(key.as_str()) {
8480                    Some(id) if authz.mask.contains_id(id) => {
8481                        // Visible: DuplicateKey — let MutPreview handle this.
8482                    }
8483                    Some(_) => {
8484                        // Hidden: indistinguishable from absent to the role.
8485                        return Err(GraphError::RoleWriteDenied {
8486                            reason: "role-bound token: target node not visible".into(),
8487                        });
8488                    }
8489                    None => {
8490                        // Absent: proceed (create).
8491                    }
8492                }
8493            }
8494
8495            // ── UPDATE-class: SetProp, RemoveProp ────────────────────────────
8496            BatchOp::SetProp { key, .. } | BatchOp::RemoveProp { key, .. } => {
8497                if batch_created.contains_key(key.as_str()) {
8498                    // Batch-created node: create gate already passed this batch.
8499                    // Updating it in the same batch is always allowed, regardless
8500                    // of update_labels (ruling §3.5: "writer just created it").
8501                } else {
8502                    let label = match node_status(key) {
8503                        NodeAuthzStatus::Visible(lbl) => lbl,
8504                        _ => {
8505                            return Err(GraphError::RoleWriteDenied {
8506                                reason: "role-bound token: target node not visible".into(),
8507                            });
8508                        }
8509                    };
8510                    if !authz.scope.update_labels.contains(&label) {
8511                        return Err(GraphError::RoleWriteDenied {
8512                            reason: format!(
8513                                "role-bound token: label '{}' not in write scope (update_labels)",
8514                                label
8515                            ),
8516                        });
8517                    }
8518                }
8519            }
8520
8521            // ── DELETE-class: DeleteNode ─────────────────────────────────────
8522            BatchOp::DeleteNode { key } => {
8523                let label = match node_status(key) {
8524                    NodeAuthzStatus::Visible(lbl) => lbl,
8525                    _ => {
8526                        return Err(GraphError::RoleWriteDenied {
8527                            reason: "role-bound token: target node not visible".into(),
8528                        });
8529                    }
8530                };
8531                if !authz.scope.delete_labels.contains(&label) {
8532                    return Err(GraphError::RoleWriteDenied {
8533                        reason: format!(
8534                            "role-bound token: label '{}' not in write scope (delete_labels)",
8535                            label
8536                        ),
8537                    });
8538                }
8539            }
8540
8541            // ── DELETE-class: DeleteEdge ─────────────────────────────────────
8542            //
8543            // Derived-edge rejection runs BEFORE the delete_edge_types scope
8544            // check (spec §3.5: "existing derived-edge rejection precedes
8545            // delete_edge_types check").
8546            BatchOp::DeleteEdge {
8547                edge_type,
8548                src_key,
8549                dst_key,
8550            } => {
8551                // Check provenance ownership BEFORE scope (spec §3.5 ordering).
8552                if let (Some(src_id), Some(dst_id), Some(et_sym)) = (
8553                    self.ids.get(src_key.as_str()),
8554                    self.ids.get(dst_key.as_str()),
8555                    self.syms.get(edge_type.as_str()),
8556                ) {
8557                    if self.engine.is_owned(et_sym, src_id, dst_id) {
8558                        return Err(GraphError::RuleOwned {
8559                            detail: format!(
8560                                "edge {edge_type} {src_key}→{dst_key} is rule-owned; \
8561                                 delete or change the owning rule"
8562                            ),
8563                        });
8564                    }
8565                    // Also check would_derive via MutPreview (empty overlay, pre-batch).
8566                    let preview = MutPreview::new(self);
8567                    if preview.would_derive(edge_type, src_key, dst_key) {
8568                        return Err(GraphError::RuleOwned {
8569                            detail: format!(
8570                                "edge {edge_type} {src_key}→{dst_key} is rule-owned; \
8571                                 delete or change the owning rule, or a live rule would \
8572                                 re-derive it"
8573                            ),
8574                        });
8575                    }
8576                }
8577                // Scope check (AFTER derived-edge check, BEFORE endpoint visibility).
8578                if !authz.scope.delete_edge_types.contains(edge_type) {
8579                    return Err(GraphError::RoleWriteDenied {
8580                        reason: format!(
8581                            "role-bound token: edge type '{}' not in write scope (delete_edge_types)",
8582                            edge_type
8583                        ),
8584                    });
8585                }
8586                // Both endpoints must be visible.
8587                for ep_key in [src_key.as_str(), dst_key.as_str()] {
8588                    match self.ids.get(ep_key) {
8589                        None => {
8590                            return Err(GraphError::RoleWriteDenied {
8591                                reason: "role-bound token: edge endpoint not visible".into(),
8592                            });
8593                        }
8594                        Some(id) if !authz.mask.contains_id(id) => {
8595                            return Err(GraphError::RoleWriteDenied {
8596                                reason: "role-bound token: edge endpoint not visible".into(),
8597                            });
8598                        }
8599                        _ => {}
8600                    }
8601                }
8602            }
8603
8604            // ── EDGE-CREATE: InsertEdge ──────────────────────────────────────
8605            //
8606            // Scope check BEFORE endpoint lookup (preserves timing symmetry).
8607            BatchOp::InsertEdge {
8608                edge_type,
8609                src_key,
8610                dst_key,
8611            } => {
8612                if !authz.scope.create_edge_types.contains(edge_type) {
8613                    return Err(GraphError::RoleWriteDenied {
8614                        reason: format!(
8615                            "role-bound token: edge type '{}' not in write scope (create_edge_types)",
8616                            edge_type
8617                        ),
8618                    });
8619                }
8620                // Both endpoints must be visible. A node created by an earlier
8621                // InsertNode in the same batch (tracked in batch_created) counts
8622                // as visible if its label passed the create-class gate.
8623                for ep_key in [src_key.as_str(), dst_key.as_str()] {
8624                    if batch_created.contains_key(ep_key) {
8625                        // Created earlier this batch — already scope-checked.
8626                        continue;
8627                    }
8628                    match self.ids.get(ep_key) {
8629                        None => {
8630                            return Err(GraphError::RoleWriteDenied {
8631                                reason: "role-bound token: edge endpoint not visible".into(),
8632                            });
8633                        }
8634                        Some(id) if !authz.mask.contains_id(id) => {
8635                            return Err(GraphError::RoleWriteDenied {
8636                                reason: "role-bound token: edge endpoint not visible".into(),
8637                            });
8638                        }
8639                        _ => {}
8640                    }
8641                }
8642            }
8643
8644            // ── EDGE-CREATE: InsertEdgeUpsert ────────────────────────────────
8645            //
8646            // Scope check first; then endpoint visibility using same-batch
8647            // placeholder awareness (spec: "a placeholder endpoint the SAME
8648            // batch creates counts as visible if its label passed the
8649            // create-class gate").
8650            BatchOp::InsertEdgeUpsert {
8651                edge_type,
8652                src_key,
8653                dst_key,
8654                placeholder_label,
8655            } => {
8656                if !authz.scope.create_edge_types.contains(edge_type) {
8657                    return Err(GraphError::RoleWriteDenied {
8658                        reason: format!(
8659                            "role-bound token: edge type '{}' not in write scope (create_edge_types)",
8660                            edge_type
8661                        ),
8662                    });
8663                }
8664                // Check placeholder label against create_labels (create-class gate).
8665                // This ensures the auto-created endpoints are scope-allowed.
8666                for ep_key in [src_key.as_str(), dst_key.as_str()] {
8667                    if !upsert_ep_visible(ep_key, placeholder_label) {
8668                        return Err(GraphError::RoleWriteDenied {
8669                            reason: "role-bound token: edge endpoint not visible".into(),
8670                        });
8671                    }
8672                }
8673                // A placeholder is created with no props, so it lands in the
8674                // default namespace. A role that cannot read `default` must not
8675                // create one there, for the same reason it may not create a node
8676                // there outright.
8677                //
8678                // The refusal is byte-identical to the hidden-endpoint one above,
8679                // and deliberately so: this arm fires only for an endpoint that
8680                // does **not** exist, and the one above only for an endpoint that
8681                // does. Two different strings would make the pair an existence
8682                // oracle — ask for an upsert and read off whether the key is
8683                // taken. Hidden ≡ absent is the rule everywhere else in this
8684                // table and it holds here too.
8685                if let Some(def) = self.role_def_for(&authz.role) {
8686                    if !def.sees_namespace(NS_DEFAULT) {
8687                        for ep_key in [src_key.as_str(), dst_key.as_str()] {
8688                            if self.ids.get(ep_key).is_none() && !batch_created.contains_key(ep_key)
8689                            {
8690                                return Err(GraphError::RoleWriteDenied {
8691                                    reason: "role-bound token: edge endpoint not visible".into(),
8692                                });
8693                            }
8694                        }
8695                    }
8696                }
8697            }
8698        }
8699        Ok(())
8700    }
8701
8702    /// Write `roles` to `roles.json` atomically and update the in-memory list.
8703    ///
8704    /// Called by `apply_schema` when roles change. Never called on unchanged
8705    /// re-apply — this preserves byte-identical idempotency.
8706    pub(crate) fn commit_roles(&mut self, roles: Vec<RoleDef>) -> Result<()> {
8707        let file = RolesFile::new_versioned(roles.clone());
8708        let bytes = serde_json::to_vec(&file).map_err(|e| GraphError::Corrupt {
8709            detail: format!("roles serialization: {e}"),
8710        })?;
8711        self.fs
8712            .write_atomic(FileId::Roles, &bytes)
8713            .map_err(GraphError::Io)?;
8714        self.roles = Some(roles);
8715        // Rewriting the sidecar is not a commit, so `commit_seq` does not move
8716        // and a memoised mask would still match its version. Install a fresh
8717        // cache instead of clearing the shared one: a reader snapshot frozen
8718        // against the old definitions keeps the old `Arc` to itself and can
8719        // never publish an answer this handle would read back.
8720        self.role_masks = Arc::new(crate::mask::RoleMaskCache::new());
8721        // Refresh the MVCC frozen overlay so that reader() immediately sees the
8722        // updated role definitions without waiting for the next K-commit fold.
8723        self.fold_now();
8724        Ok(())
8725    }
8726
8727    fn view(&self) -> GraphView<'_> {
8728        GraphView {
8729            ids: &self.ids,
8730            syms: &self.syms,
8731            labels: &self.labels,
8732            props: self.props_view(),
8733            topo: self.topo_view(),
8734            edge_props: self.edge_props_view(),
8735            mask: None,
8736            prop_index: Some(&self.prop_index),
8737        }
8738    }
8739
8740    fn view_masked<'a>(&'a self, mask: &'a crate::mask::NodeMask) -> GraphView<'a> {
8741        GraphView {
8742            ids: &self.ids,
8743            syms: &self.syms,
8744            labels: &self.labels,
8745            props: self.props_view(),
8746            topo: self.topo_view(),
8747            edge_props: self.edge_props_view(),
8748            mask: Some(&mask.visible),
8749            prop_index: Some(&self.prop_index),
8750        }
8751    }
8752
8753    /// Execute a read-only Cypher query with a node visibility mask.
8754    ///
8755    /// Only nodes whose key is in `mask` are accessible: label scans, key
8756    /// lookups, and neighbor expansions all respect the mask. Edges where
8757    /// either endpoint is hidden are silently dropped.
8758    ///
8759    /// Returns `Err` with a "masked queries are read-only" message when
8760    /// `cypher` is a write statement (CREATE / MERGE / MATCH…SET / DELETE).
8761    pub fn query_masked(
8762        &self,
8763        cypher: &str,
8764        params: &std::collections::BTreeMap<String, Value>,
8765        mask: &crate::mask::NodeMask,
8766    ) -> Result<ResultSet> {
8767        // Reject write statements up front.
8768        let tokens = lex(cypher).map_err(|e| GraphError::QueryError {
8769            detail: format!("lex: {e}"),
8770        })?;
8771        if is_write_tokens(&tokens) {
8772            return Err(GraphError::MaskedReadOnly);
8773        }
8774        let union = parse_read(&tokens).map_err(|e| GraphError::QueryError {
8775            detail: format!("parse: {e}"),
8776        })?;
8777        // Each UNION part executes against the same masked view, so the mask
8778        // applies uniformly across the chain.
8779        execute_union(&self.view_masked(mask), &union, &Params(params)).map_err(|e| {
8780            GraphError::QueryError {
8781                detail: format!("execute: {e}"),
8782            }
8783        })
8784    }
8785
8786    pub fn node_ref(&self, key: &str) -> Option<NodeRef<'_, F>> {
8787        let id = self.ids.get(key)?;
8788        Some(NodeRef { db: self, id })
8789    }
8790
8791    /// BFS neighborhood expansion restricted to visible nodes in `mask`.
8792    ///
8793    /// Hidden nodes are never used as traversal intermediaries in either
8794    /// [`MaskMode::Omit`] or [`MaskMode::Stub`] — a visible node reachable
8795    /// only through a hidden node will not appear in results.
8796    ///
8797    /// In [`MaskMode::Stub`] mode, hidden nodes that are direct neighbours of
8798    /// a visited visible node are appended to the result as stub rows
8799    /// (`label` column is `null`, same key+depth columns as visible rows).
8800    /// They are NOT added to the BFS frontier.
8801    ///
8802    /// Returns `None` when `key` does not exist (caller should 404).
8803    ///
8804    /// **SECURITY**: role-token callers always pass an Omit-mode mask, so
8805    /// stub rows are never produced on the role path.
8806    pub fn neighborhood_masked(
8807        &self,
8808        key: &str,
8809        depth: u32,
8810        edge_types: Option<&[&str]>,
8811        dir: Dir,
8812        mask: &crate::mask::NodeMask,
8813    ) -> Option<ResultSet> {
8814        let start_id = self.ids.get(key)?;
8815        let view = self.view_masked(mask);
8816        let resolved: Option<Vec<u32>> = edge_types.map(|names| {
8817            names
8818                .iter()
8819                .filter_map(|name| view.syms.get(name))
8820                .collect()
8821        });
8822        let nb = neighborhood(&view, start_id, depth, resolved.as_deref(), dir);
8823        let mut rs = ResultSet::new(vec!["key".into(), "label".into(), "depth".into()]);
8824        // Collect visible BFS results (start_id at depth 0, BFS nodes after).
8825        let mut visited: Vec<(u32, u32)> = Vec::with_capacity(nb.nodes.len() + 1);
8826        visited.push((start_id, 0));
8827        for (nid, d) in &nb.nodes {
8828            let k = view.key_of(*nid);
8829            let label = view
8830                .label_of(*nid)
8831                .expect("real nodes always have a label; u32::MAX sentinel cannot occur");
8832            rs.push_row(vec![
8833                Some(Value::Str(k.to_string())),
8834                Some(Value::Str(label.to_string())),
8835                Some(Value::Int(*d as i64)),
8836            ]);
8837            visited.push((*nid, *d));
8838        }
8839        // Stub mode: add hidden direct neighbours of each visited node as stubs.
8840        // Hidden nodes are edge-endpoints only — they are not added to the BFS
8841        // frontier, so the BFS never expands through them.
8842        if mask.mode() == crate::mask::MaskMode::Stub {
8843            let raw_view = self.view();
8844            let mut seen: std::collections::HashSet<u32> =
8845                visited.iter().map(|(id, _)| *id).collect();
8846            for (node_id, node_depth) in &visited {
8847                if *node_depth >= depth {
8848                    continue;
8849                }
8850                for e in expand(&raw_view, *node_id, resolved.as_deref(), dir) {
8851                    let nbr = if e.src == *node_id { e.dst } else { e.src };
8852                    if !mask.contains_id(nbr) && seen.insert(nbr) {
8853                        if let Some(k) = self.ids.key_of(nbr) {
8854                            rs.push_row(vec![
8855                                Some(Value::Str(k.to_string())),
8856                                None,
8857                                Some(Value::Int((*node_depth + 1) as i64)),
8858                            ]);
8859                        }
8860                    }
8861                }
8862            }
8863        }
8864        Some(rs)
8865    }
8866
8867    /// [`neighborhood_masked`](Self::neighborhood_masked) with the **subject
8868    /// check** a scoped caller needs: a start key the mask hides answers exactly
8869    /// as an absent one does.
8870    ///
8871    /// `neighborhood_masked` expands from any existing key, hidden or not,
8872    /// because a full-token caller supplying a client mask already knows which
8873    /// keys exist. A scoped caller does not, so telling it apart a hidden key
8874    /// from an absent one would be an existence oracle.
8875    ///
8876    /// Expansion itself is unchanged: hidden nodes are neither returned nor used
8877    /// as traversal intermediaries, so a visible node reachable only through a
8878    /// hidden one stays out of the result.
8879    ///
8880    /// Hidden or unknown `key` → [`GraphError::KeyNotFound`].
8881    pub fn neighborhood_scoped(
8882        &self,
8883        key: &str,
8884        depth: u32,
8885        edge_types: Option<&[&str]>,
8886        dir: Dir,
8887        mask: &crate::mask::NodeMask,
8888    ) -> Result<ResultSet> {
8889        if !mask.contains_node(self, key) {
8890            return Err(GraphError::KeyNotFound { key: key.into() });
8891        }
8892        self.neighborhood_masked(key, depth, edge_types, dir, mask)
8893            .ok_or_else(|| GraphError::KeyNotFound { key: key.into() })
8894    }
8895
8896    /// Live node's key, label, and columnar props. Unknown or tombstoned → `None`.
8897    pub fn node_info(&self, key: &str) -> Option<NodeInfo> {
8898        let n = self.node_ref(key)?;
8899        Some(NodeInfo {
8900            key: n.key().to_string(),
8901            label: n.label().to_string(),
8902            props: n.props(),
8903        })
8904    }
8905
8906    /// Look up a node with mask awareness.
8907    ///
8908    /// | Key state         | Omit mode       | Stub mode              |
8909    /// |-------------------|-----------------|------------------------|
8910    /// | does not exist    | `None` (→ 404)  | `None` (→ 404)         |
8911    /// | exists, visible   | `Some(Visible)` | `Some(Visible)`        |
8912    /// | exists, hidden    | `None` (→ 404)  | `Some(Restricted)`     |
8913    ///
8914    /// **SECURITY**: only call from client-mask (full-token) paths.
8915    /// Role-token paths must use [`node_info`] after an explicit visibility check.
8916    pub fn node_info_masked(
8917        &self,
8918        key: &str,
8919        mask: &crate::mask::NodeMask,
8920    ) -> Option<MaskedNodeResult> {
8921        let id = self.ids.get(key)?;
8922        if mask.contains_id(id) {
8923            Some(MaskedNodeResult::Visible(self.node_info(key)?))
8924        } else {
8925            match mask.mode() {
8926                crate::mask::MaskMode::Stub => Some(MaskedNodeResult::Restricted),
8927                crate::mask::MaskMode::Omit => None,
8928            }
8929        }
8930    }
8931
8932    /// Get edges for `key` with mask-aware hidden-endpoint handling.
8933    ///
8934    /// - Omit mode: edges to hidden endpoints are excluded (same as role-path filtering).
8935    /// - Stub mode: edges to hidden endpoints are included; `src_restricted`/`dst_restricted`
8936    ///   is `true` for each hidden endpoint.
8937    ///
8938    /// Unknown key → [`GraphError::KeyNotFound`].
8939    ///
8940    /// **SECURITY**: only call from client-mask (full-token) paths.
8941    pub fn node_edges_masked(
8942        &self,
8943        key: &str,
8944        mask: &crate::mask::NodeMask,
8945    ) -> Result<Vec<MaskedEdge>> {
8946        self.ensure_v8_base_sections_loaded();
8947        let id = self
8948            .ids
8949            .get(key)
8950            .ok_or_else(|| GraphError::KeyNotFound { key: key.into() })?;
8951        let derived: BTreeSet<(u32, u32, u32)> = self
8952            .engine
8953            .provenance_touching(id)
8954            .map(|(_rule, etype, src, dst)| (etype, src, dst))
8955            .collect();
8956        let mut edges = Vec::new();
8957        let tv = self.topo_view();
8958        for etype in tv.etypes() {
8959            // etype comes from the archived CSR (access_unchecked, no eager CRC).
8960            // A bit-flip in the large TOPOLOGY section can produce an etype id
8961            // that is not in the interner.  Return Corrupt rather than panic.
8962            let edge_type = self
8963                .syms
8964                .resolve(etype)
8965                .ok_or_else(|| GraphError::Corrupt {
8966                    detail: format!("v8: topology etype {etype} not in interner"),
8967                })?
8968                .to_string();
8969            for dir in [Direction::Out, Direction::In] {
8970                for &nbr in tv.neighbors(etype, dir, id).as_ref() {
8971                    let nbr_restricted = !mask.contains_id(nbr);
8972                    if nbr_restricted && mask.mode() == crate::mask::MaskMode::Omit {
8973                        continue;
8974                    }
8975                    let nbr_key = self
8976                        .ids
8977                        .key_of(nbr)
8978                        .ok_or_else(|| GraphError::Corrupt {
8979                            detail: format!("topology id {nbr} has no key"),
8980                        })?
8981                        .to_string();
8982                    let (src_id, dst_id, src_key, dst_key, src_restricted, dst_restricted) =
8983                        match dir {
8984                            Direction::Out => {
8985                                (id, nbr, key.to_string(), nbr_key, false, nbr_restricted)
8986                            }
8987                            Direction::In => {
8988                                (nbr, id, nbr_key, key.to_string(), nbr_restricted, false)
8989                            }
8990                        };
8991                    edges.push(MaskedEdge {
8992                        edge_type: edge_type.clone(),
8993                        src_key,
8994                        src_restricted,
8995                        dst_key,
8996                        dst_restricted,
8997                        derived: derived.contains(&(etype, src_id, dst_id)),
8998                    });
8999                }
9000            }
9001        }
9002        edges.sort_by(|a, b| {
9003            a.edge_type
9004                .cmp(&b.edge_type)
9005                .then(a.src_key.cmp(&b.src_key))
9006                .then(a.dst_key.cmp(&b.dst_key))
9007        });
9008        edges.dedup_by(|a, b| {
9009            a.edge_type == b.edge_type && a.src_key == b.src_key && a.dst_key == b.dst_key
9010        });
9011        Ok(edges)
9012    }
9013
9014    /// [`node_edges_masked`](Self::node_edges_masked) with the **subject check**
9015    /// a scoped caller needs, and a plain [`EdgeInfo`] list.
9016    ///
9017    /// `node_edges_masked` raises [`GraphError::KeyNotFound`] only when `key` is
9018    /// unknown; a key that exists but is hidden still yields its (filtered) edge
9019    /// list, which is correct for a full-token client mask and an existence
9020    /// oracle for a scoped one. Here a hidden subject answers exactly as an
9021    /// absent one does.
9022    ///
9023    /// Every edge naming a hidden endpoint is dropped, whatever the mask's
9024    /// [`MaskMode`](crate::mask::MaskMode): a scoped caller never sees a
9025    /// restricted stub, so there is nothing for it to render.
9026    ///
9027    /// Hidden or unknown `key` → [`GraphError::KeyNotFound`].
9028    pub fn node_edges_scoped(
9029        &self,
9030        key: &str,
9031        mask: &crate::mask::NodeMask,
9032    ) -> Result<Vec<EdgeInfo>> {
9033        if !mask.contains_node(self, key) {
9034            return Err(GraphError::KeyNotFound { key: key.into() });
9035        }
9036        Ok(self
9037            .node_edges_masked(key, mask)?
9038            .into_iter()
9039            .filter(|e| !e.src_restricted && !e.dst_restricted)
9040            .map(|e| EdgeInfo {
9041                edge_type: e.edge_type,
9042                src_key: e.src_key,
9043                dst_key: e.dst_key,
9044                derived: e.derived,
9045            })
9046            .collect())
9047    }
9048
9049    /// Every directed edge incident on `key`, both directions, every etype.
9050    ///
9051    /// Walk is `topology.etypes()` × `{Out, In}` × `neighbors()`. `derived` is
9052    /// membership in [`RuleEngine::provenance_touching`] (O(degree) via the
9053    /// Plan-8 `by_node` index). Sorted by `(edge_type, src_key, dst_key)`.
9054    /// Unknown key → [`GraphError::KeyNotFound`].
9055    pub fn node_edges(&self, key: &str) -> Result<Vec<EdgeInfo>> {
9056        self.ensure_v8_base_sections_loaded();
9057        let id = self
9058            .ids
9059            .get(key)
9060            .ok_or_else(|| GraphError::KeyNotFound { key: key.into() })?;
9061        let derived: BTreeSet<(u32, u32, u32)> = self
9062            .engine
9063            .provenance_touching(id)
9064            .map(|(_rule, etype, src, dst)| (etype, src, dst))
9065            .collect();
9066        let mut edges = Vec::new();
9067        let tv = self.topo_view();
9068        for etype in tv.etypes() {
9069            // Same guard as node_edges_masked: etype from unchecked-CRC CSR.
9070            let edge_type = self
9071                .syms
9072                .resolve(etype)
9073                .ok_or_else(|| GraphError::Corrupt {
9074                    detail: format!("v8: topology etype {etype} not in interner"),
9075                })?
9076                .to_string();
9077            for dir in [Direction::Out, Direction::In] {
9078                for &nbr in tv.neighbors(etype, dir, id).as_ref() {
9079                    let (src, dst, src_key, dst_key) = match dir {
9080                        Direction::Out => (
9081                            id,
9082                            nbr,
9083                            key.to_string(),
9084                            self.ids
9085                                .key_of(nbr)
9086                                .ok_or_else(|| GraphError::Corrupt {
9087                                    detail: format!("topology id {nbr} has no key"),
9088                                })?
9089                                .to_string(),
9090                        ),
9091                        Direction::In => (
9092                            nbr,
9093                            id,
9094                            self.ids
9095                                .key_of(nbr)
9096                                .ok_or_else(|| GraphError::Corrupt {
9097                                    detail: format!("topology id {nbr} has no key"),
9098                                })?
9099                                .to_string(),
9100                            key.to_string(),
9101                        ),
9102                    };
9103                    edges.push(EdgeInfo {
9104                        edge_type: edge_type.clone(),
9105                        src_key,
9106                        dst_key,
9107                        derived: derived.contains(&(etype, src, dst)),
9108                    });
9109                }
9110            }
9111        }
9112        edges.sort_by(|a, b| {
9113            a.edge_type
9114                .cmp(&b.edge_type)
9115                .then(a.src_key.cmp(&b.src_key))
9116                .then(a.dst_key.cmp(&b.dst_key))
9117        });
9118        // Self-loops appear in both Out and In; sort makes the pair adjacent
9119        // (sort key matches PartialEq for this case) so one pass drops the dup.
9120        edges.dedup();
9121        Ok(edges)
9122    }
9123
9124    // ── Backup ────────────────────────────────────────────────────────────────
9125
9126    /// Copy this store to `dest` as a consistent, verified snapshot.
9127    ///
9128    /// Copies every durable file in the database directory — `snapshot.bin`,
9129    /// `wal.bin`, all `wal.<N>.archive` files, `wal.floor`, `wal.genesis`, and
9130    /// `roles.json` — into a freshly created `dest` directory using OS-level
9131    /// `copy` calls (no large in-process buffers).
9132    ///
9133    /// # Consistency guarantee
9134    ///
9135    /// The guarantee is **process-local**: the caller holds `&self`, which
9136    /// prevents any concurrent writer in the **same process** from modifying
9137    /// the files during the copy.  Running `mushroomdb backup` against a
9138    /// directory that is **concurrently being written by another process** (e.g.
9139    /// `mushroomdb serve`) is **unsafe** — the copy can be torn.  The post-copy
9140    /// `verified: true` result reduces but does not eliminate the risk of a
9141    /// silent corrupt backup (CRC catches many bit-flips; it cannot catch a
9142    /// consistent mid-write snapshot).
9143    ///
9144    /// **The safe path for a live-served store is `POST /backup` on the HTTP
9145    /// server.** That handler acquires the read lock on the shared database
9146    /// before calling this method, which is the correct cross-process
9147    /// synchronisation point because the server is the single process writing
9148    /// the files.
9149    ///
9150    /// After copying, opens the destination read-only and runs the CRC section
9151    /// verifier (`verify_snapshot`) to confirm byte-for-byte integrity.
9152    /// `BackupReport::verified` reflects whether both checks passed.
9153    ///
9154    /// Returns `Err` when `self` is not backed by a `RealFs` (e.g. `SimFs`).
9155    pub fn backup_to(&self, dest: &std::path::Path) -> Result<BackupReport> {
9156        // Derive source directory from snapshot_path (RealFs only).
9157        let src_dir = match self.fs.snapshot_path() {
9158            Some(p) => p.parent().map(|d| d.to_path_buf()).ok_or_else(|| {
9159                GraphError::Io(std::io::Error::other("snapshot has no parent dir"))
9160            })?,
9161            None => {
9162                return Err(GraphError::Io(std::io::Error::other(
9163                    "backup_to requires a real filesystem (RealFs)",
9164                )))
9165            }
9166        };
9167
9168        std::fs::create_dir_all(dest)?;
9169
9170        let mut files: Vec<String> = Vec::new();
9171        let mut bytes: u64 = 0;
9172
9173        // Helper: copy src_dir/name → dest/name if the file exists.
9174        let mut try_copy = |name: &str| -> std::io::Result<()> {
9175            let src_path = src_dir.join(name);
9176            if src_path.exists() {
9177                let n = std::fs::copy(&src_path, dest.join(name))?;
9178                bytes += n;
9179                files.push(name.to_string());
9180            }
9181            Ok(())
9182        };
9183
9184        try_copy("snapshot.bin")?;
9185        try_copy("snapshot.bin.bak")?;
9186        try_copy("wal.bin")?;
9187        try_copy("wal.floor")?;
9188        try_copy("wal.genesis")?;
9189        try_copy("roles.json")?;
9190
9191        // Copy WAL archives.
9192        let archives = self.fs.list_archives()?;
9193        for n in &archives {
9194            let name = format!("wal.{n}.archive");
9195            let n_bytes = std::fs::copy(src_dir.join(&name), dest.join(&name))?;
9196            bytes += n_bytes;
9197            files.push(name);
9198        }
9199
9200        files.sort();
9201
9202        // Post-copy verification: open dest and run CRC checks.
9203        let snap_in_dest = dest.join("snapshot.bin").exists();
9204        let crc_ok = if snap_in_dest {
9205            crate::verify_snapshot(dest)
9206                .map(|results| results.iter().all(|(_, _, _, r)| r.is_ok()))
9207                .unwrap_or(false)
9208        } else {
9209            true // WAL-only store: nothing to CRC-check in snapshot
9210        };
9211        let opens_ok = GraphDb::<core_storage::fs::RealFs>::open(dest).is_ok();
9212        let verified = crc_ok && opens_ok;
9213
9214        Ok(BackupReport {
9215            files,
9216            bytes,
9217            verified,
9218        })
9219    }
9220
9221    // ── Export helpers ────────────────────────────────────────────────────────
9222
9223    /// All live nodes, sorted by key (deterministic).
9224    ///
9225    /// Reads base + WAL overlay. Tombstoned nodes are excluded.
9226    pub fn all_nodes_for_export(&self) -> Vec<NodeInfo> {
9227        self.ensure_v8_base_sections_loaded();
9228        let pv = self.props_view();
9229        let mut nodes = Vec::new();
9230        for id in 0..self.ids.len() as u32 {
9231            let Some(key) = self.ids.key_of(id) else {
9232                continue;
9233            };
9234            let Some(&sym) = self.labels.get(id as usize) else {
9235                continue;
9236            };
9237            if sym == u32::MAX {
9238                continue; // tombstoned
9239            }
9240            let Some(label) = self.syms.resolve(sym) else {
9241                continue;
9242            };
9243            let mut props = BTreeMap::new();
9244            for field in pv.field_names() {
9245                if let Some(vr) = pv.get(id, &field) {
9246                    props.insert(field, vr.into_value());
9247                }
9248            }
9249            nodes.push(NodeInfo {
9250                key: key.to_string(),
9251                label: label.to_string(),
9252                props,
9253            });
9254        }
9255        nodes.sort_by(|a, b| a.key.cmp(&b.key));
9256        nodes
9257    }
9258
9259    /// All directed edges, sorted by `(edge_type, src, dst)`. Each edge appears once.
9260    ///
9261    /// Derived edges carry `derived: true` and the creating rule's name in `rule`.
9262    /// Manual edges carry `derived: false` and `rule: None`.
9263    /// `weight` is the creating rule's `weight_prop` value read off the edge
9264    /// (numeric only), mirroring the convention used by [`GraphDb::explain`]
9265    /// and [`GraphDb::weighted_edges`]. Deterministic across runs on the same
9266    /// store state.
9267    pub fn all_edges_for_export(&self) -> Vec<ExportEdge> {
9268        self.ensure_v8_base_sections_loaded();
9269
9270        // Build (etype_sym, src_id, dst_id) → rule_name for O(1) derivation lookup.
9271        let mut prov: HashMap<(u32, u32, u32), String> = HashMap::new();
9272        for (rule_name, triples) in self.engine.provenance() {
9273            for &(etype, src, dst) in triples {
9274                prov.insert((etype, src, dst), rule_name.clone());
9275            }
9276        }
9277
9278        // rule_name → weight_prop, for O(1) lookup per derived edge.
9279        let weight_props: HashMap<&str, Option<&str>> = self
9280            .engine
9281            .rules()
9282            .map(|r| (r.name.as_str(), r.weight_prop.as_deref()))
9283            .collect();
9284
9285        let tv = self.topo_view();
9286        let ep = self.edge_props_view();
9287        let mut edges = Vec::new();
9288
9289        for id in 0..self.ids.len() as u32 {
9290            let Some(key) = self.ids.key_of(id) else {
9291                continue;
9292            };
9293            let Some(&lsym) = self.labels.get(id as usize) else {
9294                continue;
9295            };
9296            if lsym == u32::MAX {
9297                continue; // tombstoned
9298            }
9299
9300            for etype_sym in tv.etypes() {
9301                // etype from archived CSR (access_unchecked, no eager CRC).
9302                // Skip edges whose etype is not in the interner; this can only
9303                // occur with a corrupt large TOPOLOGY section (bit-flip on an
9304                // etype field in the archived data).  The function returns Vec,
9305                // not Result, so we continue rather than propagate.
9306                let Some(edge_type) = self.syms.resolve(etype_sym) else {
9307                    continue;
9308                };
9309                let edge_type = edge_type.to_string();
9310                for &nbr in tv.neighbors(etype_sym, Direction::Out, id).as_ref() {
9311                    let Some(dst_key) = self.ids.key_of(nbr) else {
9312                        continue; // skip corrupt entries
9313                    };
9314                    let prov_key = (etype_sym, id, nbr);
9315                    let rule = prov.get(&prov_key).cloned();
9316                    let derived = rule.is_some();
9317                    let weight = rule
9318                        .as_deref()
9319                        .and_then(|rn| weight_props.get(rn).copied().flatten())
9320                        .and_then(|prop| match ep.get(etype_sym, id, nbr, prop) {
9321                            Some(Value::Float(f)) => Some(f),
9322                            Some(Value::Int(i)) => Some(i as f64),
9323                            _ => None,
9324                        });
9325                    edges.push(ExportEdge {
9326                        edge_type: edge_type.clone(),
9327                        src: key.to_string(),
9328                        dst: dst_key.to_string(),
9329                        derived,
9330                        rule,
9331                        weight,
9332                    });
9333                }
9334            }
9335        }
9336
9337        edges.sort_by(|a, b| {
9338            a.edge_type
9339                .cmp(&b.edge_type)
9340                .then(a.src.cmp(&b.src))
9341                .then(a.dst.cmp(&b.dst))
9342        });
9343        edges
9344    }
9345
9346    /// What each edge type *is*, without building one record per edge.
9347    ///
9348    /// [`all_edges_for_export`](Self::all_edges_for_export) answers the same
9349    /// question by materialising every edge — three `String`s apiece, a
9350    /// provenance `HashMap` over every derived edge, and a final sort. That is
9351    /// the right shape for an export, and the wrong one for a summary: on a
9352    /// store with 1.3 M derived edges it allocates hundreds of megabytes to
9353    /// produce nine lines. This walks the topology instead, summing neighbour
9354    /// slice lengths and collecting *label symbols* rather than label strings,
9355    /// so the per-edge cost is an integer add and a set insert on a set with
9356    /// as many members as the store has labels.
9357    ///
9358    /// The rule names come off the rule *definitions*, which each declare the
9359    /// `edge_type` they derive, so naming them costs one pass over the rules
9360    /// rather than one provenance lookup per edge. That is also why `rules`
9361    /// is a list: two rules may derive the same type — the association store
9362    /// derives `INDUSTRY_ALIGNMENT` from both a talent→company and a
9363    /// talent→job rule — and naming only one of them would be a half-truth.
9364    /// A type with no rules is one written by hand.
9365    ///
9366    /// `sample` is the first edge of the type in the store's own id order,
9367    /// which is insertion order: deterministic for a given store, and not the
9368    /// same as key order, which cannot be had without resolving a key per
9369    /// edge. Sorted by `edge_type`.
9370    pub fn edge_type_census(&self) -> Vec<EdgeTypeCensus> {
9371        self.ensure_v8_base_sections_loaded();
9372
9373        let mut rules_by_type: BTreeMap<&str, BTreeSet<&str>> = BTreeMap::new();
9374        for r in self.engine.rules() {
9375            rules_by_type
9376                .entry(r.edge_type.as_str())
9377                .or_default()
9378                .insert(r.name.as_str());
9379        }
9380
9381        let tv = self.topo_view();
9382        let node_count = self.ids.len() as u32;
9383        let mut out = Vec::new();
9384        for etype_sym in tv.etypes() {
9385            // An etype the interner cannot resolve means a corrupt TOPOLOGY
9386            // section; skip it rather than name it, as `all_edges_for_export`
9387            // does for the same reason.
9388            let Some(edge_type) = self.syms.resolve(etype_sym) else {
9389                continue;
9390            };
9391            let mut edges: u64 = 0;
9392            let mut src_syms: BTreeSet<u32> = BTreeSet::new();
9393            let mut dst_syms: BTreeSet<u32> = BTreeSet::new();
9394            let mut sample: Option<(u32, u32)> = None;
9395            for id in 0..node_count {
9396                let Some(&lsym) = self.labels.get(id as usize) else {
9397                    continue;
9398                };
9399                if lsym == u32::MAX {
9400                    continue; // tombstoned
9401                }
9402                let nbrs = tv.neighbors(etype_sym, Direction::Out, id);
9403                let nbrs = nbrs.as_ref();
9404                if nbrs.is_empty() {
9405                    continue;
9406                }
9407                edges += nbrs.len() as u64;
9408                src_syms.insert(lsym);
9409                for &nbr in nbrs {
9410                    if let Some(&dsym) = self.labels.get(nbr as usize) {
9411                        if dsym != u32::MAX {
9412                            dst_syms.insert(dsym);
9413                        }
9414                    }
9415                }
9416                if sample.is_none() {
9417                    sample = Some((id, nbrs[0]));
9418                }
9419            }
9420            let resolve = |syms: &BTreeSet<u32>| -> Vec<String> {
9421                syms.iter()
9422                    .filter_map(|&s| self.syms.resolve(s))
9423                    .map(ToString::to_string)
9424                    .collect()
9425            };
9426            out.push(EdgeTypeCensus {
9427                edge_type: edge_type.to_string(),
9428                edges,
9429                src_labels: resolve(&src_syms),
9430                dst_labels: resolve(&dst_syms),
9431                rules: rules_by_type
9432                    .get(edge_type)
9433                    .map(|rs| rs.iter().map(ToString::to_string).collect())
9434                    .unwrap_or_default(),
9435                sample: sample.and_then(|(s, d)| {
9436                    Some((
9437                        self.ids.key_of(s)?.to_string(),
9438                        self.ids.key_of(d)?.to_string(),
9439                    ))
9440                }),
9441            });
9442        }
9443        out.sort_by(|a, b| a.edge_type.cmp(&b.edge_type));
9444        out
9445    }
9446
9447    /// All directed edges of `edge_type`, with the raw value of `weight_prop`
9448    /// on each edge when given.
9449    ///
9450    /// `weight` is `Some(f)` only when `weight_prop` is set and the edge
9451    /// carries that property with a numeric (`Int`/`Float`) value; otherwise
9452    /// `None` — callers that want a default weight (e.g. `1.0` for missing
9453    /// props) apply it themselves, matching the convention used internally
9454    /// by [`GraphDb::pagerank`], [`GraphDb::connected_components`],
9455    /// [`GraphDb::degree_centrality`], and [`GraphDb::communities`].
9456    ///
9457    /// Sorted by `(src, dst)` for determinism. Reads the unified topology
9458    /// (manual + rule-derived edges).  An unknown `edge_type` returns an
9459    /// empty vec.
9460    pub fn weighted_edges(
9461        &self,
9462        edge_type: &str,
9463        weight_prop: Option<&str>,
9464    ) -> Vec<(String, String, Option<f64>)> {
9465        let Some(etype_sym) = self.syms.get(edge_type) else {
9466            return Vec::new();
9467        };
9468        let tv = self.topo_view();
9469        let ep = self.edge_props_view();
9470        let mut out = Vec::new();
9471        for id in 0..self.ids.len() as u32 {
9472            let Some(key) = self.ids.key_of(id) else {
9473                continue;
9474            };
9475            let Some(&sym) = self.labels.get(id as usize) else {
9476                continue;
9477            };
9478            if sym == u32::MAX {
9479                continue; // tombstoned
9480            }
9481            for &nbr in tv.neighbors(etype_sym, Direction::Out, id).as_ref() {
9482                let Some(dst_key) = self.ids.key_of(nbr) else {
9483                    continue;
9484                };
9485                let weight = weight_prop.and_then(|prop| match ep.get(etype_sym, id, nbr, prop) {
9486                    Some(Value::Float(f)) => Some(f),
9487                    Some(Value::Int(i)) => Some(i as f64),
9488                    _ => None,
9489                });
9490                out.push((key.to_string(), dst_key.to_string(), weight));
9491            }
9492        }
9493        out.sort_by(|a, b| a.0.cmp(&b.0).then(a.1.cmp(&b.1)));
9494        out
9495    }
9496
9497    pub fn nodes_with_label(&self, label: &str) -> Vec<NodeRef<'_, F>> {
9498        self.view()
9499            .nodes_with_label(label)
9500            .into_iter()
9501            .map(|id| NodeRef { db: self, id })
9502            .collect()
9503    }
9504
9505    pub fn find_nodes(&self, label: &str, filter: &Filter) -> Vec<NodeRef<'_, F>> {
9506        let view = self.view();
9507        view.nodes_with_label(label)
9508            .into_iter()
9509            .filter(|&id| {
9510                eval_filter(filter, &|field| {
9511                    view.prop(id, field).map(|vr| vr.into_value())
9512                })
9513            })
9514            .map(|id| NodeRef { db: self, id })
9515            .collect()
9516    }
9517
9518    /// Returns `true` if any approximate (HNSW) VectorSimilar rule covers
9519    /// `field`.  Use as a capability probe: when `true`, `find_similar_vector`
9520    /// with `label = None` will use the native ANN path rather than the O(n)
9521    /// brute-force scan.
9522    pub fn has_vector_rule(&self, field: &str) -> bool {
9523        self.engine.hnsw_has_rule(field)
9524    }
9525
9526    /// How many HNSW graphs this handle has built from scratch since it was
9527    /// opened (one per side of an approximate rule).
9528    ///
9529    /// An open that restored every graph from the snapshot reports `0`.
9530    /// Exposed for tests that assert the open path reuses the persisted index
9531    /// rather than rebuilding it; not part of the stable surface.
9532    #[doc(hidden)]
9533    pub fn hnsw_build_count(&self) -> u64 {
9534        self.engine.hnsw_build_count()
9535    }
9536
9537    /// How many rules this handle still holds a lazily-decoded HNSW graph for.
9538    ///
9539    /// Zero before the first ANN query on a clean open, and again once the
9540    /// live indexes own the graphs. See [`core_rules::RuleEngine::lazy_hnsw_len`].
9541    /// Exposed for tests that assert the lazy copies are released; not part of
9542    /// the stable surface.
9543    #[doc(hidden)]
9544    pub fn lazy_hnsw_len(&self) -> usize {
9545        self.engine.lazy_hnsw_len()
9546    }
9547
9548    /// Find nodes whose `field` vector is most similar to `q` (cosine
9549    /// similarity), returning up to `k` results with similarity ≥ `min`,
9550    /// sorted descending.
9551    ///
9552    /// When `label` is `None` the search spans all labels (via
9553    /// `hnsw_search_any_dst` or a full brute-force scan); when `label` is
9554    /// `Some(lbl)` it restricts to nodes with that label.
9555    ///
9556    /// Uses the HNSW index when one is available (fast path); otherwise falls
9557    /// back to an O(n) brute-force scan.
9558    ///
9559    /// **The index supplies candidates, never scores.** Its own distances are
9560    /// `f32` (accurate to ~1e-6, so an exact duplicate scores 0.9999999), so
9561    /// every candidate is re-scored from the `f64` property vectors by
9562    /// [`exact_vector_similarity`] before `min`, the ordering and the reported
9563    /// score are decided. `k + VECTOR_RESCORE_MARGIN` candidates are fetched so
9564    /// the re-ordering cannot drop a true top-`k` member; see that constant for
9565    /// the rule. The score a caller receives is therefore the same number the
9566    /// brute-force path would have produced, to `f64` precision, and `min = 1.0`
9567    /// finds an exact duplicate.
9568    pub fn find_similar_vector(
9569        &self,
9570        field: &str,
9571        label: Option<&str>,
9572        q: &[f64],
9573        k: usize,
9574        min: f64,
9575    ) -> Vec<(String, f64)> {
9576        self.find_similar_vector_filtered(field, label, q, k, min, None, None, false)
9577            .expect("find_similar_vector_filtered is infallible without where_")
9578    }
9579
9580    /// Like [`find_similar_vector`] but restricts results to nodes visible in
9581    /// `mask`. Hidden nodes never appear in results; the mask is applied
9582    /// **before** k-truncation so a caller still receives up to `k` visible
9583    /// hits.
9584    ///
9585    /// # HNSW path (widening beam)
9586    ///
9587    /// When an HNSW index covers the request, the beam starts at an over-fetch
9588    /// of `k × n / |visible|` (plus the rescore margin) when the mask's
9589    /// selectivity is known from the index length, otherwise at `k` plus that
9590    /// margin. If fewer than `k` visible candidates remain after the mask and
9591    /// `min` filter, the beam doubles — the same ×2 loop exact `VectorSimilar`
9592    /// rules use, capped at `ef_max()` (`EF_MAX` = 4,096). Reaching the cap,
9593    /// or a beam that comes back short of its own width, falls through to the
9594    /// exhaustive masked scan rather than returning a short result.
9595    ///
9596    /// Every surviving candidate is re-scored from the `f64` property vectors,
9597    /// exactly as [`find_similar_vector`] does and for the same reason.
9598    ///
9599    /// # Brute-force path
9600    ///
9601    /// When no HNSW index covers the request, or the beam cannot admit `k`
9602    /// hits, the function builds a masked [`GraphView`] so that `nodes_all` /
9603    /// `nodes_with_label` return only visible nodes, guaranteeing exact `k`
9604    /// results (or all visible nodes if fewer than `k` exist).
9605    pub fn find_similar_vector_masked(
9606        &self,
9607        field: &str,
9608        label: Option<&str>,
9609        q: &[f64],
9610        k: usize,
9611        min: f64,
9612        mask: &crate::mask::NodeMask,
9613    ) -> Vec<(String, f64)> {
9614        self.find_similar_vector_filtered(field, label, q, k, min, Some(mask), None, false)
9615            .expect("find_similar_vector_filtered is infallible without where_")
9616    }
9617
9618    /// Exact or ANN kNN with optional key-list `mask` and property `where_`.
9619    ///
9620    /// `where_` present and failing [`PropPredicate::validate_named`] `"where"`
9621    /// → `QueryError`. `exact=true` or `where_=Some` skip HNSW and GEMM-brute
9622    /// the candidate set (`label ∩ mask ∩ holds(where)`). `mask` alone still
9623    /// uses HNSW when an index covers the field.
9624    #[allow(clippy::too_many_arguments)]
9625    pub fn find_similar_vector_filtered(
9626        &self,
9627        field: &str,
9628        label: Option<&str>,
9629        q: &[f64],
9630        k: usize,
9631        min: f64,
9632        mask: Option<&crate::mask::NodeMask>,
9633        where_: Option<&PropPredicate>,
9634        exact: bool,
9635    ) -> Result<Vec<(String, f64)>> {
9636        self.find_similar_vector_as(
9637            field,
9638            label,
9639            q,
9640            k,
9641            min,
9642            mask,
9643            where_,
9644            exact,
9645            ExactnessCaller::Vector,
9646        )
9647    }
9648
9649    /// [`find_similar_vector_filtered`](Self::find_similar_vector_filtered)
9650    /// with the caller shape named, so the exactness warning can advise the
9651    /// signature that actually reached it. Everything else is identical.
9652    #[allow(clippy::too_many_arguments)]
9653    fn find_similar_vector_as(
9654        &self,
9655        field: &str,
9656        label: Option<&str>,
9657        q: &[f64],
9658        k: usize,
9659        min: f64,
9660        mask: Option<&crate::mask::NodeMask>,
9661        where_: Option<&PropPredicate>,
9662        exact: bool,
9663        caller: ExactnessCaller,
9664    ) -> Result<Vec<(String, f64)>> {
9665        if let Some(pred) = where_ {
9666            pred.validate_named("where")
9667                .map_err(|detail| GraphError::QueryError { detail })?;
9668        }
9669
9670        // Ensure any HNSW blobs retained from the snapshot are deserialized
9671        // before the first ANN query on a clean-open (no-WAL) path.  The
9672        // section read has to come first: on a clean open nothing else has
9673        // called it, so without it `retained_hnsw_blobs` is empty,
9674        // `ensure_hnsw_loaded` caches an empty map in its `OnceLock`, and every
9675        // approximate query on the handle runs brute force — correct results,
9676        // silently off the index.  Both calls are idempotent and cheap once hot.
9677        self.ensure_v8_base_sections_loaded();
9678        self.engine.ensure_hnsw_loaded();
9679        let norm: f64 = q.iter().map(|x| x * x).sum::<f64>().sqrt();
9680        if norm == 0.0 {
9681            return Ok(vec![]);
9682        }
9683        if let Some(m) = mask {
9684            if k == 0 || m.is_empty() {
9685                return Ok(vec![]);
9686            }
9687        }
9688        let q_unit: Vec<f64> = q.iter().map(|x| x / norm).collect();
9689
9690        // `where` implies exact: a predicate must not ride a silent ANN.
9691        let skip_hnsw = exact || where_.is_some();
9692        if !skip_hnsw {
9693            if let Some(mask) = mask {
9694                if let Some(out) =
9695                    self.find_similar_hnsw_masked(field, label, &q_unit, k, min, mask, caller)
9696                {
9697                    return Ok(out);
9698                }
9699            } else if let Some(out) = self.find_similar_hnsw(field, label, &q_unit, k, min) {
9700                return Ok(out);
9701            }
9702        }
9703
9704        let view = match mask {
9705            Some(m) => self.view_masked(m),
9706            None => self.view(),
9707        };
9708        let candidate_ids = Self::vector_candidates(&view, label, where_);
9709        Ok(self.brute_vector_hits(&view, candidate_ids, field, &q_unit, k, min))
9710    }
9711
9712    /// Unmasked HNSW path. `None` when no populated index covers the request.
9713    fn find_similar_hnsw(
9714        &self,
9715        field: &str,
9716        label: Option<&str>,
9717        q_unit: &[f64],
9718        k: usize,
9719        min: f64,
9720    ) -> Option<Vec<(String, f64)>> {
9721        // Try HNSW fast path.
9722        // `None` label searches across all VectorSimilar rules covering `field`
9723        // (merging their results); `Some(lbl)` restricts to rules whose
9724        // dst_label matches.  Returns `None` when no populated HNSW index
9725        // covers the request — the O(n) brute-force fallback handles that case.
9726        let over_k = k.saturating_add(VECTOR_RESCORE_MARGIN);
9727        let hits = match label {
9728            Some(lbl) => self.engine.hnsw_search_dst(field, lbl, q_unit, over_k)?,
9729            None => self.engine.hnsw_search_any_dst(field, q_unit, over_k)?,
9730        };
9731        // Candidates only: the index's `f32` similarity is discarded and
9732        // each hit is re-scored against the `f64` vectors.
9733        let view = self.view();
9734        let mut out: Vec<(String, f64)> = hits
9735            .into_iter()
9736            .filter_map(|(id, _)| {
9737                let sim = exact_vector_similarity(&view, id, field, q_unit)?;
9738                if sim < min {
9739                    return None;
9740                }
9741                Some((self.ids.key_of(id)?.to_string(), sim))
9742            })
9743            .collect();
9744        out.sort_by(|a, b| {
9745            b.1.partial_cmp(&a.1)
9746                .unwrap_or(std::cmp::Ordering::Equal)
9747                .then_with(|| a.0.cmp(&b.0))
9748        });
9749        out.truncate(k);
9750        Some(out)
9751    }
9752
9753    /// Masked HNSW widening beam. `None` when no index covers the request or
9754    /// the beam cannot admit `k` visible hits (caller falls through to brute).
9755    #[allow(clippy::too_many_arguments)]
9756    fn find_similar_hnsw_masked(
9757        &self,
9758        field: &str,
9759        label: Option<&str>,
9760        q_unit: &[f64],
9761        k: usize,
9762        min: f64,
9763        mask: &crate::mask::NodeMask,
9764        caller: ExactnessCaller,
9765    ) -> Option<Vec<(String, f64)>> {
9766        let index_len = match label {
9767            Some(lbl) => self.engine.hnsw_dst_len(field, lbl, q_unit.len()),
9768            None => self.engine.hnsw_any_dst_len(field, q_unit.len()),
9769        };
9770        let n = index_len?;
9771        // The `?` above is the coverage test: past it, an index exists and this
9772        // masked, non-exact call is about to ride it.
9773        self.note_ambiguous_exactness(field, label, caller);
9774        // Same ceiling the exact-rule widening loop in `hnsw_candidates`
9775        // consults — including the `with_ef_max` test hook.
9776        let cap = ef_max();
9777        let visible = mask.len();
9778        let mut ef = k.saturating_add(VECTOR_RESCORE_MARGIN);
9779        if visible > 0 && n > 0 {
9780            let over = k
9781                .saturating_mul(n)
9782                .div_ceil(visible)
9783                .saturating_add(VECTOR_RESCORE_MARGIN);
9784            ef = ef.max(over);
9785        }
9786        loop {
9787            let hits = match label {
9788                Some(lbl) => self
9789                    .engine
9790                    .hnsw_search_dst_with_ef(field, lbl, q_unit, ef, ef),
9791                None => self
9792                    .engine
9793                    .hnsw_search_any_dst_with_ef(field, q_unit, ef, ef),
9794            };
9795            let hits = hits?;
9796            let full = hits.len() == ef;
9797            let mut out = self.score_masked_hnsw_hits(&hits, field, q_unit, min, mask);
9798            if out.len() >= k {
9799                out.truncate(k);
9800                return Some(out);
9801            }
9802            // Short of its width (frontier exhausted) or at the ceiling:
9803            // a wider beam reaches nothing new, so the scan answers.
9804            if !full || ef >= cap {
9805                return None;
9806            }
9807            ef = ef.saturating_mul(2);
9808        }
9809    }
9810
9811    /// Say once, per `(field, label)` index and caller shape, that a masked
9812    /// search is answering approximately.
9813    ///
9814    /// A mask narrows *which nodes may be returned*. It does not choose a
9815    /// kernel — `exact=true` and a `where=` predicate do, and nothing else
9816    /// does. A caller who needed exact answers, passed `mask=` alone, and read
9817    /// the mask as a promise of exhaustiveness gets a correct-looking
9818    /// approximate answer and no signal at all; that is a silent wrong answer,
9819    /// and it has cost an integration team real time.
9820    ///
9821    /// The fix is a question, not a behaviour change. Making a mask imply
9822    /// `exact` would turn every existing masked caller's ANN into an O(n) GEMM
9823    /// without asking them, which is a worse trade than the ambiguity.
9824    ///
9825    /// Printed once per index for the reason the dimension-mismatch skip in
9826    /// `core_rules::hnsw` is: a line on every call is a line callers learn to
9827    /// scroll past.
9828    ///
9829    /// `caller` decides the advice. The same leg is reached from two signatures
9830    /// and only one of them has an `exact` argument to pass; see
9831    /// [`ExactnessCaller`].
9832    fn note_ambiguous_exactness(&self, field: &str, label: Option<&str>, caller: ExactnessCaller) {
9833        let entry = (field.to_string(), label.unwrap_or("").to_string(), caller);
9834        let first = match self.warned_ambiguous_exactness.lock() {
9835            Ok(mut seen) => seen.insert(entry),
9836            Err(poisoned) => poisoned.into_inner().insert(entry),
9837        };
9838        if !first {
9839            return;
9840        }
9841        AMBIGUOUS_EXACTNESS_WARNS.with(|c| c.set(c.get().saturating_add(1)));
9842        let which = match label {
9843            Some(lbl) => format!(" (label `{lbl}`)"),
9844            None => String::new(),
9845        };
9846        let subject = caller.subject();
9847        let advice = caller.advice();
9848        let line = format!(
9849            "mushroomdb: {subject} on field `{field}`{which} is answering \
9850             approximately. A mask narrows which nodes may be returned; it does not \
9851             change which kernel runs, and an index covers this field. For an exact \
9852             answer over the same visible candidate set, {advice} Further masked \
9853             searches of this shape on this index are silent."
9854        );
9855        AMBIGUOUS_EXACTNESS_LAST.with(|c| *c.borrow_mut() = Some(line.clone()));
9856        eprintln!("{line}");
9857    }
9858
9859    /// `label ∩ mask ∩ holds(where)`. Index fast path when `label` is `Some`
9860    /// and `(label, where.field)` is enabled; otherwise scan with `visible()`.
9861    fn vector_candidates(
9862        view: &GraphView<'_>,
9863        label: Option<&str>,
9864        where_: Option<&PropPredicate>,
9865    ) -> Vec<u32> {
9866        if let (Some(lbl), Some(pred)) = (label, where_) {
9867            let indexed = view
9868                .prop_index
9869                .is_some_and(|idx| idx.is_enabled(lbl, &pred.field));
9870            if indexed {
9871                match (&pred.eq, &pred.in_) {
9872                    (Some(eq), None) => {
9873                        if let Some(ids) = view.nodes_with_prop(lbl, &pred.field, eq) {
9874                            return ids;
9875                        }
9876                    }
9877                    (None, Some(allowed)) => {
9878                        let mut seen = HashSet::new();
9879                        let mut out = Vec::new();
9880                        for v in allowed {
9881                            if let Some(ids) = view.nodes_with_prop(lbl, &pred.field, v) {
9882                                for id in ids {
9883                                    if seen.insert(id) {
9884                                        out.push(id);
9885                                    }
9886                                }
9887                            }
9888                        }
9889                        return out;
9890                    }
9891                    _ => {}
9892                }
9893            }
9894        }
9895
9896        let mut ids: Vec<u32> = match label {
9897            Some(lbl) => view
9898                .nodes_with_label(lbl)
9899                .into_iter()
9900                .filter(|&id| view.visible(id))
9901                .collect(),
9902            None => view.nodes_all(),
9903        };
9904        if let Some(pred) = where_ {
9905            ids.retain(|&id| match view.prop(id, &pred.field) {
9906                None => pred.holds(None),
9907                Some(vr) => pred.holds(Some(vr.as_value())),
9908            });
9909        }
9910        ids
9911    }
9912
9913    /// Exact brute kNN: pack candidates at `q_unit`'s dim, GEMV, keep
9914    /// `score >= min`, sort `(sim desc, key asc)`, truncate to `k`.
9915    fn brute_vector_hits(
9916        &self,
9917        view: &GraphView<'_>,
9918        candidate_ids: impl IntoIterator<Item = u32>,
9919        field: &str,
9920        q_unit: &[f64],
9921        k: usize,
9922        min: f64,
9923    ) -> Vec<(String, f64)> {
9924        let rows: Vec<(u32, std::borrow::Cow<'_, [f64]>)> = candidate_ids
9925            .into_iter()
9926            .filter_map(|id| crate::exact_knn::vector_f64(view, id, field).map(|v| (id, v)))
9927            .collect();
9928        let packed =
9929            crate::exact_knn::pack(rows.iter().map(|(id, v)| (*id, v.as_ref())), q_unit.len());
9930        let scores = crate::exact_knn::gemv(&packed, q_unit);
9931        let mut scored: Vec<(String, f64)> = packed
9932            .ids
9933            .iter()
9934            .zip(scores.iter())
9935            .filter_map(|(&id, &sim)| {
9936                if sim < min {
9937                    return None;
9938                }
9939                let key = self.ids.key_of(id)?.to_string();
9940                Some((key, sim))
9941            })
9942            .collect();
9943        scored.sort_by(|a, b| {
9944            b.1.partial_cmp(&a.1)
9945                .unwrap_or(std::cmp::Ordering::Equal)
9946                .then_with(|| a.0.cmp(&b.0))
9947        });
9948        scored.truncate(k);
9949        scored
9950    }
9951
9952    /// Exact cosine top-k for each key in `keys`, scored only against `keys`.
9953    ///
9954    /// `min` is cosine similarity in [-1, 1], inclusive (`score >= min`), the
9955    /// same unit and inequality as `find_similar_vector`. Self-matches are
9956    /// excluded. Unknown keys, keys with no `field`, zero-norm or wrong-dim
9957    /// embeddings are omitted as both query and candidate. Duplicate keys are
9958    /// collapsed, first-seen order. Empty `keys` → empty `Ok(vec![])`. Never
9959    /// uses HNSW. `n > PAIRWISE_MAX_N` → `QueryError`.
9960    #[allow(clippy::type_complexity)]
9961    pub fn pairwise_similar(
9962        &self,
9963        keys: &[&str],
9964        field: &str,
9965        k: usize,
9966        min: f64,
9967    ) -> Result<Vec<(String, Vec<(String, f64)>)>> {
9968        let mut seen = HashSet::new();
9969        let mut unique_ids = Vec::new();
9970        for key in keys {
9971            let Some(id) = self.ids.get(key) else {
9972                continue;
9973            };
9974            if seen.insert(id) {
9975                unique_ids.push(id);
9976            }
9977        }
9978        let max_n = crate::exact_knn::pairwise_max_n();
9979        if unique_ids.len() > max_n {
9980            return Err(GraphError::QueryError {
9981                detail: format!(
9982                    "pairwise_similar: n={} exceeds PAIRWISE_MAX_N ({max_n})",
9983                    unique_ids.len()
9984                ),
9985            });
9986        }
9987        if unique_ids.is_empty() {
9988            return Ok(Vec::new());
9989        }
9990
9991        let view = self.view();
9992        let mut rows: Vec<(u32, std::borrow::Cow<'_, [f64]>)> = Vec::new();
9993        let mut counts: HashMap<usize, usize> = HashMap::new();
9994        for id in unique_ids {
9995            let Some(v) = crate::exact_knn::vector_f64(&view, id, field) else {
9996                continue;
9997            };
9998            let norm: f64 = v.iter().map(|x| x * x).sum::<f64>().sqrt();
9999            if norm == 0.0 {
10000                continue;
10001            }
10002            *counts.entry(v.len()).or_default() += 1;
10003            rows.push((id, v));
10004        }
10005        if rows.is_empty() {
10006            return Ok(Vec::new());
10007        }
10008        let dim = counts
10009            .into_iter()
10010            .max_by_key(|&(d, c)| (c, d))
10011            .map(|(d, _)| d)
10012            .expect("rows non-empty");
10013        let packed = crate::exact_knn::pack(rows.iter().map(|(id, v)| (*id, v.as_ref())), dim);
10014        let n = packed.ids.len();
10015        if n == 0 {
10016            return Ok(Vec::new());
10017        }
10018        let src_keys: Vec<String> = packed
10019            .ids
10020            .iter()
10021            .map(|&id| self.ids.key_of(id).unwrap_or("").to_string())
10022            .collect();
10023
10024        let mut out = Vec::with_capacity(n);
10025        if n <= crate::exact_knn::pairwise_gram_max() {
10026            let sims = crate::exact_knn::gram(&packed);
10027            for i in 0..n {
10028                out.push(Self::topk_from_row(
10029                    &src_keys,
10030                    i,
10031                    &sims[i * n..(i + 1) * n],
10032                    k,
10033                    min,
10034                ));
10035            }
10036        } else {
10037            for i in 0..n {
10038                let row = &packed.data[i * packed.dim..(i + 1) * packed.dim];
10039                let scores = crate::exact_knn::gemv(&packed, row);
10040                out.push(Self::topk_from_row(&src_keys, i, &scores, k, min));
10041            }
10042        }
10043        Ok(out)
10044    }
10045
10046    /// [`pairwise_similar`](Self::pairwise_similar) over the keys the mask
10047    /// admits — intersected **before** the matmul, never filtered after it.
10048    ///
10049    /// A hidden vector packed into the Gram is a row every visible key is
10050    /// scored against. It can take a visible neighbour's place in the top-`k`,
10051    /// and because the packed dimension is a majority vote over the candidate
10052    /// rows it can decide whether a visible pair is scored at all. Dropping
10053    /// hidden names from the finished answer leaves both effects standing, so
10054    /// the intersection happens first and the answer is byte-for-byte the one
10055    /// `pairwise_similar` gives for the visible keys alone.
10056    ///
10057    /// The caps therefore measure the **post-filter** count: a key set over
10058    /// [`PAIRWISE_MAX_N`](crate::PAIRWISE_MAX_N) unscoped can come under it
10059    /// scoped and succeed, because the work the cap refuses is work this call
10060    /// no longer does. A filtered count still over the cap is still refused.
10061    #[allow(clippy::type_complexity)]
10062    pub fn pairwise_similar_scoped(
10063        &self,
10064        keys: &[&str],
10065        field: &str,
10066        k: usize,
10067        min: f64,
10068        mask: &crate::mask::NodeMask,
10069    ) -> Result<Vec<(String, Vec<(String, f64)>)>> {
10070        let visible: Vec<&str> = keys
10071            .iter()
10072            .copied()
10073            .filter(|key| mask.contains_node(self, key))
10074            .collect();
10075        self.pairwise_similar(&visible, field, k, min)
10076    }
10077
10078    /// Neighbours of packed row `i`: drop self, keep `score >= min`, sort
10079    /// `(sim desc, key asc)`, truncate to `k`. Packed srcs with no survivors
10080    /// still appear as `(src, [])`.
10081    fn topk_from_row(
10082        src_keys: &[String],
10083        i: usize,
10084        scores: &[f64],
10085        k: usize,
10086        min: f64,
10087    ) -> (String, Vec<(String, f64)>) {
10088        let mut neigh: Vec<(String, f64)> = scores
10089            .iter()
10090            .enumerate()
10091            .filter_map(|(j, &sim)| {
10092                if i == j || sim < min {
10093                    return None;
10094                }
10095                Some((src_keys[j].clone(), sim))
10096            })
10097            .collect();
10098        neigh.sort_by(|a, b| {
10099            b.1.partial_cmp(&a.1)
10100                .unwrap_or(std::cmp::Ordering::Equal)
10101                .then_with(|| a.0.cmp(&b.0))
10102        });
10103        neigh.truncate(k);
10104        (src_keys[i].clone(), neigh)
10105    }
10106
10107    /// Re-score HNSW candidates from the `f64` vectors, drop hidden / below-`min`
10108    /// hits, order by score then key. The index's own `f32` similarity is discarded.
10109    fn score_masked_hnsw_hits(
10110        &self,
10111        hits: &[(u32, f64)],
10112        field: &str,
10113        q_unit: &[f64],
10114        min: f64,
10115        mask: &crate::mask::NodeMask,
10116    ) -> Vec<(String, f64)> {
10117        let view = self.view_masked(mask);
10118        let mut out: Vec<(String, f64)> = hits
10119            .iter()
10120            .copied()
10121            .filter(|&(id, _)| mask.contains_id(id))
10122            .filter_map(|(id, _)| {
10123                let sim = exact_vector_similarity(&view, id, field, q_unit)?;
10124                if sim < min {
10125                    return None;
10126                }
10127                Some((self.ids.key_of(id)?.to_string(), sim))
10128            })
10129            .collect();
10130        out.sort_by(|a, b| {
10131            b.1.partial_cmp(&a.1)
10132                .unwrap_or(std::cmp::Ordering::Equal)
10133                .then_with(|| a.0.cmp(&b.0))
10134        });
10135        out
10136    }
10137
10138    /// Read a single property from an edge.
10139    ///
10140    /// Returns `None` when the edge does not exist, the field is absent, or any
10141    /// of the string keys cannot be resolved to interned ids.  Only edge props
10142    /// written by rules (weight fields) are accessible without a `set_edge_prop`
10143    /// binding; topology-only edges (no props set) return `None` for every field.
10144    pub fn get_edge_prop(
10145        &self,
10146        edge_type: &str,
10147        src_key: &str,
10148        dst_key: &str,
10149        field: &str,
10150    ) -> Option<Value> {
10151        let etype = self.syms.get(edge_type)?;
10152        let src = self.ids.get(src_key)?;
10153        let dst = self.ids.get(dst_key)?;
10154        self.edge_props_view().get(etype, src, dst, field)
10155    }
10156
10157    /// Lex → parse → plan → execute `cypher` over a read-only view.
10158    /// Every pipeline `Err(String)` becomes `GraphError::QueryError` with a
10159    /// stage prefix (`lex:` / `parse:` / `plan:` / `execute:`).
10160    pub fn query(&self, cypher: &str, params: &BTreeMap<String, Value>) -> Result<ResultSet> {
10161        let tokens = lex(cypher).map_err(|e| GraphError::QueryError {
10162            detail: format!("lex: {e}"),
10163        })?;
10164        let union = parse_read(&tokens).map_err(|e| GraphError::QueryError {
10165            detail: format!("parse: {e}"),
10166        })?;
10167        let t0 = std::time::Instant::now();
10168        let result = execute_union(&self.view(), &union, &Params(params)).map_err(|e| {
10169            GraphError::QueryError {
10170                detail: format!("execute: {e}"),
10171            }
10172        });
10173        let elapsed_ms = t0.elapsed().as_millis() as u64;
10174        let threshold = self.slow_query_threshold_ms;
10175        if threshold > 0 && elapsed_ms >= threshold {
10176            eprintln!("[mushroomdb] slow query ({elapsed_ms}ms): {cypher}");
10177            let entry = SlowQueryEntry {
10178                ms: elapsed_ms,
10179                query: cypher.to_string(),
10180                at_commit: self.commit_seq,
10181            };
10182            if let Ok(mut log) = self.slow_queries.lock() {
10183                if log.entries.len() == SLOW_QUERY_RING_CAP {
10184                    log.entries.pop_front();
10185                }
10186                log.entries.push_back(entry);
10187                log.total += 1;
10188            }
10189        }
10190        result
10191    }
10192
10193    /// Convenience entry-point that accepts a slice of `(name, value)` pairs
10194    /// instead of a pre-built `BTreeMap`.  Equivalent to building the map and
10195    /// calling [`GraphDb::query`].
10196    pub fn query_with_params(&self, cypher: &str, params: &[(&str, Value)]) -> Result<ResultSet> {
10197        let map: BTreeMap<String, Value> = params
10198            .iter()
10199            .map(|(k, v)| (k.to_string(), v.clone()))
10200            .collect();
10201        self.query(cypher, &map)
10202    }
10203
10204    /// Execute a Cypher write statement (CREATE / MATCH…SET / MATCH…DELETE / MERGE).
10205    ///
10206    /// All mutations flow through the same `insert_node` / `set_prop` /
10207    /// `delete_edge` / `insert_edge` path as the Rust API so the rule engine
10208    /// fires and the WAL captures everything with one fsync per statement.
10209    ///
10210    /// Returns a one-row [`ResultSet`] with columns `created`, `properties_set`,
10211    /// and `deleted` matching the write-result contract.
10212    ///
10213    /// **Mutation routing**: mutations are collected into a single
10214    /// [`BatchBuilder`] and committed atomically (one WAL `Batch` frame, one
10215    /// fsync). The MATCH phase for SET/DELETE uses a read-only `execute` call
10216    /// over `self.view()` — the borrow is dropped before the batch is opened.
10217    ///
10218    /// **Limitations (v1)**:
10219    /// - SET RHS must be a literal, `$param`, or arithmetic; bare property copy → named error.
10220    /// - `DETACH DELETE n` → calls `delete_node` for each matched node (removes all edges).
10221    /// - Bare `DELETE n` → error if n has any incident edges; succeeds for isolated nodes.
10222    /// - MERGE supports `ON CREATE SET` / `ON MATCH SET` in the same write batch.
10223    /// - Deleting a derived edge → named error "cannot delete derived edge".
10224    pub fn query_write(
10225        &mut self,
10226        cypher: &str,
10227        params: &BTreeMap<String, Value>,
10228    ) -> Result<ResultSet> {
10229        let tokens = lex(cypher).map_err(|e| GraphError::QueryError {
10230            detail: format!("lex: {e}"),
10231        })?;
10232        let stmt = parse_write(&tokens).map_err(|e| GraphError::QueryError {
10233            detail: format!("parse: {e}"),
10234        })?;
10235        self.exec_write_stmt(stmt, params)
10236    }
10237
10238    fn exec_write_stmt(
10239        &mut self,
10240        stmt: WriteStatement,
10241        params: &BTreeMap<String, Value>,
10242    ) -> Result<ResultSet> {
10243        match stmt {
10244            WriteStatement::Create(s) => self.exec_create(s, params),
10245            WriteStatement::MatchSet(s) => self.exec_match_set(s, params),
10246            WriteStatement::MatchDelete(s) => self.exec_match_delete(s, params),
10247            WriteStatement::MatchDeleteNode(s) => self.exec_match_delete_node(s, params),
10248            WriteStatement::Merge(s) => self.exec_merge(s, params),
10249        }
10250    }
10251
10252    fn exec_create(
10253        &mut self,
10254        stmt: core_query::cypher::CreateStmt,
10255        params: &BTreeMap<String, Value>,
10256    ) -> Result<ResultSet> {
10257        // Extract the node key from props: require a string-valued `id` field.
10258        let mut var_to_key: BTreeMap<String, String> = BTreeMap::new();
10259        for node in &stmt.nodes {
10260            let var = node.var.as_deref().unwrap_or("_cn0");
10261            let key = node
10262                .props
10263                .iter()
10264                .find(|(f, _)| f == "id")
10265                .and_then(|(_, v)| {
10266                    if let Value::Str(s) = v {
10267                        Some(s.clone())
10268                    } else {
10269                        None
10270                    }
10271                })
10272                .ok_or_else(|| GraphError::QueryError {
10273                    detail: format!(
10274                        "CREATE node ({}:{}) requires a string 'id' property",
10275                        var, node.label
10276                    ),
10277                })?;
10278            var_to_key.insert(var.to_string(), key);
10279        }
10280
10281        let mut batch = self.batch();
10282        let mut created: usize = 0;
10283        for node in &stmt.nodes {
10284            let var = node.var.as_deref().unwrap_or("_cn0");
10285            let key = &var_to_key[var];
10286            batch.insert_node(&node.label, key, node.props.clone());
10287            created += 1;
10288        }
10289        for edge in &stmt.edges {
10290            let src_key = var_to_key
10291                .get(&edge.src_var)
10292                .ok_or_else(|| GraphError::QueryError {
10293                    detail: format!("CREATE edge src variable '{}' is not bound", edge.src_var),
10294                })?;
10295            let dst_key = var_to_key
10296                .get(&edge.dst_var)
10297                .ok_or_else(|| GraphError::QueryError {
10298                    detail: format!("CREATE edge dst variable '{}' is not bound", edge.dst_var),
10299                })?;
10300            batch.insert_edge(&edge.etype, src_key, dst_key);
10301        }
10302        batch.commit()?;
10303
10304        // Optional RETURN clause: project created bindings as a read result.
10305        if let Some(returns) = stmt.returns {
10306            // Each created node is looked up by its key via a separate MATCH pattern.
10307            // Multiple single-node patterns cross-join to produce 1 output row with
10308            // all variables bound (each pattern returns exactly 1 row).
10309            let patterns: Vec<Pattern> = stmt
10310                .nodes
10311                .iter()
10312                .map(|node| {
10313                    let var = node.var.as_deref().unwrap_or("_cn0");
10314                    let key = var_to_key[var].clone();
10315                    Pattern {
10316                        start: NodePat {
10317                            var: Some(var.to_string()),
10318                            label: Some(node.label.clone()),
10319                            props: vec![("id".to_string(), Operand::Lit(Value::Str(key)))],
10320                        },
10321                        chain: vec![],
10322                        shortest: false,
10323                    }
10324                })
10325                .collect();
10326            let q = Query {
10327                matches: patterns,
10328                optional_clauses: vec![],
10329                where_expr: None,
10330                unwinds: vec![],
10331                post_unwind_where: None,
10332                stages: vec![],
10333                returns,
10334                distinct: false,
10335                order_by: vec![],
10336                skip: None,
10337                limit: None,
10338            };
10339            let ops = plan(&q).map_err(|e| GraphError::QueryError {
10340                detail: format!("plan: {e}"),
10341            })?;
10342            return execute(&self.view(), &ops, &Params(params)).map_err(|e| {
10343                GraphError::QueryError {
10344                    detail: format!("execute: {e}"),
10345                }
10346            });
10347        }
10348
10349        let mut rs = write_result_set();
10350        rs.push_row(vec![
10351            Some(Value::Int(created as i64)),
10352            Some(Value::Int(0)),
10353            Some(Value::Int(0)),
10354        ]);
10355        Ok(rs)
10356    }
10357
10358    fn exec_match_set(
10359        &mut self,
10360        stmt: core_query::cypher::MatchSetStmt,
10361        params: &BTreeMap<String, Value>,
10362    ) -> Result<ResultSet> {
10363        let project_returns = stmt.returns.clone();
10364        // Collect unique node vars targeted by SET clauses, plus RETURN bindings
10365        // so the post-write projection can look them up by key.
10366        let mut set_vars: Vec<String> = Vec::new();
10367        for s in &stmt.sets {
10368            if !set_vars.contains(&s.var) {
10369                set_vars.push(s.var.clone());
10370            }
10371        }
10372        let rel_vars = pattern_rel_vars(&stmt.matches);
10373        // `count` is the engine's, on an edge: it is the insert-count §5.13
10374        // maintains, and a `SET` that overwrote it would make the number mean
10375        // whatever the last writer said rather than how many times the pair was
10376        // inserted. Refused by name here, before the match runs, so the caller
10377        // is told what is actually wrong instead of meeting the executor's
10378        // generic "did not resolve to a node key" — and so the answer does not
10379        // depend on whether the pattern happened to match a row. The same name
10380        // on a *node* is an ordinary property and is untouched.
10381        for s in &stmt.sets {
10382            if s.field == EDGE_COUNT_PROP && rel_vars.iter().any(|r| r == &s.var) {
10383                return Err(GraphError::QueryError {
10384                    detail: format!(
10385                        "cannot SET {}.{EDGE_COUNT_PROP}: `{EDGE_COUNT_PROP}` is a reserved edge \
10386                         property holding the pair's insert count",
10387                        s.var
10388                    ),
10389                });
10390            }
10391        }
10392        let mut lookup_vars = set_vars.clone();
10393        for v in pattern_node_vars(&stmt.matches) {
10394            add_var(&mut lookup_vars, &v);
10395        }
10396        if let Some(ref returns) = project_returns {
10397            for v in ret_node_vars(returns) {
10398                if !rel_vars.iter().any(|r| r == &v) {
10399                    add_var(&mut lookup_vars, &v);
10400                }
10401            }
10402        }
10403
10404        // Synthesize a read query: MATCH … WHERE … RETURN <lookup_vars>, <set_values…>
10405        // SET values are projected as ScalarExpr items so that arithmetic expressions
10406        // (e.g. `SET n.score = n.score * 1.5`) are evaluated in the matched-row context.
10407        let mut set_returns: Vec<RetItem> = lookup_vars
10408            .iter()
10409            .map(|v| RetItem {
10410                value: RetVal::Var(v.clone()),
10411                alias: None,
10412            })
10413            .collect();
10414        // One computed column per SET clause; alias is `__sv_<i>`.
10415        let set_val_cols: Vec<String> = stmt
10416            .sets
10417            .iter()
10418            .enumerate()
10419            .map(|(i, _)| format!("__sv_{i}"))
10420            .collect();
10421        for (sc, col) in stmt.sets.iter().zip(&set_val_cols) {
10422            set_returns.push(RetItem {
10423                value: RetVal::ScalarExpr(sc.value.clone()),
10424                alias: Some(col.clone()),
10425            });
10426        }
10427        // Capture relationship types while r is bound; SET does not change them.
10428        for r in &rel_vars {
10429            set_returns.push(RetItem {
10430                value: RetVal::FuncCall {
10431                    name: "type".into(),
10432                    args: vec![Operand::Var(r.clone())],
10433                },
10434                alias: Some(rel_type_alias(r)),
10435            });
10436        }
10437
10438        let read_q = Query {
10439            matches: stmt.matches.clone(),
10440            optional_clauses: vec![],
10441            where_expr: stmt.where_expr.clone(),
10442            unwinds: vec![],
10443            post_unwind_where: None,
10444            stages: vec![],
10445            returns: set_returns,
10446            distinct: false,
10447            order_by: vec![],
10448            skip: None,
10449            limit: None,
10450        };
10451        let ops = plan(&read_q).map_err(|e| GraphError::QueryError {
10452            detail: format!("plan: {e}"),
10453        })?;
10454        // MATCH phase is read-only; borrow ends before batch opens.
10455        //
10456        // When a role-scoped write is in flight, run the MATCH read through
10457        // view_masked so hidden nodes are invisible → hidden ≡ absent ≡
10458        // zero-rows (no SetProp ops generated, no existence-oracle 403).
10459        // Full-authority writes (pending_write_authz=None) keep view().
10460        let match_rs = {
10461            let mask_opt = self.pending_write_authz.as_ref().map(|a| a.mask.clone());
10462            if let Some(ref mask) = mask_opt {
10463                execute(&self.view_masked(mask), &ops, &Params(params))
10464            } else {
10465                execute(&self.view(), &ops, &Params(params))
10466            }
10467        }
10468        .map_err(|e| GraphError::QueryError {
10469            detail: format!("execute: {e}"),
10470        })?;
10471
10472        // Collect (key, field, value) for each matched row × each SET clause.
10473        let mut set_ops: Vec<(String, String, Value)> = Vec::new();
10474        for row_i in 0..match_rs.len() {
10475            for (sc, col) in stmt.sets.iter().zip(&set_val_cols) {
10476                let key = match match_rs.get(row_i, &sc.var) {
10477                    Some(Value::Str(k)) => k.clone(),
10478                    _ => {
10479                        return Err(GraphError::QueryError {
10480                            detail: format!(
10481                                "SET variable '{}' did not resolve to a node key",
10482                                sc.var
10483                            ),
10484                        })
10485                    }
10486                };
10487                // The SET value was already evaluated by the executor.
10488                let value = match match_rs.get(row_i, col) {
10489                    Some(v) => v.clone(),
10490                    None => {
10491                        return Err(GraphError::QueryError {
10492                            detail: format!(
10493                                "SET value for {}.{} evaluated to null",
10494                                sc.var, sc.field
10495                            ),
10496                        })
10497                    }
10498                };
10499                set_ops.push((key, sc.field.clone(), value));
10500            }
10501        }
10502
10503        // Apply as one atomic batch.
10504        let props_set = set_ops.len();
10505        let mut batch = self.batch();
10506        for (key, field, value) in set_ops {
10507            batch.set_prop(&key, &field, value);
10508        }
10509        batch.commit()?;
10510
10511        if let Some(returns) = project_returns {
10512            return project_set_return_rows(self, &rel_vars, &match_rs, &returns, params);
10513        }
10514
10515        let mut rs = write_result_set();
10516        rs.push_row(vec![
10517            Some(Value::Int(0)),
10518            Some(Value::Int(props_set as i64)),
10519            Some(Value::Int(0)),
10520        ]);
10521        Ok(rs)
10522    }
10523
10524    fn exec_match_delete(
10525        &mut self,
10526        stmt: core_query::cypher::MatchDeleteStmt,
10527        params: &BTreeMap<String, Value>,
10528    ) -> Result<ResultSet> {
10529        // Collect unique node vars needed to identify edge endpoints.
10530        let mut node_vars: Vec<String> = Vec::new();
10531        for ed in &stmt.deletes {
10532            if !node_vars.contains(&ed.src_var) {
10533                node_vars.push(ed.src_var.clone());
10534            }
10535            if !node_vars.contains(&ed.dst_var) {
10536                node_vars.push(ed.dst_var.clone());
10537            }
10538        }
10539
10540        // Synthesize read query.
10541        let returns: Vec<RetItem> = node_vars
10542            .iter()
10543            .map(|v| RetItem {
10544                value: RetVal::Var(v.clone()),
10545                alias: None,
10546            })
10547            .collect();
10548        let read_q = Query {
10549            matches: stmt.matches,
10550            optional_clauses: vec![],
10551            where_expr: stmt.where_expr,
10552            unwinds: vec![],
10553            post_unwind_where: None,
10554            stages: vec![],
10555            returns,
10556            distinct: false,
10557            order_by: vec![],
10558            skip: None,
10559            limit: None,
10560        };
10561        let ops = plan(&read_q).map_err(|e| GraphError::QueryError {
10562            detail: format!("plan: {e}"),
10563        })?;
10564        // Role-scoped writes: mask the MATCH read phase so hidden nodes are
10565        // invisible → hidden ≡ absent ≡ zero-rows (spec §3.1, hidden ≡ absent).
10566        let match_rs = {
10567            let mask_opt = self.pending_write_authz.as_ref().map(|a| a.mask.clone());
10568            if let Some(ref mask) = mask_opt {
10569                execute(&self.view_masked(mask), &ops, &Params(params))
10570            } else {
10571                execute(&self.view(), &ops, &Params(params))
10572            }
10573        }
10574        .map_err(|e| GraphError::QueryError {
10575            detail: format!("execute: {e}"),
10576        })?;
10577
10578        // Collect (etype, src_key, dst_key) for each row × each delete target.
10579        let mut del_ops: Vec<(String, String, String)> = Vec::new();
10580        for row_i in 0..match_rs.len() {
10581            for ed in &stmt.deletes {
10582                let src_key = match match_rs.get(row_i, &ed.src_var) {
10583                    Some(Value::Str(k)) => k.clone(),
10584                    _ => {
10585                        return Err(GraphError::QueryError {
10586                            detail: format!(
10587                                "DELETE src variable '{}' did not resolve to a node key",
10588                                ed.src_var
10589                            ),
10590                        })
10591                    }
10592                };
10593                let dst_key = match match_rs.get(row_i, &ed.dst_var) {
10594                    Some(Value::Str(k)) => k.clone(),
10595                    _ => {
10596                        return Err(GraphError::QueryError {
10597                            detail: format!(
10598                                "DELETE dst variable '{}' did not resolve to a node key",
10599                                ed.dst_var
10600                            ),
10601                        })
10602                    }
10603                };
10604                del_ops.push((ed.etype.clone(), src_key, dst_key));
10605            }
10606        }
10607
10608        // Apply as one atomic batch.
10609        let deleted = del_ops.len();
10610        let mut batch = self.batch();
10611        for (etype, src_key, dst_key) in del_ops {
10612            batch.delete_edge(&etype, &src_key, &dst_key);
10613        }
10614        batch.commit().map_err(|e| match e {
10615            GraphError::RuleOwned { .. } => GraphError::QueryError {
10616                detail: "cannot delete derived edge; retract via the rule or change the property"
10617                    .to_string(),
10618            },
10619            other => other,
10620        })?;
10621
10622        let mut rs = write_result_set();
10623        rs.push_row(vec![
10624            Some(Value::Int(0)),
10625            Some(Value::Int(0)),
10626            Some(Value::Int(deleted as i64)),
10627        ]);
10628        Ok(rs)
10629    }
10630
10631    /// Execute `MATCH … [DETACH] DELETE <node_var> [, …]`.
10632    ///
10633    /// Collects the matching node keys via an ephemeral read query, then calls
10634    /// `delete_node` on each one.  When `stmt.detach` is `false` (bare DELETE)
10635    /// the executor first checks that the node has no incident edges; if any
10636    /// remain it returns a named error matching openCypher semantics.
10637    fn exec_match_delete_node(
10638        &mut self,
10639        stmt: MatchDeleteNodeStmt,
10640        params: &BTreeMap<String, Value>,
10641    ) -> Result<ResultSet> {
10642        // Build a read query returning only the node keys we need.
10643        let returns: Vec<RetItem> = stmt
10644            .node_vars
10645            .iter()
10646            .map(|v| RetItem {
10647                value: RetVal::Var(v.clone()),
10648                alias: None,
10649            })
10650            .collect();
10651        let read_q = Query {
10652            matches: stmt.matches,
10653            optional_clauses: vec![],
10654            where_expr: stmt.where_expr,
10655            unwinds: vec![],
10656            post_unwind_where: None,
10657            stages: vec![],
10658            returns,
10659            distinct: false,
10660            order_by: vec![],
10661            skip: None,
10662            limit: None,
10663        };
10664        let ops = plan(&read_q).map_err(|e| GraphError::QueryError {
10665            detail: format!("plan: {e}"),
10666        })?;
10667        // Role-scoped writes: mask the MATCH read phase so hidden nodes are
10668        // invisible → hidden ≡ absent ≡ zero-rows (spec §3.1, hidden ≡ absent).
10669        let match_rs = {
10670            let mask_opt = self.pending_write_authz.as_ref().map(|a| a.mask.clone());
10671            if let Some(ref mask) = mask_opt {
10672                execute(&self.view_masked(mask), &ops, &Params(params))
10673            } else {
10674                execute(&self.view(), &ops, &Params(params))
10675            }
10676        }
10677        .map_err(|e| GraphError::QueryError {
10678            detail: format!("execute: {e}"),
10679        })?;
10680
10681        // Collect unique node keys to delete (deduplicate across rows × vars).
10682        let mut keys: Vec<String> = Vec::new();
10683        for row_i in 0..match_rs.len() {
10684            for var in &stmt.node_vars {
10685                if let Some(Value::Str(k)) = match_rs.get(row_i, var) {
10686                    if !keys.contains(k) {
10687                        keys.push(k.clone());
10688                    }
10689                }
10690            }
10691        }
10692
10693        if !stmt.detach {
10694            // openCypher bare DELETE: error if any matched node has incident edges.
10695            for key in &keys {
10696                if let Some(id) = self.ids.get(key) {
10697                    let tv = self.topo_view();
10698                    let has_edges = tv.etypes().any(|et| {
10699                        !tv.neighbors(et, Direction::Out, id).is_empty()
10700                            || !tv.neighbors(et, Direction::In, id).is_empty()
10701                    });
10702                    if has_edges {
10703                        return Err(GraphError::QueryError {
10704                            detail: format!(
10705                                "Cannot delete node `{key}` because it still has incident edges. \
10706                                 Use DETACH DELETE to remove the node and all its edges."
10707                            ),
10708                        });
10709                    }
10710                }
10711            }
10712        }
10713
10714        let mut nodes_deleted = 0i64;
10715        let mut edges_deleted = 0i64;
10716        for key in keys {
10717            match self.delete_node(&key) {
10718                Ok(report) => {
10719                    nodes_deleted += 1;
10720                    edges_deleted += (report.manual_edges + report.derived_edges) as i64;
10721                }
10722                Err(GraphError::KeyNotFound { .. }) => {
10723                    // Node may have been deleted by an earlier iteration (e.g., via
10724                    // multiple MATCH rows for the same node).  Safe to skip.
10725                }
10726                Err(e) => return Err(e),
10727            }
10728        }
10729
10730        let mut rs = write_result_set();
10731        rs.push_row(vec![
10732            Some(Value::Int(0)),
10733            Some(Value::Int(0)),
10734            Some(Value::Int(nodes_deleted + edges_deleted)),
10735        ]);
10736        Ok(rs)
10737    }
10738
10739    /// Props the MERGE create arm inserts: the identifying key, plus `ns` when
10740    /// the pattern named one, or the executing role's sole namespace when it
10741    /// did not. A role bound to two or more namespaces cannot choose, and is
10742    /// refused with [`MERGE_CREATE_NEEDS_ONE_NAMESPACE`]. The authorizer still
10743    /// refuses a named `ns` the role cannot write.
10744    fn merge_create_props(
10745        &self,
10746        key_field: &str,
10747        key_value: &Value,
10748        named_ns: Option<&Value>,
10749    ) -> Result<Vec<(String, Value)>> {
10750        let mut props = vec![(key_field.to_string(), key_value.clone())];
10751        if let Some(ns) = named_ns {
10752            props.push((NS_PROP.to_string(), ns.clone()));
10753            return Ok(props);
10754        }
10755        if let Some(ns) = self.merge_create_stamp_ns()? {
10756            props.push((NS_PROP.to_string(), Value::Str(ns)));
10757        }
10758        Ok(props)
10759    }
10760
10761    /// The namespace a role-scoped MERGE create stamps when the pattern does
10762    /// not name `ns`. `None` = unscoped / full authority, so the node lands in
10763    /// `default`.
10764    fn merge_create_stamp_ns(&self) -> Result<Option<String>> {
10765        let Some(authz) = self.pending_write_authz.as_ref() else {
10766            return Ok(None);
10767        };
10768        let Some(def) = self.role_def_for(&authz.role) else {
10769            return Ok(None);
10770        };
10771        match def.namespaces.as_deref() {
10772            Some([only]) => Ok(Some(only.clone())),
10773            Some(_) => Err(GraphError::RoleWriteDenied {
10774                reason: MERGE_CREATE_NEEDS_ONE_NAMESPACE.to_string(),
10775            }),
10776            None => Ok(None),
10777        }
10778    }
10779
10780    fn exec_merge(
10781        &mut self,
10782        stmt: core_query::cypher::MergeStmt,
10783        params: &BTreeMap<String, Value>,
10784    ) -> Result<ResultSet> {
10785        // MERGE: check if a node with the given key already exists.
10786        let key = match &stmt.key_value {
10787            Value::Str(s) => s.clone(),
10788            _ => {
10789                return Err(GraphError::QueryError {
10790                    detail: format!(
10791                        "MERGE key value must be a string (got {:?})",
10792                        stmt.key_value
10793                    ),
10794                })
10795            }
10796        };
10797
10798        if let Some(var) = stmt.var.as_deref() {
10799            for sc in stmt.on_create.iter().chain(&stmt.on_match) {
10800                if sc.var != var {
10801                    return Err(GraphError::QueryError {
10802                        detail: format!(
10803                            "SET variable '{}' does not match MERGE variable '{var}'",
10804                            sc.var
10805                        ),
10806                    });
10807                }
10808            }
10809        }
10810
10811        // ── MERGE authz pre-check (when role-scoped) ─────────────────────────
10812        //
10813        // MERGE scope precondition: check create OR update scope for the
10814        // declared label BEFORE calling `has_node` (timing-oracle closure,
10815        // spec §6.2 "MERGE visibility oracle" item: hidden ≡ absent for
10816        // unscoped roles — the scope denial fires without touching the key store).
10817        //
10818        // Clone to avoid holding a borrow on `self.pending_write_authz` while
10819        // also calling `self.ids.get(key)`.
10820        let merge_existed: bool = if let Some(authz) = self.pending_write_authz.clone() {
10821            let has_create = authz.scope.create_labels.contains(&stmt.label);
10822            let has_update = authz.scope.update_labels.contains(&stmt.label);
10823            if !has_create && !has_update {
10824                // Scope-before-lookup: 403 without has_node call (timing oracle
10825                // closure — see test_merge_unscoped_no_key_lookup).
10826                return Err(GraphError::RoleWriteDenied {
10827                    reason: format!(
10828                        "role-bound token: label '{}' not in write scope (create_labels)",
10829                        stmt.label
10830                    ),
10831                });
10832            }
10833            // Key lookup under mask.
10834            match self.ids.get(key.as_str()) {
10835                Some(id) if authz.mask.contains_id(id) => {
10836                    // Visible: must have update scope to proceed to match arm.
10837                    if !has_update {
10838                        return Err(GraphError::RoleWriteDenied {
10839                            reason: format!(
10840                                "role-bound token: label '{}' not in write scope (update_labels)",
10841                                stmt.label
10842                            ),
10843                        });
10844                    }
10845                    true // existed = true → match arm
10846                }
10847                Some(_) => {
10848                    // Hidden: same error as absent to the role (spec §3.1/§3.3).
10849                    return Err(GraphError::RoleWriteDenied {
10850                        reason: "role-bound token: target node not visible".into(),
10851                    });
10852                }
10853                None => {
10854                    // Absent: must have create scope to proceed to the create arm.
10855                    //
10856                    // Update-only roles (create_labels empty, update_labels set):
10857                    // return the SAME "not visible" error as the hidden-key branch
10858                    // so hidden ≡ absent — no distinguishing oracle (spec §6.1
10859                    // "confirm existence of hidden nodes: No").
10860                    //
10861                    // Create-scoped roles (has_create=true): absent → create arm
10862                    // as before.  The accepted structural key-existence disclosure
10863                    // (§THREAT-MODEL) applies only when the role holds create scope.
10864                    if !has_create {
10865                        return Err(GraphError::RoleWriteDenied {
10866                            reason: "role-bound token: target node not visible".into(),
10867                        });
10868                    }
10869                    false // existed = false → create arm
10870                }
10871            }
10872        } else {
10873            // Full authority: use the existing non-masked has_node check.
10874            self.has_node(&key)
10875        };
10876
10877        let existed = merge_existed;
10878        let create_props = if existed {
10879            None
10880        } else {
10881            Some(self.merge_create_props(&stmt.key_field, &stmt.key_value, stmt.ns.as_ref())?)
10882        };
10883        let mut created = 0i64;
10884        if create_props.is_some() || !stmt.on_match.is_empty() {
10885            let mut batch = self.batch();
10886            if let Some(props) = create_props {
10887                batch.insert_node(&stmt.label, &key, props);
10888                for sc in &stmt.on_create {
10889                    let value = resolve_merge_set_value(&sc.value, params)?;
10890                    batch.set_prop(&key, &sc.field, value);
10891                }
10892                created = 1;
10893            } else {
10894                for sc in &stmt.on_match {
10895                    let value = resolve_merge_set_value(&sc.value, params)?;
10896                    batch.set_prop(&key, &sc.field, value);
10897                }
10898            }
10899            batch.commit()?;
10900        }
10901
10902        // Refresh the role mask so the just-created node is visible to this
10903        // statement's RETURN (read-after-write). Safe: create_labels ⊆ read labels
10904        // (apply_schema subset rule), so the new node's label is already in the
10905        // role's read scope — this never widens beyond the role's declared labels.
10906        if !existed {
10907            if let Some(role) = self.pending_write_authz.as_ref().map(|a| a.role.clone()) {
10908                let new_mask = self.mask_for_role(&role)?;
10909                if let Some(a) = self.pending_write_authz.as_mut() {
10910                    a.mask = new_mask;
10911                }
10912            }
10913        }
10914
10915        // Optional RETURN clause: project the node (created or matched) as a read result.
10916        if let Some(returns) = stmt.returns {
10917            let var = stmt.var.as_deref().unwrap_or("_mn0");
10918            let q = Query {
10919                matches: vec![Pattern {
10920                    start: NodePat {
10921                        var: Some(var.to_string()),
10922                        label: Some(stmt.label.clone()),
10923                        props: vec![("id".to_string(), Operand::Lit(stmt.key_value.clone()))],
10924                    },
10925                    chain: vec![],
10926                    shortest: false,
10927                }],
10928                optional_clauses: vec![],
10929                where_expr: None,
10930                unwinds: vec![],
10931                post_unwind_where: None,
10932                stages: vec![],
10933                returns,
10934                distinct: false,
10935                order_by: vec![],
10936                skip: None,
10937                limit: None,
10938            };
10939            let ops = plan(&q).map_err(|e| GraphError::QueryError {
10940                detail: format!("plan: {e}"),
10941            })?;
10942            // Use view_masked when a role-scoped write is in flight so the
10943            // post-merge projection is consistent with the masked read phase.
10944            let mask_opt = self.pending_write_authz.as_ref().map(|a| a.mask.clone());
10945            return (if let Some(ref mask) = mask_opt {
10946                execute(&self.view_masked(mask), &ops, &Params(params))
10947            } else {
10948                execute(&self.view(), &ops, &Params(params))
10949            })
10950            .map_err(|e| GraphError::QueryError {
10951                detail: format!("execute: {e}"),
10952            });
10953        }
10954
10955        let mut rs = write_result_set();
10956        rs.push_row(vec![
10957            Some(Value::Int(created)),
10958            Some(Value::Int(0)),
10959            Some(Value::Int(0)),
10960        ]);
10961        Ok(rs)
10962    }
10963
10964    /// Return all rule-owned edges between `key_a` and `key_b` (either direction),
10965    /// annotated with rule name, edge type, direction, and weight.
10966    /// Results are sorted by (rule, edge_type).
10967    /// Returns `Err(KeyNotFound)` if either key is unknown.
10968    pub fn explain(&self, key_a: &str, key_b: &str) -> Result<Vec<Explanation>> {
10969        self.ensure_v8_base_sections_loaded();
10970        let id_a = self
10971            .ids
10972            .get(key_a)
10973            .ok_or_else(|| GraphError::KeyNotFound { key: key_a.into() })?;
10974        let id_b = self
10975            .ids
10976            .get(key_b)
10977            .ok_or_else(|| GraphError::KeyNotFound { key: key_b.into() })?;
10978
10979        let mut results = Vec::new();
10980
10981        // Walk the smaller incident set so explain is O(min(deg(a), deg(b)))
10982        // rather than O(total provenance).
10983        let scan = if self.engine.provenance_touching_len(id_a)
10984            <= self.engine.provenance_touching_len(id_b)
10985        {
10986            id_a
10987        } else {
10988            id_b
10989        };
10990        for (rule_name, etype, src, dst) in self.engine.provenance_touching(scan) {
10991            if !((src == id_a && dst == id_b) || (src == id_b && dst == id_a)) {
10992                continue;
10993            }
10994            let Some(rule_def) = self.engine.rules().find(|r| r.name == rule_name) else {
10995                continue;
10996            };
10997            let edge_type = match self.syms.resolve(etype) {
10998                Some(s) => s.to_string(),
10999                None => continue,
11000            };
11001            // Provenance (src, dst) ids come from the archived PROVENANCE section
11002            // (large, no eager CRC).  A corrupt section can produce ids that are
11003            // out of range; return Corrupt rather than panic.
11004            let src_key = self
11005                .ids
11006                .key_of(src)
11007                .ok_or_else(|| GraphError::Corrupt {
11008                    detail: format!("v8: provenance src id {src} not in id table"),
11009                })?
11010                .to_string();
11011            let dst_key = self
11012                .ids
11013                .key_of(dst)
11014                .ok_or_else(|| GraphError::Corrupt {
11015                    detail: format!("v8: provenance dst id {dst} not in id table"),
11016                })?
11017                .to_string();
11018            let stored = rule_def.weight_prop.as_deref().and_then(|prop| {
11019                self.edge_props_view()
11020                    .get(etype, src, dst, prop)
11021                    .and_then(|v| {
11022                        if let Value::Float(f) = v {
11023                            Some(f)
11024                        } else {
11025                            None
11026                        }
11027                    })
11028            });
11029            // Rules that store no weight (KeyMatch/FieldEqual defaults, auto-FK)
11030            // still have a score: recompute it from the predicate so explain
11031            // never reports "no score" for an edge the engine scored.  Via-hop
11032            // rules score over their via set, not over (src, dst), so leave
11033            // those None rather than report a number the rule did not produce.
11034            let weight = stored.or_else(|| {
11035                if rule_def.via_edge.is_some() {
11036                    return None;
11037                }
11038                let props_view = build_props_view(&self.props, &self.base);
11039                let src_get = |field: &str| props_view.get(src, field).map(|vr| vr.into_value());
11040                let dst_get = |field: &str| props_view.get(dst, field).map(|vr| vr.into_value());
11041                let src_view = NodeView {
11042                    key: &src_key,
11043                    props: &src_get,
11044                };
11045                let dst_view = NodeView {
11046                    key: &dst_key,
11047                    props: &dst_get,
11048                };
11049                evaluate(&rule_def.predicate, &src_view, &dst_view)
11050            });
11051            results.push(Explanation {
11052                rule: rule_name.to_string(),
11053                edge_type,
11054                src_key,
11055                dst_key,
11056                weight,
11057                predicate: PredicateSummary {
11058                    approximate: rule_def.approximate,
11059                    ..PredicateSummary::from(&rule_def.predicate)
11060                },
11061                via_edge: rule_def.via_edge.clone(),
11062            });
11063        }
11064
11065        results.sort_by(|a, b| a.rule.cmp(&b.rule).then(a.edge_type.cmp(&b.edge_type)));
11066        Ok(results)
11067    }
11068
11069    /// [`explain`](Self::explain) with the scoped read contract (§5.3): both
11070    /// endpoints are subject-checked, and any explanation whose evidence runs
11071    /// through a hidden node is **dropped entirely, not redacted**.
11072    ///
11073    /// A plain two-node rule's evidence is the pair itself, so once both
11074    /// subjects are visible there is nothing left to hide. A **via-hop** rule is
11075    /// different: it fires `src → dst` because some node carrying `via_label`
11076    /// sits between them, and [`Explanation`] carries the hop's edge *type*
11077    /// (`via_edge`) and never the hop's key. There is no field to blank, so a
11078    /// redacted explanation would still say "these two are linked through
11079    /// something you cannot see" — which discloses that the something exists.
11080    /// The explanation is therefore kept only when at least one **visible** via
11081    /// node satisfies the rule on its own.
11082    ///
11083    /// The weight is the **visible corpus's** number, not the store's: a via-hop
11084    /// rule stores the max over every via it hopped through, so the stored value
11085    /// can be a score only a hidden via produced. It is recomputed over the
11086    /// visible vias alone.
11087    ///
11088    /// Hidden or unknown `key_a` or `key_b` → [`GraphError::KeyNotFound`].
11089    pub fn explain_scoped(
11090        &self,
11091        key_a: &str,
11092        key_b: &str,
11093        mask: &crate::mask::NodeMask,
11094    ) -> Result<Vec<Explanation>> {
11095        for key in [key_a, key_b] {
11096            if !mask.contains_node(self, key) {
11097                return Err(GraphError::KeyNotFound { key: key.into() });
11098            }
11099        }
11100        Ok(self
11101            .explain(key_a, key_b)?
11102            .into_iter()
11103            .filter_map(|e| self.scoped_explanation(e, mask))
11104            .collect())
11105    }
11106
11107    /// `e` as a caller limited to `mask` may have it, or `None` when it must be
11108    /// dropped entirely.
11109    ///
11110    /// Every non-via-hop explanation passes through untouched: its only nodes
11111    /// are the two subjects, which [`explain_scoped`](Self::explain_scoped) has
11112    /// already checked, and its weight is scored over that pair alone.
11113    ///
11114    /// A via-hop explanation is kept only when some via node the caller may see
11115    /// satisfies the rule on its own — and then its weight is recomputed as the
11116    /// max over exactly those vias. The engine writes the max over **all** of
11117    /// them (`core-rules::engine`, `best = prev.max(score)`), so passing the
11118    /// stored number through would let a hidden node set a figure the caller
11119    /// reads: the same disclosure dropping the explanation exists to prevent.
11120    ///
11121    /// A rule that stores no weight still reports none. The recomputed score is
11122    /// a sanitised version of a number `explain` already returned, never a new
11123    /// one — a scoped read must not say more than the unscoped read it narrows.
11124    fn scoped_explanation(
11125        &self,
11126        e: Explanation,
11127        mask: &crate::mask::NodeMask,
11128    ) -> Option<Explanation> {
11129        let Some(via_edge) = e.via_edge.clone() else {
11130            return Some(e);
11131        };
11132        let Some(rule_def) = self.engine.rules().find(|r| r.name == e.rule) else {
11133            // The rule is gone but its provenance is not; nothing can vouch for
11134            // the hop, so nothing is shown.
11135            return None;
11136        };
11137        let Some(via_label) = rule_def.via_label.as_deref() else {
11138            return Some(e);
11139        };
11140        let (Some(src), Some(dst)) = (self.ids.get(&e.src_key), self.ids.get(&e.dst_key)) else {
11141            return None;
11142        };
11143        let (Some(via_etype), Some(via_sym)) = (self.syms.get(&via_edge), self.syms.get(via_label))
11144        else {
11145            return None;
11146        };
11147        let via_dir = rule_def.via_dir.unwrap_or(Direction::Out);
11148        let props_view = build_props_view(&self.props, &self.base);
11149        let dst_get = |field: &str| props_view.get(dst, field).map(|vr| vr.into_value());
11150        let dst_view = NodeView {
11151            key: &e.dst_key,
11152            props: &dst_get,
11153        };
11154        // The rule's own namespace test, the one the engine applies to each via
11155        // candidate (`core-rules::engine::rule_sees_node`). Without it a
11156        // visible, out-of-namespace via — right label, satisfying predicate —
11157        // vouches for a hop the engine never made, and an explanation whose real
11158        // evidence is a hidden in-namespace node is kept.
11159        let rule_sees = |id: u32| match rule_def.namespace.as_deref() {
11160            None => true,
11161            Some(ns) => {
11162                let value = props_view.get(id, NS_PROP).map(|vr| vr.into_value());
11163                namespace_of_value(value.as_ref()) == ns
11164            }
11165        };
11166        let best = self
11167            .topo_view()
11168            .neighbors(via_etype, via_dir, src)
11169            .iter()
11170            .copied()
11171            .filter_map(|via| {
11172                if !mask.contains_id(via) {
11173                    return None;
11174                }
11175                if self.labels.get(via as usize).copied() != Some(via_sym) {
11176                    return None;
11177                }
11178                if !rule_sees(via) {
11179                    return None;
11180                }
11181                let via_key = self.ids.key_of(via)?;
11182                let via_get = |field: &str| props_view.get(via, field).map(|vr| vr.into_value());
11183                let via_view = NodeView {
11184                    key: via_key,
11185                    props: &via_get,
11186                };
11187                evaluate(&rule_def.predicate, &via_view, &dst_view)
11188            })
11189            .fold(None::<f64>, |best, score| {
11190                Some(match best {
11191                    None => score,
11192                    Some(prev) => prev.max(score),
11193                })
11194            })?;
11195        let weight = e.weight.map(|_| best);
11196        Some(Explanation { weight, ..e })
11197    }
11198
11199    pub fn neighbors(&self, key: &str, edge_type: &str, dir: Direction) -> Result<Vec<String>> {
11200        let id = self
11201            .ids
11202            .get(key)
11203            .ok_or_else(|| GraphError::KeyNotFound { key: key.into() })?;
11204        let Some(sym) = self.syms.get(edge_type) else {
11205            return Ok(Vec::new());
11206        };
11207        self.topo_view()
11208            .neighbors(sym, dir, id)
11209            .iter()
11210            .map(|&n| {
11211                self.ids
11212                    .key_of(n)
11213                    .map(|k| k.to_string())
11214                    .ok_or_else(|| GraphError::Corrupt {
11215                        detail: format!("topology id {n} has no key"),
11216                    })
11217            })
11218            .collect::<Result<Vec<_>>>()
11219    }
11220
11221    /// Unique directed degree of `key`. Unknown key → [`GraphError::KeyNotFound`].
11222    /// Unknown `edge_type` → 0. [`crate::algo::AlgoDir::Both`] is out + in (sum).
11223    pub fn degree(
11224        &self,
11225        key: &str,
11226        edge_type: Option<&str>,
11227        direction: crate::algo::AlgoDir,
11228    ) -> Result<u64> {
11229        let id = self
11230            .ids
11231            .get(key)
11232            .ok_or_else(|| GraphError::KeyNotFound { key: key.into() })?;
11233        let topo = self.topo_view();
11234        Ok(Self::unique_directed_degree(
11235            &topo, &self.syms, id, edge_type, direction,
11236        ))
11237    }
11238
11239    /// [`degree`](Self::degree) summing each pair's **insert count** instead of
11240    /// counting each pair once (§5.13).
11241    ///
11242    /// The unique degree asks how many neighbours there are; this asks how many
11243    /// times they were inserted. A pair with no recorded count contributes 1,
11244    /// so on a store that never called
11245    /// [`enable_multiplicity`](Self::enable_multiplicity) this returns exactly
11246    /// what [`degree`](Self::degree) returns rather than erroring — the
11247    /// distinction is a readout preference, not a demand the store cannot meet.
11248    ///
11249    /// `AlgoDir::Both` still sums out + in, so a pair visible on both sides
11250    /// still contributes twice: multiplicity changes what a pair is worth, never
11251    /// how a direction is counted.
11252    pub fn degree_multiplicity(
11253        &self,
11254        key: &str,
11255        edge_type: Option<&str>,
11256        direction: crate::algo::AlgoDir,
11257    ) -> Result<u64> {
11258        let id = self
11259            .ids
11260            .get(key)
11261            .ok_or_else(|| GraphError::KeyNotFound { key: key.into() })?;
11262        Ok(Self::multiplicity_directed_degree(
11263            &self.topo_view(),
11264            &self.edge_props_view(),
11265            &self.syms,
11266            id,
11267            edge_type,
11268            direction,
11269            None,
11270        ))
11271    }
11272
11273    /// [`degree_multiplicity`](Self::degree_multiplicity) under a scope.
11274    ///
11275    /// The sum covers **visible pairs only**. A hidden neighbour's inserts stay
11276    /// out of it for the reason
11277    /// [`degree_scoped`](Self::degree_scoped) documents, and more sharply: an
11278    /// unscoped multiplicity count discloses not only that a hidden neighbour
11279    /// exists but how often it was written.
11280    ///
11281    /// Hidden or unknown `key` → [`GraphError::KeyNotFound`].
11282    pub fn degree_scoped_multiplicity(
11283        &self,
11284        key: &str,
11285        edge_type: Option<&str>,
11286        direction: crate::algo::AlgoDir,
11287        mask: &crate::mask::NodeMask,
11288    ) -> Result<u64> {
11289        let id = self
11290            .ids
11291            .get(key)
11292            .filter(|&id| mask.contains_id(id))
11293            .ok_or_else(|| GraphError::KeyNotFound { key: key.into() })?;
11294        Ok(Self::multiplicity_directed_degree(
11295            &self.topo_view(),
11296            &self.edge_props_view(),
11297            &self.syms,
11298            id,
11299            edge_type,
11300            direction,
11301            Some(mask),
11302        ))
11303    }
11304
11305    /// [`degree`](Self::degree) counting **only neighbours the mask admits**.
11306    ///
11307    /// The filter is a correctness requirement, not an optimisation: an
11308    /// unfiltered count discloses the existence of a hidden neighbour to a
11309    /// caller who cannot see it, which is the same leak
11310    /// [`node_edges_scoped`](Self::node_edges_scoped) exists to prevent —
11311    /// reached by arithmetic instead of by name.
11312    ///
11313    /// Hidden or unknown `key` → [`GraphError::KeyNotFound`]. Unknown
11314    /// `edge_type` is still 0, as it is unscoped.
11315    pub fn degree_scoped(
11316        &self,
11317        key: &str,
11318        edge_type: Option<&str>,
11319        direction: crate::algo::AlgoDir,
11320        mask: &crate::mask::NodeMask,
11321    ) -> Result<u64> {
11322        let id = self
11323            .ids
11324            .get(key)
11325            .filter(|&id| mask.contains_id(id))
11326            .ok_or_else(|| GraphError::KeyNotFound { key: key.into() })?;
11327        let topo = self.topo_view();
11328        Ok(Self::visible_directed_degree(
11329            &topo, &self.syms, id, edge_type, direction, mask,
11330        ))
11331    }
11332
11333    /// Unique directed degree for a subset or a label scan.
11334    ///
11335    /// Unknown keys in `keys` are omitted (mask-like). `keys = Some(&[])` →
11336    /// empty `Ok(vec![])`. `limit` is applied after sorting degree desc, key
11337    /// asc, and only when `Some`. Invalid `where_` → `QueryError`.
11338    #[allow(clippy::too_many_arguments)]
11339    pub fn degrees(
11340        &self,
11341        keys: Option<&[String]>,
11342        label: Option<&str>,
11343        where_: Option<&PropPredicate>,
11344        edge_type: Option<&str>,
11345        direction: crate::algo::AlgoDir,
11346        limit: Option<usize>,
11347    ) -> Result<Vec<(String, u64)>> {
11348        self.degrees_inner(
11349            keys, label, where_, edge_type, direction, limit, None, false,
11350        )
11351    }
11352
11353    /// [`degrees`](Self::degrees) reporting each row's **insert-count** sum
11354    /// instead of its unique neighbour count, as
11355    /// [`degree_multiplicity`](Self::degree_multiplicity) does for one key.
11356    ///
11357    /// The sort is still degree descending, key ascending — over the counts this
11358    /// reading produces — and `limit` still applies after it.
11359    #[allow(clippy::too_many_arguments)]
11360    pub fn degrees_multiplicity(
11361        &self,
11362        keys: Option<&[String]>,
11363        label: Option<&str>,
11364        where_: Option<&PropPredicate>,
11365        edge_type: Option<&str>,
11366        direction: crate::algo::AlgoDir,
11367        limit: Option<usize>,
11368    ) -> Result<Vec<(String, u64)>> {
11369        self.degrees_inner(keys, label, where_, edge_type, direction, limit, None, true)
11370    }
11371
11372    /// [`degrees_scoped`](Self::degrees_scoped) reporting insert counts.
11373    ///
11374    /// Both filters apply: a hidden key stays out of the result, and every
11375    /// row's sum covers its **visible** pairs only.
11376    #[allow(clippy::too_many_arguments)]
11377    pub fn degrees_scoped_multiplicity(
11378        &self,
11379        keys: Option<&[String]>,
11380        label: Option<&str>,
11381        where_: Option<&PropPredicate>,
11382        edge_type: Option<&str>,
11383        direction: crate::algo::AlgoDir,
11384        limit: Option<usize>,
11385        mask: &crate::mask::NodeMask,
11386    ) -> Result<Vec<(String, u64)>> {
11387        self.degrees_inner(
11388            keys,
11389            label,
11390            where_,
11391            edge_type,
11392            direction,
11393            limit,
11394            Some(mask),
11395            true,
11396        )
11397    }
11398
11399    /// [`degrees`](Self::degrees) with the scope applied on both sides: a hidden
11400    /// key is omitted from the input — whether it arrived in `keys` or came out
11401    /// of the `label`/`where_` scan — and every row's count is the count of its
11402    /// **visible** neighbours, for the reason
11403    /// [`degree_scoped`](Self::degree_scoped) documents.
11404    ///
11405    /// Unlike `degree_scoped`, a hidden key here is not
11406    /// [`GraphError::KeyNotFound`]: `degrees` already drops unknown keys
11407    /// silently, so hidden and absent stay one answer by staying out of the
11408    /// result. `limit` still applies after the sort, and so counts visible rows.
11409    #[allow(clippy::too_many_arguments)]
11410    pub fn degrees_scoped(
11411        &self,
11412        keys: Option<&[String]>,
11413        label: Option<&str>,
11414        where_: Option<&PropPredicate>,
11415        edge_type: Option<&str>,
11416        direction: crate::algo::AlgoDir,
11417        limit: Option<usize>,
11418        mask: &crate::mask::NodeMask,
11419    ) -> Result<Vec<(String, u64)>> {
11420        self.degrees_inner(
11421            keys,
11422            label,
11423            where_,
11424            edge_type,
11425            direction,
11426            limit,
11427            Some(mask),
11428            false,
11429        )
11430    }
11431
11432    /// The body shared by [`degrees`](Self::degrees) and
11433    /// [`degrees_scoped`](Self::degrees_scoped). `mask = None` is the unscoped
11434    /// contract unchanged.
11435    #[allow(clippy::too_many_arguments)]
11436    fn degrees_inner(
11437        &self,
11438        keys: Option<&[String]>,
11439        label: Option<&str>,
11440        where_: Option<&PropPredicate>,
11441        edge_type: Option<&str>,
11442        direction: crate::algo::AlgoDir,
11443        limit: Option<usize>,
11444        mask: Option<&crate::mask::NodeMask>,
11445        multiplicity: bool,
11446    ) -> Result<Vec<(String, u64)>> {
11447        if let Some(pred) = where_ {
11448            pred.validate_named("where")
11449                .map_err(|detail| GraphError::QueryError { detail })?;
11450        }
11451        if matches!(keys, Some(ks) if ks.is_empty()) {
11452            return Ok(Vec::new());
11453        }
11454        let view = self.view();
11455        let ids: Vec<u32> = match keys {
11456            Some(ks) => {
11457                let mut seen = HashSet::new();
11458                let mut out = Vec::new();
11459                for k in ks {
11460                    let Some(id) = view.ids.get(k) else {
11461                        continue;
11462                    };
11463                    if !seen.insert(id) {
11464                        continue;
11465                    }
11466                    if let Some(pred) = where_ {
11467                        let holds = match view.prop(id, &pred.field) {
11468                            None => pred.holds(None),
11469                            Some(vr) => pred.holds(Some(vr.as_value())),
11470                        };
11471                        if !holds {
11472                            continue;
11473                        }
11474                    }
11475                    out.push(id);
11476                }
11477                out
11478            }
11479            None => Self::vector_candidates(&view, label, where_),
11480        };
11481        let mut out: Vec<(String, u64)> = ids
11482            .into_iter()
11483            // A hidden candidate leaves as quietly as an unknown key does.
11484            .filter(|&id| mask.is_none_or(|m| m.contains_id(id)))
11485            .filter_map(|id| {
11486                let key = self.ids.key_of(id)?.to_string();
11487                let deg = match (multiplicity, mask) {
11488                    // The same `view` the unique arms read, so the per-row
11489                    // rebuild F9 measured is gone and all three arms agree on
11490                    // the state they are reading.
11491                    (true, m) => Self::multiplicity_directed_degree(
11492                        &view.topo,
11493                        &view.edge_props,
11494                        view.syms,
11495                        id,
11496                        edge_type,
11497                        direction,
11498                        m,
11499                    ),
11500                    (false, Some(m)) => Self::visible_directed_degree(
11501                        &view.topo, view.syms, id, edge_type, direction, m,
11502                    ),
11503                    (false, None) => Self::unique_directed_degree(
11504                        &view.topo, view.syms, id, edge_type, direction,
11505                    ),
11506                };
11507                Some((key, deg))
11508            })
11509            .collect();
11510        out.sort_by(|a, b| b.1.cmp(&a.1).then_with(|| a.0.cmp(&b.0)));
11511        if let Some(lim) = limit {
11512            out.truncate(lim);
11513        }
11514        Ok(out)
11515    }
11516
11517    /// Unique neighbour count for `id` across `edge_type` (or all types) and
11518    /// `direction`. Unknown `edge_type` → 0. `Both` sums out + in.
11519    fn unique_directed_degree(
11520        topo: &TopologyView<'_>,
11521        syms: &Interner,
11522        id: u32,
11523        edge_type: Option<&str>,
11524        direction: crate::algo::AlgoDir,
11525    ) -> u64 {
11526        let dirs: &[Direction] = match direction {
11527            crate::algo::AlgoDir::Out => &[Direction::Out],
11528            crate::algo::AlgoDir::In => &[Direction::In],
11529            crate::algo::AlgoDir::Both => &[Direction::Out, Direction::In],
11530        };
11531        match edge_type {
11532            Some(name) => {
11533                let Some(et) = syms.get(name) else {
11534                    return 0;
11535                };
11536                dirs.iter().map(|&d| topo.degree(et, d, id) as u64).sum()
11537            }
11538            None => topo
11539                .etypes()
11540                .map(|et| {
11541                    dirs.iter()
11542                        .map(|&d| topo.degree(et, d, id) as u64)
11543                        .sum::<u64>()
11544                })
11545                .sum(),
11546        }
11547    }
11548
11549    /// [`unique_directed_degree`](Self::unique_directed_degree) counting only
11550    /// neighbours `mask` admits.
11551    ///
11552    /// Same shape, one substitution: `topo.degree` is a length, so it cannot be
11553    /// filtered; the neighbour list it measures can. `Both` still sums out + in,
11554    /// so a node visible on both sides still counts twice — the filter changes
11555    /// which neighbours are counted, never how a degree is defined.
11556    fn visible_directed_degree(
11557        topo: &TopologyView<'_>,
11558        syms: &Interner,
11559        id: u32,
11560        edge_type: Option<&str>,
11561        direction: crate::algo::AlgoDir,
11562        mask: &crate::mask::NodeMask,
11563    ) -> u64 {
11564        let dirs: &[Direction] = match direction {
11565            crate::algo::AlgoDir::Out => &[Direction::Out],
11566            crate::algo::AlgoDir::In => &[Direction::In],
11567            crate::algo::AlgoDir::Both => &[Direction::Out, Direction::In],
11568        };
11569        let visible = |et: u32| -> u64 {
11570            dirs.iter()
11571                .map(|&d| {
11572                    topo.neighbors(et, d, id)
11573                        .iter()
11574                        .filter(|&&n| mask.contains_id(n))
11575                        .count() as u64
11576                })
11577                .sum()
11578        };
11579        match edge_type {
11580            Some(name) => syms.get(name).map_or(0, visible),
11581            None => topo.etypes().map(visible).sum(),
11582        }
11583    }
11584
11585    /// Sum of the insert counts of `id`'s pairs (§5.13), over `edge_type` (or
11586    /// all types) and `direction`, restricted to what `mask` admits when one is
11587    /// given.
11588    ///
11589    /// The same neighbour lists the unique reading measures, with each entry
11590    /// worth its pair's count rather than worth 1 — so the filter decides which
11591    /// pairs are in the sum and the count decides what each contributes. A
11592    /// direction decides which way round the pair is addressed: an `In`
11593    /// neighbour `n` of `id` is the pair `(et, n, id)`.
11594    ///
11595    /// Takes its views as parameters, exactly as the unique helpers do, because
11596    /// it is called once per row from a label scan. `edge_props_view()` reaches
11597    /// into the mmap'd base's rkyv section on every call, so building the two
11598    /// views inside made an N-row `degrees(multiplicity=True)` do N section
11599    /// accesses where the unique reading does one: worth 2.57 ms of 16.68 ms
11600    /// over 20 000 rows, about 0.13 us per row (defect #29,
11601    /// `tests/f9_bench.rs`). Most of that call's cost is the per-neighbour
11602    /// count lookup and is inherent, so this is a hoist, not a rescue.
11603    ///
11604    /// The views are exactly `self.view()`'s own `topo` and `edge_props`, so a
11605    /// caller that already has a view passes its halves and reads the same
11606    /// state it reads everything else from.
11607    fn multiplicity_directed_degree(
11608        topo: &TopologyView<'_>,
11609        edge_props: &EdgePropsView<'_>,
11610        syms: &Interner,
11611        id: u32,
11612        edge_type: Option<&str>,
11613        direction: crate::algo::AlgoDir,
11614        mask: Option<&crate::mask::NodeMask>,
11615    ) -> u64 {
11616        let dirs: &[Direction] = match direction {
11617            crate::algo::AlgoDir::Out => &[Direction::Out],
11618            crate::algo::AlgoDir::In => &[Direction::In],
11619            crate::algo::AlgoDir::Both => &[Direction::Out, Direction::In],
11620        };
11621        let count_of = |et: u32, src: u32, dst: u32| -> u64 {
11622            match edge_props.get(et, src, dst, EDGE_COUNT_PROP) {
11623                Some(Value::Int(n)) if n > 0 => n as u64,
11624                _ => 1,
11625            }
11626        };
11627        let per_etype = |et: u32| -> u64 {
11628            dirs.iter()
11629                .map(|&d| {
11630                    topo.neighbors(et, d, id)
11631                        .iter()
11632                        .filter(|&&n| mask.is_none_or(|m| m.contains_id(n)))
11633                        .map(|&n| match d {
11634                            Direction::Out => count_of(et, id, n),
11635                            Direction::In => count_of(et, n, id),
11636                        })
11637                        .sum::<u64>()
11638                })
11639                .sum()
11640        };
11641        match edge_type {
11642            Some(name) => syms.get(name).map_or(0, per_etype),
11643            None => topo.etypes().map(per_etype).sum(),
11644        }
11645    }
11646
11647    /// Return the last-change commit sequence for `key`, or `None` if the node
11648    /// does not exist or has never been mutated since the last V5-V7 snapshot
11649    /// (horizon-bounded for legacy stores).
11650    ///
11651    /// The returned sequence is a monotonically increasing counter that starts
11652    /// at 1 for the first commit after `open` and increments with every
11653    /// successful write.  WAL replay at open also assigns sequences (1..N for N
11654    /// replayed frames), so sequences are consistent across snapshot+WAL cycles.
11655    ///
11656    /// For V5-V7 stores opened without a V8 snapshot, nodes that were present
11657    /// in the snapshot but not touched by any WAL frame will return `None`
11658    /// (horizon-bounded: CAS against such nodes is only safe after the first
11659    /// V8 snapshot or after the node is next mutated).
11660    pub fn last_changed(&self, key: &str) -> Option<u64> {
11661        let id = self.ids.get(key)?;
11662        self.last_change.get(&id).copied()
11663    }
11664
11665    /// Which loaded store this handle is.
11666    ///
11667    /// Paired with [`commit_seq`](GraphDb::commit_seq) it identifies a graph
11668    /// state outright, which `commit_seq` alone does not: two stores of the same
11669    /// age share a sequence, and a reload can return to one. Memos of dense
11670    /// node ids that the handle does not own are stamped with both; see
11671    /// [`StoreStamp`](crate::mask::StoreStamp).
11672    pub(crate) fn store_id(&self) -> crate::mask::StoreId {
11673        self.store_id
11674    }
11675
11676    /// The current commit sequence (number of successful commits since open,
11677    /// including WAL replay frames).  Useful for recording a baseline before
11678    /// a read-modify-write cycle.
11679    pub fn commit_seq(&self) -> u64 {
11680        self.commit_seq
11681    }
11682
11683    /// Check that all `preconds` are satisfied against the current db state.
11684    /// Returns `Err(GraphError::CasConflict)` on the first failing precondition.
11685    pub(crate) fn check_preconditions(&self, preconds: &[Precondition]) -> Result<()> {
11686        for precond in preconds {
11687            match precond {
11688                Precondition::NodeUnchangedSince { key, expected } => {
11689                    // Missing entry means the node predates the WAL window or
11690                    // does not exist; treat as 0 (before any commit).
11691                    let actual = self.last_changed(key).unwrap_or_default();
11692                    if actual != *expected {
11693                        return Err(GraphError::CasConflict {
11694                            key: key.clone(),
11695                            expected: *expected,
11696                            actual,
11697                        });
11698                    }
11699                }
11700                Precondition::NodeAbsent { key } => {
11701                    // Node must not exist (not live).
11702                    if self.ids.get(key).is_some() {
11703                        let actual = self.last_changed(key).unwrap_or(0);
11704                        return Err(GraphError::CasConflict {
11705                            key: key.clone(),
11706                            expected: u64::MAX,
11707                            actual,
11708                        });
11709                    }
11710                }
11711            }
11712        }
11713        Ok(())
11714    }
11715
11716    /// Apply a batch of mutations with compare-and-set preconditions.
11717    ///
11718    /// All preconditions are checked atomically before any operation is applied.
11719    /// If any precondition fails, the entire batch is rejected with
11720    /// [`GraphError::CasConflict`] and no WAL frame is written.
11721    ///
11722    /// # Returns
11723    /// `(nodes_inserted, edges_inserted)` on success, same as [`write_batch`].
11724    ///
11725    /// # Errors
11726    /// - [`GraphError::CasConflict`] if any precondition is not satisfied.
11727    /// - Any error that [`write_batch`] would return for the ops themselves.
11728    pub fn write_batch_cas(
11729        &mut self,
11730        preconds: Vec<Precondition>,
11731        ops: Vec<BatchOp>,
11732    ) -> Result<(usize, usize)> {
11733        self.check_preconditions(&preconds)?;
11734        self.commit_logged_batch(ops, None, None).map(inserted_pair)
11735    }
11736
11737    /// Update the per-node last-change map for a WAL record at commit `seq`.
11738    ///
11739    /// Called after a successful apply to record which nodes were touched.
11740    /// For replay, called with the WAL-frame's replayed seq.
11741    ///
11742    /// Touch definition (see [`Precondition`] doc):
11743    /// - InsertNode / InsertNodeId / SetProp / SetPropId / RemoveProp → the node.
11744    /// - InsertEdge / InsertEdgeId / DeleteEdge → both src and dst.
11745    /// - DeleteNode → node tombstoned; last_changed() returns None so no update needed.
11746    /// - DerivedEdge markers, Intern, rule/view records → no-ops.
11747    /// - Batch → recurse into inner records.
11748    fn update_last_change_from_rec(&mut self, rec: &WalRecord, seq: u64) {
11749        match rec {
11750            WalRecord::InsertNode { key, .. }
11751            | WalRecord::SetProp { key, .. }
11752            | WalRecord::RemoveProp { key, .. } => {
11753                if let Some(id) = self.ids.get(key) {
11754                    self.last_change.insert(id, seq);
11755                }
11756            }
11757            WalRecord::InsertNodeId { key, .. } => {
11758                if let Some(id) = self.ids.get(key) {
11759                    self.last_change.insert(id, seq);
11760                }
11761            }
11762            WalRecord::SetPropId { id, .. } => {
11763                self.last_change.insert(*id, seq);
11764            }
11765            WalRecord::InsertEdge {
11766                src_key, dst_key, ..
11767            }
11768            | WalRecord::DeleteEdge {
11769                src_key, dst_key, ..
11770            } => {
11771                if let Some(src_id) = self.ids.get(src_key) {
11772                    self.last_change.insert(src_id, seq);
11773                }
11774                if let Some(dst_id) = self.ids.get(dst_key) {
11775                    self.last_change.insert(dst_id, seq);
11776                }
11777            }
11778            WalRecord::InsertEdgeId { src, dst, .. } => {
11779                self.last_change.insert(*src, seq);
11780                self.last_change.insert(*dst, seq);
11781            }
11782            // A count record touches the pair, so it touches both endpoints —
11783            // the same reading `InsertEdgeId` gets, because a duplicate insert
11784            // that raises the count *is* a mutation of that pair. The opt-in
11785            // declaration touches nothing.
11786            WalRecord::SetEdgeCount { src, dst, .. } if !rec.is_multiplicity_decl() => {
11787                self.last_change.insert(*src, seq);
11788                self.last_change.insert(*dst, seq);
11789            }
11790            WalRecord::SetEdgeCount { .. } => {}
11791            // DeleteNode: node is tombstoned; last_changed(key) returns None for
11792            // deleted keys (ids.get() returns None post-tombstone), so no update needed.
11793            // History markers: state no-ops; the underlying mutation already
11794            // touched the relevant nodes' last_change entries.
11795            WalRecord::DeleteNode { .. }
11796            | WalRecord::DerivedEdgeAdded { .. }
11797            | WalRecord::DerivedEdgeRetracted { .. }
11798            | WalRecord::Intern { .. }
11799            | WalRecord::CreateRule { .. }
11800            | WalRecord::DeleteRule { .. }
11801            | WalRecord::RebuildRule { .. }
11802            | WalRecord::CreateView { .. }
11803            | WalRecord::DeleteView { .. }
11804            | WalRecord::EnableFulltext { .. }
11805            | WalRecord::DisableFulltext { .. }
11806            | WalRecord::EnableIndex { .. }
11807            | WalRecord::DisableIndex { .. } => {}
11808            // RenameNode: node id is stable; update last_change via the new key.
11809            // Called after apply(), so ids already reflects new_key.
11810            WalRecord::RenameNode { new_key, .. } => {
11811                if let Some(id) = self.ids.get(new_key) {
11812                    self.last_change.insert(id, seq);
11813                }
11814            }
11815            WalRecord::Batch(inner) => {
11816                for inner_rec in inner {
11817                    self.update_last_change_from_rec(inner_rec, seq);
11818                }
11819            }
11820        }
11821    }
11822
11823    pub fn node_count(&self) -> usize {
11824        self.ids.len()
11825    }
11826
11827    /// Configure archive retention: keep the `N` newest WAL archives at each
11828    /// [`snapshot_with`] call when `archive_wal: true`.
11829    ///
11830    /// `Some(N)` where N > 0 → prune oldest archives keeping the newest N.
11831    /// `Some(0)` or `None` → unlimited (no pruning).
11832    ///
11833    /// Pruning only ever happens inside [`snapshot_with`]; this method only
11834    /// stores the policy.  Archives below the retention limit are deleted
11835    /// oldest-first.  The horizon floor is updated so that
11836    /// [`was_linked`] / history APIs return `CommitOutOfRange` for commits
11837    /// in pruned archives rather than silently returning wrong data.
11838    pub fn set_wal_archive_retention(&mut self, keep: Option<u32>) {
11839        self.wal_archive_retention = keep;
11840    }
11841
11842    /// Delete any WAL archives that are fully below the current horizon floor.
11843    ///
11844    /// Orphaned archives arise when the floor is written first during retention
11845    /// pruning and then a crash interrupts the archive-delete sequence.  The
11846    /// opening cleanup ensures no subsequent read path sees stale data.
11847    ///
11848    /// Under the monotonic naming scheme, the archive name N equals the
11849    /// cumulative end-frame index of the archive in global commit space (i.e.
11850    /// the archive covers global frames `[prev_n, N)`).  An archive is
11851    /// fully orphaned when `N <= wal_horizon_floor`: all of its frames fall
11852    /// below the floor and have already been counted in it.
11853    fn cleanup_orphaned_archives(&mut self) -> Result<()> {
11854        if self.wal_horizon_floor == 0 {
11855            // Floor at 0 means no pruning has ever occurred; nothing to clean.
11856            return Ok(());
11857        }
11858        let archive_ns = self.fs.list_archives()?;
11859        for n in archive_ns {
11860            if n <= self.wal_horizon_floor {
11861                // Archive N ends at global frame N; all its frames are below
11862                // the floor (floor already accounts for them) → orphaned.
11863                self.fs.delete_archive(n).map_err(GraphError::Io)?;
11864            } else {
11865                // Archives are sorted ascending; first one above floor stops scan.
11866                break;
11867            }
11868        }
11869        Ok(())
11870    }
11871
11872    /// Collect all WAL frames from surviving archives (oldest-first) then the
11873    /// live WAL into one flat list, and return the total along with the number
11874    /// of archive frames at the front of the list.
11875    ///
11876    /// Commit indices into the returned list are LOCAL (0 = first frame of
11877    /// oldest surviving archive).  To obtain the GLOBAL index add
11878    /// `self.wal_horizon_floor`.
11879    fn all_frames(&self) -> Result<(Vec<WalRecord>, u64)> {
11880        let archive_ns = self.fs.list_archives()?;
11881        let mut all: Vec<WalRecord> = Vec::new();
11882        for n in archive_ns {
11883            let bytes = self.fs.read_archive(n)?;
11884            let (frames, _) = decode_all(&bytes);
11885            all.extend(frames);
11886        }
11887        let archive_count = all.len() as u64;
11888        let live_bytes = self.fs.read(FileId::Wal)?;
11889        let (live_frames, _) = decode_all(&live_bytes);
11890        all.extend(live_frames);
11891        Ok((all, archive_count))
11892    }
11893
11894    /// Return the total number of committed WAL frames visible in the current
11895    /// horizon window, including frames in surviving WAL archives.
11896    ///
11897    /// This is the exclusive upper bound for valid `at_commit` indices in
11898    /// `was_linked`. Valid indices are `wal_horizon_floor()..wal_total_commits()`.
11899    ///
11900    /// Returns the horizon floor when all surviving history is empty.
11901    pub fn wal_total_commits(&self) -> Result<u64> {
11902        let (frames, _) = self.all_frames()?;
11903        Ok(self.wal_horizon_floor + frames.len() as u64)
11904    }
11905
11906    /// The global frame index of the first commit reachable through surviving
11907    /// archives (0 when no archives have been pruned).
11908    pub fn wal_horizon_floor(&self) -> u64 {
11909        self.wal_horizon_floor
11910    }
11911
11912    /// Return the per-node change history for `key` by scanning the on-disk WAL.
11913    ///
11914    /// ## Horizon
11915    ///
11916    /// History reaches back only to the last WAL-truncating snapshot, exactly like `open_at`.
11917    /// Snapshots written with `keep_wal: true` preserve deeper history. This is the honest,
11918    /// zero-cost contract; a durable history log is out of scope.
11919    ///
11920    /// ## Derived edges
11921    ///
11922    /// Rule-created (derived) edges are **not** in the WAL and therefore do not appear in
11923    /// history. Only edges written directly by the application are recorded.
11924    ///
11925    /// ## Deleted nodes
11926    ///
11927    /// For nodes that have been deleted, dense-id records (SetPropId, InsertEdgeId) that
11928    /// predate the deletion may not resolve (the id is tombstoned in the live map). The
11929    /// string-keyed `DeleteNode` record still matches and produces a `NodeDeleted` entry.
11930    /// Prop/edge history of a deleted node may therefore be partially unresolvable.
11931    ///
11932    /// ## Dense-id edge entries and tombstoned partners
11933    ///
11934    /// Edge entries from dense-id WAL records (`InsertEdgeId`) are omitted when the partner
11935    /// endpoint's dense id is tombstoned. As a result, a live node's history can contain an
11936    /// `EdgeRemoved` (string-keyed, always resolves) without a corresponding `EdgeAdded`.
11937    /// Build commit-bounded alias intervals for `queried_key`.
11938    ///
11939    /// Returns a list of `(key, valid_from_inclusive, valid_until_exclusive)` tuples.
11940    /// A record written under `key` at commit `c` matches the queried identity iff
11941    /// `c >= valid_from && (valid_until.is_none() || c < valid_until)`.
11942    ///
11943    /// Each alias entry carries both a lower and an upper bound so that key-reuse
11944    /// after a rename is handled correctly: if "a" is renamed to "b" at commit 5,
11945    /// then a NEW node is created as "a" at commit 7 and renamed to "c" at commit 10,
11946    /// querying "c" must NOT surface identity-1's events (commits 0–4 under "a");
11947    /// only identity-2's events (commits 7–9 under "a") are in scope.
11948    ///
11949    /// Only **forward aliasing**: querying the *new* key surfaces events written
11950    /// under the *old* key.  The reverse direction is not supported.
11951    fn build_key_alias_intervals(
11952        &self,
11953        frames: &[core_storage::wal::WalRecord],
11954        queried_key: &str,
11955    ) -> Vec<(String, u64, Option<u64>)> {
11956        use core_storage::wal::WalRecord;
11957
11958        // Pre-pass: build reverse_rename and key_starts maps.
11959        let mut reverse_rename: HashMap<String, (String, u64)> = HashMap::new();
11960        let mut key_starts: HashMap<String, Vec<u64>> = HashMap::new();
11961
11962        for (local_i, frame) in frames.iter().enumerate() {
11963            let commit = self.wal_horizon_floor + local_i as u64;
11964            let records: &[WalRecord] = match frame {
11965                WalRecord::Batch(inner) => inner.as_slice(),
11966                single => std::slice::from_ref(single),
11967            };
11968            for rec in records {
11969                match rec {
11970                    WalRecord::InsertNode { key, .. } | WalRecord::InsertNodeId { key, .. } => {
11971                        key_starts.entry(key.clone()).or_default().push(commit);
11972                    }
11973                    WalRecord::RenameNode { old_key, new_key } => {
11974                        // new_key came into existence at this commit.
11975                        key_starts.entry(new_key.clone()).or_default().push(commit);
11976                        // Record the reverse rename: new_key was introduced by renaming old_key.
11977                        reverse_rename.insert(new_key.clone(), (old_key.clone(), commit));
11978                    }
11979                    _ => {}
11980                }
11981            }
11982        }
11983
11984        // Build alias intervals by following the reverse rename chain.
11985        let mut result: Vec<(String, u64, Option<u64>)> = Vec::new();
11986        let mut current_key = queried_key.to_string();
11987        let mut current_valid_until: Option<u64> = None;
11988
11989        loop {
11990            // valid_from: the most recent commit where current_key was assigned to this
11991            // identity.  For aliases (valid_until = Some(vu)), find the last start event
11992            // for the key strictly before vu — this is where the alias's occupancy by
11993            // this identity began, correctly excluding prior identities that reused the key.
11994            let valid_from = if let Some(vu) = current_valid_until {
11995                key_starts
11996                    .get(&current_key)
11997                    .and_then(|starts| starts.iter().rev().find(|&&s| s < vu).copied())
11998                    .unwrap_or(self.wal_horizon_floor)
11999            } else {
12000                // Queried key — no upper bound; may have been introduced at any commit.
12001                self.wal_horizon_floor
12002            };
12003
12004            result.push((current_key.clone(), valid_from, current_valid_until));
12005
12006            match reverse_rename.get(&current_key) {
12007                Some((old_key, rename_commit)) => {
12008                    current_valid_until = Some(*rename_commit);
12009                    current_key = old_key.clone();
12010                }
12011                None => break,
12012            }
12013        }
12014
12015        result
12016    }
12017
12018    /// Returns true if `record_key` matches any alias interval that covers `commit`.
12019    fn aliases_match(
12020        intervals: &[(String, u64, Option<u64>)],
12021        record_key: &str,
12022        commit: u64,
12023    ) -> bool {
12024        intervals
12025            .iter()
12026            .any(|(k, vf, vu)| k == record_key && commit >= *vf && vu.is_none_or(|u| commit < u))
12027    }
12028
12029    /// Return the change history of node `key` by scanning the on-disk WAL.
12030    ///
12031    /// ## Horizon
12032    ///
12033    /// History reaches back only as far as the retained WAL. The returned
12034    /// [`HistoryResult`](crate::history::HistoryResult) carries `total_commits`
12035    /// (the exclusive upper bound for valid commit indices) and `horizon` (the
12036    /// oldest commit still reachable). When `horizon > 0`, older events were
12037    /// pruned and are not in `items`.
12038    pub fn node_history(
12039        &self,
12040        key: &str,
12041    ) -> Result<crate::history::HistoryResult<crate::history::HistoryEntry>> {
12042        use crate::history::{HistoryChange, HistoryEntry, HistoryResult};
12043        use core_storage::wal::WalRecord;
12044
12045        let (frames, _) = self.all_frames()?;
12046        let total_commits = self.wal_horizon_floor + frames.len() as u64;
12047
12048        // Resolve commit-bounded alias intervals for `key` (handles renames in the WAL).
12049        let alias_intervals = self.build_key_alias_intervals(&frames, key);
12050
12051        let mut out: Vec<HistoryEntry> = Vec::new();
12052
12053        for (local_i, frame) in frames.iter().enumerate() {
12054            let commit = self.wal_horizon_floor + local_i as u64;
12055            // Collect the inner records to process — Batch is one commit, single records are one commit.
12056            let records: &[WalRecord] = match frame {
12057                WalRecord::Batch(inner) => inner.as_slice(),
12058                single => std::slice::from_ref(single),
12059            };
12060
12061            for rec in records {
12062                let change = match rec {
12063                    WalRecord::InsertNode { label, key: k, .. }
12064                        if Self::aliases_match(&alias_intervals, k, commit) =>
12065                    {
12066                        Some(HistoryChange::NodeInserted {
12067                            label: label.clone(),
12068                        })
12069                    }
12070                    WalRecord::InsertNodeId { label, key: k, .. }
12071                        if Self::aliases_match(&alias_intervals, k, commit) =>
12072                    {
12073                        let label_str = match self.syms.resolve(*label) {
12074                            Some(s) => s.to_string(),
12075                            None => continue,
12076                        };
12077                        Some(HistoryChange::NodeInserted { label: label_str })
12078                    }
12079                    WalRecord::SetProp {
12080                        key: k,
12081                        field,
12082                        value,
12083                    } if Self::aliases_match(&alias_intervals, k, commit) => {
12084                        Some(HistoryChange::PropSet {
12085                            field: field.clone(),
12086                            value: value.clone(),
12087                        })
12088                    }
12089                    WalRecord::SetPropId { id, field, value } => {
12090                        // Use key_of_historical (not key_of) so a node's prop_set
12091                        // events remain visible after the node is later deleted:
12092                        // key_of returns None for a tombstoned id, which would
12093                        // silently drop every PropSet between insert and delete.
12094                        // Mirrors the InsertEdgeId arm below and edge_history's
12095                        // own id-keyed arms.
12096                        match self.ids.key_of_historical(*id) {
12097                            // key_of_historical returns the last-known (possibly
12098                            // post-rename, possibly post-delete) key; compare to queried key.
12099                            Some(resolved) if resolved == key => {
12100                                let field_str = match self.syms.resolve(*field) {
12101                                    Some(s) => s.to_string(),
12102                                    None => continue,
12103                                };
12104                                Some(HistoryChange::PropSet {
12105                                    field: field_str,
12106                                    value: value.clone(),
12107                                })
12108                            }
12109                            _ => None,
12110                        }
12111                    }
12112                    WalRecord::RemoveProp { key: k, field }
12113                        if Self::aliases_match(&alias_intervals, k, commit) =>
12114                    {
12115                        Some(HistoryChange::PropRemoved {
12116                            field: field.clone(),
12117                        })
12118                    }
12119                    WalRecord::InsertEdge {
12120                        edge_type,
12121                        src_key,
12122                        dst_key,
12123                    } => {
12124                        if Self::aliases_match(&alias_intervals, src_key, commit) {
12125                            Some(HistoryChange::EdgeAdded {
12126                                edge_type: edge_type.clone(),
12127                                other: dst_key.clone(),
12128                                outgoing: true,
12129                            })
12130                        } else if Self::aliases_match(&alias_intervals, dst_key, commit) {
12131                            Some(HistoryChange::EdgeAdded {
12132                                edge_type: edge_type.clone(),
12133                                other: src_key.clone(),
12134                                outgoing: false,
12135                            })
12136                        } else {
12137                            None
12138                        }
12139                    }
12140                    WalRecord::InsertEdgeId { etype, src, dst } => {
12141                        let etype_str = match self.syms.resolve(*etype) {
12142                            Some(s) => s.to_string(),
12143                            None => continue,
12144                        };
12145                        // key_of_historical (not key_of): an edge added before
12146                        // either endpoint was later deleted must still resolve —
12147                        // see the SetPropId arm above and edge_history's
12148                        // InsertEdgeId arm, which use the same lookup for the
12149                        // same reason.
12150                        let src_key = self.ids.key_of_historical(*src);
12151                        let dst_key = self.ids.key_of_historical(*dst);
12152                        if src_key == Some(key) {
12153                            let other = match dst_key {
12154                                Some(s) => s.to_string(),
12155                                None => continue,
12156                            };
12157                            Some(HistoryChange::EdgeAdded {
12158                                edge_type: etype_str,
12159                                other,
12160                                outgoing: true,
12161                            })
12162                        } else if dst_key == Some(key) {
12163                            let other = match src_key {
12164                                Some(s) => s.to_string(),
12165                                None => continue,
12166                            };
12167                            Some(HistoryChange::EdgeAdded {
12168                                edge_type: etype_str,
12169                                other,
12170                                outgoing: false,
12171                            })
12172                        } else {
12173                            None
12174                        }
12175                    }
12176                    WalRecord::DeleteEdge {
12177                        edge_type,
12178                        src_key,
12179                        dst_key,
12180                    } => {
12181                        if Self::aliases_match(&alias_intervals, src_key, commit) {
12182                            Some(HistoryChange::EdgeRemoved {
12183                                edge_type: edge_type.clone(),
12184                                other: dst_key.clone(),
12185                                outgoing: true,
12186                            })
12187                        } else if Self::aliases_match(&alias_intervals, dst_key, commit) {
12188                            Some(HistoryChange::EdgeRemoved {
12189                                edge_type: edge_type.clone(),
12190                                other: src_key.clone(),
12191                                outgoing: false,
12192                            })
12193                        } else {
12194                            None
12195                        }
12196                    }
12197                    WalRecord::DeleteNode { key: k }
12198                        if Self::aliases_match(&alias_intervals, k, commit) =>
12199                    {
12200                        Some(HistoryChange::NodeDeleted)
12201                    }
12202                    // Skip: rule/view/fulltext/intern metadata; Batch wrapper handled above.
12203                    _ => None,
12204                };
12205
12206                if let Some(change) = change {
12207                    out.push(HistoryEntry { commit, change });
12208                }
12209            }
12210        }
12211
12212        Ok(HistoryResult {
12213            items: out,
12214            total_commits,
12215            horizon: self.wal_horizon_floor,
12216        })
12217    }
12218
12219    /// Return the per-edge change history between nodes `a` and `b` by scanning
12220    /// the on-disk WAL.
12221    ///
12222    /// ## Horizon
12223    ///
12224    /// History reaches back only to the last WAL-truncating snapshot, exactly
12225    /// like `node_history` and `open_at`. The returned [`HistoryResult`] carries
12226    /// `total_commits` (= number of WAL frames), which is the exclusive upper
12227    /// bound for valid commit indices.
12228    ///
12229    /// ## Derived edges
12230    ///
12231    /// Rule-derived edges appear via `DerivedEdgeAdded` / `DerivedEdgeRetracted`
12232    /// WAL markers written by `log_then_apply_with` after each rule-firing
12233    /// mutation. The `rule` field of those events carries the rule name.
12234    ///
12235    /// ## DeleteNode
12236    ///
12237    /// When a node is deleted, its manual incident edges are swept inline without
12238    /// individual `DeleteEdge` WAL records. `edge_history` detects `DeleteNode`
12239    /// events for either endpoint and synthesises `Retracted(rule:None)` events
12240    /// for each manual edge that was active at that point. Derived edges active at
12241    /// the time of deletion are handled by the `DerivedEdgeRetracted` marker that
12242    /// the engine appends immediately after the `DeleteNode` record; those events
12243    /// carry correct rule attribution and are emitted by the marker arm, not the
12244    /// synthetic sweep.
12245    ///
12246    /// ## Masks
12247    ///
12248    /// Like `node_history`, this method has no mask parameter and returns WAL
12249    /// history regardless of any role mask. For masked history semantics, apply
12250    /// the mask at the caller level.
12251    pub fn edge_history(
12252        &self,
12253        a: &str,
12254        b: &str,
12255    ) -> Result<crate::history::HistoryResult<crate::history::EdgeHistoryEvent>> {
12256        use crate::history::{EdgeEvent, EdgeHistoryEvent, HistoryResult};
12257        use core_storage::wal::WalRecord;
12258
12259        let (frames, _) = self.all_frames()?;
12260        let total_commits = self.wal_horizon_floor + frames.len() as u64;
12261
12262        // Resolve all historical names for a and b (handles RenameNode in the WAL).
12263        // Intervals are commit-bounded so recycled keys don't contaminate histories.
12264        let alias_a = self.build_key_alias_intervals(&frames, a);
12265        let alias_b = self.build_key_alias_intervals(&frames, b);
12266
12267        // Active edges between a and b tracked as (edge_type, src_key, dst_key, is_derived).
12268        // The is_derived flag is used by the DeleteNode sweep: manual edges are
12269        // swept with a synthetic Retracted(rule:None); derived edges are skipped
12270        // because the engine writes a DerivedEdgeRetracted marker immediately after
12271        // the DeleteNode record, which carries the correct rule attribution.
12272        let mut active: Vec<(String, String, String, bool)> = Vec::new();
12273        let mut out: Vec<EdgeHistoryEvent> = Vec::new();
12274
12275        for (local_i, frame) in frames.iter().enumerate() {
12276            let commit = self.wal_horizon_floor + local_i as u64;
12277            let records: &[WalRecord] = match frame {
12278                WalRecord::Batch(inner) => inner.as_slice(),
12279                single => std::slice::from_ref(single),
12280            };
12281
12282            for rec in records {
12283                match rec {
12284                    WalRecord::InsertEdge {
12285                        edge_type,
12286                        src_key,
12287                        dst_key,
12288                    } => {
12289                        let is_ab = Self::aliases_match(&alias_a, src_key, commit)
12290                            && Self::aliases_match(&alias_b, dst_key, commit);
12291                        let is_ba = Self::aliases_match(&alias_b, src_key, commit)
12292                            && Self::aliases_match(&alias_a, dst_key, commit);
12293                        if is_ab || is_ba {
12294                            active.push((
12295                                edge_type.clone(),
12296                                src_key.clone(),
12297                                dst_key.clone(),
12298                                false,
12299                            ));
12300                            out.push(EdgeHistoryEvent {
12301                                edge_type: edge_type.clone(),
12302                                commit,
12303                                event: EdgeEvent::Added,
12304                                rule: None,
12305                            });
12306                        }
12307                    }
12308                    WalRecord::InsertEdgeId { etype, src, dst } => {
12309                        let etype_str = match self.syms.resolve(*etype) {
12310                            Some(s) => s.to_string(),
12311                            None => continue,
12312                        };
12313                        // Use key_of_historical so tombstoned nodes (deleted
12314                        // later in the WAL) still resolve during the scan.
12315                        let src_key = self.ids.key_of_historical(*src);
12316                        let dst_key = self.ids.key_of_historical(*dst);
12317                        let is_ab = src_key == Some(a) && dst_key == Some(b);
12318                        let is_ba = src_key == Some(b) && dst_key == Some(a);
12319                        if is_ab || is_ba {
12320                            let src_str = src_key.unwrap().to_string();
12321                            let dst_str = dst_key.unwrap().to_string();
12322                            active.push((etype_str.clone(), src_str, dst_str, false));
12323                            out.push(EdgeHistoryEvent {
12324                                edge_type: etype_str,
12325                                commit,
12326                                event: EdgeEvent::Added,
12327                                rule: None,
12328                            });
12329                        }
12330                    }
12331                    WalRecord::DeleteEdge {
12332                        edge_type,
12333                        src_key,
12334                        dst_key,
12335                    } => {
12336                        let is_ab = Self::aliases_match(&alias_a, src_key, commit)
12337                            && Self::aliases_match(&alias_b, dst_key, commit);
12338                        let is_ba = Self::aliases_match(&alias_b, src_key, commit)
12339                            && Self::aliases_match(&alias_a, dst_key, commit);
12340                        if is_ab || is_ba {
12341                            // Remove the first matching active entry (flag ignored).
12342                            if let Some(pos) = active.iter().position(|(et, s, d, _)| {
12343                                et == edge_type && s == src_key && d == dst_key
12344                            }) {
12345                                active.remove(pos);
12346                            }
12347                            out.push(EdgeHistoryEvent {
12348                                edge_type: edge_type.clone(),
12349                                commit,
12350                                event: EdgeEvent::Retracted,
12351                                rule: None,
12352                            });
12353                        }
12354                    }
12355                    WalRecord::DeleteNode { key: k }
12356                        if Self::aliases_match(&alias_a, k, commit)
12357                            || Self::aliases_match(&alias_b, k, commit) =>
12358                    {
12359                        // Sweep: implicitly retract only MANUAL active edges.
12360                        // Derived active edges are skipped here because the rule
12361                        // engine appends a DerivedEdgeRetracted marker immediately
12362                        // after this DeleteNode record; that marker produces the
12363                        // single correctly-attributed Retracted event.  Derived
12364                        // entries are dropped from `active` (the marker arm's
12365                        // idempotent retain finds nothing to remove).
12366                        for (et, _, _, is_derived) in active.drain(..) {
12367                            if !is_derived {
12368                                out.push(EdgeHistoryEvent {
12369                                    edge_type: et,
12370                                    commit,
12371                                    event: EdgeEvent::Retracted,
12372                                    rule: None,
12373                                });
12374                            }
12375                            // Derived: drop silently; marker carries the Retracted event.
12376                        }
12377                    }
12378                    WalRecord::DerivedEdgeAdded {
12379                        rule,
12380                        edge_type: et,
12381                        src_key,
12382                        dst_key,
12383                    } => {
12384                        let is_ab = Self::aliases_match(&alias_a, src_key, commit)
12385                            && Self::aliases_match(&alias_b, dst_key, commit);
12386                        let is_ba = Self::aliases_match(&alias_b, src_key, commit)
12387                            && Self::aliases_match(&alias_a, dst_key, commit);
12388                        if is_ab || is_ba {
12389                            active.push((et.clone(), src_key.clone(), dst_key.clone(), true));
12390                            out.push(EdgeHistoryEvent {
12391                                edge_type: et.clone(),
12392                                commit,
12393                                event: EdgeEvent::Added,
12394                                rule: Some(rule.clone()),
12395                            });
12396                        }
12397                    }
12398                    WalRecord::DerivedEdgeRetracted {
12399                        rule,
12400                        edge_type: et,
12401                        src_key,
12402                        dst_key,
12403                    } => {
12404                        let is_ab = Self::aliases_match(&alias_a, src_key, commit)
12405                            && Self::aliases_match(&alias_b, dst_key, commit);
12406                        let is_ba = Self::aliases_match(&alias_b, src_key, commit)
12407                            && Self::aliases_match(&alias_a, dst_key, commit);
12408                        if is_ab || is_ba {
12409                            // Push unconditionally: a derived edge whose Added marker
12410                            // predates the history horizon has no `active` entry, but
12411                            // the retraction is still a real in-window event.
12412                            // Remove from active idempotently if present.
12413                            active.retain(|(aet, s, d, _)| {
12414                                !(aet == et && s == src_key && d == dst_key)
12415                            });
12416                            out.push(EdgeHistoryEvent {
12417                                edge_type: et.clone(),
12418                                commit,
12419                                event: EdgeEvent::Retracted,
12420                                rule: Some(rule.clone()),
12421                            });
12422                        }
12423                    }
12424                    // All other records (InsertNode, SetProp, CreateRule, etc.)
12425                    // do not affect edges between a and b.
12426                    _ => {}
12427                }
12428            }
12429        }
12430
12431        Ok(HistoryResult {
12432            items: out,
12433            total_commits,
12434            horizon: self.wal_horizon_floor,
12435        })
12436    }
12437
12438    /// Return `true` iff an edge of `edge_type` existed between `a` and `b`
12439    /// (in either direction) at the WAL commit `at_commit`.
12440    ///
12441    /// ## Horizon
12442    ///
12443    /// Valid commit indices are `0..total_commits` where `total_commits` is the
12444    /// number of WAL frames. An `at_commit >= total_commits` is outside the
12445    /// visible horizon and returns [`GraphError::CommitOutOfRange`].
12446    ///
12447    /// ## Derived edges
12448    ///
12449    /// Rule-derived edges are tracked via `DerivedEdgeAdded` / `DerivedEdgeRetracted`
12450    /// WAL markers appended at firing time (Task 1). `was_linked` reads these markers
12451    /// and therefore includes derived edges in its point-in-time evaluation,
12452    /// matching `edge_history`'s fidelity.
12453    pub fn was_linked(&self, a: &str, b: &str, edge_type: &str, at_commit: u64) -> Result<bool> {
12454        use core_storage::wal::WalRecord;
12455
12456        let (frames, _) = self.all_frames()?;
12457        let total_commits = self.wal_horizon_floor + frames.len() as u64;
12458
12459        // Horizon floor: commits in pruned archives are unreachable.
12460        if at_commit < self.wal_horizon_floor {
12461            return Err(GraphError::CommitOutOfRange {
12462                commit: at_commit,
12463                total: total_commits,
12464                floor: self.wal_horizon_floor,
12465            });
12466        }
12467        if at_commit >= total_commits {
12468            return Err(GraphError::CommitOutOfRange {
12469                commit: at_commit,
12470                total: total_commits,
12471                floor: self.wal_horizon_floor,
12472            });
12473        }
12474
12475        // Resolve all historical names for a and b (handles RenameNode in the WAL).
12476        // Intervals are commit-bounded so recycled keys don't contaminate point-in-time reads.
12477        let alias_a = self.build_key_alias_intervals(&frames, a);
12478        let alias_b = self.build_key_alias_intervals(&frames, b);
12479
12480        // Local index into surviving frames (0 = first frame of oldest archive).
12481        let local_commit = at_commit - self.wal_horizon_floor;
12482
12483        // Replay local frames 0..=local_commit, tracking active edges.
12484        let mut active: BTreeSet<(String, String, String)> = BTreeSet::new();
12485
12486        for (local_i, frame) in frames.iter().enumerate().take((local_commit + 1) as usize) {
12487            let commit = self.wal_horizon_floor + local_i as u64;
12488            let records: &[WalRecord] = match frame {
12489                WalRecord::Batch(inner) => inner.as_slice(),
12490                single => std::slice::from_ref(single),
12491            };
12492
12493            for rec in records {
12494                match rec {
12495                    WalRecord::InsertEdge {
12496                        edge_type: et,
12497                        src_key,
12498                        dst_key,
12499                    } => {
12500                        let is_ab = Self::aliases_match(&alias_a, src_key, commit)
12501                            && Self::aliases_match(&alias_b, dst_key, commit);
12502                        let is_ba = Self::aliases_match(&alias_b, src_key, commit)
12503                            && Self::aliases_match(&alias_a, dst_key, commit);
12504                        if is_ab || is_ba {
12505                            active.insert((et.clone(), src_key.clone(), dst_key.clone()));
12506                        }
12507                    }
12508                    WalRecord::InsertEdgeId { etype, src, dst } => {
12509                        let etype_str = match self.syms.resolve(*etype) {
12510                            Some(s) => s.to_string(),
12511                            None => continue,
12512                        };
12513                        // Use key_of_historical so tombstoned nodes resolve.
12514                        let src_key = self.ids.key_of_historical(*src);
12515                        let dst_key = self.ids.key_of_historical(*dst);
12516                        let is_ab = src_key == Some(a) && dst_key == Some(b);
12517                        let is_ba = src_key == Some(b) && dst_key == Some(a);
12518                        if is_ab || is_ba {
12519                            active.insert((
12520                                etype_str,
12521                                src_key.unwrap().to_string(),
12522                                dst_key.unwrap().to_string(),
12523                            ));
12524                        }
12525                    }
12526                    WalRecord::DeleteEdge {
12527                        edge_type: et,
12528                        src_key,
12529                        dst_key,
12530                    } => {
12531                        let is_ab = Self::aliases_match(&alias_a, src_key, commit)
12532                            && Self::aliases_match(&alias_b, dst_key, commit);
12533                        let is_ba = Self::aliases_match(&alias_b, src_key, commit)
12534                            && Self::aliases_match(&alias_a, dst_key, commit);
12535                        if is_ab || is_ba {
12536                            active.remove(&(et.clone(), src_key.clone(), dst_key.clone()));
12537                        }
12538                    }
12539                    WalRecord::DeleteNode { key: k }
12540                        if Self::aliases_match(&alias_a, k, commit)
12541                            || Self::aliases_match(&alias_b, k, commit) =>
12542                    {
12543                        // All edges touching the deleted node are gone.
12544                        active.retain(|(_, s, d)| s != k && d != k);
12545                    }
12546                    WalRecord::DerivedEdgeAdded {
12547                        edge_type: et,
12548                        src_key,
12549                        dst_key,
12550                        ..
12551                    } => {
12552                        let is_ab = Self::aliases_match(&alias_a, src_key, commit)
12553                            && Self::aliases_match(&alias_b, dst_key, commit);
12554                        let is_ba = Self::aliases_match(&alias_b, src_key, commit)
12555                            && Self::aliases_match(&alias_a, dst_key, commit);
12556                        if is_ab || is_ba {
12557                            active.insert((et.clone(), src_key.clone(), dst_key.clone()));
12558                        }
12559                    }
12560                    WalRecord::DerivedEdgeRetracted {
12561                        edge_type: et,
12562                        src_key,
12563                        dst_key,
12564                        ..
12565                    } => {
12566                        let is_ab = Self::aliases_match(&alias_a, src_key, commit)
12567                            && Self::aliases_match(&alias_b, dst_key, commit);
12568                        let is_ba = Self::aliases_match(&alias_b, src_key, commit)
12569                            && Self::aliases_match(&alias_a, dst_key, commit);
12570                        if is_ab || is_ba {
12571                            active.remove(&(et.clone(), src_key.clone(), dst_key.clone()));
12572                        }
12573                    }
12574                    _ => {}
12575                }
12576            }
12577        }
12578
12579        Ok(active.iter().any(|(et, _, _)| et == edge_type))
12580    }
12581
12582    /// Every edge incident to `key` — either endpoint — that existed at WAL
12583    /// commit `commit`, from ONE scan of the WAL.
12584    ///
12585    /// This is the bulk form of [`was_linked`](GraphDb::was_linked): answering
12586    /// "what did K's relationships look like at commit C" with one call instead
12587    /// of one [`edge_history`](GraphDb::edge_history) per candidate partner.
12588    /// The two agree edge for edge.
12589    ///
12590    /// Results are sorted by `(edge_type, src_key, dst_key)`.
12591    ///
12592    /// ## Horizon
12593    ///
12594    /// Valid commit indices are `wal_horizon_floor()..wal_total_commits()`;
12595    /// anything outside is [`GraphError::CommitOutOfRange`], exactly like
12596    /// `was_linked`. An unknown key is not an error — it simply had no edges.
12597    ///
12598    /// ## Derived edges
12599    ///
12600    /// `DerivedEdgeAdded` / `DerivedEdgeRetracted` markers carry rule
12601    /// attribution, so a rule-owned edge comes back with `derived: true` and
12602    /// `rule: Some(name)`.
12603    ///
12604    /// ## Renames
12605    ///
12606    /// `key` is matched through the same commit-bounded alias intervals
12607    /// `edge_history` uses, so querying a node's *current* key surfaces edges
12608    /// written under an earlier name. Endpoint keys in the result are reported
12609    /// under the name the node carries today, so they can be fed straight back
12610    /// into `node_info`, `explain` or another `edges_at`.
12611    ///
12612    /// ## Masks
12613    ///
12614    /// Like `edge_history` and `node_history`, this reads the WAL regardless of
12615    /// any role mask. Apply masking at the caller level.
12616    pub fn edges_at(&self, key: &str, commit: u64) -> Result<Vec<EdgeAt>> {
12617        use core_storage::wal::WalRecord;
12618
12619        let (frames, _) = self.all_frames()?;
12620        let total_commits = self.wal_horizon_floor + frames.len() as u64;
12621
12622        // Horizon floor: commits in pruned archives are unreachable.
12623        if commit < self.wal_horizon_floor || commit >= total_commits {
12624            return Err(GraphError::CommitOutOfRange {
12625                commit,
12626                total: total_commits,
12627                floor: self.wal_horizon_floor,
12628            });
12629        }
12630
12631        // Commit-bounded historical names of `key` (handles RenameNode).
12632        let alias = self.build_key_alias_intervals(&frames, key);
12633
12634        // Forward rename chain, for reporting endpoints under their current
12635        // names: old key → [(commit, new key)] in ascending commit order.
12636        // Built over the whole WAL, not just the prefix up to `commit`, because
12637        // a rename after `commit` still changes what the node is called today.
12638        let mut renames: HashMap<String, Vec<(u64, String)>> = HashMap::new();
12639        for (local_i, frame) in frames.iter().enumerate() {
12640            let c = self.wal_horizon_floor + local_i as u64;
12641            let records: &[WalRecord] = match frame {
12642                WalRecord::Batch(inner) => inner.as_slice(),
12643                single => std::slice::from_ref(single),
12644            };
12645            for rec in records {
12646                if let WalRecord::RenameNode { old_key, new_key } = rec {
12647                    renames
12648                        .entry(old_key.clone())
12649                        .or_default()
12650                        .push((c, new_key.clone()));
12651                }
12652            }
12653        }
12654
12655        // The name a node written as `k` at commit `from` carries today.
12656        // Follows the first rename at or after `from`, then keeps going. The
12657        // iteration cap bounds a rename cycle inside a single batch.
12658        let canon = |k: &str, from: u64| -> String {
12659            if renames.is_empty() {
12660                return k.to_string();
12661            }
12662            let mut cur = k.to_string();
12663            let mut at = from;
12664            for _ in 0..64 {
12665                match renames
12666                    .get(&cur)
12667                    .and_then(|v| v.iter().find(|(c, _)| *c >= at))
12668                {
12669                    Some((c, new)) => {
12670                        at = *c;
12671                        cur = new.clone();
12672                    }
12673                    None => break,
12674                }
12675            }
12676            cur
12677        };
12678
12679        let local_commit = commit - self.wal_horizon_floor;
12680        // (edge_type, src_key, dst_key) → (derived, rule)
12681        let mut active: BTreeMap<(String, String, String), (bool, Option<String>)> =
12682            BTreeMap::new();
12683
12684        for (local_i, frame) in frames.iter().enumerate().take((local_commit + 1) as usize) {
12685            let c = self.wal_horizon_floor + local_i as u64;
12686            let records: &[WalRecord] = match frame {
12687                WalRecord::Batch(inner) => inner.as_slice(),
12688                single => std::slice::from_ref(single),
12689            };
12690
12691            for rec in records {
12692                match rec {
12693                    WalRecord::InsertEdge {
12694                        edge_type,
12695                        src_key,
12696                        dst_key,
12697                    } => {
12698                        if Self::aliases_match(&alias, src_key, c)
12699                            || Self::aliases_match(&alias, dst_key, c)
12700                        {
12701                            active.insert(
12702                                (edge_type.clone(), canon(src_key, c), canon(dst_key, c)),
12703                                (false, None),
12704                            );
12705                        }
12706                    }
12707                    WalRecord::InsertEdgeId { etype, src, dst } => {
12708                        let Some(etype_str) = self.syms.resolve(*etype) else {
12709                            continue;
12710                        };
12711                        // `key_of_historical` resolves tombstoned ids too, and
12712                        // already returns the node's current key — no rename
12713                        // canonicalisation needed on this arm.
12714                        let (Some(src_key), Some(dst_key)) = (
12715                            self.ids.key_of_historical(*src),
12716                            self.ids.key_of_historical(*dst),
12717                        ) else {
12718                            continue;
12719                        };
12720                        if src_key == key || dst_key == key {
12721                            active.insert(
12722                                (
12723                                    etype_str.to_string(),
12724                                    src_key.to_string(),
12725                                    dst_key.to_string(),
12726                                ),
12727                                (false, None),
12728                            );
12729                        }
12730                    }
12731                    WalRecord::DeleteEdge {
12732                        edge_type,
12733                        src_key,
12734                        dst_key,
12735                    } => {
12736                        if Self::aliases_match(&alias, src_key, c)
12737                            || Self::aliases_match(&alias, dst_key, c)
12738                        {
12739                            active.remove(&(
12740                                edge_type.clone(),
12741                                canon(src_key, c),
12742                                canon(dst_key, c),
12743                            ));
12744                        }
12745                    }
12746                    WalRecord::DeleteNode { key: k } => {
12747                        if active.is_empty() {
12748                            continue;
12749                        }
12750                        if Self::aliases_match(&alias, k, c) {
12751                            // Our node is gone; every incident edge goes with it.
12752                            active.clear();
12753                        } else {
12754                            // A partner is gone; its edges to us go with it.
12755                            let ck = canon(k, c);
12756                            active.retain(|(_, s, d), _| *s != ck && *d != ck);
12757                        }
12758                    }
12759                    WalRecord::DerivedEdgeAdded {
12760                        rule,
12761                        edge_type,
12762                        src_key,
12763                        dst_key,
12764                    } => {
12765                        if Self::aliases_match(&alias, src_key, c)
12766                            || Self::aliases_match(&alias, dst_key, c)
12767                        {
12768                            active.insert(
12769                                (edge_type.clone(), canon(src_key, c), canon(dst_key, c)),
12770                                (true, Some(rule.clone())),
12771                            );
12772                        }
12773                    }
12774                    WalRecord::DerivedEdgeRetracted {
12775                        edge_type,
12776                        src_key,
12777                        dst_key,
12778                        ..
12779                    } => {
12780                        if Self::aliases_match(&alias, src_key, c)
12781                            || Self::aliases_match(&alias, dst_key, c)
12782                        {
12783                            active.remove(&(
12784                                edge_type.clone(),
12785                                canon(src_key, c),
12786                                canon(dst_key, c),
12787                            ));
12788                        }
12789                    }
12790                    // InsertNode, SetProp, CreateRule, … do not move edges.
12791                    _ => {}
12792                }
12793            }
12794        }
12795
12796        // BTreeMap iteration is already (edge_type, src, dst) order.
12797        Ok(active
12798            .into_iter()
12799            .map(|((edge_type, src_key, dst_key), (derived, rule))| EdgeAt {
12800                edge_type,
12801                src_key,
12802                dst_key,
12803                derived,
12804                rule,
12805            })
12806            .collect())
12807    }
12808
12809    /// The derived edges that would be retracted and derived if `key.field`
12810    /// were set to `value` — computed WITHOUT writing anything.
12811    ///
12812    /// Nothing is committed and nothing on `self` is mutated: the rule engine's
12813    /// provenance, its candidate indexes, the topology and the property columns
12814    /// are all cloned first, the change is applied to the clone, and the real
12815    /// per-node re-derivation (`RuleEngine::on_node_changed` — the same call
12816    /// `set_prop` makes during apply) runs against it. The derived-edge deltas
12817    /// it emits are the answer, so rule semantics — predicates, top-k,
12818    /// via-hops, chaining, weights — are the engine's, not a re-implementation.
12819    ///
12820    /// Works on a read-only handle.
12821    ///
12822    /// **While a rule's vector index is still building** (`RuleStats::building`)
12823    /// the clone carries no pending-build state, so this reports the edges that
12824    /// rule would derive — which the live store will not derive until its
12825    /// backfill runs. Right about the end state, early about the timing.
12826    ///
12827    /// Returns `Err(KeyNotFound)` for an unknown or tombstoned key and
12828    /// `Err(ViewPropReadOnly)` for a field a view owns — matching
12829    /// [`set_prop`](GraphDb::set_prop)'s validation. A change with no effect
12830    /// (the node already holds `value`, or no rule watches `field`) returns
12831    /// empty lists.
12832    ///
12833    /// ## Cost
12834    ///
12835    /// One clone of the property columns, the topology overlay, the symbol
12836    /// interner, the edge properties and the provenance map, plus one candidate
12837    /// re-index (O(nodes × rules)). That is much cheaper than copying the store
12838    /// directory, but it is not free — this is an interactive "what if", not a
12839    /// hot path.
12840    pub fn what_if_set_prop(&self, key: &str, field: &str, value: Value) -> Result<WhatIf> {
12841        // The engine's provenance, HNSW and IVF state live in the mmap'd base
12842        // until something asks for them. On a store opened cold from a snapshot
12843        // this is the first ask, and without it the clone below starts from an
12844        // empty provenance map: nothing to retract, so `lost` comes back empty.
12845        self.ensure_v8_base_sections_loaded();
12846
12847        let empty = WhatIf {
12848            lost: Vec::new(),
12849            gained: Vec::new(),
12850        };
12851
12852        if let Some(view_name) = self.view_store.view_for_prop(field) {
12853            return Err(GraphError::ViewPropReadOnly {
12854                view_name: view_name.to_string(),
12855            });
12856        }
12857        MutPreview::new(self).check_live_key(key)?;
12858        let id = self
12859            .ids
12860            .get(key)
12861            .ok_or_else(|| GraphError::KeyNotFound { key: key.into() })?;
12862
12863        let rules: Vec<RuleDef> = self.engine.rules().cloned().collect();
12864        if rules.is_empty() {
12865            return Ok(empty);
12866        }
12867
12868        // No rule watches this field → no derivation can change.
12869        if !rules.iter().any(|r| r.watched_fields().contains(field)) {
12870            return Ok(empty);
12871        }
12872
12873        let old_value = build_props_view(&self.props, &self.base)
12874            .get(id, field)
12875            .map(|vr| vr.into_value());
12876        if old_value.as_ref() == Some(&value) {
12877            return Ok(empty);
12878        }
12879
12880        // --- Clone every piece of state the re-derivation writes to. ---
12881        let mut props = self.props.clone();
12882        let mut topo = self.topo.clone();
12883        let mut syms = self.syms.clone();
12884        let mut edge_props = self.edge_props.clone();
12885
12886        let mut tripped: BTreeMap<String, bool> = BTreeMap::new();
12887        let mut fires: BTreeMap<String, u64> = BTreeMap::new();
12888        for r in &rules {
12889            tripped.insert(r.name.clone(), self.engine.is_tripped(&r.name));
12890            fires.insert(r.name.clone(), self.engine.fire_count(&r.name));
12891        }
12892        // `provenance()` decodes retained snapshot bytes on first use; the
12893        // engine clone needs the real map, not an empty one.
12894        let provenance = self.engine.provenance().clone();
12895        let mut engine = core_rules::RuleEngine::from_persist(rules, provenance, tripped, fires);
12896
12897        // Build the candidate indexes from the state BEFORE the change, exactly
12898        // as apply() sees them: `on_node_changed` withdraws the node under its
12899        // old value and refiles it under the new one, so the index must not
12900        // already reflect the change.
12901        engine.reindex_all_load_state(
12902            &self.ids,
12903            &syms,
12904            &self.labels,
12905            build_props_view(&self.props, &self.base),
12906            self.engine.export_ivf_state(),
12907            self.engine.export_hnsw_state_passthrough(),
12908        );
12909        engine.set_emit_deltas(true);
12910
12911        // --- Apply the hypothetical change and re-derive. ---
12912        props.set(id, field, value);
12913        {
12914            let mut gm = make_graph_mut(
12915                &self.ids,
12916                &mut syms,
12917                &self.labels,
12918                build_props_view(&props, &self.base),
12919                &mut topo,
12920                &self.base,
12921                &mut edge_props,
12922            );
12923            engine.on_node_changed(id, Some((field, old_value)), &mut gm);
12924        }
12925
12926        let mut lost: BTreeSet<EdgeAt> = BTreeSet::new();
12927        let mut gained: BTreeSet<EdgeAt> = BTreeSet::new();
12928        for d in engine.drain_deltas() {
12929            let edge = EdgeAt {
12930                edge_type: d.edge_type,
12931                src_key: d.src_key,
12932                dst_key: d.dst_key,
12933                derived: true,
12934                rule: Some(d.rule),
12935            };
12936            if d.fired {
12937                gained.insert(edge);
12938            } else {
12939                lost.insert(edge);
12940            }
12941        }
12942        // An edge retracted and re-derived within the same re-derivation (top-k
12943        // churn) is not a change the caller would see.
12944        let churn: Vec<EdgeAt> = lost.intersection(&gained).cloned().collect();
12945        for e in churn {
12946            lost.remove(&e);
12947            gained.remove(&e);
12948        }
12949
12950        Ok(WhatIf {
12951            lost: lost.into_iter().collect(),
12952            gained: gained.into_iter().collect(),
12953        })
12954    }
12955
12956    pub fn edge_count(&self) -> u64 {
12957        self.topo_view().edge_count()
12958    }
12959
12960    /// Live/tombstone/edge counts plus per-rule provenance size, trip latch,
12961    /// and fire counter (includes rebuild evaluations). Rules are sorted by name.
12962    pub fn stats(&self) -> Stats {
12963        self.ensure_v8_base_sections_loaded();
12964        let building = self.engine.builds_in_progress();
12965        let rules: Vec<RuleStats> = self
12966            .engine
12967            .rules()
12968            .map(|r| RuleStats {
12969                name: r.name.clone(),
12970                edges: self
12971                    .engine
12972                    .provenance()
12973                    .get(&r.name)
12974                    .map(|s| s.len() as u64)
12975                    .unwrap_or(0),
12976                tripped: self.engine.is_tripped(&r.name),
12977                fires: self.engine.fire_count(&r.name),
12978                approximate: r.approximate,
12979                building: building.iter().find(|b| b.rule == r.name).cloned(),
12980            })
12981            .collect();
12982        Stats {
12983            nodes_live: self.ids.live_len(),
12984            nodes_tombstoned: self.ids.len() - self.ids.live_len(),
12985            edges: self.topo_view().edge_count(),
12986            rules,
12987            chain_truncations: self.engine.chain_truncations(),
12988            history_floor: self.wal_horizon_floor,
12989            namespaces: self.namespace_stats(),
12990        }
12991    }
12992
12993    /// On-disk size of the WAL file in bytes.
12994    ///
12995    /// Reads file metadata without loading WAL contents.  Returns `Err` for
12996    /// in-memory (`SimFs`) databases where no WAL file exists on disk.
12997    pub fn wal_size_bytes(&self) -> std::io::Result<u64> {
12998        let path = self.fs.wal_path().ok_or_else(|| {
12999            std::io::Error::new(
13000                std::io::ErrorKind::Unsupported,
13001                "wal_path not available for this Fs implementation",
13002            )
13003        })?;
13004        Ok(std::fs::metadata(path)?.len())
13005    }
13006
13007    /// Set the slow-query threshold.  Queries whose execution time equals or
13008    /// exceeds `ms` milliseconds are logged.  Pass `0` to disable.
13009    ///
13010    /// Use this setter in tests — the environment variable
13011    /// `MUSHROOMDB_SLOW_QUERY_MS` is process-global and races parallel test
13012    /// threads.
13013    pub fn set_slow_query_threshold_ms(&mut self, ms: u64) {
13014        self.slow_query_threshold_ms = ms;
13015    }
13016
13017    /// Snapshot of the slow-query ring buffer and lifetime counter.
13018    pub fn slow_query_snapshot(&self) -> SlowQuerySnapshot {
13019        let log = self.slow_queries.lock().unwrap_or_else(|e| e.into_inner());
13020        SlowQuerySnapshot {
13021            threshold_ms: self.slow_query_threshold_ms,
13022            count: log.total,
13023            last: log.entries.iter().cloned().collect(),
13024        }
13025    }
13026
13027    /// Instant the database was opened.  Used by consumers (e.g. `/metrics`)
13028    /// to compute uptime.
13029    pub fn started_at(&self) -> std::time::Instant {
13030        self.started_at
13031    }
13032
13033    /// The on-disk snapshot version a store that has opted in to nothing
13034    /// writes — the **floor**, not the whole answer.
13035    ///
13036    /// It is not "the version this binary writes", and it is not "the version
13037    /// this binary reads". Since v0.6.10 this binary writes 9 **or** 10
13038    /// depending on the store — [`snapshot::version_for`] decides, and a store
13039    /// that has called [`enable_multiplicity`](Self::enable_multiplicity)
13040    /// writes 10 — and it reads 5 through 10. A caller comparing a store's
13041    /// stamp against this value must use `>=`, not `==`, or it will report an
13042    /// opted-in store as needing a migration *down*; `cli::run_migrate` is the
13043    /// worked example.
13044    ///
13045    /// The name is kept for compatibility: it is public API reachable from the
13046    /// CLI and from any embedder, and respelling it would break them for a
13047    /// doc-level clarification.
13048    ///
13049    /// [`snapshot::version_for`]: core_storage::snapshot::version_for
13050    pub fn format_version() -> u16 {
13051        core_storage::snapshot::VERSION
13052    }
13053
13054    /// Test-support: total bytes appended (SimFs only usage).
13055    pub fn fs_total_appended(&self) -> usize
13056    where
13057        F: FsIntrospect,
13058    {
13059        self.fs.total_appended()
13060    }
13061
13062    /// Test-support: successful `Fs::sync` calls (SimFs / counting fs).
13063    pub fn fs_sync_count(&self) -> usize
13064    where
13065        F: FsIntrospect,
13066    {
13067        self.fs.sync_count()
13068    }
13069
13070    /// Consume the db, returning its fs (for crash simulation).
13071    pub fn into_fs(self) -> F {
13072        self.fs
13073    }
13074
13075    pub fn snapshot(&mut self) -> Result<()> {
13076        self.snapshot_with(SnapshotOptions::default())
13077    }
13078
13079    /// Snapshot with explicit options.
13080    ///
13081    /// # `keep_wal`
13082    ///
13083    /// When `keep_wal` is `false` (the default, same as [`snapshot`]):
13084    ///   - The WAL is replaced with a minimal baseline containing one
13085    ///     `EnableFulltext` record per active declaration.  All pre-snapshot
13086    ///     history is discarded; `open_at` can only reach post-snapshot commits.
13087    ///
13088    /// When `keep_wal` is `true`:
13089    ///   - The WAL is left intact.  All pre-snapshot commits remain reachable
13090    ///     via `open_at`.  The existing WAL already contains the original
13091    ///     `EnableFulltext` records, so no baseline re-write is needed; the
13092    ///     recovery guards in `apply()` silently skip any duplicate records on
13093    ///     replay.
13094    ///   - Crash window: a crash after the snapshot write but before the next
13095    ///     WAL write leaves the full pre-snapshot WAL intact.  On reopen the
13096    ///     snapshot is loaded and the WAL replayed idempotently over it — safe
13097    ///     because every `apply()` arm is idempotent when replayed over an
13098    ///     already-current snapshot.
13099    pub fn snapshot_with(&mut self, opts: SnapshotOptions) -> Result<()> {
13100        if self.read_only {
13101            return Err(GraphError::ReadOnly);
13102        }
13103        // A snapshot rewrites `wal.bin` through a tmp+rename, so a peer that is
13104        // appending ends up holding a descriptor on an unlinked inode and loses
13105        // commits it believes durable. Snapshotting therefore requires the
13106        // cross-process write lock, exactly as appending does. Unlike the WAL
13107        // append path this does not go through `log_then_apply_with`, so both
13108        // guards are repeated here.
13109        if self.degraded {
13110            return Err(GraphError::Io(std::io::Error::other(
13111                "database degraded after group-commit fsync failure; reopen required",
13112            )));
13113        }
13114        if self.lock_denied {
13115            return Err(GraphError::Busy { holder: None });
13116        }
13117        // Capture whether snapshot.bin already existed BEFORE this snapshot write.
13118        // Used by the archive path's conservative genesis-chain check: if a prior
13119        // snapshot exists but wal.truncated does not, we cannot distinguish a
13120        // legacy store (may have been truncated in an older code version) from a
13121        // new store that only used keep_wal=true.  Conservative: refuse genesis in
13122        // both cases.  Must be sampled here, before the snapshot write below.
13123        //
13124        // `snapshot_preserved_history` is the one case where the answer is not a
13125        // guess: a snapshot *this handle* took, on a store that had none when it
13126        // opened, and that kept the WAL. The proxy defers to it, because
13127        // otherwise `enable_multiplicity` — whose forced snapshot is exactly
13128        // that — would permanently disqualify the store from a genesis chain it
13129        // is fully entitled to (defect #23).
13130        let had_prior_snapshot = self.fs.snapshot_path().map(|p| p.exists()).unwrap_or(false)
13131            && !self.snapshot_preserved_history;
13132        // Which version this store writes. V9 unless it has opted in to
13133        // multiplicity, in which case V10 — the stamp that makes a reader which
13134        // does not know WAL discriminant 23 refuse the open instead of
13135        // truncating the WAL at the first such frame. The container is
13136        // identical either way; only these two header bytes move.
13137        let snapshot_version = core_storage::snapshot::version_for(self.multiplicity);
13138        self.ensure_v8_base_sections_loaded();
13139        // Ensure provenance is decoded before to_persist() clones it.
13140        self.engine.ensure_provenance_loaded_mut();
13141        let (rule_defs_typed, provenance, rule_tripped, rule_fires) = self.engine.to_persist();
13142        let rule_defs = rule_defs_typed
13143            .iter()
13144            .map(|r| bincode::serialize(r).expect("RuleDef serialize cannot fail"))
13145            .collect();
13146        // Collect HNSW state and IVF state.  When indexes are not yet
13147        // populated (clean open, no mutation since open), pass the retained
13148        // raw bytes through directly so that migrate/snapshot does not
13149        // silently discard fitted approximate-rule indexes.
13150        let hnsw_state = self.engine.export_hnsw_state_passthrough();
13151        let ivf_bytes = if !self.engine.indexes_populated() {
13152            // Pass retained IVF bytes through unchanged (no re-encode).
13153            self.engine.retained_ivf_bytes_clone().unwrap_or_default()
13154        } else {
13155            // Indexes live: encode from current state.
13156            let raw_ivf = self.engine.export_ivf_state();
13157            let ivf_state_map: BTreeMap<String, core_storage::snapshot::PerRuleIvfState> = raw_ivf
13158                .into_iter()
13159                .map(|(name, ((sc, sa, sd), (dc, da, dd)))| {
13160                    (
13161                        name,
13162                        core_storage::snapshot::PerRuleIvfState {
13163                            src: core_storage::snapshot::SideIvfState {
13164                                centroids: sc,
13165                                clusters: sa,
13166                                drift: sd,
13167                            },
13168                            dst: core_storage::snapshot::SideIvfState {
13169                                centroids: dc,
13170                                clusters: da,
13171                                drift: dd,
13172                            },
13173                        },
13174                    )
13175                })
13176                .collect();
13177            if ivf_state_map.is_empty() {
13178                Vec::new()
13179            } else {
13180                bincode::serialize(&ivf_state_map).expect("IVF state serialize cannot fail")
13181            }
13182        };
13183        let view_defs: Vec<Vec<u8>> = self
13184            .view_store
13185            .views()
13186            .map(|v| bincode::serialize(v).expect("ViewDef serialize cannot fail"))
13187            .collect();
13188        if self.base.is_some() {
13189            // V8 merge-snapshot path: encode base+overlay into a new V8 snapshot,
13190            // write it atomically, remap it as the new base, then clear the overlay.
13191            let meta = V8Meta {
13192                labels: self.labels.clone(),
13193                edge_props: self.edge_props.clone(),
13194                rule_defs,
13195                provenance,
13196                rule_tripped,
13197                rule_fires,
13198                ivf_bytes,
13199                view_defs,
13200                wal_truncated: !opts.keep_wal,
13201                hnsw: hnsw_state,
13202                last_change: self.last_change.clone(),
13203            };
13204            let mut buf: Vec<u8> = Vec::new();
13205            {
13206                // Clone the Arc so the old base stays alive while we encode.
13207                // The borrow of archived_csr (into old_base's mmap) is released
13208                // at the end of this block, before we replace self.base.
13209                let old_base = self.base.clone().expect("is_some checked above");
13210                let archived_csr = old_base.topology().map_err(|e| GraphError::Corrupt {
13211                    detail: format!("v8 snapshot: topology section: {e:?}"),
13212                })?;
13213                let archived_cols = old_base.columns().map_err(|e| GraphError::Corrupt {
13214                    detail: format!("v8 snapshot: columns section: {e:?}"),
13215                })?;
13216                // `None` when the base predates V9 — the migration path: its
13217                // string columns still carry their own tables and this snapshot
13218                // is the rewrite that collapses them into section 12.
13219                let archived_strings =
13220                    old_base
13221                        .string_table()
13222                        .transpose()
13223                        .map_err(|e| GraphError::Corrupt {
13224                            detail: format!("v8 snapshot: strings section: {e:?}"),
13225                        })?;
13226                let archived_edge_props =
13227                    old_base
13228                        .edge_props_section()
13229                        .map_err(|e| GraphError::Corrupt {
13230                            detail: format!("v8 snapshot: edge_props section: {e:?}"),
13231                        })?;
13232                let edge_props_raw =
13233                    old_base
13234                        .edge_props_raw_bytes()
13235                        .map_err(|e| GraphError::Corrupt {
13236                            detail: format!("v8 snapshot: edge_props raw bytes: {e:?}"),
13237                        })?;
13238                let prov_raw =
13239                    old_base
13240                        .provenance_raw_bytes()
13241                        .map_err(|e| GraphError::Corrupt {
13242                            detail: format!("v8 snapshot: provenance raw bytes: {e:?}"),
13243                        })?;
13244                encode_v8(
13245                    Some(archived_csr),
13246                    Some(archived_cols),
13247                    archived_strings,
13248                    Some((archived_edge_props, edge_props_raw)),
13249                    Some(prov_raw),
13250                    &self.topo,
13251                    &self.props,
13252                    &self.ids,
13253                    &self.syms,
13254                    &meta,
13255                    &mut buf,
13256                )?;
13257            }
13258            core_storage::snapshot::stamp_container_version(&mut buf, snapshot_version)?;
13259            self.fs.write_atomic(FileId::Snapshot, &buf)?;
13260            // Remap the freshly-written snapshot as the new base.
13261            // C2: use file mmap on RealFs; fall back to from_bytes on SimFs.
13262            let new_base = if let Some(snap_path) = self.fs.snapshot_path() {
13263                core_storage::v8::MappedBase::map(&snap_path)
13264            } else {
13265                core_storage::v8::MappedBase::from_bytes(buf)
13266            }
13267            .map_err(|e| GraphError::Corrupt {
13268                detail: format!("v8 snapshot: remap new base: {e:?}"),
13269            })?;
13270            self.base = Some(Arc::new(new_base));
13271            // Clear the overlay and prop tombstones — all data is now in the new base.
13272            self.topo = Topology::new();
13273            self.props = core_storage::columns::ColumnStore::new();
13274        } else {
13275            // Legacy path (V5–V7 stores without a V8 base).
13276            //
13277            // Memory-diet path: build V8Meta directly from &self — no SnapshotState
13278            // clone and no encode_v8_from_state intermediate clones.  The big
13279            // structures (self.topo, self.props) are borrowed, not cloned.
13280            // self.edge_props is moved (not cloned) because we immediately clear it
13281            // when we remap the new V8 snapshot as self.base (see below).
13282            //
13283            // Eliminates from peak RSS vs. the old SnapshotState path:
13284            //   • self.topo.clone()      (~topology HashMap footprint)
13285            //   • self.props.clone()     (~column-store footprint)
13286            //   • encode_v8_from_state V8Meta secondary clones (labels, edge_props, …)
13287            let meta = V8Meta {
13288                labels: self.labels.clone(),
13289                wal_truncated: !opts.keep_wal,
13290                // Move edge_props out so the large overlay is freed when meta
13291                // drops at end of this block (self.edge_props is now empty; reads
13292                // after base assignment go through the mmap'd base section).
13293                edge_props: std::mem::take(&mut self.edge_props),
13294                rule_defs,
13295                provenance,
13296                rule_tripped,
13297                rule_fires,
13298                ivf_bytes,
13299                view_defs,
13300                hnsw: hnsw_state,
13301                last_change: self.last_change.clone(),
13302            };
13303            let mut buf = Vec::new();
13304            encode_v8(
13305                None,
13306                None,
13307                None,
13308                None,
13309                None,
13310                &self.topo,
13311                &self.props,
13312                &self.ids,
13313                &self.syms,
13314                &meta,
13315                &mut buf,
13316            )?;
13317            // meta (and the moved edge_props inside it) is no longer needed;
13318            // drop it before the write to keep the peak window narrow.
13319            drop(meta);
13320            core_storage::snapshot::stamp_container_version(&mut buf, snapshot_version)?;
13321            self.fs.write_atomic(FileId::Snapshot, &buf)?;
13322            // Remap the freshly-written V8 snapshot as self.base.
13323            // On RealFs: drop the encode buffer before mmap to recover ~1.9 GiB.
13324            // On SimFs (tests): pass buf to from_bytes.
13325            let new_base = if let Some(snap_path) = self.fs.snapshot_path() {
13326                drop(buf);
13327                core_storage::v8::MappedBase::map(&snap_path)
13328            } else {
13329                core_storage::v8::MappedBase::from_bytes(buf)
13330            }
13331            .map_err(|e| GraphError::Corrupt {
13332                detail: format!("v8 snapshot: remap new base (legacy path): {e:?}"),
13333            })?;
13334            self.base = Some(Arc::new(new_base));
13335            // Free the large heap-allocated decoded state — all data is now in the
13336            // mmap'd base.  Mirrors the V8 merge-snapshot path (see above).
13337            // self.edge_props was already moved into meta and is effectively empty.
13338            self.topo = Topology::new();
13339            self.props = core_storage::columns::ColumnStore::new();
13340        }
13341
13342        if opts.archive_wal {
13343            // History-preserving snapshot (Task 4):
13344            //   1. Snapshot already written above (write_atomic → fsynced).
13345            //   2. Rename WAL → wal.<commit_seq>.archive  (atomic, same fs).
13346            //      Crash window B: crash here leaves archive present, WAL
13347            //      absent.  Reopen: snapshot loaded (full state), no WAL
13348            //      replay.  Archive is NOT replayed into live state — it is
13349            //      pre-snapshot by construction.  Safe.
13350            //   3. Optionally write genesis marker (first archive only, no
13351            //      prior WAL truncation).
13352            //   4. Prune old archives (retention), update horizon floor.
13353            //      Pruning invalidates the genesis chain; delete marker.
13354            //   5. Write new minimal baseline WAL (write_atomic).
13355            //      Crash window C: crash here leaves new archive plus no live
13356            //      WAL.  Same as window B — handled above.
13357            //
13358            // Sample existing archives BEFORE the rename so we can detect
13359            // whether this is the first archive.
13360            let existing_archives = self.fs.list_archives()?;
13361            let is_first_archive = existing_archives.is_empty();
13362
13363            // Compute a globally-monotonic archive name: the name equals the
13364            // cumulative end-frame index of the archive in global commit space.
13365            //
13366            // Using `commit_seq` directly is UNSOUND across sessions: on reopen
13367            // commit_seq is seeded from max(last_change), which underestimates
13368            // the WAL depth when trailing commits (e.g. insert_edge) do not
13369            // update last_change.  A session-2 archive could then receive a name
13370            // ≤ the session-1 archive, causing incorrect sort order or collision.
13371            //
13372            // Instead: read and decode the live WAL here (before the rename) to
13373            // get its exact frame count, then add it to the last known global
13374            // end-frame index (the name of the most recent existing archive, or
13375            // wal_horizon_floor if no archives exist).  This is O(WAL size) but
13376            // snapshot is already serialising the full graph state, so the cost
13377            // is dominated.
13378            let live_wal_bytes_for_name = self.fs.read(FileId::Wal)?;
13379            let (live_frames_for_name, _) = decode_all(&live_wal_bytes_for_name);
13380            let archive_n = existing_archives
13381                .last()
13382                .copied()
13383                .unwrap_or(self.wal_horizon_floor)
13384                + live_frames_for_name.len() as u64;
13385            self.fs.archive_wal(archive_n)?;
13386
13387            // The replacement WAL goes in **immediately**, with no fallible call
13388            // between it and the rename above.
13389            //
13390            // The rename is what removes the store's live declarations — the
13391            // multiplicity opt-in, and every `EnableFulltext` / `EnableIndex` —
13392            // and this write is what puts them back. Every call that used to sit
13393            // in between (the genesis marker, the retention sweep's reads, the
13394            // floor write, the archive deletes) was a `?` that could leave the
13395            // store with neither, so a single transient `Err` was enough to lose
13396            // a declaration that no rebuild can recover (defect #22).
13397            //
13398            // Ordering alone cannot close the crash window between two
13399            // filesystem calls; for the multiplicity declaration the V10 stamp
13400            // does that on the open path. What ordering does close is the much
13401            // wider window in which an ordinary I/O error did it — and that half
13402            // covers all three declarations, not just the one with a stamp.
13403            let mut baseline_wal: Vec<u8> = Vec::new();
13404            // The multiplicity opt-in is a declaration like the two below it,
13405            // and it is re-emitted for the same reason: truncation must not
13406            // silently opt the store back out and stop counting.
13407            if self.multiplicity {
13408                baseline_wal
13409                    .extend_from_slice(&encode_record(&core_storage::wal::MULTIPLICITY_ENABLED));
13410            }
13411            for (label, field) in self.fulltext.enabled_pairs() {
13412                let rec = WalRecord::EnableFulltext {
13413                    label: label.clone(),
13414                    field: field.clone(),
13415                };
13416                baseline_wal.extend_from_slice(&encode_record(&rec));
13417            }
13418            for (label, field) in self.prop_index.enabled_pairs() {
13419                let rec = WalRecord::EnableIndex {
13420                    label: label.clone(),
13421                    field: field.clone(),
13422                };
13423                baseline_wal.extend_from_slice(&encode_record(&rec));
13424            }
13425            self.fs.write_atomic(FileId::Wal, &baseline_wal)?;
13426
13427            // Genesis marker: written once when the first archive is taken
13428            // from a store that has never undergone a WAL-truncating snapshot.
13429            // When present, `open_at` may replay archive-resident commits from
13430            // empty state (the archive chain covers from global index 0).
13431            //
13432            // Two conditions must ALL hold:
13433            //   1. This is the first archive (existing_archives was empty).
13434            //   2. No snapshot.bin existed before this operation (had_prior_snapshot=false).
13435            //      A WAL-truncating snapshot (keep_wal=false) always writes snapshot.bin
13436            //      before truncating the WAL, so if any prior truncating snapshot was taken
13437            //      — even in a previous session — snapshot.bin is present and this condition
13438            //      is false.  This subsumes the cross-session truncation case without
13439            //      requiring a separate wal.truncated sidecar file.
13440            //      For legacy stores (snapshot.bin written by an older code version that
13441            //      may have truncated the WAL), the same conservative refusal applies:
13442            //      we cannot prove the chain is complete, so we refuse genesis (cost =
13443            //      no as-of-through-archives; never silent wrong data).
13444            //      The one exception is a snapshot this handle took itself, on a store
13445            //      that had none when it opened, with the WAL kept: there the answer is
13446            //      known rather than guessed, and `snapshot_preserved_history` says so.
13447            //      Without that exception `enable_multiplicity`'s forced keep_wal
13448            //      snapshot would disqualify the store forever (defect #23).
13449            //      On SimFs (snapshot_path() == None) had_prior_snapshot is always false,
13450            //      so SimFs always passes this check.
13451            if is_first_archive && !had_prior_snapshot {
13452                self.fs.write_genesis_marker()?;
13453                self.archive_genesis_chain = true;
13454            }
13455
13456            // Retention pruning: keep newest `keep` archives; delete oldest.
13457            // Pruning is the ONLY deletion site for archives.
13458            //
13459            // Crash-safety ordering (C1 fix):
13460            //   1. Count frames in surplus archives (reads only — no mutation).
13461            //   2. Advance and PERSIST the horizon floor FIRST via write-then-
13462            //      rename (atomic).  A crash after this point leaves orphaned
13463            //      archives on disk, but the floor is correct.  The opening
13464            //      cleanup sweep (`cleanup_orphaned_archives`) removes them on
13465            //      the next open, so the store is always safe to reopen.
13466            //   3. Delete the genesis marker (floor > 0 already blocks open_at
13467            //      via the conjunctive gate; marker cleanup is belt-and-suspenders).
13468            //   4. Delete surplus archives.  A crash between any two deletes
13469            //      leaves the floor committed and orphaned archives cleaned at
13470            //      next open — never a stale floor with a missing archive prefix.
13471            if let Some(keep) = self.wal_archive_retention {
13472                if keep > 0 {
13473                    let archives = self.fs.list_archives()?;
13474                    // archives is sorted ascending (oldest first)
13475                    if archives.len() as u32 > keep {
13476                        let surplus = archives.len() - keep as usize;
13477                        // Step 1: count pruned frames (reads, no mutation).
13478                        let mut pruned_frames = 0u64;
13479                        for &n in &archives[..surplus] {
13480                            let bytes = self.fs.read_archive(n)?;
13481                            let (frames, _) = decode_all(&bytes);
13482                            pruned_frames += frames.len() as u64;
13483                        }
13484                        // Step 2: advance and persist floor FIRST.
13485                        self.wal_horizon_floor += pruned_frames;
13486                        self.fs.write_horizon_floor(self.wal_horizon_floor)?;
13487                        // Step 3: delete genesis marker (floor > 0 already
13488                        // blocks open_at; this is belt-and-suspenders cleanup).
13489                        if pruned_frames > 0 && self.archive_genesis_chain {
13490                            self.fs.delete_genesis_marker()?;
13491                            self.archive_genesis_chain = false;
13492                        }
13493                        // Step 4: delete surplus archives.  Crash here →
13494                        // orphaned archives; cleaned at next open.
13495                        for &n in &archives[..surplus] {
13496                            self.fs.delete_archive(n)?;
13497                        }
13498                    }
13499                }
13500            }
13501        } else if opts.keep_wal {
13502            // keep_wal=true: WAL is left untouched.  The existing WAL already
13503            // contains the EnableFulltext records from the original enable calls;
13504            // replay is idempotent (guards in apply() skip already-live entries).
13505            // No baseline re-write is needed or safe here — the full WAL history
13506            // must remain intact for open_at to reach pre-snapshot commits.
13507        } else {
13508            // keep_wal=false (default): truncate by replacing the WAL with a
13509            // minimal baseline of one EnableFulltext record per active pair.
13510            //
13511            // Crash-ordering: write_atomic is atomic.
13512            //   • Crash before snapshot write  → WAL unchanged.  Safe.
13513            //   • Crash after snapshot write but before this WAL write → full
13514            //     pre-snapshot WAL still present; open_with replays idempotently.
13515            //   • Crash after both writes → normal post-snapshot state.
13516            //
13517            // Genesis chain: a WAL-truncating snapshot breaks the archive chain
13518            // for any archives taken AFTER this point (their WAL slices would
13519            // not start at genesis).  Delete any existing genesis marker so that
13520            // open_at refuses archive-resident commits.  Future sessions are
13521            // covered by had_prior_snapshot: snapshot.bin written here persists
13522            // across sessions and prevents a later archiving session from
13523            // incorrectly claiming a complete genesis chain.
13524            if self.archive_genesis_chain {
13525                self.fs.delete_genesis_marker()?;
13526                self.archive_genesis_chain = false;
13527            }
13528            // And this handle can no longer prove the WAL is whole: it is about
13529            // to truncate it itself. Same-session archives after this point get
13530            // the conservative answer, exactly as cross-session ones do.
13531            self.snapshot_preserved_history = false;
13532            let mut baseline_wal: Vec<u8> = Vec::new();
13533            // The multiplicity opt-in is a declaration like the two below it,
13534            // and it is re-emitted for the same reason: truncation must not
13535            // silently opt the store back out and stop counting.
13536            if self.multiplicity {
13537                baseline_wal
13538                    .extend_from_slice(&encode_record(&core_storage::wal::MULTIPLICITY_ENABLED));
13539            }
13540            for (label, field) in self.fulltext.enabled_pairs() {
13541                let rec = WalRecord::EnableFulltext {
13542                    label: label.clone(),
13543                    field: field.clone(),
13544                };
13545                baseline_wal.extend_from_slice(&encode_record(&rec));
13546            }
13547            for (label, field) in self.prop_index.enabled_pairs() {
13548                let rec = WalRecord::EnableIndex {
13549                    label: label.clone(),
13550                    field: field.clone(),
13551                };
13552                baseline_wal.extend_from_slice(&encode_record(&rec));
13553            }
13554            self.fs.write_atomic(FileId::Wal, &baseline_wal)?;
13555        }
13556        // After snapshot the overlay may have changed (V8 merge path clears
13557        // self.topo and self.props). Refresh the MVCC fold so future readers
13558        // see the post-snapshot state rather than stale overlay data.
13559        self.fold_now();
13560        // We wrote the snapshot and (unless keep_wal) replaced the WAL, so both
13561        // markers this handle uses to detect other processes' work must be
13562        // re-taken from disk. Skipping this would make our own snapshot look
13563        // like a peer's on the next staleness check and force a needless
13564        // reload.
13565        self.wal_consumed = self.fs.wal_len().map_err(GraphError::Io)?;
13566        self.snapshot_ident = self.fs.snapshot_ident().map_err(GraphError::Io)?;
13567        Ok(())
13568    }
13569}
13570
13571/// What a batch node insert does when its key is already taken.
13572///
13573/// A mirror rebuild writes a frame onto a store that already has content, so
13574/// "the key exists" is a routine answer rather than a failure. The decision is
13575/// made during the batch's existing validate pass, from one id-map lookup per
13576/// row, so the frame stays atomic and re-ingest stays O(n).
13577#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)]
13578pub enum OnConflict {
13579    /// Refuse the whole frame with [`GraphError::DuplicateKey`]. The default,
13580    /// and the only behaviour before v0.6.10.
13581    #[default]
13582    Error,
13583    /// Leave the stored node exactly as it is — properties, label and edges —
13584    /// and count it in [`BatchOutcome::skipped`].
13585    Skip,
13586    /// Keep the key and make the node's properties **exactly** the supplied
13587    /// props: supplied fields are set, fields absent from the supplied props
13588    /// are removed. A supplied label that differs from the stored one, and a
13589    /// supplied `ns` that would move the node, are row errors — relabelling is
13590    /// [`GraphDb::rename_node`], not a side effect of a rebuild.
13591    ///
13592    /// Two properties are outside "exactly", both because they are not the
13593    /// caller's to supply:
13594    ///
13595    /// - `ns` is immutable, so an omitted `ns` leaves the node where it is
13596    ///   rather than moving it to `default`;
13597    /// - a property a **view** owns is kept, not removed. Supplying one is a
13598    ///   row error, so omitting it cannot be a request to delete it, and
13599    ///   refusing the row instead would make `Replace` impossible for the whole
13600    ///   population a view has written to. Each such field kept is counted in
13601    ///   [`BatchOutcome::kept_view_owned`] — the row is still `replaced` and
13602    ///   still raises no row error, so that count is the only signal a caller
13603    ///   gets that the stored node carries a field its frame did not describe.
13604    Replace,
13605}
13606
13607/// What one [`OnConflict::Replace`] row resolves to.
13608///
13609/// The property writes that make the node exactly the supplied props —
13610/// `Some(value)` is a set, `None` a removal — paired with how many view-owned
13611/// fields the row kept instead of removing, which is the one way the result is
13612/// not exactly the supplied props. See [`MutPreview::plan_replace`].
13613type ReplacePlan = (Vec<(String, Option<Value>)>, usize);
13614
13615/// What one committed batch did.
13616///
13617/// [`BatchBuilder::commit`] returns the first two fields as a tuple; the rest
13618/// exist for [`OnConflict`] and are always zero / empty without it.
13619#[derive(Clone, Debug, Default, PartialEq, Eq)]
13620pub struct BatchOutcome {
13621    /// Node records actually written.
13622    pub nodes_inserted: usize,
13623    /// Edge records actually written. A duplicate edge is a silent no-op under
13624    /// every policy — adjacency is a set — and is not counted.
13625    pub edges_inserted: usize,
13626    /// Rows whose key was taken and whose policy was [`OnConflict::Skip`].
13627    pub skipped: usize,
13628    /// Rows whose key was taken and whose policy was [`OnConflict::Replace`].
13629    pub replaced: usize,
13630    /// View-owned properties an [`OnConflict::Replace`] row **kept** although
13631    /// the caller did not supply them — counted per field, so one row that
13632    /// keeps two contributes two.
13633    ///
13634    /// This is the one respect in which `Replace` does not make a node's props
13635    /// exactly the supplied ones (see [`OnConflict::Replace`]). Those rows
13636    /// still count in `replaced` and still raise no `row_errors`, because
13637    /// nothing went wrong: a view's property is not the caller's to supply or
13638    /// to remove. A mirror rebuild that needs its copy to be byte-exact reads
13639    /// this to learn that the store kept fields its frame did not describe.
13640    pub kept_view_owned: usize,
13641    /// `(row, why)` for rows an [`OnConflict::Replace`] refused. `row` counts
13642    /// node-insert ops in this batch from zero, which for a caller that queues
13643    /// its nodes in order is the index of the offending node. The rest of the
13644    /// frame still commits; the refused row changes nothing.
13645    pub row_errors: Vec<(usize, String)>,
13646}
13647
13648/// The `(nodes_inserted, edges_inserted)` pair every pre-0.6.10 commit entry
13649/// point returns. Keeps those signatures unchanged now that the validate pass
13650/// produces a [`BatchOutcome`].
13651fn inserted_pair(outcome: BatchOutcome) -> (usize, usize) {
13652    (outcome.nodes_inserted, outcome.edges_inserted)
13653}
13654
13655/// One entry of a frame the validate pass has decided on, before
13656/// [`GraphDb::rewrite_wal_dense_planned`] turns it into dense-id records.
13657///
13658/// Almost every entry is already a finished [`WalRecord`]. The exception is a
13659/// duplicate edge insert: its count names a dense triple, and on the batch path
13660/// the endpoints and the edge type may all be created by earlier records in the
13661/// *same* frame, so no id for them exists until the dense rewrite allocates it.
13662/// Carrying the keys this far and resolving them there is what lets the count
13663/// survive the shape a mirror rebuild writes (defect #24).
13664enum PlannedRec {
13665    Rec(WalRecord),
13666    DuplicateCount {
13667        edge_type: String,
13668        src_key: String,
13669        dst_key: String,
13670    },
13671}
13672
13673/// Queued mutation for a [`BatchBuilder`] or [`GraphDb::commit_group`].
13674///
13675/// The `submit_batch` / `commit_group` APIs accept `Vec<BatchOp>` so that
13676/// callers can build a set of mutations without holding `&mut GraphDb` and
13677/// hand them off to the group-committing writer for durable, batched I/O.
13678pub enum BatchOp {
13679    InsertNode {
13680        label: String,
13681        key: String,
13682        props: Vec<(String, Value)>,
13683    },
13684    InsertEdge {
13685        edge_type: String,
13686        src_key: String,
13687        dst_key: String,
13688    },
13689    SetProp {
13690        key: String,
13691        field: String,
13692        value: Value,
13693    },
13694    RemoveProp {
13695        key: String,
13696        field: String,
13697    },
13698    DeleteEdge {
13699        edge_type: String,
13700        src_key: String,
13701        dst_key: String,
13702    },
13703    DeleteNode {
13704        key: String,
13705    },
13706    CreateRule(RuleDef),
13707    DeleteRule {
13708        name: String,
13709    },
13710    /// Rename a node's key. Validated: old must exist, new must not.
13711    RenameNode {
13712        old_key: String,
13713        new_key: String,
13714    },
13715    /// Insert an edge, auto-creating any missing endpoint as a plain node with
13716    /// `placeholder_label` and no props. Rules fire and last-change is updated
13717    /// for each created endpoint (normal InsertNode semantics in the batch frame).
13718    InsertEdgeUpsert {
13719        edge_type: String,
13720        src_key: String,
13721        dst_key: String,
13722        placeholder_label: String,
13723    },
13724    /// Insert `key`, or — when the key is already taken — do what `on_conflict`
13725    /// says. Queued by [`BatchBuilder::insert_node_on_conflict`]; `Error`
13726    /// queues a plain [`BatchOp::InsertNode`] instead, so this variant only
13727    /// ever carries `Skip` or `Replace`.
13728    InsertNodeOnConflict {
13729        label: String,
13730        key: String,
13731        props: Vec<(String, Value)>,
13732        on_conflict: OnConflict,
13733    },
13734}
13735
13736/// Three-way node visibility status used by `check_single_op_authz`.
13737enum NodeAuthzStatus {
13738    /// Node exists in the store and is in the role's read mask.
13739    Visible(String), // carries the node's label
13740    /// Node exists in the store but is NOT in the role's read mask.
13741    Hidden,
13742    /// Node does not exist in the store.
13743    Absent,
13744}
13745
13746/// Overlay of ops already accepted earlier in the same batch. Never written
13747/// back to the database — validation only.
13748#[derive(Default)]
13749struct Overlay {
13750    extra_keys: BTreeSet<String>,
13751    /// Label of each node inserted earlier in this batch. The store does not
13752    /// have these keys yet, so `label_of` cannot answer for them, and
13753    /// `OnConflict::Replace` has to compare labels.
13754    extra_labels: BTreeMap<String, String>,
13755    deleted_keys: BTreeSet<String>,
13756    extra_props: BTreeMap<(String, String), Value>,
13757    removed_props: BTreeSet<(String, String)>,
13758    extra_edges: BTreeSet<(String, String, String)>,
13759    deleted_edges: BTreeSet<(String, String, String)>,
13760    extra_rules: BTreeSet<String>,
13761    deleted_rules: BTreeSet<String>,
13762    /// `rule name → (via_edge, edge_type)` for every via-hop rule accepted
13763    /// earlier in this batch. Feeds the rule-chain cycle check, which otherwise
13764    /// sees only the rules already committed to the engine. Keyed by name so a
13765    /// later `DeleteRule` in the same batch drops the arc with the rule.
13766    extra_rule_arcs: BTreeMap<String, (String, String)>,
13767}
13768
13769/// Read-only view of live db state plus a batch overlay. Shared by single-op
13770/// public methods (empty overlay) and `commit_batch`.
13771struct MutPreview<'a, F: Fs> {
13772    db: &'a GraphDb<F>,
13773    overlay: Overlay,
13774}
13775
13776/// Shortest path from `start` to `target` following `arcs` (`from → to`), or
13777/// `None` if `target` is unreachable.
13778///
13779/// Used for rule-chain cycle detection, where an arc is "a rule hops over
13780/// `from` and writes `to`". Breadth-first over BTree-ordered adjacency, so the
13781/// reported path is stable for a given rule set, and iterative so a pathological
13782/// rule graph cannot overflow the stack.
13783fn find_cycle_through(arcs: &[(String, String)], start: &str, target: &str) -> Option<Vec<String>> {
13784    let mut adj: BTreeMap<&str, BTreeSet<&str>> = BTreeMap::new();
13785    for (from, to) in arcs {
13786        adj.entry(from.as_str()).or_default().insert(to.as_str());
13787    }
13788    let mut parent: BTreeMap<&str, &str> = BTreeMap::new();
13789    let mut visited: BTreeSet<&str> = BTreeSet::new();
13790    let mut queue: std::collections::VecDeque<&str> = std::collections::VecDeque::new();
13791    visited.insert(start);
13792    queue.push_back(start);
13793    while let Some(node) = queue.pop_front() {
13794        if node == target {
13795            let mut path = vec![node.to_string()];
13796            let mut cur = node;
13797            while let Some(&p) = parent.get(cur) {
13798                path.push(p.to_string());
13799                cur = p;
13800            }
13801            path.reverse();
13802            return Some(path);
13803        }
13804        for &next in adj.get(node).into_iter().flatten() {
13805            if visited.insert(next) {
13806                parent.insert(next, node);
13807                queue.push_back(next);
13808            }
13809        }
13810    }
13811    None
13812}
13813
13814impl<'a, F: Fs> MutPreview<'a, F> {
13815    fn new(db: &'a GraphDb<F>) -> Self {
13816        Self {
13817            db,
13818            overlay: Overlay::default(),
13819        }
13820    }
13821
13822    fn has_key(&self, key: &str) -> bool {
13823        if self.overlay.extra_keys.contains(key) {
13824            return true;
13825        }
13826        if self.overlay.deleted_keys.contains(key) {
13827            return false;
13828        }
13829        self.db.ids.get(key).is_some()
13830    }
13831
13832    fn has_prop(&self, key: &str, field: &str) -> bool {
13833        if !self.has_key(key) {
13834            return false;
13835        }
13836        let k = (key.to_string(), field.to_string());
13837        if self.overlay.removed_props.contains(&k) {
13838            return false;
13839        }
13840        if self.overlay.extra_props.contains_key(&k) {
13841            return true;
13842        }
13843        // Fresh identity (first insert in this batch, or delete+reinsert):
13844        // ignore props still sitting on the soon-to-be-tombstoned slot.
13845        if self.overlay.extra_keys.contains(key) {
13846            return false;
13847        }
13848        self.db.get_prop(key, field).is_some()
13849    }
13850
13851    fn has_edge(&self, edge_type: &str, src_key: &str, dst_key: &str) -> bool {
13852        let k = (
13853            edge_type.to_string(),
13854            src_key.to_string(),
13855            dst_key.to_string(),
13856        );
13857        if self.overlay.deleted_edges.contains(&k) {
13858            return false;
13859        }
13860        if self.overlay.extra_edges.contains(&k) {
13861            return true;
13862        }
13863        // A key created in this batch (including reinsert) has no db edges.
13864        if self.overlay.extra_keys.contains(src_key) || self.overlay.extra_keys.contains(dst_key) {
13865            return false;
13866        }
13867        if self.overlay.deleted_keys.contains(src_key)
13868            || self.overlay.deleted_keys.contains(dst_key)
13869        {
13870            return false;
13871        }
13872        let Some(src) = self.db.ids.get(src_key) else {
13873            return false;
13874        };
13875        let Some(dst) = self.db.ids.get(dst_key) else {
13876            return false;
13877        };
13878        let Some(sym) = self.db.syms.get(edge_type) else {
13879            return false;
13880        };
13881        self.db
13882            .topo_view()
13883            .neighbors(sym, Direction::Out, src)
13884            .binary_search(&dst)
13885            .is_ok()
13886    }
13887
13888    fn has_rule(&self, name: &str) -> bool {
13889        if self.overlay.extra_rules.contains(name) {
13890            return true;
13891        }
13892        if self.overlay.deleted_rules.contains(name) {
13893            return false;
13894        }
13895        self.db.engine.rules().any(|r| r.name == name)
13896    }
13897
13898    fn is_rule_owned(&self, edge_type: &str, src_key: &str, dst_key: &str) -> bool {
13899        if self.overlay.extra_keys.contains(src_key) || self.overlay.extra_keys.contains(dst_key) {
13900            return false;
13901        }
13902        if self.overlay.deleted_keys.contains(src_key)
13903            || self.overlay.deleted_keys.contains(dst_key)
13904        {
13905            return false;
13906        }
13907        let Some(src) = self.db.ids.get(src_key) else {
13908            return false;
13909        };
13910        let Some(dst) = self.db.ids.get(dst_key) else {
13911            return false;
13912        };
13913        let Some(et) = self.db.syms.get(edge_type) else {
13914            return false;
13915        };
13916        // extra_rules is deliberately not consulted: a CreateRule earlier in
13917        // this batch has not fired, so it contributes no provenance. That is
13918        // the documented rule-window gap (see GraphDb::batch).
13919        if self.overlay.deleted_rules.is_empty() {
13920            return self.db.engine.is_owned(et, src, dst);
13921        }
13922        for (rule, triples) in self.db.engine.provenance() {
13923            if self.overlay.deleted_rules.contains(rule) {
13924                continue;
13925            }
13926            if triples.contains(&(et, src, dst)) {
13927                return true;
13928            }
13929        }
13930        false
13931    }
13932
13933    /// The refusals a node creation makes, in the order it makes them.
13934    ///
13935    /// A view owns its property, and creating a node that carries one is a
13936    /// write to it exactly as `set_prop` is — so it is refused here, at the one
13937    /// choke-point `GraphDb::insert_node`, `BatchOp::InsertNode` and the
13938    /// no-conflict arm of `BatchOp::InsertNodeOnConflict` all pass through.
13939    ///
13940    /// Leaving creation exempt was not harmless. The value was stored and
13941    /// served: a created node the view has no reason to revisit keeps the
13942    /// caller's number for the life of the handle, and the backfill at the next
13943    /// open overwrites it — so the store answered `deg = 777` before a restart
13944    /// and `deg = 0` after, for a property every other surface calls read-only.
13945    /// It also split one op two ways: supplying a view-owned field under
13946    /// `OnConflict::Replace` was already a row error on a taken key while the
13947    /// same field on a fresh key was accepted.
13948    ///
13949    /// Checked before the key, like [`MutPreview::prepare_remove_prop`], so the
13950    /// answer does not depend on whether the key exists.
13951    fn check_insert_node(&self, key: &str, props: &[(String, Value)]) -> Result<()> {
13952        for (field, _) in props {
13953            if let Some(view_name) = self.db.view_store.view_for_prop(field) {
13954                return Err(GraphError::ViewPropReadOnly {
13955                    view_name: view_name.to_string(),
13956                });
13957            }
13958        }
13959        if self.has_key(key) {
13960            Err(GraphError::DuplicateKey { key: key.into() })
13961        } else {
13962            Ok(())
13963        }
13964    }
13965
13966    fn check_live_key(&self, key: &str) -> Result<()> {
13967        if self.has_key(key) {
13968            Ok(())
13969        } else {
13970            Err(GraphError::KeyNotFound { key: key.into() })
13971        }
13972    }
13973
13974    fn prepare_insert_edge(&self, edge_type: &str, src_key: &str, dst_key: &str) -> Result<bool> {
13975        for k in [src_key, dst_key] {
13976            if !self.has_key(k) {
13977                return Err(GraphError::KeyNotFound { key: k.into() });
13978            }
13979        }
13980        if self.is_rule_owned(edge_type, src_key, dst_key) {
13981            return Err(GraphError::RuleOwned {
13982                detail: format!("edge {edge_type} {src_key}→{dst_key} is rule-owned"),
13983            });
13984        }
13985        // A user-written edge stays inside one namespace. Derived edges do not
13986        // come through here — the engine adds them directly — and the rule
13987        // scoping check is what keeps those pure.
13988        let src_ns = self.namespace_in_batch(src_key);
13989        let dst_ns = self.namespace_in_batch(dst_key);
13990        if src_ns != dst_ns {
13991            return Err(GraphError::CrossNamespace {
13992                src: src_key.to_string(),
13993                src_ns,
13994                dst: dst_key.to_string(),
13995                dst_ns,
13996            });
13997        }
13998        Ok(!self.has_edge(edge_type, src_key, dst_key))
13999    }
14000
14001    fn prepare_remove_prop(&self, key: &str, field: &str) -> Result<bool> {
14002        // A view owns its property, and the refusal has to live here rather
14003        // than on `GraphDb::remove_prop`: `BatchOp::RemoveProp` never meets
14004        // that one, and it is what the HTTP `DELETE /node/{key}/prop/{field}`
14005        // route, `Batch::remove_prop` and the CLI all submit. This is the one
14006        // choke-point every removal passes, exactly as it is for `ns` below.
14007        // Checked before the key, so the answer does not depend on whether the
14008        // key exists — which is also what `GraphDb::remove_prop` answered when
14009        // it carried the only copy of this guard.
14010        if let Some(view_name) = self.db.view_store.view_for_prop(field) {
14011            return Err(GraphError::ViewPropReadOnly {
14012                view_name: view_name.to_string(),
14013            });
14014        }
14015        self.check_live_key(key)?;
14016        // Removing `ns` is changing the namespace — to `default`, the namespace
14017        // an absent property names. It goes through this one choke-point and NOT
14018        // through `rewrite_wal_dense` (a `RemoveProp` needs no dense rewrite), so
14019        // the immutability rule has to be stated here as well. Without it the
14020        // node silently lands in `default` on the next open: the cross-namespace
14021        // edge guard is defeated and a default-bound role reads a tenant's node.
14022        if field == NS_PROP {
14023            let from = self.namespace_in_batch(key);
14024            if from != NS_DEFAULT {
14025                return Err(GraphError::NamespaceImmutable {
14026                    key: key.to_string(),
14027                    from,
14028                    to: NS_DEFAULT.to_string(),
14029                });
14030            }
14031            // Already in `default`: the removal changes no namespace. It is the
14032            // no-op `set_prop` to the current namespace is, not an error.
14033            return Ok(false);
14034        }
14035        Ok(self.has_prop(key, field))
14036    }
14037
14038    fn prepare_delete_edge(&self, edge_type: &str, src_key: &str, dst_key: &str) -> Result<bool> {
14039        for k in [src_key, dst_key] {
14040            if !self.has_key(k) {
14041                return Err(GraphError::KeyNotFound { key: k.into() });
14042            }
14043        }
14044        // Provenance-owned OR a live rule would derive this pair. User-first
14045        // edges that a later rule matches are not in `owned`, but deleting
14046        // them would leave a hole `rebuild_rule` immediately fills.
14047        if self.is_rule_owned(edge_type, src_key, dst_key) {
14048            return Err(GraphError::RuleOwned {
14049                detail: format!(
14050                    "edge {edge_type} {src_key}→{dst_key} is rule-owned; \
14051                     delete or change the owning rule"
14052                ),
14053            });
14054        }
14055        if self.would_derive(edge_type, src_key, dst_key) {
14056            return Err(GraphError::RuleOwned {
14057                detail: format!(
14058                    "edge {edge_type} {src_key}→{dst_key} is rule-owned; \
14059                     delete or change the owning rule, or a live rule would re-derive it"
14060                ),
14061            });
14062        }
14063        Ok(self.has_edge(edge_type, src_key, dst_key))
14064    }
14065
14066    /// True if any live rule (minus overlay-deleted names) would derive
14067    /// `(edge_type, src, dst)` from current overlay-visible props/labels.
14068    /// CreateRule names in `extra_rules` are ignored — same documented
14069    /// same-batch rule-window as [`Self::is_rule_owned`].
14070    fn would_derive(&self, edge_type: &str, src_key: &str, dst_key: &str) -> bool {
14071        if src_key == dst_key {
14072            return false;
14073        }
14074        let Some(src_label) = self.label_of(src_key) else {
14075            return false;
14076        };
14077        let Some(dst_label) = self.label_of(dst_key) else {
14078            return false;
14079        };
14080        for rule in self.db.engine.rules() {
14081            if self.overlay.deleted_rules.contains(&rule.name) {
14082                continue;
14083            }
14084            if rule.edge_type != edge_type {
14085                continue;
14086            }
14087            if rule.src_label != src_label || rule.dst_label != dst_label {
14088                continue;
14089            }
14090            let src_props = |f: &str| self.prop_value(src_key, f);
14091            let dst_props = |f: &str| self.prop_value(dst_key, f);
14092            let src_view = NodeView {
14093                key: src_key,
14094                props: &src_props,
14095            };
14096            let dst_view = NodeView {
14097                key: dst_key,
14098                props: &dst_props,
14099            };
14100            if evaluate(&rule.predicate, &src_view, &dst_view).is_some() {
14101                return true;
14102            }
14103        }
14104        false
14105    }
14106
14107    fn label_of(&self, key: &str) -> Option<String> {
14108        if self.overlay.deleted_keys.contains(key) {
14109            return None;
14110        }
14111        // Fresh identities created in this batch have no stored label in the
14112        // overlay; they cannot be provenance-owned yet either.
14113        let id = self.db.ids.get(key)?;
14114        let sym = self.db.labels.get(id as usize).copied()?;
14115        if sym == u32::MAX {
14116            return None;
14117        }
14118        self.db.syms.resolve(sym).map(str::to_string)
14119    }
14120
14121    /// The label `key` carries as this batch sees it — including a node
14122    /// inserted earlier in the same batch, which the store does not have yet.
14123    fn label_in_batch(&self, key: &str) -> Option<String> {
14124        if self.overlay.deleted_keys.contains(key) {
14125            return None;
14126        }
14127        if let Some(label) = self.overlay.extra_labels.get(key) {
14128            return Some(label.clone());
14129        }
14130        self.label_of(key)
14131    }
14132
14133    /// The property writes that make `key`'s props exactly `props`, or why the
14134    /// row is refused.
14135    ///
14136    /// `Some(value)` is a set and `None` is a removal. `store_fields` is every
14137    /// field name the store knows, hoisted by the caller so a frame of N
14138    /// replaces reads the field list once rather than N times.
14139    ///
14140    /// The second half of the pair is how many view-owned fields this row kept
14141    /// rather than removed — the one part of "exactly the supplied props" that
14142    /// does not hold, and the caller's only signal that it did not.
14143    ///
14144    /// The refusals are row errors, not frame errors: a mirror rebuild should
14145    /// learn which of its rows disagree with the store without losing the rows
14146    /// that agree.
14147    fn plan_replace(
14148        &self,
14149        label: &str,
14150        key: &str,
14151        props: &[(String, Value)],
14152        store_fields: &[String],
14153    ) -> std::result::Result<ReplacePlan, String> {
14154        // A different label is a relabel, and a rebuild does not relabel: that
14155        // is `rename_node` or an explicit write, never a side effect here.
14156        let stored = self.label_in_batch(key).unwrap_or_default();
14157        if stored != label {
14158            return Err(format!(
14159                "node {key}: on_conflict=\"replace\" will not relabel {stored:?} to {label:?}; \
14160                 relabelling is rename_node or an explicit write"
14161            ));
14162        }
14163        // `ns` is immutable. Replace removes what the supplied props omit, so
14164        // an omitted `ns` is a move to `default` exactly as a different `ns` is
14165        // a move to that one; both are the same refusal.
14166        let from = self.namespace_in_batch(key);
14167        let to = match props.iter().find(|(field, _)| field == NS_PROP) {
14168            Some((_, Value::Str(ns))) => ns.clone(),
14169            Some((_, value)) => {
14170                return Err(format!(
14171                    "node {key}: {NS_PROP} must be a string naming a namespace, got {value:?}"
14172                ));
14173            }
14174            None => NS_DEFAULT.to_string(),
14175        };
14176        if to != from {
14177            return Err(format!(
14178                "node {key}: {NS_PROP} is immutable; on_conflict=\"replace\" cannot move it \
14179                 from {from:?} to {to:?}"
14180            ));
14181        }
14182
14183        if let Some(why) = self.supplied_view_owned_prop(key, props) {
14184            return Err(why);
14185        }
14186
14187        let supplied: BTreeSet<&str> = props.iter().map(|(field, _)| field.as_str()).collect();
14188        let mut writes = Vec::new();
14189        for (field, value) in props {
14190            // `ns` names the namespace the node is already in, so the write is
14191            // the no-op the dense-rewrite seam would drop anyway.
14192            if field == NS_PROP {
14193                continue;
14194            }
14195            // Already exactly this value: a rebuild of an unchanged row should
14196            // cost no WAL record.
14197            if self.prop_value(key, field).as_ref() == Some(value) {
14198                continue;
14199            }
14200            writes.push((field.clone(), Some(value.clone())));
14201        }
14202        // Everything the node still carries that the supplied props do not.
14203        // `ns` is never removed: it is immutable, and the check above has
14204        // already established the node stays where it is.
14205        let overlay_fields = self
14206            .overlay
14207            .extra_props
14208            .keys()
14209            .filter(|(k, _)| k == key)
14210            .map(|(_, field)| field.as_str());
14211        //
14212        // A view-owned field is filtered out rather than refused. It is not the
14213        // caller's to supply (supplying one is still the row error above) and
14214        // so it is not part of what "exactly the supplied ones" ranges over:
14215        // omitting it is not a request to delete it. Refusing here instead
14216        // would make `replace` impossible for every node a view has written to
14217        // — which on a store carrying a view is the whole population a mirror
14218        // rebuild has to cover.
14219        let omitted: BTreeSet<&str> = store_fields
14220            .iter()
14221            .map(String::as_str)
14222            .chain(overlay_fields)
14223            .filter(|field| {
14224                *field != NS_PROP && !supplied.contains(field) && self.has_prop(key, field)
14225            })
14226            .collect();
14227        // The view-owned half is kept, and counted: the row still commits and
14228        // still reports no error, so without this number a mirror rebuild is
14229        // told it got exactly what it asked for when it did not (defect #18).
14230        let (stale, kept): (Vec<&str>, Vec<&str>) = omitted
14231            .into_iter()
14232            .partition(|field| self.db.view_store.view_for_prop(field).is_none());
14233        writes.extend(stale.into_iter().map(|field| (field.to_string(), None)));
14234        Ok((writes, kept.len()))
14235    }
14236
14237    /// The row error a supplied view-owned field earns, or `None`.
14238    ///
14239    /// Shared by [`MutPreview::plan_replace`] and the no-conflict arm of
14240    /// `BatchOp::InsertNodeOnConflict` so that one op answers a supplied
14241    /// view-owned field the same way whether or not the key was already taken.
14242    fn supplied_view_owned_prop(&self, key: &str, props: &[(String, Value)]) -> Option<String> {
14243        props.iter().find_map(|(field, _)| {
14244            self.db.view_store.view_for_prop(field).map(|view_name| {
14245                format!(
14246                    "node {key}: property {field:?} is owned by view {view_name:?} and is \
14247                     read-only"
14248                )
14249            })
14250        })
14251    }
14252
14253    /// The namespace `key` is in as this batch sees it — including a node
14254    /// inserted earlier in the same batch, which the store does not have yet.
14255    fn namespace_in_batch(&self, key: &str) -> String {
14256        namespace_of_value(self.prop_value(key, NS_PROP).as_ref()).to_string()
14257    }
14258
14259    fn prop_value(&self, key: &str, field: &str) -> Option<Value> {
14260        if !self.has_key(key) {
14261            return None;
14262        }
14263        let k = (key.to_string(), field.to_string());
14264        if self.overlay.removed_props.contains(&k) {
14265            return None;
14266        }
14267        if let Some(v) = self.overlay.extra_props.get(&k) {
14268            return Some(v.clone());
14269        }
14270        if self.overlay.extra_keys.contains(key) {
14271            return None;
14272        }
14273        self.db.get_prop(key, field)
14274    }
14275
14276    fn check_create_rule(&self, def: &RuleDef) -> Result<()> {
14277        def.validate()
14278            .map_err(|e| GraphError::RuleInvalid { detail: e })?;
14279        if self.has_rule(&def.name) {
14280            return Err(GraphError::RuleInvalid {
14281                detail: format!("rule {:?} already exists", def.name),
14282            });
14283        }
14284        // Rule-chain cycle rejection. Derived edges feed via-hop rules, so a
14285        // rule set forms a graph whose arcs are "hops over `via_edge`, writes
14286        // `edge_type`". A cycle in that graph is a rule set that would re-fire
14287        // itself forever; the engine's depth cap would silently truncate it
14288        // instead, leaving an arbitrary partial result. Reject it here, the one
14289        // place that sees the whole rule set.
14290        //
14291        // Rules accepted earlier in the same batch count too: the overlay
14292        // carries their arcs, so a cycle cannot be assembled one op at a time.
14293        if let Some(via) = def.via_edge.as_deref() {
14294            if via == def.edge_type {
14295                return Err(GraphError::RuleInvalid {
14296                    detail: format!("rule chain cycle: {} -> {}", via, def.edge_type),
14297                });
14298            }
14299            let mut arcs: Vec<(String, String)> = self
14300                .db
14301                .engine
14302                .rules()
14303                .filter(|r| !self.overlay.deleted_rules.contains(&r.name))
14304                .filter_map(|r| r.via_edge.clone().map(|v| (v, r.edge_type.clone())))
14305                .collect();
14306            arcs.extend(self.overlay.extra_rule_arcs.values().cloned());
14307            arcs.push((via.to_string(), def.edge_type.clone()));
14308            if let Some(path) = find_cycle_through(&arcs, &def.edge_type, via) {
14309                return Err(GraphError::RuleInvalid {
14310                    detail: format!("rule chain cycle: {} -> {}", via, path.join(" -> ")),
14311                });
14312            }
14313        }
14314        Ok(())
14315    }
14316
14317    fn check_delete_rule(&self, name: &str) -> Result<()> {
14318        if self.has_rule(name) {
14319            Ok(())
14320        } else {
14321            Err(GraphError::RuleNotFound { name: name.into() })
14322        }
14323    }
14324
14325    fn note_insert_node(&mut self, label: &str, key: &str, props: &[(String, Value)]) {
14326        self.overlay.deleted_keys.remove(key);
14327        self.overlay.extra_keys.insert(key.to_string());
14328        self.overlay
14329            .extra_labels
14330            .insert(key.to_string(), label.to_string());
14331        self.overlay.extra_props.retain(|(k, _), _| k != key);
14332        self.overlay.removed_props.retain(|(k, _)| k != key);
14333        for (field, value) in props {
14334            self.overlay
14335                .extra_props
14336                .insert((key.to_string(), field.clone()), value.clone());
14337        }
14338    }
14339
14340    fn note_insert_edge(&mut self, edge_type: &str, src_key: &str, dst_key: &str) {
14341        let k = (
14342            edge_type.to_string(),
14343            src_key.to_string(),
14344            dst_key.to_string(),
14345        );
14346        self.overlay.deleted_edges.remove(&k);
14347        self.overlay.extra_edges.insert(k);
14348    }
14349
14350    fn note_set_prop(&mut self, key: &str, field: &str, value: &Value) {
14351        let k = (key.to_string(), field.to_string());
14352        self.overlay.removed_props.remove(&k);
14353        self.overlay.extra_props.insert(k, value.clone());
14354    }
14355
14356    fn note_remove_prop(&mut self, key: &str, field: &str) {
14357        let k = (key.to_string(), field.to_string());
14358        self.overlay.extra_props.remove(&k);
14359        self.overlay.removed_props.insert(k);
14360    }
14361
14362    fn note_delete_edge(&mut self, edge_type: &str, src_key: &str, dst_key: &str) {
14363        let k = (
14364            edge_type.to_string(),
14365            src_key.to_string(),
14366            dst_key.to_string(),
14367        );
14368        self.overlay.extra_edges.remove(&k);
14369        self.overlay.deleted_edges.insert(k);
14370    }
14371
14372    fn note_delete_node(&mut self, key: &str) {
14373        self.overlay.extra_keys.remove(key);
14374        self.overlay.deleted_keys.insert(key.to_string());
14375        self.overlay.extra_props.retain(|(k, _), _| k != key);
14376        self.overlay.removed_props.retain(|(k, _)| k != key);
14377        self.overlay
14378            .extra_edges
14379            .retain(|(_, s, d)| s != key && d != key);
14380        self.overlay
14381            .deleted_edges
14382            .retain(|(_, s, d)| s != key && d != key);
14383    }
14384
14385    fn note_create_rule(&mut self, def: &RuleDef) {
14386        self.overlay.deleted_rules.remove(&def.name);
14387        self.overlay.extra_rules.insert(def.name.clone());
14388        // Rules accepted earlier in this batch are not in the engine yet, so
14389        // the cycle check would not see their arcs. Keep the arc, not just the
14390        // name, so a batch cannot smuggle in a cycle one op at a time.
14391        if let Some(via) = def.via_edge.clone() {
14392            self.overlay
14393                .extra_rule_arcs
14394                .insert(def.name.clone(), (via, def.edge_type.clone()));
14395        }
14396    }
14397
14398    fn check_rename_node(&self, old: &str, new: &str) -> Result<()> {
14399        if !self.has_key(old) {
14400            return Err(GraphError::KeyNotFound { key: old.into() });
14401        }
14402        if self.has_key(new) {
14403            return Err(GraphError::DuplicateKey { key: new.into() });
14404        }
14405        Ok(())
14406    }
14407
14408    fn note_rename_node(&mut self, old: &str, new: &str) {
14409        // Mark old as deleted so subsequent batch ops cannot reference it.
14410        self.overlay.extra_keys.remove(old);
14411        self.overlay.deleted_keys.insert(old.to_string());
14412        // Mark new as extra so subsequent batch ops can reference it.
14413        self.overlay.deleted_keys.remove(new);
14414        self.overlay.extra_keys.insert(new.to_string());
14415        // Migrate any overlay props from old key to new key.
14416        let new_str = new.to_string();
14417        let transferred: Vec<((String, String), Value)> = self
14418            .overlay
14419            .extra_props
14420            .iter()
14421            .filter(|((k, _), _)| k.as_str() == old)
14422            .map(|((_, f), v)| ((new_str.clone(), f.clone()), v.clone()))
14423            .collect();
14424        self.overlay
14425            .extra_props
14426            .retain(|(k, _), _| k.as_str() != old);
14427        for (k, v) in transferred {
14428            self.overlay.extra_props.insert(k, v);
14429        }
14430        // Migrate removed_props.
14431        let transferred_removed: Vec<(String, String)> = self
14432            .overlay
14433            .removed_props
14434            .iter()
14435            .filter(|(k, _)| k.as_str() == old)
14436            .map(|(_, f)| (new_str.clone(), f.clone()))
14437            .collect();
14438        self.overlay
14439            .removed_props
14440            .retain(|(k, _)| k.as_str() != old);
14441        for k in transferred_removed {
14442            self.overlay.removed_props.insert(k);
14443        }
14444    }
14445
14446    fn note_delete_rule(&mut self, name: &str) {
14447        self.overlay.extra_rules.remove(name);
14448        // Drop its chain arc too: a rule created and then deleted in the same
14449        // batch must not make a later, legal rule look like a cycle.
14450        self.overlay.extra_rule_arcs.remove(name);
14451        self.overlay.deleted_rules.insert(name.to_string());
14452        // Treat the deleted rule's current provenance as gone so a later
14453        // delete_edge of those triples is a no-op (matches sequential).
14454        if let Some(triples) = self.db.engine.provenance().get(name) {
14455            for &(et, s, d) in triples {
14456                let Some(etype) = self.db.syms.resolve(et) else {
14457                    continue;
14458                };
14459                let Some(src) = self.db.ids.key_of(s) else {
14460                    continue;
14461                };
14462                let Some(dst) = self.db.ids.key_of(d) else {
14463                    continue;
14464                };
14465                let k = (etype.to_string(), src.to_string(), dst.to_string());
14466                self.overlay.extra_edges.remove(&k);
14467                self.overlay.deleted_edges.insert(k);
14468            }
14469        }
14470    }
14471}
14472
14473/// Collects mutations and commits them as one WAL `Batch` frame.
14474///
14475/// Holds `&mut GraphDb` for its lifetime. Queue with the same method names
14476/// as [`GraphDb`]; call [`commit`](Self::commit) to validate, log, and apply.
14477/// See [`GraphDb::batch`] for validation and atomicity rules.
14478pub struct BatchBuilder<'a, F: Fs> {
14479    db: &'a mut GraphDb<F>,
14480    ops: Vec<BatchOp>,
14481}
14482
14483impl<'a, F: Fs> BatchBuilder<'a, F> {
14484    pub fn insert_node(
14485        &mut self,
14486        label: &str,
14487        key: &str,
14488        props: Vec<(String, Value)>,
14489    ) -> &mut Self {
14490        self.ops.push(BatchOp::InsertNode {
14491            label: label.into(),
14492            key: key.into(),
14493            props,
14494        });
14495        self
14496    }
14497
14498    /// Queue a node insert whose answer to a taken key is `on_conflict`.
14499    ///
14500    /// [`OnConflict::Error`] queues exactly the op [`insert_node`](Self::insert_node)
14501    /// does, so the default path is unchanged.
14502    pub fn insert_node_on_conflict(
14503        &mut self,
14504        label: &str,
14505        key: &str,
14506        props: Vec<(String, Value)>,
14507        on_conflict: OnConflict,
14508    ) -> &mut Self {
14509        self.ops.push(match on_conflict {
14510            OnConflict::Error => BatchOp::InsertNode {
14511                label: label.into(),
14512                key: key.into(),
14513                props,
14514            },
14515            on_conflict => BatchOp::InsertNodeOnConflict {
14516                label: label.into(),
14517                key: key.into(),
14518                props,
14519                on_conflict,
14520            },
14521        });
14522        self
14523    }
14524
14525    pub fn insert_edge(&mut self, edge_type: &str, src_key: &str, dst_key: &str) -> &mut Self {
14526        self.ops.push(BatchOp::InsertEdge {
14527            edge_type: edge_type.into(),
14528            src_key: src_key.into(),
14529            dst_key: dst_key.into(),
14530        });
14531        self
14532    }
14533
14534    pub fn set_prop(&mut self, key: &str, field: &str, value: Value) -> &mut Self {
14535        self.ops.push(BatchOp::SetProp {
14536            key: key.into(),
14537            field: field.into(),
14538            value,
14539        });
14540        self
14541    }
14542
14543    pub fn remove_prop(&mut self, key: &str, field: &str) -> &mut Self {
14544        self.ops.push(BatchOp::RemoveProp {
14545            key: key.into(),
14546            field: field.into(),
14547        });
14548        self
14549    }
14550
14551    pub fn delete_edge(&mut self, edge_type: &str, src_key: &str, dst_key: &str) -> &mut Self {
14552        self.ops.push(BatchOp::DeleteEdge {
14553            edge_type: edge_type.into(),
14554            src_key: src_key.into(),
14555            dst_key: dst_key.into(),
14556        });
14557        self
14558    }
14559
14560    pub fn delete_node(&mut self, key: &str) -> &mut Self {
14561        self.ops.push(BatchOp::DeleteNode { key: key.into() });
14562        self
14563    }
14564
14565    pub fn create_rule(&mut self, def: RuleDef) -> &mut Self {
14566        self.ops.push(BatchOp::CreateRule(def));
14567        self
14568    }
14569
14570    pub fn delete_rule(&mut self, name: &str) -> &mut Self {
14571        self.ops.push(BatchOp::DeleteRule { name: name.into() });
14572        self
14573    }
14574
14575    /// Queue a node-rename in this batch.
14576    ///
14577    /// Validation (old exists, new not taken) runs at commit time.
14578    pub fn rename_node(&mut self, old_key: &str, new_key: &str) -> &mut Self {
14579        self.ops.push(BatchOp::RenameNode {
14580            old_key: old_key.into(),
14581            new_key: new_key.into(),
14582        });
14583        self
14584    }
14585
14586    /// Queue an edge insert with endpoint auto-creation.
14587    ///
14588    /// Any missing endpoint is created as a plain node `{key, label:
14589    /// placeholder_label, no props}` inside this batch frame. Rules fire and
14590    /// last-change is updated for each auto-created node.
14591    pub fn insert_edge_upsert(
14592        &mut self,
14593        edge_type: &str,
14594        src_key: &str,
14595        dst_key: &str,
14596        placeholder_label: &str,
14597    ) -> &mut Self {
14598        self.ops.push(BatchOp::InsertEdgeUpsert {
14599            edge_type: edge_type.into(),
14600            src_key: src_key.into(),
14601            dst_key: dst_key.into(),
14602            placeholder_label: placeholder_label.into(),
14603        });
14604        self
14605    }
14606
14607    /// Validate every queued op, then log one `Batch` frame and apply.
14608    /// Empty / all-noop batches return `Ok(())` without writing the WAL.
14609    /// A second `commit()` after a successful one is an empty-batch no-op
14610    /// (queued ops were taken).
14611    /// Takes `&mut self` so it chains after the queue methods (`b.insert_node(..).commit()`)
14612    /// and also works as `let mut b = db.batch(); b.insert_node(..); b.commit()`.
14613    ///
14614    /// **Rule-window limitation:** batch validation cannot see edges that a
14615    /// rule created earlier in the *same* batch will derive at apply time, so
14616    /// a `delete_edge` / `insert_edge` in that window is silently no-oped
14617    /// where sequential calls would return `Err(RuleOwned)`. State integrity
14618    /// is unaffected (idempotent apply, provenance intact). Create rules in
14619    /// their own batch, or sequentially, when later ops may touch derived
14620    /// edges.
14621    /// Validate every queued op and commit atomically.
14622    ///
14623    /// Returns `(nodes_inserted, edges_inserted)` — the counts of node and edge
14624    /// WAL records actually written (duplicate edges are silent no-ops and are
14625    /// NOT counted). Both are 0 when the batch is empty or all-noop.
14626    pub fn commit(&mut self) -> Result<(usize, usize)> {
14627        let ops = std::mem::take(&mut self.ops);
14628        self.db.commit_batch(ops)
14629    }
14630
14631    /// [`commit`](Self::commit) with the full [`BatchOutcome`] — the counts a
14632    /// caller needs when its rows carry an [`OnConflict`] policy.
14633    pub fn commit_outcome(&mut self) -> Result<BatchOutcome> {
14634        let ops = std::mem::take(&mut self.ops);
14635        self.db.commit_logged_batch(ops, None, None)
14636    }
14637
14638    /// Same as [`commit`](Self::commit) but tail the inner events with
14639    /// [`MutationEvent::Ingested`] instead of [`MutationEvent::BatchApplied`].
14640    pub(crate) fn commit_ingest(&mut self, label: &str, inserted: usize) -> Result<(usize, usize)> {
14641        let ops = std::mem::take(&mut self.ops);
14642        self.db
14643            .commit_logged_batch(ops, Some((label.to_string(), inserted)), None)
14644            .map(inserted_pair)
14645    }
14646}
14647
14648pub struct NodeRef<'a, F: Fs> {
14649    db: &'a GraphDb<F>,
14650    id: u32,
14651}
14652
14653impl<'a, F: Fs> NodeRef<'a, F> {
14654    pub fn key(&self) -> &str {
14655        self.db.ids.key_of(self.id).expect("dense ids")
14656    }
14657
14658    pub fn label(&self) -> &str {
14659        let sym = self
14660            .db
14661            .labels
14662            .get(self.id as usize)
14663            .copied()
14664            .filter(|&s| s != u32::MAX)
14665            .expect("real nodes always have a label; u32::MAX sentinel cannot occur");
14666        self.db.syms.resolve(sym).expect("interned label symbol")
14667    }
14668
14669    pub fn prop(&self, field: &str) -> Option<Value> {
14670        self.db
14671            .props_view()
14672            .get(self.id, field)
14673            .map(|vr| vr.into_value())
14674    }
14675
14676    /// All stored fields for this node, sorted by field name.
14677    ///
14678    /// Reads from the full base+overlay view so that props stored only in the
14679    /// V8 snapshot base (i.e. before any post-snapshot WAL writes) are visible.
14680    pub fn props(&self) -> BTreeMap<String, Value> {
14681        let mut out = BTreeMap::new();
14682        let pv = self.db.props_view();
14683        for field in pv.field_names() {
14684            if let Some(vr) = pv.get(self.id, &field) {
14685                out.insert(field, vr.into_value());
14686            }
14687        }
14688        out
14689    }
14690
14691    /// depth-N BFS as a ResultSet: columns ["key","label","depth"], BFS order.
14692    pub fn neighborhood(&self, depth: u32, edge_types: Option<&[&str]>, dir: Dir) -> ResultSet {
14693        let view = self.db.view();
14694        let resolved: Option<Vec<u32>> = edge_types.map(|names| {
14695            names
14696                .iter()
14697                .filter_map(|name| view.syms.get(name))
14698                .collect()
14699        });
14700        let nb = neighborhood(&view, self.id, depth, resolved.as_deref(), dir);
14701        let mut rs = ResultSet::new(vec!["key".into(), "label".into(), "depth".into()]);
14702        for (nid, d) in nb.nodes {
14703            let key = view.key_of(nid);
14704            let label = view
14705                .label_of(nid)
14706                .expect("real nodes always have a label; u32::MAX sentinel cannot occur");
14707            rs.push_row(vec![
14708                Some(Value::Str(key.to_string())),
14709                Some(Value::Str(label.to_string())),
14710                Some(Value::Int(d as i64)),
14711            ]);
14712        }
14713        rs
14714    }
14715
14716    /// 1-hop, Both directions: edge-type name → sorted unique neighbor keys.
14717    pub fn grouped_by_edge_type(&self) -> BTreeMap<String, Vec<String>> {
14718        let view = self.db.view();
14719        let mut groups: BTreeMap<String, BTreeSet<String>> = BTreeMap::new();
14720        for e in expand(&view, self.id, None, Dir::Both) {
14721            // Skip edges with unknown etypes (only possible from corrupt large
14722            // TOPOLOGY section; function returns BTreeMap not Result).
14723            let Some(etype) = view.syms.resolve(e.etype) else {
14724                continue;
14725            };
14726            let etype = etype.to_string();
14727            let nbr = if e.src == self.id { e.dst } else { e.src };
14728            groups
14729                .entry(etype)
14730                .or_default()
14731                .insert(view.key_of(nbr).to_string());
14732        }
14733        groups
14734            .into_iter()
14735            .map(|(k, v)| (k, v.into_iter().collect()))
14736            .collect()
14737    }
14738}
14739
14740#[cfg(test)]
14741mod tests {
14742    use super::*;
14743    use core_rules::Predicate;
14744
14745    fn tmp_dir(name: &str) -> std::path::PathBuf {
14746        let d =
14747            std::env::temp_dir().join(format!("graphdb-db-unit-{}-{}", name, std::process::id()));
14748        let _ = std::fs::remove_dir_all(&d);
14749        d
14750    }
14751
14752    fn fk_rule() -> RuleDef {
14753        RuleDef {
14754            name: "works_at".into(),
14755            src_label: "Person".into(),
14756            dst_label: "Org".into(),
14757            predicate: Predicate::KeyMatch {
14758                field: "org_id".into(),
14759            },
14760            edge_type: "WORKS_AT".into(),
14761            weight_prop: None,
14762            max_edges: None,
14763            approximate: false,
14764            via_label: None,
14765            via_edge: None,
14766            via_dir: None,
14767            namespace: None,
14768        }
14769    }
14770
14771    /// Regression guard for the no-views delta-copy fast path.
14772    ///
14773    /// When no views are defined, `pending_deltas_since().to_vec()` must never
14774    /// be called — even during a large CreateRule backfill. The DELTA_COPY_COUNT
14775    /// thread-local is incremented inside every `if !view_store.is_empty()` block;
14776    /// a count of 0 after the entire sequence proves the guard fires correctly.
14777    #[test]
14778    fn no_delta_copy_when_no_views() {
14779        DELTA_COPY_COUNT.with(|c| c.set(0));
14780        let dir = tmp_dir("no-delta-copy");
14781        {
14782            let mut db = GraphDb::open(&dir).unwrap();
14783            // Insert 50 Org + 50 Person nodes with FK links.
14784            for i in 0..50u32 {
14785                db.insert_node("Org", &format!("o{i}"), vec![]).unwrap();
14786            }
14787            for i in 0..50u32 {
14788                db.insert_node(
14789                    "Person",
14790                    &format!("p{i}"),
14791                    vec![("org_id".into(), Value::Str(format!("o{i}")))],
14792                )
14793                .unwrap();
14794            }
14795            // CreateRule backfill should NOT invoke to_vec() when no views are defined.
14796            db.create_rule(fk_rule()).unwrap();
14797
14798            // Counter must stay 0 — no views, no copies.
14799            let copies = DELTA_COPY_COUNT.with(|c| c.get());
14800            assert_eq!(
14801                copies, 0,
14802                "pending_deltas_since().to_vec() called despite no views"
14803            );
14804
14805            // Derived edges must still be correct (the guard skips only the
14806            // empty delta propagation loop, not the rule application itself).
14807            let nbrs = db.neighbors("p0", "WORKS_AT", Direction::Out).unwrap();
14808            assert_eq!(
14809                nbrs,
14810                vec!["o0"],
14811                "rule must derive edges even with no views"
14812            );
14813        }
14814        let _ = std::fs::remove_dir_all(&dir);
14815    }
14816
14817    /// Gating regression: subscribe AFTER a backfill must see no stale events.
14818    /// subscribe BEFORE a backfill must see every edge-fire event.
14819    #[test]
14820    fn subscribe_after_backfill_no_stale_events() {
14821        let dir = tmp_dir("sub-after-backfill");
14822        {
14823            let mut db = GraphDb::open(&dir).unwrap();
14824            for i in 0..10u32 {
14825                db.insert_node("Org", &format!("o{i}"), vec![]).unwrap();
14826                db.insert_node(
14827                    "Person",
14828                    &format!("p{i}"),
14829                    vec![("org_id".into(), Value::Str(format!("o{i}")))],
14830                )
14831                .unwrap();
14832            }
14833            // Create rule BEFORE subscribing — emit_deltas is false during backfill.
14834            db.create_rule(fk_rule()).unwrap();
14835
14836            // Subscribe AFTER the backfill — queue must be empty (no stale events).
14837            let sub = db.subscribe_all_rules().unwrap();
14838            // No events should have queued for the prior backfill.
14839            assert!(
14840                sub.try_recv().is_none(),
14841                "subscribe after backfill must see no stale events"
14842            );
14843
14844            // Inserting a new node now should fire an event (emit_deltas is now true).
14845            db.insert_node("Org", "o_new", vec![]).unwrap();
14846            db.insert_node(
14847                "Person",
14848                "p_new",
14849                vec![("org_id".into(), Value::Str("o_new".into()))],
14850            )
14851            .unwrap();
14852            let ev = sub.recv_timeout(std::time::Duration::from_millis(200));
14853            assert!(
14854                ev.is_some(),
14855                "edge-fire event must arrive after subscribe (emit_deltas=true)"
14856            );
14857        }
14858        let _ = std::fs::remove_dir_all(&dir);
14859    }
14860
14861    /// Gating regression: subscribe BEFORE a backfill → events flow.
14862    #[test]
14863    fn subscribe_before_backfill_events_flow() {
14864        let dir = tmp_dir("sub-before-backfill");
14865        {
14866            let mut db = GraphDb::open(&dir).unwrap();
14867            // Subscribe FIRST — emit_deltas becomes true.
14868            let sub = db.subscribe_all_rules().unwrap();
14869
14870            for i in 0..5u32 {
14871                db.insert_node("Org", &format!("o{i}"), vec![]).unwrap();
14872                db.insert_node(
14873                    "Person",
14874                    &format!("p{i}"),
14875                    vec![("org_id".into(), Value::Str(format!("o{i}")))],
14876                )
14877                .unwrap();
14878            }
14879            // Backfill fires with emit_deltas=true → events queued.
14880            db.create_rule(fk_rule()).unwrap();
14881
14882            // Should receive at least one edge-fired event from the backfill.
14883            let mut received = 0usize;
14884            while sub.try_recv().is_some() {
14885                received += 1;
14886            }
14887            assert!(
14888                received > 0,
14889                "subscribe before backfill must receive edge-fire events (got 0)"
14890            );
14891        }
14892        let _ = std::fs::remove_dir_all(&dir);
14893    }
14894
14895    /// Companion: when a view IS defined, the delta path fires and view values update.
14896    #[test]
14897    fn delta_copy_fires_when_view_exists() {
14898        use core_rules::ViewSource;
14899        DELTA_COPY_COUNT.with(|c| c.set(0));
14900        let dir = tmp_dir("delta-copy-with-view");
14901        {
14902            let mut db = GraphDb::open(&dir).unwrap();
14903            db.insert_node("Org", "o1", vec![]).unwrap();
14904            db.insert_node(
14905                "Person",
14906                "p1",
14907                vec![("org_id".into(), Value::Str("o1".into()))],
14908            )
14909            .unwrap();
14910            // Declare a Degree view so is_empty() returns false.
14911            db.create_view(ViewDef {
14912                name: "degree_out".into(),
14913                label: "Person".into(),
14914                view_prop: "degree_out".into(),
14915                source: ViewSource::Degree {
14916                    edge_type: "WORKS_AT".into(),
14917                    direction: Direction::Out,
14918                },
14919            })
14920            .unwrap();
14921            db.create_rule(fk_rule()).unwrap();
14922
14923            // At least one delta copy should have happened (CreateRule backfill).
14924            let copies = DELTA_COPY_COUNT.with(|c| c.get());
14925            assert!(
14926                copies > 0,
14927                "expected delta copy to fire when a view is defined"
14928            );
14929
14930            // View value should be computed: p1 has one WORKS_AT out-edge.
14931            let info = db.node_info("p1").unwrap();
14932            let degree = info.props.get("degree_out");
14933            assert!(
14934                degree.is_some(),
14935                "view prop should be written to node props"
14936            );
14937        }
14938        let _ = std::fs::remove_dir_all(&dir);
14939    }
14940
14941    /// Regression: `open_at_with` must call `rebuild_all` after WAL replay so
14942    /// derived-edge-driven view values reflect the as-of state rather than just
14943    /// the initial backfill written at `CreateView` time.
14944    ///
14945    /// Base WAL frames (indices 0..=5 before history markers):
14946    ///   0: insert Org "o1"
14947    ///   1: create_view "employee_count" (Degree / WORKS_AT / In) on Org
14948    ///   2: create_rule fk_rule (WORKS_AT, Person→Org via org_id)
14949    ///   3: insert Person "p1" → rule fires WORKS_AT p1→o1 (degree = 1)  ← mid
14950    ///   4: insert Person "p2" → rule fires WORKS_AT p2→o1 (degree = 2)
14951    ///   5: insert Person "p3" → rule fires WORKS_AT p3→o1 (degree = 3)  ← latest
14952    ///
14953    /// Each rule-fire also appends a DerivedEdgeAdded history-marker frame (state
14954    /// no-op), so the total commit count is higher than the base frame count.
14955    /// The "latest" open_at commit is computed dynamically via `wal_commit_count_at`.
14956    ///
14957    /// Without `rebuild_all`, the as-of instance's "emp" view stays at the
14958    /// initial backfill value (0) instead of reflecting the replayed derived edges.
14959    #[test]
14960    fn open_at_derived_edge_view_values_correct() {
14961        use core_rules::ViewSource;
14962        let dir = tmp_dir("open-at-view-rebuild");
14963        {
14964            let mut db = GraphDb::open(&dir).unwrap();
14965            // frame 0
14966            db.insert_node("Org", "o1", vec![]).unwrap();
14967            // frame 1: create view — initial backfill sees 0 derived edges (none fired yet)
14968            db.create_view(ViewDef {
14969                name: "employee_count".into(),
14970                label: "Org".into(),
14971                view_prop: "emp".into(),
14972                source: ViewSource::Degree {
14973                    edge_type: "WORKS_AT".into(),
14974                    direction: Direction::In,
14975                },
14976            })
14977            .unwrap();
14978            // frame 2: create rule — no Persons yet; backfill is a no-op
14979            db.create_rule(fk_rule()).unwrap();
14980            // frame 3: p1 — rule fires WORKS_AT p1→o1; degree = 1
14981            db.insert_node(
14982                "Person",
14983                "p1",
14984                vec![("org_id".into(), Value::Str("o1".into()))],
14985            )
14986            .unwrap();
14987            // frame 4: p2 — degree = 2
14988            db.insert_node(
14989                "Person",
14990                "p2",
14991                vec![("org_id".into(), Value::Str("o1".into()))],
14992            )
14993            .unwrap();
14994            // frame 5: p3 — degree = 3
14995            db.insert_node(
14996                "Person",
14997                "p3",
14998                vec![("org_id".into(), Value::Str("o1".into()))],
14999            )
15000            .unwrap();
15001            // Sanity: normal open sees degree = 3.
15002            assert_eq!(
15003                db.get_view_prop("o1", "emp"),
15004                Some(Value::Int(3)),
15005                "normal db must show degree 3 after 3 derived edges"
15006            );
15007        } // WAL flushed
15008
15009        // Re-open normally to get the authoritative reference value.
15010        let normal_db = GraphDb::open(&dir).unwrap();
15011        let normal_emp = normal_db.get_view_prop("o1", "emp");
15012        assert_eq!(
15013            normal_emp,
15014            Some(Value::Int(3)),
15015            "re-opened normal db must show degree 3"
15016        );
15017
15018        // Latest as-of (last WAL commit): must match the normal open.
15019        // History-marker frames are appended after each rule-fire, so the total
15020        // commit count is computed dynamically rather than hardcoded.
15021        let total = crate::wal_commit_count_at(&dir).unwrap();
15022        let aof_latest = GraphDb::open_at(&dir, total - 1).unwrap();
15023        assert_eq!(
15024            aof_latest.get_view_prop("o1", "emp"),
15025            normal_emp,
15026            "open_at latest: derived-edge view must equal normal open (rebuild_all required)"
15027        );
15028
15029        // Mid-history as-of (commit 3 = p1 insert Batch frame): only p1; degree = 1.
15030        // The DerivedEdgeAdded marker for p1 is at frame 4 (state no-op on replay),
15031        // so replaying 0..=3 correctly re-derives only the p1→o1 edge.
15032        let aof_mid = GraphDb::open_at(&dir, 3).unwrap();
15033        assert_eq!(
15034            aof_mid.get_view_prop("o1", "emp"),
15035            Some(Value::Int(1)),
15036            "open_at mid-history: only p1 exists at frame 3, degree must be 1"
15037        );
15038
15039        let _ = std::fs::remove_dir_all(&dir);
15040    }
15041
15042    /// Pin: subscribe_* on an as-of instance must return Err(ReadOnly) —
15043    /// as-of instances never commit, so distribute_events never runs and any
15044    /// subscription would wait forever.
15045    #[test]
15046    fn subscribe_on_as_of_returns_read_only_error() {
15047        let dir = tmp_dir("sub-as-of-read-only");
15048        {
15049            let mut db = GraphDb::open(&dir).unwrap();
15050            db.insert_node("Org", "o1", vec![]).unwrap();
15051            db.create_rule(fk_rule()).unwrap();
15052        }
15053        let mut aof = GraphDb::open_at(&dir, 0).unwrap();
15054
15055        assert!(
15056            matches!(
15057                aof.subscribe_all_rules(),
15058                Err(core_storage::GraphError::ReadOnly)
15059            ),
15060            "subscribe_all_rules on as-of must return ReadOnly"
15061        );
15062        assert!(
15063            matches!(
15064                aof.subscribe_writes(),
15065                Err(core_storage::GraphError::ReadOnly)
15066            ),
15067            "subscribe_writes on as-of must return ReadOnly"
15068        );
15069        assert!(
15070            matches!(
15071                aof.subscribe_rule("works_at"),
15072                Err(core_storage::GraphError::ReadOnly)
15073            ),
15074            "subscribe_rule on as-of must return ReadOnly"
15075        );
15076        let _ = std::fs::remove_dir_all(&dir);
15077    }
15078
15079    /// Regression: a failed dense WAL rewrite must not leave speculative
15080    /// interns in `syms`. If it does, the next successful mutation logs an
15081    /// `Intern` record with an inflated id; replay (which never saw the
15082    /// orphans) assigns a smaller id and the WAL becomes unreplayable.
15083    #[test]
15084    fn dense_rewrite_error_rolls_back_speculative_interns() {
15085        let dir = tmp_dir("dense-rewrite-rollback");
15086        {
15087            let mut db = GraphDb::open(&dir).unwrap();
15088            db.insert_node("Person", "a", vec![]).unwrap();
15089
15090            // Bypass MutPreview validation to hit the rewrite's own error path
15091            // (same shape as an id-exhaustion failure mid-rewrite). The
15092            // InsertEdge arm interns the edge type before it resolves keys.
15093            let err = db.rewrite_wal_dense(vec![WalRecord::InsertEdge {
15094                edge_type: "ORPHAN_TYPE".into(),
15095                src_key: "missing".into(),
15096                dst_key: "a".into(),
15097            }]);
15098            assert!(err.is_err(), "rewrite of a missing src key must fail");
15099            assert_eq!(
15100                db.syms.get("ORPHAN_TYPE"),
15101                None,
15102                "failed rewrite must roll back speculative interns"
15103            );
15104
15105            // A later successful mutation must produce a replayable WAL.
15106            db.set_prop("a", "later_field", Value::Int(2)).unwrap();
15107        }
15108        let db = GraphDb::open(&dir).expect("WAL must replay after failed rewrite");
15109        assert_eq!(db.get_prop("a", "later_field"), Some(Value::Int(2)));
15110        let _ = std::fs::remove_dir_all(&dir);
15111    }
15112}