Skip to main content

core_api/
db.rs

1use crate::ingest::{IngestOptions, IngestReport};
2use crate::roles::{PropPredicate, RoleDef, RolesFile, WriteScope};
3use crate::subscription::{
4    event_matches, DbEvent, SubEntry, SubFilter, SubInner, Subscription, DEFAULT_SUB_CAPACITY,
5};
6use core_query::cypher::ast::{ret_val_label, ArithOp};
7use core_query::cypher::{
8    execute, execute_union, is_subscribable, is_write_tokens, lex, parse, parse_read, parse_write,
9    plan, Expr, MatchDeleteNodeStmt, NodePat, Operand, Params, Pattern, PlanOp, Query, RetItem,
10    RetVal, WriteStatement,
11};
12use core_query::{eval_cmp, eval_filter, expand, neighborhood, Dir, Filter, GraphView, ResultSet};
13use core_rules::{
14    decode_rule_def, ef_max, evaluate, BuildProgress, EngineEdgeDelta, GraphMut, NodeView,
15    Predicate, RuleDef, RuleEngine, ViewDef, ViewStore,
16};
17use core_storage::fs::{FileId, Fs, FsIntrospect, RealFs};
18use core_storage::fulltext::FulltextIndex;
19use core_storage::property_index::PropertyIndex;
20use core_storage::v8::encode::{
21    archived_hnsw_to_owned, archived_rules_meta_to_owned, archived_to_idmap, archived_to_interner,
22    archived_views_to_owned, decode_last_change_bytes, decode_meta, encode_v8, V8Meta,
23};
24use core_storage::v8::seam::TopologyView;
25use core_storage::wal::{decode_all, encode_record, WalRecord};
26use core_storage::EdgePropsView;
27use core_storage::{
28    namespace_of_value, ColumnStore, Direction, EdgeProps, GraphError, IdMap, Interner, Result,
29    Topology, Value,
30};
31pub use core_storage::{valid_namespace, NS_DEFAULT, NS_MAX_LEN, NS_PROP};
32
33/// Index of [`NS_DEFAULT`] in `GraphDb::ns_names` — always zero, so the
34/// open-time pass over a store with no `ns` column fills `node_ns` with one
35/// constant and allocates no names.
36const NS_DEFAULT_IDX: u32 = 0;
37
38/// The reserved edge property holding a pair's insert count (§5.13).
39///
40/// Absent means 1 — the count is written only from the second insert of a
41/// triple onward, and only on a store that called
42/// [`GraphDb::enable_multiplicity`]. The engine owns the name: Cypher `SET` on
43/// it is refused, as the other reserved names are.
44pub const EDGE_COUNT_PROP: &str = "count";
45use serde::{Deserialize, Serialize};
46use std::collections::{BTreeMap, BTreeSet, HashMap, HashSet};
47use std::sync::Arc;
48
49/// Print a timing checkpoint when MUSHROOMDB_TRACE_OPEN is set.
50/// Zero-cost when the env var is absent (the var check is O(1) after first call).
51macro_rules! trace_open {
52    ($phase:literal, $t:expr) => {
53        if std::env::var("MUSHROOMDB_TRACE_OPEN").is_ok() {
54            eprintln!(
55                "[MUSHROOMDB_TRACE_OPEN] {:40} {:>9.3?}",
56                $phase,
57                $t.elapsed()
58            );
59        }
60    };
61}
62
63/// Print a migration phase checkpoint when MUSHROOMDB_TRACE_MIGRATE is set.
64/// Zero-cost when the env var is absent (the var check is O(1) after first call).
65macro_rules! trace_migrate {
66    ($phase:literal, $t:expr) => {
67        if std::env::var("MUSHROOMDB_TRACE_MIGRATE").is_ok() {
68            eprintln!(
69                "[MUSHROOMDB_TRACE_MIGRATE] {:40} {:>9.3?}",
70                $phase,
71                $t.elapsed()
72            );
73        }
74    };
75}
76
77// Test-only: counts how many times `pending_deltas_since().to_vec()` actually
78// executes (i.e., at least one view is defined). Used to verify the fast-path
79// guard skips the allocation when `view_store.is_empty()`.
80#[cfg(test)]
81thread_local! {
82    static DELTA_COPY_COUNT: std::cell::Cell<usize> = const { std::cell::Cell::new(0) };
83}
84
85// Per-thread count of query-subscription `execute` calls in `distribute_events`.
86//
87// Incremented each time a query subscription actually runs its plan (i.e.,
88// the label-skip fast-path did not fire). Because `distribute_events` is
89// called synchronously on the writer thread, this thread-local correctly
90// isolates each test thread's count even when integration tests run in
91// parallel. Read via [`query_sub_exec_count`].
92thread_local! {
93    static QUERY_SUB_EXECS_TL: std::cell::Cell<usize> = const { std::cell::Cell::new(0) };
94}
95
96/// Return the number of query-subscription re-executions logged on this
97/// thread since the process started (or since last reset via
98/// [`reset_query_sub_exec_count`]).
99///
100/// Primarily for integration tests that verify the label-skip fast-path.
101#[doc(hidden)]
102pub fn query_sub_exec_count() -> usize {
103    QUERY_SUB_EXECS_TL.with(|c| c.get())
104}
105
106/// Reset the per-thread query-subscription execution counter to zero.
107#[doc(hidden)]
108pub fn reset_query_sub_exec_count() {
109    QUERY_SUB_EXECS_TL.with(|c| c.set(0));
110}
111
112// Exact-versus-approximate warnings emitted on this thread. Thread-local for
113// the same reason [`QUERY_SUB_EXECS_TL`] is: integration tests run in parallel
114// and each gets its own thread, so a neighbour's masked search cannot be
115// mistaken for this test's.
116thread_local! {
117    static AMBIGUOUS_EXACTNESS_WARNS: std::cell::Cell<u64> = const { std::cell::Cell::new(0) };
118    static AMBIGUOUS_EXACTNESS_LAST: std::cell::RefCell<Option<String>> =
119        const { std::cell::RefCell::new(None) };
120}
121
122/// How many times a masked, non-exact vector search has explained itself on
123/// this thread since the last [`ambiguous_exactness_warns_reset`].
124///
125/// The line itself is the product; this counter exists so a test can assert it
126/// is printed **once per index** rather than once per call.
127///
128/// **Single-threaded assertions only.** The suppression set this counts is a
129/// `Mutex<HashSet<_>>` on the `GraphDb` — shared by every thread — while the
130/// counter is thread-local. Under a concurrent caller (`serve`, which is the
131/// deployment the warning exists for) the thread that prints the line is not
132/// necessarily the thread that asked, so a zero here does not mean the line was
133/// not printed and a one does not mean it was printed once. It answers
134/// "once per index" only in a test that owns the store.
135#[doc(hidden)]
136pub fn ambiguous_exactness_warns() -> u64 {
137    AMBIGUOUS_EXACTNESS_WARNS.with(|c| c.get())
138}
139
140/// The most recent exactness warning printed on this thread, verbatim.
141///
142/// The advice a caller reads has to be advice that caller can act on, which the
143/// counter alone cannot witness — see
144/// `the_hybrid_path_advises_a_call_the_hybrid_caller_can_make`. Carries the
145/// same single-threaded caveat as [`ambiguous_exactness_warns`].
146#[doc(hidden)]
147pub fn ambiguous_exactness_last_warning() -> Option<String> {
148    AMBIGUOUS_EXACTNESS_LAST.with(|c| c.borrow().clone())
149}
150
151/// Reset this thread's exactness-warning counter and recorded line.
152#[doc(hidden)]
153pub fn ambiguous_exactness_warns_reset() {
154    AMBIGUOUS_EXACTNESS_WARNS.with(|c| c.set(0));
155    AMBIGUOUS_EXACTNESS_LAST.with(|c| *c.borrow_mut() = None);
156}
157
158/// Which signature reached the approximate masked vector leg.
159///
160/// One code path, two entry points, and the advice cannot be the same: a
161/// warning that names an argument the caller's function does not take sends
162/// them looking for a parameter that is not there. `search_hybrid` and
163/// `search_hybrid_scoped` take `(text_field, query_text, vector_field,
164/// query_vec, label, k[, mask])` — no `exact`, no `where`.
165///
166/// The warning is not suppressed on the hybrid path. The approximation is the
167/// same one, and a caller who read `mask=` as a promise of exhaustiveness is
168/// the reader it was written for whichever door they came in by; only the
169/// remedy differs, so only the remedy changes.
170#[derive(Clone, Copy, PartialEq, Eq, Hash, Debug)]
171enum ExactnessCaller {
172    /// `find_similar` / `find_similar_vector_*` — `exact` and `where` are its
173    /// own parameters.
174    Vector,
175    /// `search_hybrid` / `search_hybrid_scoped` — neither argument exists, and
176    /// the leg is one half of a fusion.
177    Hybrid,
178}
179
180impl ExactnessCaller {
181    fn subject(self) -> &'static str {
182        match self {
183            Self::Vector => "a masked vector search",
184            Self::Hybrid => "the vector leg of a masked hybrid search",
185        }
186    }
187
188    fn advice(self) -> &'static str {
189        match self {
190            Self::Vector => "pass exact=True or a where= predicate.",
191            // Names the call that does take the argument, because this one
192            // does not: the caller's own next step, not a parameter hunt.
193            Self::Hybrid => {
194                "run the vector leg on its own with find_similar(field, vector, mask=…, \
195                 exact=True) and fuse it with search() yourself — search_hybrid itself \
196                 takes no exactness argument."
197            }
198        }
199    }
200}
201
202/// Internal state for a single `subscribe_query` subscription.
203///
204/// On every commit, `distribute_events` re-executes `ops` against the current
205/// graph state, diffs the result against `prev_rows`, and pushes
206/// `DbEvent::QueryRowAdded` / `QueryRowRemoved` events to `inner`.
207///
208/// **Full re-run per commit; use LIMIT to bound execution cost.**
209/// (Differential evaluation is roadmap / Phase 5.)
210pub(crate) struct QuerySubEntry {
211    /// Compiled plan for the subscribed Cypher query.
212    ops: Vec<PlanOp>,
213    /// Column names from the first execution (fixed for the subscription lifetime).
214    columns: Vec<String>,
215    /// Serialized (JSON) row key → row data, representing the result set at
216    /// the end of the last commit. Used to diff against the new result.
217    prev_row_map: std::collections::HashMap<String, Vec<Option<Value>>>,
218    /// Weak pointer to the subscriber queue; dead Weak → subscription dropped.
219    inner: std::sync::Weak<SubInner>,
220    /// Interned label sym captured at subscribe time from the plan's leading scan
221    /// (`ScanLabel`, `IndexScan`, or `IndexIntersect` with a concrete label).
222    ///
223    /// `None` means the plan has an `Expand` op (or no recognizable leading scan
224    /// with a concrete label), and this subscription must re-execute on every
225    /// commit without skipping. This is the conservative v0.4.3 boundary: Expand
226    /// queries are never skipped because edges can alter join results regardless
227    /// of which node labels were written.
228    scan_label: Option<u32>,
229}
230
231/// A post-commit mutation notification.
232///
233/// Emitted from `log_then_apply` after the WAL append, fsync, and
234/// in-memory `apply` all succeed. Never emitted for rejected operations
235/// (validation errors, [`GraphError::RuleOwned`], duplicate keys, no-op
236/// deletes/removes). Event payloads carry user keys and rule names, never
237/// internal ids.
238///
239/// **Replay:** [`GraphDb::open`] / [`GraphDb::open_with`] replay the WAL via
240/// `apply` only. Emission lives exclusively in `log_then_apply`, so
241/// recovery is silent even if a sink were installed (it cannot be: the
242/// sink is in-memory and set after open).
243///
244/// **Ordering:** a `Batch` WAL frame emits one event per inner record, then
245/// [`MutationEvent::BatchApplied`]. An ingest commit emits those same inner
246/// events, then [`MutationEvent::Ingested`] (not `BatchApplied`). An empty
247/// or all-noop batch writes no WAL and emits nothing (including no summary).
248///
249/// **Derived edges:** rule-created or retracted edges are not individually
250/// evented — they are recoverable from the triggering mutation plus the live
251/// rule set. Only the triggering record is emitted.
252///
253/// **Wire form:** externally tagged snake_case JSON
254/// (`{"node_inserted":{"label":"A","key":"k"}}`).
255#[derive(Debug, Clone, PartialEq, Eq, Serialize)]
256#[serde(rename_all = "snake_case")]
257pub enum MutationEvent {
258    NodeInserted {
259        label: String,
260        key: String,
261    },
262    PropSet {
263        key: String,
264        field: String,
265    },
266    PropRemoved {
267        key: String,
268        field: String,
269    },
270    EdgeInserted {
271        edge_type: String,
272        src: String,
273        dst: String,
274    },
275    EdgeDeleted {
276        edge_type: String,
277        src: String,
278        dst: String,
279    },
280    NodeDeleted {
281        key: String,
282    },
283    RuleCreated {
284        name: String,
285    },
286    RuleDeleted {
287        name: String,
288    },
289    RuleRebuilt {
290        name: String,
291    },
292    BatchApplied {
293        ops: usize,
294    },
295    Ingested {
296        label: String,
297        inserted: usize,
298    },
299}
300
301fn event_from_record(rec: &WalRecord, intern: &Interner, ids: &IdMap) -> Option<MutationEvent> {
302    match rec {
303        WalRecord::InsertNode { label, key, .. } => Some(MutationEvent::NodeInserted {
304            label: label.clone(),
305            key: key.clone(),
306        }),
307        WalRecord::InsertNodeId { label, key, .. } => Some(MutationEvent::NodeInserted {
308            label: intern.resolve(*label)?.to_string(),
309            key: key.clone(),
310        }),
311        WalRecord::SetProp { key, field, .. } => Some(MutationEvent::PropSet {
312            key: key.clone(),
313            field: field.clone(),
314        }),
315        WalRecord::SetPropId { id, field, .. } => Some(MutationEvent::PropSet {
316            key: ids.key_of(*id)?.to_string(),
317            field: intern.resolve(*field)?.to_string(),
318        }),
319        WalRecord::RemoveProp { key, field } => Some(MutationEvent::PropRemoved {
320            key: key.clone(),
321            field: field.clone(),
322        }),
323        WalRecord::InsertEdge {
324            edge_type,
325            src_key,
326            dst_key,
327        } => Some(MutationEvent::EdgeInserted {
328            edge_type: edge_type.clone(),
329            src: src_key.clone(),
330            dst: dst_key.clone(),
331        }),
332        WalRecord::InsertEdgeId { etype, src, dst } => Some(MutationEvent::EdgeInserted {
333            edge_type: intern.resolve(*etype)?.to_string(),
334            src: ids.key_of(*src)?.to_string(),
335            dst: ids.key_of(*dst)?.to_string(),
336        }),
337        WalRecord::DeleteEdge {
338            edge_type,
339            src_key,
340            dst_key,
341        } => Some(MutationEvent::EdgeDeleted {
342            edge_type: edge_type.clone(),
343            src: src_key.clone(),
344            dst: dst_key.clone(),
345        }),
346        WalRecord::DeleteNode { key } => Some(MutationEvent::NodeDeleted { key: key.clone() }),
347        WalRecord::CreateRule { def_bytes } => {
348            let def: RuleDef = decode_rule_def(def_bytes).ok()?;
349            Some(MutationEvent::RuleCreated { name: def.name })
350        }
351        WalRecord::DeleteRule { name } => Some(MutationEvent::RuleDeleted { name: name.clone() }),
352        WalRecord::RebuildRule { name } => Some(MutationEvent::RuleRebuilt { name: name.clone() }),
353        WalRecord::Batch(_)
354        | WalRecord::CreateView { .. }
355        | WalRecord::DeleteView { .. }
356        | WalRecord::EnableFulltext { .. }
357        | WalRecord::DisableFulltext { .. }
358        | WalRecord::EnableIndex { .. }
359        | WalRecord::DisableIndex { .. }
360        | WalRecord::Intern { .. }
361        // History markers are no-ops for mutation events — they carry no new
362        // state and rules re-derive deterministically on replay.
363        | WalRecord::DerivedEdgeAdded { .. }
364        | WalRecord::DerivedEdgeRetracted { .. }
365        // A count changes neither the node nor the edge population: the pair it
366        // counts was already there, which is why it is written at all.
367        | WalRecord::SetEdgeCount { .. }
368        // RenameNode carries no node/edge count change; no special event.
369        | WalRecord::RenameNode { .. } => None,
370    }
371}
372
373/// Database-wide counters plus per-rule budget/fire stats.
374#[derive(Debug, Clone, PartialEq, Eq, Serialize)]
375pub struct Stats {
376    pub nodes_live: usize,
377    pub nodes_tombstoned: usize,
378    pub edges: u64,
379    pub rules: Vec<RuleStats>,
380    /// How many writes hit the rule-chaining depth cap with work still pending,
381    /// since this handle was opened. Non-zero means some derived edges beyond
382    /// the cap are stale and no single later write will repair them: split the
383    /// rule chain or shorten it. Never persisted, so it resets on reopen.
384    #[serde(default)]
385    pub chain_truncations: u64,
386    /// The oldest commit index history still reaches (the WAL horizon floor).
387    /// `0` means nothing has been pruned and history is complete; a non-zero
388    /// value means events before that commit were pruned and are gone.
389    #[serde(default)]
390    pub history_floor: u64,
391    /// Live node counts per namespace, in name order. Always carries
392    /// `default` — a store is at least its default namespace — so a
393    /// single-tenant store reads `[{"name":"default", …}]` and a reader can
394    /// tell "no namespaces in use" from one entry.
395    #[serde(default)]
396    pub namespaces: Vec<NamespaceStats>,
397}
398
399/// Live node count for one namespace; one entry of [`Stats::namespaces`].
400#[derive(Debug, Clone, PartialEq, Eq, Serialize)]
401pub struct NamespaceStats {
402    pub name: String,
403    pub nodes_live: usize,
404}
405
406/// One rule's provenance size, trip latch, and fire counter.
407///
408/// `tripped` is a one-way latch: once set, the engine adds no new edges for
409/// that rule until [`GraphDb::rebuild_rule`] (and only if the full desired
410/// set then fits). `fires` counts `on_node_changed` evaluations plus
411/// backfill/rebuild participant ticks (rebuild counts even when it is a
412/// provenance no-op).
413#[derive(Debug, Clone, PartialEq, Eq, Serialize)]
414pub struct RuleStats {
415    pub name: String,
416    pub edges: u64,
417    pub tripped: bool,
418    pub fires: u64,
419    /// Whether this rule uses the approximate IVF-Flat candidate path.
420    pub approximate: bool,
421    /// `Some` while this rule's vector index is still being built.
422    ///
423    /// The rule derives **no** edges until it is `None`: the backfill is one
424    /// commit that runs after the index is whole, so a caller never sees a
425    /// partial edge set. Absent from the JSON when the rule is not building,
426    /// which is every rule created over a corpus at or below
427    /// [`core_rules::HNSW_BUILD_BATCH`] vectors.
428    #[serde(default, skip_serializing_if = "Option::is_none")]
429    pub building: Option<BuildProgress>,
430}
431
432/// One entry in the slow-query ring buffer.
433#[derive(Debug, Clone, Serialize)]
434pub struct SlowQueryEntry {
435    /// Execution time in whole milliseconds.
436    pub ms: u64,
437    /// The Cypher query string that was slow.
438    pub query: String,
439    /// The commit sequence number at the time the query ran.
440    pub at_commit: u64,
441}
442
443/// Snapshot of the slow-query log returned by [`GraphDb::slow_query_snapshot`].
444#[derive(Debug, Clone, Serialize)]
445pub struct SlowQuerySnapshot {
446    /// Current threshold in milliseconds (0 = disabled).
447    pub threshold_ms: u64,
448    /// Total number of slow queries ever recorded (not capped by ring size).
449    pub count: u64,
450    /// Most-recent slow queries (up to 16), oldest first.
451    pub last: Vec<SlowQueryEntry>,
452}
453
454/// Internal ring-buffer state protected by a `Mutex` so `query(&self)` can
455/// write to it without a mutable borrow.
456struct SlowQueryLog {
457    entries: std::collections::VecDeque<SlowQueryEntry>,
458    total: u64,
459}
460
461/// Maximum number of entries kept in the slow-query ring buffer.
462const SLOW_QUERY_RING_CAP: usize = 16;
463
464/// Wire summary of a [`Predicate`]. JSON only — `Explanation` is never
465/// bincode-persisted (WAL/snapshots store `RuleDef` bytes, not this type).
466#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
467pub struct PredicateSummary {
468    pub kind: String,
469    pub fields: Vec<String>,
470    pub min: Option<f64>,
471    pub tolerance: Option<f64>,
472    pub km: Option<f64>,
473    pub parts: Option<Vec<PredicateSummary>>,
474    /// True when the owning rule has `approximate=true` (IVF-Flat candidate path).
475    /// Always false for predicates reported without rule context (sub-predicates in `parts`).
476    #[serde(default)]
477    pub approximate: bool,
478}
479
480impl From<&Predicate> for PredicateSummary {
481    fn from(p: &Predicate) -> Self {
482        match p {
483            Predicate::KeyMatch { field } => PredicateSummary {
484                kind: "key_match".into(),
485                fields: vec![field.clone()],
486                min: None,
487                tolerance: None,
488                km: None,
489                parts: None,
490                approximate: false,
491            },
492            Predicate::FieldEqual { field } => PredicateSummary {
493                kind: "field_equal".into(),
494                fields: vec![field.clone()],
495                min: None,
496                tolerance: None,
497                km: None,
498                parts: None,
499                approximate: false,
500            },
501            Predicate::Overlap { field, min } => PredicateSummary {
502                kind: "overlap".into(),
503                fields: vec![field.clone()],
504                min: Some(*min),
505                tolerance: None,
506                km: None,
507                parts: None,
508                approximate: false,
509            },
510            Predicate::NumericWithin { field, tolerance } => PredicateSummary {
511                kind: "numeric_within".into(),
512                fields: vec![field.clone()],
513                min: None,
514                tolerance: Some(*tolerance),
515                km: None,
516                parts: None,
517                approximate: false,
518            },
519            Predicate::GeoRadius { field, km } => PredicateSummary {
520                kind: "geo_radius".into(),
521                fields: vec![field.clone()],
522                min: None,
523                tolerance: None,
524                km: Some(*km),
525                parts: None,
526                approximate: false,
527            },
528            Predicate::VectorSimilar { field, min } => PredicateSummary {
529                kind: "vector_similar".into(),
530                fields: vec![field.clone()],
531                min: Some(*min),
532                tolerance: None,
533                km: None,
534                parts: None,
535                approximate: false,
536            },
537            Predicate::All(inner) => {
538                let parts: Vec<PredicateSummary> = inner.iter().map(Self::from).collect();
539                let mut fields = Vec::new();
540                for part in &parts {
541                    for f in &part.fields {
542                        if !fields.contains(f) {
543                            fields.push(f.clone());
544                        }
545                    }
546                }
547                PredicateSummary {
548                    kind: "all".into(),
549                    fields,
550                    min: None,
551                    tolerance: None,
552                    km: None,
553                    parts: Some(parts),
554                    approximate: false,
555                }
556            }
557            Predicate::Any(inner) => {
558                let parts: Vec<PredicateSummary> = inner.iter().map(Self::from).collect();
559                let mut fields = Vec::new();
560                for part in &parts {
561                    for f in &part.fields {
562                        if !fields.contains(f) {
563                            fields.push(f.clone());
564                        }
565                    }
566                }
567                PredicateSummary {
568                    kind: "any".into(),
569                    fields,
570                    min: None,
571                    tolerance: None,
572                    km: None,
573                    parts: Some(parts),
574                    approximate: false,
575                }
576            }
577        }
578    }
579}
580
581/// Snapshot of a live node's key, label, and columnar properties.
582///
583/// `props` is a [`BTreeMap`] so field order is deterministic (sorted by name)
584/// regardless of insert order or the columnar store's `HashMap` iteration.
585///
586/// Deliberately does not derive `Serialize`: `Value`'s serde form is
587/// internally tagged. Wire JSON is built by `value_to_json` in the server.
588#[derive(Debug, Clone, PartialEq)]
589pub struct NodeInfo {
590    pub key: String,
591    pub label: String,
592    pub props: BTreeMap<String, Value>,
593}
594
595/// Counts returned by [`GraphDb::delete_node`].
596#[derive(Debug, Clone, PartialEq, Eq, Default)]
597pub struct DeleteReport {
598    /// Number of manual (user-inserted) edges removed.
599    pub manual_edges: u64,
600    /// Number of derived (rule-owned) edges retracted.
601    pub derived_edges: u64,
602}
603
604/// One directed edge incident on a node, with provenance membership.
605///
606/// `derived` is true iff `(edge_type, src, dst)` is in the rule engine's
607/// Plan-8 `by_node` provenance index.
608#[derive(Debug, Clone, PartialEq, Eq, Serialize)]
609pub struct EdgeInfo {
610    pub edge_type: String,
611    pub src_key: String,
612    pub dst_key: String,
613    pub derived: bool,
614}
615
616/// One directed edge incident on a node at a point in WAL history, with the
617/// rule that derived it when it is rule-owned.
618///
619/// Returned by [`GraphDb::edges_at`] (sorted by `(edge_type, src_key, dst_key)`)
620/// and by [`GraphDb::what_if_set_prop`].
621#[derive(Debug, Clone, PartialEq, Eq, PartialOrd, Ord, Serialize)]
622pub struct EdgeAt {
623    pub edge_type: String,
624    pub src_key: String,
625    pub dst_key: String,
626    /// `true` when a rule wrote the edge (`DerivedEdgeAdded` in the WAL, or a
627    /// live provenance entry).
628    pub derived: bool,
629    /// The rule that derived the edge. `None` for a manual edge.
630    pub rule: Option<String>,
631}
632
633/// The derived edges a hypothetical property change would retract and derive.
634///
635/// Returned by [`GraphDb::what_if_set_prop`]. Both lists are sorted by
636/// `(edge_type, src_key, dst_key)` and every entry is rule-derived.
637#[derive(Debug, Clone, PartialEq, Eq, Serialize)]
638pub struct WhatIf {
639    /// Derived edges that exist now and would be retracted.
640    pub lost: Vec<EdgeAt>,
641    /// Derived edges that do not exist now and would be derived.
642    pub gained: Vec<EdgeAt>,
643}
644
645/// An edge with mask-aware endpoint visibility.
646///
647/// Returned by [`GraphDb::node_edges_masked`] in [`crate::mask::MaskMode::Stub`]
648/// mode — hidden endpoints carry `*_restricted: true`.
649#[derive(Debug, Clone, PartialEq, Eq)]
650pub struct MaskedEdge {
651    pub edge_type: String,
652    pub src_key: String,
653    /// `true` when `src_key` is in the DB but hidden from the mask.
654    pub src_restricted: bool,
655    pub dst_key: String,
656    /// `true` when `dst_key` is in the DB but hidden from the mask.
657    pub dst_restricted: bool,
658    pub derived: bool,
659}
660
661/// Result of a mask-aware node lookup via [`GraphDb::node_info_masked`].
662///
663/// `None` from that method means the key does not exist (→ 404).
664/// `Some(Restricted)` is only produced when `mask.mode() == MaskMode::Stub`.
665#[derive(Debug, PartialEq)]
666pub enum MaskedNodeResult {
667    Visible(NodeInfo),
668    /// Node exists in the DB but is hidden from this mask.
669    Restricted,
670}
671
672/// One rule-owned edge between two nodes, with the rule name, edge type,
673/// direction (src_key → dst_key), and weight if the rule stores one.
674#[derive(Debug, Clone, PartialEq, Serialize)]
675pub struct Explanation {
676    pub rule: String,
677    pub edge_type: String,
678    pub src_key: String,
679    pub dst_key: String,
680    pub weight: Option<f64>,
681    pub predicate: PredicateSummary,
682    /// For a via-hop rule, the edge type the rule hops over to reach its
683    /// candidates. `None` for a plain two-node rule. A via-hop rule whose
684    /// `via_edge` is itself rule-derived is the chaining case: the hop edge
685    /// was written by another rule in the same commit.
686    #[serde(default)]
687    pub via_edge: Option<String>,
688}
689
690/// Report returned by [`GraphDb::backup_to`].
691#[derive(Debug, Clone)]
692pub struct BackupReport {
693    /// Filenames copied into the destination directory (sorted ascending).
694    pub files: Vec<String>,
695    /// Total bytes written across all copied files.
696    pub bytes: u64,
697    /// `true` when the destination opened cleanly and passed post-copy checks.
698    ///
699    /// For stores that have a `snapshot.bin` this means: all V8 section CRCs
700    /// matched **and** the destination opened without error.
701    ///
702    /// For WAL-only stores (no `snapshot.bin`) there is no snapshot to
703    /// CRC-check; `verified` is `true` when the destination opened and
704    /// replayed the WAL without error (record-level checksums in the WAL
705    /// provide the integrity signal, not section CRCs).
706    pub verified: bool,
707}
708
709/// One directed edge in export form, with optional rule attribution for derived edges.
710///
711/// Returned by [`GraphDb::all_edges_for_export`].
712///
713/// Does not derive `Eq`/`Ord`: `weight` is an `f64` and NaN breaks a total
714/// order. Callers that need a stable edge ordering already sort by
715/// `(edge_type, src, dst)` explicitly (see `all_edges_for_export`).
716#[derive(Debug, Clone, PartialEq, PartialOrd)]
717pub struct ExportEdge {
718    pub edge_type: String,
719    pub src: String,
720    pub dst: String,
721    pub derived: bool,
722    /// Rule name that created this edge, if derived. `None` for manual edges.
723    pub rule: Option<String>,
724    /// The creating rule's declared `weight_prop`, read off this edge, when
725    /// derived and numeric (`Int`/`Float`). `None` for manual edges, derived
726    /// edges whose rule declares no `weight_prop`, or a non-numeric value.
727    pub weight: Option<f64>,
728}
729
730/// One edge type's shape, as [`GraphDb::edge_type_census`] counts it.
731///
732/// Deliberately per *type* and not per edge: everything here is a summary a
733/// caller can print in one line, and none of it costs a record per edge.
734#[derive(Debug, Clone, PartialEq, Eq)]
735pub struct EdgeTypeCensus {
736    pub edge_type: String,
737    /// Directed edges of this type. Counted the way
738    /// [`GraphDb::edge_count`] counts: each edge once, from its source.
739    pub edges: u64,
740    /// Every label seen on a source of this type, sorted.
741    pub src_labels: Vec<String>,
742    /// Every label seen on a destination of this type, sorted.
743    pub dst_labels: Vec<String>,
744    /// The rules that declare this `edge_type`, sorted. Empty for a type
745    /// written by hand.
746    pub rules: Vec<String>,
747    /// `(src key, dst key)` of the first edge of this type in the store's own
748    /// id order — a real pair to quote in an example.
749    pub sample: Option<(String, String)>,
750}
751
752/// Construct the standard write-query result set (columns: created, properties_set, deleted).
753fn write_result_set() -> ResultSet {
754    ResultSet::new(vec![
755        "created".into(),
756        "properties_set".into(),
757        "deleted".into(),
758    ])
759}
760
761fn resolve_merge_set_value(op: &Operand, params: &BTreeMap<String, Value>) -> Result<Value> {
762    match op {
763        Operand::Lit(v) => Ok(v.clone()),
764        Operand::Param(name) => params
765            .get(name)
766            .cloned()
767            .ok_or_else(|| GraphError::QueryError {
768                detail: format!("missing parameter `{name}`"),
769            }),
770        _ => Err(GraphError::QueryError {
771            detail: "ON CREATE/ON MATCH SET value must be a literal or $parameter".into(),
772        }),
773    }
774}
775
776fn operand_node_vars(op: &Operand, out: &mut Vec<String>) {
777    match op {
778        Operand::Prop { var, .. } | Operand::Var(var) => {
779            if !out.contains(var) {
780                out.push(var.clone());
781            }
782        }
783        Operand::FuncCall { args, .. } => {
784            for arg in args {
785                operand_node_vars(arg, out);
786            }
787        }
788        Operand::BinArith { left, right, .. } => {
789            operand_node_vars(left, out);
790            operand_node_vars(right, out);
791        }
792        Operand::Case { branches, default } => {
793            // Branch conditions reference vars already bound (and mask-filtered)
794            // by the MATCH phase, so collecting from the value operands + ELSE
795            // is sufficient for RETURN-projection var discovery.
796            for (_, value) in branches {
797                operand_node_vars(value, out);
798            }
799            if let Some(d) = default {
800                operand_node_vars(d, out);
801            }
802        }
803        Operand::Index { base, index } => {
804            operand_node_vars(base, out);
805            operand_node_vars(index, out);
806        }
807        Operand::Lit(_) | Operand::Param(_) => {}
808    }
809}
810
811fn ret_node_vars(items: &[RetItem]) -> Vec<String> {
812    let mut out = Vec::new();
813    for item in items {
814        match &item.value {
815            RetVal::Var(v) | RetVal::Prop { var: v, .. } => {
816                if !out.contains(v) {
817                    out.push(v.clone());
818                }
819            }
820            RetVal::FuncCall { args, .. } => {
821                for arg in args {
822                    operand_node_vars(arg, &mut out);
823                }
824            }
825            RetVal::ScalarExpr(op) => operand_node_vars(op, &mut out),
826            RetVal::Agg { .. } => {}
827        }
828    }
829    out
830}
831
832fn add_var(out: &mut Vec<String>, v: &str) {
833    if !out.iter().any(|x| x == v) {
834        out.push(v.to_string());
835    }
836}
837
838fn pattern_node_vars(pats: &[Pattern]) -> Vec<String> {
839    let mut out = Vec::new();
840    for p in pats {
841        if let Some(v) = &p.start.var {
842            add_var(&mut out, v);
843        }
844        for (_, dest) in &p.chain {
845            if let Some(v) = &dest.var {
846                add_var(&mut out, v);
847            }
848        }
849    }
850    out
851}
852
853fn pattern_rel_vars(pats: &[Pattern]) -> Vec<String> {
854    let mut out = Vec::new();
855    for p in pats {
856        for (rel, _) in &p.chain {
857            if rel.hops.is_none() {
858                if let Some(v) = &rel.var {
859                    add_var(&mut out, v);
860                }
861            }
862        }
863    }
864    out
865}
866
867fn rel_type_alias(var: &str) -> String {
868    format!("__rt_{var}")
869}
870
871fn ret_column_name(item: &RetItem) -> String {
872    if let Some(alias) = &item.alias {
873        return alias.clone();
874    }
875    // The same naming rule the planner and the executor use, so a
876    // write-statement RETURN names its columns exactly as a read query does.
877    // An aggregate is not legal in a write-statement RETURN; it keeps the
878    // placeholder it always had.
879    ret_val_label(&item.value).unwrap_or_else(|| "<agg>".to_string())
880}
881
882fn eval_set_return_operand<F: Fs>(
883    db: &GraphDb<F>,
884    match_rs: &ResultSet,
885    row: usize,
886    rel_vars: &[String],
887    op: &Operand,
888    params: &BTreeMap<String, Value>,
889) -> Result<Option<Value>> {
890    match op {
891        Operand::Lit(v) => Ok(Some(v.clone())),
892        Operand::Param(name) => params.get(name).cloned().ok_or_else(|| GraphError::QueryError {
893            detail: format!("missing parameter `{name}`"),
894        }).map(Some),
895        Operand::Var(name) if rel_vars.iter().any(|r| r == name) => Err(GraphError::QueryError {
896            detail: format!(
897                "cannot return relationship variable '{name}' bare; return its properties ({name}.field) instead"
898            ),
899        }),
900        Operand::Var(name) => Ok(match_rs.get(row, name).cloned()),
901        Operand::Prop { var, field } => {
902            if rel_vars.iter().any(|r| r == var) {
903                return Ok(None);
904            }
905            let Some(Value::Str(key)) = match_rs.get(row, var) else {
906                return Ok(None);
907            };
908            if let Some(v) = db.get_prop(key, field) {
909                return Ok(Some(v));
910            }
911            // Same stored-wins identity fallback as the read path:
912            // n.key / n.id / n.label, not only get_prop.
913            Ok(match field.as_str() {
914                "key" | "id" => Some(Value::Str(key.clone())),
915                "label" => db
916                    .node_ref(key)
917                    .map(|n| Value::Str(n.label().to_owned())),
918                _ => None,
919            })
920        }
921        Operand::FuncCall { name, args } => {
922            eval_set_return_func(db, match_rs, row, rel_vars, name, args, params)
923        }
924        Operand::BinArith { op, left, right } => {
925            let lv = eval_set_return_operand(db, match_rs, row, rel_vars, left, params)?;
926            let rv = eval_set_return_operand(db, match_rs, row, rel_vars, right, params)?;
927            eval_set_return_arith(op, lv, rv)
928        }
929        Operand::Case { branches, default } => {
930            for (cond, value) in branches {
931                if eval_set_return_expr(db, match_rs, row, rel_vars, cond, params, 0)? {
932                    return eval_set_return_operand(db, match_rs, row, rel_vars, value, params);
933                }
934            }
935            match default {
936                Some(d) => eval_set_return_operand(db, match_rs, row, rel_vars, d, params),
937                None => Ok(None),
938            }
939        }
940        Operand::Index { base, index } => {
941            let base_val = eval_set_return_operand(db, match_rs, row, rel_vars, base, params)?;
942            let idx_val = eval_set_return_operand(db, match_rs, row, rel_vars, index, params)?;
943            Ok(core_query::value_ops::index_list(base_val, idx_val))
944        }
945    }
946}
947
948fn eval_set_return_expr<F: Fs>(
949    db: &GraphDb<F>,
950    match_rs: &ResultSet,
951    row: usize,
952    rel_vars: &[String],
953    expr: &Expr,
954    params: &BTreeMap<String, Value>,
955    depth: u32,
956) -> Result<bool> {
957    if depth > 256 {
958        return Err(GraphError::QueryError {
959            detail: "expression nesting too deep".into(),
960        });
961    }
962    match expr {
963        Expr::And(lhs, rhs) => {
964            let l = eval_set_return_expr(db, match_rs, row, rel_vars, lhs, params, depth + 1)?;
965            let r = eval_set_return_expr(db, match_rs, row, rel_vars, rhs, params, depth + 1)?;
966            Ok(l && r)
967        }
968        Expr::Or(lhs, rhs) => {
969            let l = eval_set_return_expr(db, match_rs, row, rel_vars, lhs, params, depth + 1)?;
970            let r = eval_set_return_expr(db, match_rs, row, rel_vars, rhs, params, depth + 1)?;
971            Ok(l || r)
972        }
973        Expr::Not(inner) => Ok(!eval_set_return_expr(
974            db,
975            match_rs,
976            row,
977            rel_vars,
978            inner,
979            params,
980            depth + 1,
981        )?),
982        Expr::Cmp { lhs, op, rhs } => {
983            let l = eval_set_return_operand(db, match_rs, row, rel_vars, lhs, params)?;
984            let r = eval_set_return_operand(db, match_rs, row, rel_vars, rhs, params)?;
985            match (l, r) {
986                (Some(a), Some(b)) => Ok(eval_cmp(op, &a, &b)),
987                _ => Ok(false),
988            }
989        }
990        Expr::Truthy(op) => {
991            let val = eval_set_return_operand(db, match_rs, row, rel_vars, op, params)?;
992            Ok(match val {
993                None => false,
994                Some(Value::Bool(b)) => b,
995                Some(Value::Int(n)) => n != 0,
996                Some(Value::Float(f)) => f != 0.0,
997                Some(Value::Str(s)) => !s.is_empty(),
998                Some(Value::List(v)) => !v.is_empty(),
999                Some(Value::Map(m)) => !m.is_empty(),
1000            })
1001        }
1002        Expr::IsNull(op) => {
1003            let val = eval_set_return_operand(db, match_rs, row, rel_vars, op, params)?;
1004            Ok(val.is_none())
1005        }
1006        Expr::IsNotNull(op) => {
1007            let val = eval_set_return_operand(db, match_rs, row, rel_vars, op, params)?;
1008            Ok(val.is_some())
1009        }
1010        Expr::In { expr, list } => {
1011            let Some(needle) = eval_set_return_operand(db, match_rs, row, rel_vars, expr, params)?
1012            else {
1013                return Ok(false);
1014            };
1015            for item_op in list {
1016                match eval_set_return_operand(db, match_rs, row, rel_vars, item_op, params)? {
1017                    None => {}
1018                    Some(Value::List(items)) => {
1019                        for item in items {
1020                            if eval_cmp(&core_query::CmpOp::Eq, &needle, &item) {
1021                                return Ok(true);
1022                            }
1023                        }
1024                    }
1025                    Some(item) if eval_cmp(&core_query::CmpOp::Eq, &needle, &item) => {
1026                        return Ok(true);
1027                    }
1028                    Some(_) => {}
1029                }
1030            }
1031            Ok(false)
1032        }
1033    }
1034}
1035
1036fn eval_set_return_arith(
1037    op: &ArithOp,
1038    lv: Option<Value>,
1039    rv: Option<Value>,
1040) -> Result<Option<Value>> {
1041    match (lv, rv) {
1042        (None, _) | (_, None) => Ok(None),
1043        (Some(Value::Int(a)), Some(Value::Int(b))) => {
1044            let result = match op {
1045                ArithOp::Sub => a.saturating_sub(b),
1046                ArithOp::Mul => a.saturating_mul(b),
1047                ArithOp::Add => a.saturating_add(b),
1048                ArithOp::Div => {
1049                    if b == 0 {
1050                        return Err(GraphError::QueryError {
1051                            detail: "division by zero".into(),
1052                        });
1053                    }
1054                    a.checked_div(b).unwrap_or(i64::MAX)
1055                }
1056            };
1057            Ok(Some(Value::Int(result)))
1058        }
1059        (Some(lv), Some(rv)) => {
1060            let a = match &lv {
1061                Value::Float(f) => *f,
1062                Value::Int(i) => *i as f64,
1063                _ => {
1064                    return Err(GraphError::QueryError {
1065                        detail: format!("arithmetic operand must be numeric, got {lv:?}"),
1066                    })
1067                }
1068            };
1069            let b = match &rv {
1070                Value::Float(f) => *f,
1071                Value::Int(i) => *i as f64,
1072                _ => {
1073                    return Err(GraphError::QueryError {
1074                        detail: format!("arithmetic operand must be numeric, got {rv:?}"),
1075                    })
1076                }
1077            };
1078            let result = match op {
1079                ArithOp::Sub => a - b,
1080                ArithOp::Mul => a * b,
1081                ArithOp::Add => a + b,
1082                ArithOp::Div => {
1083                    if b == 0.0 {
1084                        return Err(GraphError::QueryError {
1085                            detail: "division by zero".into(),
1086                        });
1087                    }
1088                    a / b
1089                }
1090            };
1091            Ok(Some(Value::Float(result)))
1092        }
1093    }
1094}
1095
1096fn eval_set_return_func<F: Fs>(
1097    db: &GraphDb<F>,
1098    match_rs: &ResultSet,
1099    row: usize,
1100    rel_vars: &[String],
1101    name: &str,
1102    args: &[Operand],
1103    params: &BTreeMap<String, Value>,
1104) -> Result<Option<Value>> {
1105    let norm = name.to_ascii_lowercase();
1106    if norm == "type" {
1107        if args.len() != 1 {
1108            return Err(GraphError::QueryError {
1109                detail: format!("type() requires exactly 1 argument, got {}", args.len()),
1110            });
1111        }
1112        let Operand::Var(rel) = &args[0] else {
1113            return Err(GraphError::QueryError {
1114                detail: "type() argument must be a relationship variable (e.g. type(r))".into(),
1115            });
1116        };
1117        return Ok(match_rs.get(row, &rel_type_alias(rel)).cloned());
1118    }
1119    if norm == "key" || norm == "id" {
1120        let fname = if norm == "id" { "id" } else { "key" };
1121        if args.len() != 1 {
1122            return Err(GraphError::QueryError {
1123                detail: format!("{fname}() requires exactly 1 argument, got {}", args.len()),
1124            });
1125        }
1126        let Operand::Var(var) = &args[0] else {
1127            return Err(GraphError::QueryError {
1128                detail: format!("{fname}() argument must be a node variable (e.g. {fname}(n))"),
1129            });
1130        };
1131        if rel_vars.iter().any(|r| r == var) {
1132            return Err(GraphError::QueryError {
1133                detail: format!("{fname}() argument `{var}` is a relationship, not a node"),
1134            });
1135        }
1136        // MATCH rows bind node variables to their key string, so the column
1137        // value *is* the key. `id()` aliases `key()`.
1138        return Ok(match_rs.get(row, var).cloned());
1139    }
1140    let mut vals = Vec::with_capacity(args.len());
1141    for arg in args {
1142        vals.push(eval_set_return_operand(
1143            db, match_rs, row, rel_vars, arg, params,
1144        )?);
1145    }
1146    match norm.as_str() {
1147        "tolower" => {
1148            if vals.len() != 1 {
1149                return Err(GraphError::QueryError {
1150                    detail: format!("toLower() requires exactly 1 argument, got {}", vals.len()),
1151                });
1152            }
1153            Ok(vals[0].clone().map(|val| match val {
1154                Value::Str(s) => Value::Str(s.to_ascii_lowercase()),
1155                other => other,
1156            }))
1157        }
1158        "toupper" => {
1159            if vals.len() != 1 {
1160                return Err(GraphError::QueryError {
1161                    detail: format!("toUpper() requires exactly 1 argument, got {}", vals.len()),
1162                });
1163            }
1164            Ok(vals[0].clone().map(|val| match val {
1165                Value::Str(s) => Value::Str(s.to_ascii_uppercase()),
1166                other => other,
1167            }))
1168        }
1169        "size" => match vals.first().cloned().flatten() {
1170            None => Ok(None),
1171            Some(Value::Str(s)) => Ok(Some(Value::Int(s.len() as i64))),
1172            Some(Value::List(items)) => Ok(Some(Value::Int(items.len() as i64))),
1173            Some(_) => Ok(None),
1174        },
1175        "coalesce" => Ok(vals.into_iter().flatten().next()),
1176        "abs" => match vals.first().cloned().flatten() {
1177            None => Ok(None),
1178            Some(Value::Int(n)) => Ok(Some(Value::Int(n.saturating_abs()))),
1179            Some(Value::Float(f)) => Ok(Some(Value::Float(f.abs()))),
1180            Some(_) => Ok(None),
1181        },
1182        "round" => match vals.first().cloned().flatten() {
1183            None => Ok(None),
1184            Some(Value::Float(f)) => Ok(Some(Value::Float(f.round()))),
1185            Some(Value::Int(n)) => Ok(Some(Value::Int(n))),
1186            Some(_) => Ok(None),
1187        },
1188        "decay" => {
1189            if vals.len() != 3 {
1190                return Err(GraphError::QueryError {
1191                    detail: format!("decay() requires exactly 3 arguments, got {}", vals.len()),
1192                });
1193            }
1194            match (vals[0].clone(), vals[1].clone(), vals[2].clone()) {
1195                (None, _, _) | (_, None, _) | (_, _, None) => Ok(None),
1196                (Some(b), Some(a), Some(h)) => {
1197                    let numeric = |v: Value| -> Result<f64> {
1198                        match v {
1199                            Value::Int(n) => Ok(n as f64),
1200                            Value::Float(f) => Ok(f),
1201                            other => Err(GraphError::QueryError {
1202                                detail: format!(
1203                                    "decay() requires numeric arguments, got {other:?}"
1204                                ),
1205                            }),
1206                        }
1207                    };
1208                    let b = numeric(b)?;
1209                    let a = numeric(a)?;
1210                    let h = numeric(h)?;
1211                    if h <= 0.0 {
1212                        return Err(GraphError::QueryError {
1213                            detail: "decay() requires halflife > 0".into(),
1214                        });
1215                    }
1216                    Ok(Some(Value::Float(b * 0.5f64.powf(a / h))))
1217                }
1218            }
1219        }
1220        _ => Err(GraphError::QueryError {
1221            detail: format!(
1222                "unknown function `{name}`; supported: toLower, toUpper, size, coalesce, type, abs, round, decay, key, id"
1223            ),
1224        }),
1225    }
1226}
1227
1228fn eval_set_return_item<F: Fs>(
1229    db: &GraphDb<F>,
1230    match_rs: &ResultSet,
1231    row: usize,
1232    rel_vars: &[String],
1233    item: &RetItem,
1234    params: &BTreeMap<String, Value>,
1235) -> Result<Option<Value>> {
1236    match &item.value {
1237        RetVal::Var(v) => eval_set_return_operand(
1238            db,
1239            match_rs,
1240            row,
1241            rel_vars,
1242            &Operand::Var(v.clone()),
1243            params,
1244        ),
1245        RetVal::Prop { var, field } => eval_set_return_operand(
1246            db,
1247            match_rs,
1248            row,
1249            rel_vars,
1250            &Operand::Prop {
1251                var: var.clone(),
1252                field: field.clone(),
1253            },
1254            params,
1255        ),
1256        RetVal::FuncCall { name, args } => {
1257            eval_set_return_func(db, match_rs, row, rel_vars, name, args, params)
1258        }
1259        RetVal::ScalarExpr(op) => eval_set_return_operand(db, match_rs, row, rel_vars, op, params),
1260        RetVal::Agg { .. } => Err(GraphError::QueryError {
1261            detail: "aggregates are not supported in MATCH … SET … RETURN".into(),
1262        }),
1263    }
1264}
1265
1266/// Project user RETURN from original MATCH rows after SET. No rematch.
1267fn project_set_return_rows<F: Fs>(
1268    db: &GraphDb<F>,
1269    rel_vars: &[String],
1270    match_rs: &ResultSet,
1271    returns: &[RetItem],
1272    params: &BTreeMap<String, Value>,
1273) -> Result<ResultSet> {
1274    let columns: Vec<String> = returns.iter().map(ret_column_name).collect();
1275    let mut out = ResultSet::new(columns);
1276    for row in 0..match_rs.len() {
1277        let mut cells = Vec::with_capacity(returns.len());
1278        for item in returns {
1279            cells.push(eval_set_return_item(
1280                db, match_rs, row, rel_vars, item, params,
1281            )?);
1282        }
1283        out.push_row(cells);
1284    }
1285    Ok(out)
1286}
1287
1288/// Single construction point for a `GraphMut` view over the split-borrowed graph fields.
1289/// Callers use `std::mem::take` on the engine before calling this, then restore it after.
1290/// Extract a `Vec<f64>` from a `Value::List` whose items are all numeric.
1291/// Returns `None` for non-list values or lists with non-numeric elements.
1292/// Extra candidates pulled from an approximate index before re-scoring, over and
1293/// above the `k` asked for.
1294///
1295/// The index orders candidates by `f32` distances, which agree with the exact
1296/// `f64` cosine to about 1e-6. Re-scoring can therefore only reshuffle
1297/// candidates inside a band that narrow — it cannot move a hit past one that is
1298/// further away by more than 1e-6 — so the only way a true top-`k` member can be
1299/// lost is if the index ranked it just outside `k` on the `f32` order. Fetching
1300/// `k + 16` covers any such band up to 16 members wide, which at 1e-6 means 16
1301/// vectors within a millionth of each other in cosine: a duplicate cluster, and
1302/// then the members are interchangeable anyway. `min` is applied to the exact
1303/// score, never to the index's, so a hit sitting on the threshold is decided
1304/// exactly.
1305const VECTOR_RESCORE_MARGIN: usize = 16;
1306
1307/// Cosine similarity between an already-unit query and node `id`'s `field`
1308/// vector, read from the **`f64`** properties. `None` when the node has no
1309/// numeric-list vector there, or its norm is zero.
1310///
1311/// The single definition of the score this API reports. Both the brute-force
1312/// scan and the re-scoring step that follows an index lookup go through it, so
1313/// the two paths cannot disagree — which is the property
1314/// `index_and_brute_force_agree_on_scores` pins.
1315fn exact_vector_similarity(
1316    view: &GraphView<'_>,
1317    id: u32,
1318    field: &str,
1319    q_unit: &[f64],
1320) -> Option<f64> {
1321    let v = view.prop(id, field)?;
1322    let xs = value_as_float_list(&v.into_value())?;
1323    let v_norm: f64 = xs.iter().map(|x| x * x).sum::<f64>().sqrt();
1324    if v_norm == 0.0 {
1325        return None;
1326    }
1327    Some(
1328        q_unit
1329            .iter()
1330            .zip(xs.iter())
1331            .map(|(a, b)| a * (b / v_norm))
1332            .sum(),
1333    )
1334}
1335
1336fn value_as_float_list(v: &Value) -> Option<Vec<f64>> {
1337    match v {
1338        Value::List(items) => items
1339            .iter()
1340            .map(|item| match item {
1341                Value::Float(f) => Some(*f),
1342                Value::Int(i) => Some(*i as f64),
1343                _ => None,
1344            })
1345            .collect(),
1346        _ => None,
1347    }
1348}
1349
1350fn make_graph_mut<'a>(
1351    ids: &'a IdMap,
1352    syms: &'a mut Interner,
1353    labels: &'a [u32],
1354    props: core_storage::v8::seam::ColumnsView<'a>,
1355    topo: &'a mut Topology,
1356    base: &'a Option<std::sync::Arc<core_storage::v8::MappedBase>>,
1357    edge_props: &'a mut EdgeProps,
1358) -> GraphMut<'a> {
1359    GraphMut {
1360        ids,
1361        syms,
1362        labels,
1363        props,
1364        topo,
1365        base_topo: base_csr(base),
1366        edge_props,
1367    }
1368}
1369
1370/// The archived CSR of an open V8 snapshot, for the rule engine's graph reads.
1371///
1372/// A store opened from a snapshot keeps its edges in the mapping and its
1373/// overlay empty, so a rule that reads the graph's shape has to see both.
1374fn base_csr(
1375    base: &Option<std::sync::Arc<core_storage::v8::MappedBase>>,
1376) -> Option<&core_storage::v8::layout::ArchivedCsr> {
1377    base.as_ref().map(|b| {
1378        b.topology()
1379            .expect("base topology section bounds validated at open")
1380    })
1381}
1382
1383/// Build a `ColumnsView` from the disjoint `props` overlay and optional V8 base.
1384///
1385/// Takes explicit field references rather than `&self` so the caller can hold
1386/// simultaneous mutable borrows of other fields (e.g. `syms`, `topo`).
1387fn build_props_view<'a>(
1388    props: &'a ColumnStore,
1389    base: &'a Option<std::sync::Arc<core_storage::v8::MappedBase>>,
1390) -> core_storage::v8::seam::ColumnsView<'a> {
1391    match base {
1392        None => core_storage::v8::seam::ColumnsView::owned(props),
1393        Some(b) => {
1394            let archived = b
1395                .columns()
1396                .expect("base columns section bounds validated at open");
1397            core_storage::v8::seam::ColumnsView::with_base_cached(props, archived, b.mixed_cache())
1398                .with_shared_strings(base_string_table(b))
1399        }
1400    }
1401}
1402
1403/// The base columns section paired with the string table that resolves its
1404/// string ids — what `ViewStore` needs to read a neighbour's string property
1405/// out of a V9 snapshot.
1406fn base_columns(
1407    base: &Option<std::sync::Arc<core_storage::v8::MappedBase>>,
1408) -> Option<core_storage::v8::seam::BaseColumns<'_>> {
1409    base.as_ref().map(|b| core_storage::v8::seam::BaseColumns {
1410        cols: b
1411            .columns()
1412            .expect("base columns section bounds validated at open"),
1413        strings: base_string_table(b),
1414    })
1415}
1416
1417/// The shared string table of a V9 base, or `None` for a pre-V9 one.
1418///
1419/// Every `ColumnsView` built over a base must carry it: without it a V9
1420/// snapshot's string columns, whose own tables are empty, read back as absent.
1421fn base_string_table(
1422    base: &core_storage::v8::MappedBase,
1423) -> Option<&core_storage::v8::layout::ArchivedStringTable> {
1424    base.string_table()
1425        .transpose()
1426        .expect("base strings section bounds validated at open")
1427}
1428
1429fn build_topo_view<'a>(
1430    overlay: &'a Topology,
1431    base: &'a Option<std::sync::Arc<core_storage::v8::MappedBase>>,
1432) -> core_storage::v8::seam::TopologyView<'a> {
1433    match base {
1434        None => core_storage::v8::seam::TopologyView::owned(overlay),
1435        Some(b) => {
1436            let archived_csr = b
1437                .topology()
1438                .expect("base topology section bounds validated at open");
1439            core_storage::v8::seam::TopologyView::with_base(overlay, archived_csr)
1440        }
1441    }
1442}
1443
1444/// When [`GraphDb`] calls `Fs::sync` after a WAL append.
1445///
1446/// Default is [`Strict`](FsyncPolicy::Strict): every `log_then_apply_with`
1447/// fsyncs (single `insert_node` / `set_prop`). Ingest and `write_batch`
1448/// emit one `WalRecord::Batch` and fsync once at that frame (Batched).
1449/// [`Relaxed`](FsyncPolicy::Relaxed) skips WAL sync; [`GraphDb::snapshot`]
1450/// is still durable via `write_atomic`. Crash-recovery DST stays Strict.
1451#[derive(Clone, Copy, Debug, PartialEq, Eq, Default)]
1452pub enum FsyncPolicy {
1453    /// Every WAL commit calls `fs.sync` (today's behavior).
1454    #[default]
1455    Strict,
1456    /// Sync only at a `Batch` frame end. Single-op path stays Strict unless
1457    /// this policy is set on the database.
1458    Batched,
1459    /// Never call `fs.sync`. [`GraphDb::snapshot`] still syncs via `write_atomic`.
1460    Relaxed,
1461}
1462
1463/// A precondition for a compare-and-set batch write.
1464///
1465/// All preconditions in a [`GraphDb::write_batch_cas`] or
1466/// [`crate::SharedDb::submit_batch_cas`] call are checked atomically before
1467/// any operation in the batch is applied.  If any precondition fails, the
1468/// entire batch is rejected with [`GraphError::CasConflict`] and no WAL frame
1469/// is written.
1470///
1471/// # Touch definition
1472///
1473/// A node's last-change commit (`last_changed`) is updated when any of the
1474/// following state-changing WAL records touch it:
1475///
1476/// - `InsertNode` / `InsertNodeId` — the newly-inserted node.
1477/// - `SetProp` / `SetPropId` / `RemoveProp` — the property-bearing node.
1478/// - `InsertEdge` / `InsertEdgeId` / `DeleteEdge` — **both** src and dst
1479///   endpoints (an edge change touches both sides).
1480/// - `DeleteNode` — the node is tombstoned; `last_changed` returns `None`
1481///   for deleted keys so the pre-deletion entry is never observed.
1482///
1483/// History markers (`DerivedEdgeAdded` / `DerivedEdgeRetracted`) are
1484/// state no-ops.  The underlying mutation that triggered rule firing already
1485/// updated the relevant nodes' last-change entries.  Rule-management records
1486/// (`CreateRule`, `DeleteRule`, `RebuildRule`) and view/full-text declarations
1487/// do not touch any node's last-change.
1488#[derive(Debug, Clone, PartialEq, Eq)]
1489pub enum Precondition {
1490    /// The node's last-change commit must equal `expected`.
1491    ///
1492    /// Fails with [`GraphError::CasConflict`] when:
1493    /// - The node does not exist (`last_changed` returns `None`), or
1494    /// - The recorded commit seq does not match `expected`.
1495    NodeUnchangedSince { key: String, expected: u64 },
1496    /// The node must not exist (not inserted, or already deleted).
1497    ///
1498    /// Fails with [`GraphError::CasConflict`] (expected=`u64::MAX`,
1499    /// actual=`last_changed(key).unwrap_or(0)`) when the node is live.
1500    NodeAbsent { key: String },
1501}
1502
1503pub struct GraphDb<F: Fs> {
1504    fs: F,
1505    ids: Arc<IdMap>,
1506    syms: Arc<Interner>,
1507    topo: Arc<Topology>,
1508    props: Arc<ColumnStore>,
1509    labels: Arc<Vec<u32>>, // node id -> label symbol
1510    /// Namespace names by index; index [`NS_DEFAULT_IDX`] is always
1511    /// [`NS_DEFAULT`]. Derived beside [`Self::node_ns`], never persisted.
1512    ///
1513    /// A private table rather than the shared [`Interner`]: interning
1514    /// `"default"` at open would add a symbol to the store's symbol table and
1515    /// change the bytes of the next snapshot of a store that has no namespaces
1516    /// at all.
1517    ns_names: Vec<String>,
1518    /// Namespace index per dense node id, into [`Self::ns_names`];
1519    /// [`NS_DEFAULT_IDX`] for a node with no `ns` property.
1520    ///
1521    /// Derived: built by one pass over the `ns` column at open (which reads
1522    /// nothing when the column does not exist) and maintained at every node
1523    /// insert. Never written to a snapshot or the WAL, because the property it
1524    /// mirrors already is. A namespace cannot change, so no other record shape
1525    /// can move a node between namespaces.
1526    node_ns: Vec<u32>,
1527    edge_props: Arc<EdgeProps>,
1528    engine: RuleEngine,
1529    view_store: ViewStore,
1530    /// Incremental inverted index for full-text-lite search.
1531    /// Rebuild-on-open: populated from WAL replay + rebuild_all at open end.
1532    fulltext: Arc<FulltextIndex>,
1533    /// Opt-in equality index over scalar node properties.
1534    /// Rebuild-on-open: declarations replay from the WAL, postings rebuild at
1535    /// open end (mirrors `fulltext`).
1536    prop_index: PropertyIndex,
1537    /// Whether this store records insert-count multiplicity (§5.13).
1538    ///
1539    /// Declared like `prop_index`'s enabled pairs — a WAL record replayed at
1540    /// open, re-emitted into the baseline by a truncating snapshot — but it
1541    /// gates a *format* step rather than an index: `WalRecord::SetEdgeCount`
1542    /// (discriminant 23) is written only when this is `true`, so a store that
1543    /// never opts in stays readable by a binary that predates the record.
1544    multiplicity: bool,
1545    event_sink: Option<Box<dyn Fn(MutationEvent) + Send + Sync>>,
1546    /// WAL fsync cadence. Default [`FsyncPolicy::Strict`].
1547    fsync: FsyncPolicy,
1548    /// Monotonically increasing per-commit counter.  A single `log_then_apply_with`
1549    /// call increments this once; all events emitted from that call share the same
1550    /// `commit_seq` value.
1551    commit_seq: u64,
1552    /// Commit → wall-clock map, loaded from the `commit_times.bin` sidecar at
1553    /// open and appended to by `log_then_apply_with` — the one place a commit
1554    /// is born. Replay does **not** stamp: `apply_frames` re-applies commits
1555    /// that already happened, and `SystemTime::now()` there would record replay
1556    /// time as commit time. Empty on a store written before v0.6.11, which
1557    /// makes every date query answer `NoRecordedTime` rather than guess.
1558    commit_times: core_storage::commit_times::CommitTimes,
1559    /// When set, subsequent commits are recorded at this instant instead of the
1560    /// system clock.
1561    ///
1562    /// Sticky on purpose. A backfill replays history that happened over months,
1563    /// and a day's worth of rows genuinely share one instant — a one-shot flag
1564    /// would mean setting it before every row of a bulk load, and forgetting one
1565    /// would stamp that row "now" in the middle of 2026-06. Sticky makes the
1566    /// failure visible instead: forget to move it and every commit carries the
1567    /// same timestamp, which a date query answers oddly and an inspection shows
1568    /// at once.
1569    commit_time_override: Option<i64>,
1570    /// `true` only while the open path is replaying, where `load_from_disk`
1571    /// calls `fulltext.rebuild_all` unconditionally afterwards.
1572    ///
1573    /// Replaying an `EnableFulltext` record backfills its pair with a full
1574    /// `0..ids.len()` scan, and every snapshot re-emits one such record per
1575    /// enabled pair — so on a snapshotted store the open does that scan once per
1576    /// pair and then `rebuild_all` clears every posting and does it all again.
1577    /// The backfill is pure waste *when a rebuild follows*, which is true of the
1578    /// open path and **false** of `refresh()`: refresh applies peer frames and
1579    /// then only folds, so its backfill is the only thing that indexes them.
1580    fulltext_rebuild_follows: bool,
1581    /// `true` when `commit_times.bin` was present but would not decode.
1582    ///
1583    /// Mirrors `roles: None`: a damaged map must not read as "this store
1584    /// records no times", because that is also what an honest pre-v0.6.11 store
1585    /// says. Date queries answer `Corrupt` instead, and nothing is appended to
1586    /// a file already known to be damaged.
1587    commit_times_poisoned: bool,
1588    /// RBAC role definitions loaded from `roles.json` at open.
1589    ///
1590    /// `Some(roles)` — loaded successfully (may be empty when no roles are defined).
1591    /// `None` — `roles.json` was present but corrupt; `mask_for_role` returns
1592    /// `Err` for any request (fail-loud, never silently grant empty visibility).
1593    roles: Option<Vec<RoleDef>>,
1594    /// Memo for [`mask_for_role`](GraphDb::mask_for_role), keyed by
1595    /// `(role, commit_seq)` — a scoped reader between two writes resolves once.
1596    ///
1597    /// Shared by `Arc` with every [`ReaderSnapshot`](crate::reader::ReaderSnapshot)
1598    /// taken from this handle. Replaced (not cleared) whenever the role
1599    /// definitions change or the store is reloaded, which `commit_seq` does not
1600    /// record; see [`RoleMaskCache`](crate::mask::RoleMaskCache).
1601    role_masks: Arc<crate::mask::RoleMaskCache>,
1602    /// Which loaded store this handle is, for memos that outlive it.
1603    ///
1604    /// `role_masks` needs no such thing — the handle owns it and replaces it —
1605    /// but a [`Scope`](crate::mask::Scope) is the caller's, so its resolved key
1606    /// leg is stamped with this alongside `commit_seq`. Minted fresh here and
1607    /// again in [`reset_for_reload`](GraphDb::reset_for_reload), at exactly the
1608    /// two points a fresh `RoleMaskCache` is installed; see
1609    /// [`StoreStamp`](crate::mask::StoreStamp) for the invariant.
1610    store_id: crate::mask::StoreId,
1611    /// Live subscriptions.  Entries with a dead `Weak` are pruned on the next
1612    /// distribute_events call.
1613    subscriptions: Vec<SubEntry>,
1614    /// Live query subscriptions. Re-executed on every commit when non-empty.
1615    /// Dead `Weak` entries are pruned inside `distribute_events`.
1616    query_subscriptions: Vec<QuerySubEntry>,
1617    /// Queue capacity for new subscriptions created by this db.  Default is
1618    /// [`DEFAULT_SUB_CAPACITY`]; can be overridden via [`set_sub_capacity`]
1619    /// to test Lagged behaviour with small queues.
1620    sub_capacity: usize,
1621    /// True for as-of instances opened via [`GraphDb::open_at`].
1622    /// Every mutation method and `snapshot()` returns [`GraphError::ReadOnly`]
1623    /// when this flag is set.
1624    read_only: bool,
1625    /// Total WAL commit count at the time [`open_at`] was called.
1626    /// 0 for normal (non-as-of) instances.
1627    total_wal_commits: u64,
1628    /// Immutable mmap-backed base snapshot (V8).  When `Some`, `self.topo` is
1629    /// the WAL-replay overlay (empty at open time, populated by apply()) and
1630    /// reads go through a merged `TopologyView`.  `self.props` is always
1631    /// fully materialized (base + WAL replay) for HNSW/IVF and view compat.
1632    base: Option<Arc<core_storage::v8::MappedBase>>,
1633    // ── MVCC epoch reader state ───────────────────────────────────────────────
1634    /// Most-recent full overlay clone.  Initialized at end of `open_with` /
1635    /// `open_at_with`; refreshed every `FOLD_EVERY_K` commits.
1636    /// `None` only between struct creation and the first fold.
1637    fold_overlay: Option<Arc<crate::reader::FrozenOverlay>>,
1638    /// Per-commit deltas accumulated since the last fold.
1639    delta_tail: Vec<Arc<crate::reader::CommitDelta>>,
1640    /// How many commits have occurred since the last fold.
1641    commits_since_fold: usize,
1642    /// When true, `log_then_apply_with` buffers event notifications instead of
1643    /// firing them immediately.  Used by the group-commit drain thread to defer
1644    /// events until after the group fsync (R2: durability before notification).
1645    /// Cleared to false once the drain thread flushes or discards the buffer.
1646    defer_events: bool,
1647    /// Buffered events accumulated while `defer_events` is true.
1648    deferred_events: Vec<DeferredEvent>,
1649    /// Set to true by the group-commit drain thread when a group fsync fails
1650    /// after WAL truncation.  All subsequent mutation attempts return an IO
1651    /// error until the database is reopened.
1652    degraded: bool,
1653    /// Set to `true` after `ensure_v8_base_sections_loaded` has read provenance,
1654    /// HNSW, and IVF sections from the mmap base into the engine's retained
1655    /// fields.  `false` on all opens until first use; always `true` for non-V8
1656    /// opens (base is None, fast-path sets flag immediately).
1657    v8_sections_loaded: std::sync::atomic::AtomicBool,
1658    /// Serializes the one-time section population in `ensure_v8_base_sections_loaded`.
1659    v8_sections_mutex: std::sync::Mutex<()>,
1660    /// Per-node last-change commit sequence.  `last_change[node_id] = seq` means
1661    /// the node was last modified by commit `seq`.
1662    ///
1663    /// Loaded from V8 section 11 at open; updated on every state-changing commit
1664    /// and WAL replay frame.  V5-V7 stores start with an empty map; pre-WAL-horizon
1665    /// nodes return `None` from `last_changed` until they are next mutated.
1666    ///
1667    /// See [`Precondition`] for the full touch definition.
1668    last_change: HashMap<u32, u64>,
1669    /// WAL archive retention policy set by [`set_wal_archive_retention`].
1670    /// `None` = unlimited (keep all archives); `Some(N)` = keep N newest archives,
1671    /// pruning older ones at snapshot time.  0 is treated as unlimited.
1672    wal_archive_retention: Option<u32>,
1673    /// Global frame index of the first commit that is still reachable through
1674    /// surviving archives.  Persisted to `wal.floor` sidecar when pruning occurs.
1675    /// Default 0 = all history reachable.
1676    wal_horizon_floor: u64,
1677    /// True when the surviving archive chain forms a continuous WAL history
1678    /// starting from the store's first commit (the genesis chain).
1679    ///
1680    /// `open_at` may replay archive-resident commits from empty state only when
1681    /// this flag is true AND `wal_horizon_floor == 0`.  Cleared whenever:
1682    ///   - a WAL-truncating snapshot (`keep_wal=false`) is taken after archives
1683    ///     already exist (breaks the chain for subsequent archives), or
1684    ///   - any archive is pruned (floor advances past zero).
1685    ///
1686    /// Persisted via the `wal.genesis` marker file; loaded from it at open.
1687    archive_genesis_chain: bool,
1688    /// True when this handle can *prove* the live WAL has never been truncated:
1689    /// there was no `snapshot.bin` when it opened the store, and it has taken no
1690    /// truncating snapshot since.
1691    ///
1692    /// The archive path's genesis check asks "did a snapshot exist before this
1693    /// one?" as a proxy for "was the WAL ever truncated". The proxy is sound
1694    /// across sessions — this binary cannot tell a history-preserving snapshot
1695    /// from a truncating one once the handle that took it is gone — but inside
1696    /// one session it is not, and `enable_multiplicity` made that visible: its
1697    /// forced `keep_wal` snapshot left the WAL entirely intact and yet
1698    /// permanently disqualified the store from ever receiving a genesis marker
1699    /// (defect #23). This flag is what the proxy defers to when the answer is
1700    /// actually known.
1701    snapshot_preserved_history: bool,
1702    /// Transient write-authz context set by `write_batch_authz` /
1703    /// `query_write_authz` for the duration of ONE mutation call.
1704    /// Always `None` at rest.  Never serialized, never WAL-replayed.
1705    pending_write_authz: Option<WriteAuthz>,
1706    /// Slow-query threshold in milliseconds.  0 = disabled.
1707    /// Seeded from `MUSHROOMDB_SLOW_QUERY_MS` at open; override via
1708    /// [`GraphDb::set_slow_query_threshold_ms`] (tests must use the setter
1709    /// — env vars are process-global and race parallel test threads).
1710    slow_query_threshold_ms: u64,
1711    /// Ring buffer of recent slow queries (interior-mutable so `query(&self)`
1712    /// can record entries without requiring `&mut self`).
1713    slow_queries: std::sync::Mutex<SlowQueryLog>,
1714    /// `(field, label, caller)` triples whose exact-versus-approximate
1715    /// ambiguity this handle has already explained once. See
1716    /// [`note_ambiguous_exactness`](GraphDb::note_ambiguous_exactness).
1717    /// The caller shape is part of the key because the two shapes give
1718    /// different advice — silencing one with the other would leave a caller
1719    /// reading advice meant for a signature it does not have.
1720    /// Advice bookkeeping, not graph state: a reload keeps it, as the
1721    /// slow-query log does.
1722    warned_ambiguous_exactness: std::sync::Mutex<HashSet<(String, String, ExactnessCaller)>>,
1723    /// Instant at which the database was opened (used by `/metrics` uptime).
1724    started_at: std::time::Instant,
1725    // ── Multi-process state (cross-process lock + WAL tailing) ────────────────
1726    /// Byte offset of the WAL prefix already applied to in-memory state.
1727    ///
1728    /// Advanced by exactly the encoded length of every frame this handle
1729    /// appends, and by the decoded byte count of every tail
1730    /// [`refresh`](GraphDb::refresh) absorbs. Rewound by
1731    /// [`set_wal_consumed`](GraphDb::set_wal_consumed) when the group-commit
1732    /// drain thread truncates a failed group. Compared against the WAL's
1733    /// on-disk length to decide staleness.
1734    wal_consumed: u64,
1735    /// The **global 0-based frame index the next appended WAL frame will
1736    /// occupy** — `wal_horizon_floor` plus every frame currently reachable
1737    /// through archives and the live WAL.
1738    ///
1739    /// This is the space every history surface addresses: `edges_at`,
1740    /// `was_linked`, both history readouts and `open_at` all index the sequence
1741    /// [`all_frames`](GraphDb::all_frames) returns, and
1742    /// [`wal_total_commits`](GraphDb::wal_total_commits) counts it.
1743    ///
1744    /// It exists because **a commit is not a frame**. `commit_seq` counts
1745    /// commits; a commit whose rules fire appends a *second* frame — the
1746    /// derived-edge history marker — that no counter of commits ever sees. The
1747    /// two diverge by one frame per rule-firing commit, cumulatively, so
1748    /// deriving a frame index from `commit_seq` under-reports by more and more
1749    /// as history grows and resolves every date to an earlier graph. Silently:
1750    /// an older graph is a plausible answer, not an error.
1751    ///
1752    /// Maintained in lockstep with [`wal_consumed`](GraphDb::wal_consumed) —
1753    /// the same appends advance both, one in frames and one in bytes — so the
1754    /// two are seeded and rewound at exactly the same places. Keep it that way.
1755    wal_frames_written: u64,
1756    /// Identity of the snapshot this handle's base state came from, as
1757    /// `(len, mtime_nanos)`. A different value means another process replaced
1758    /// the snapshot and the WAL no longer continues our state: refresh reloads.
1759    snapshot_ident: Option<(u64, u64)>,
1760    /// The options this handle was opened with. Replayed verbatim when
1761    /// `refresh` has to rebuild from disk.
1762    open_opts: OpenOptions,
1763    /// True when this handle holds the cross-process write lock for its whole
1764    /// lifetime (a plain read-write open). Per-write lock acquisition is a
1765    /// no-op on such a handle, and never releases the lock.
1766    holds_lifetime_lock: bool,
1767    /// True between a failed lock acquisition and the end of the write scope
1768    /// that failed. Makes every WAL-appending mutation in that scope return
1769    /// [`GraphError::Busy`] instead of writing.
1770    lock_denied: bool,
1771    /// True for an as-of view opened via [`GraphDb::open_at`]. Such a view is
1772    /// pinned to one commit, so it is never stale and never refreshes — later
1773    /// commits by any process are deliberately invisible to it.
1774    pinned: bool,
1775}
1776
1777/// One group of deferred event notifications, held until the group fsync
1778/// completes.  Replayed by [`GraphDb::flush_deferred_events`].
1779struct DeferredEvent {
1780    rec: core_storage::WalRecord,
1781    engine_deltas: Vec<EngineEdgeDelta>,
1782    seq: u64,
1783    ingest: Option<(String, usize)>,
1784}
1785
1786/// Options for [`GraphDb::open_with_options`].
1787#[derive(Clone, Copy, Debug)]
1788pub struct OpenOptions {
1789    /// Rewrite an old-format snapshot to the current VERSION after a
1790    /// successful load (default `true`). The old snapshot is kept as
1791    /// `snapshot.bin.bak` until the next clean open at the current version,
1792    /// at which point the `.bak` is deleted.
1793    ///
1794    /// Set to `false` to open a store without touching any on-disk files
1795    /// (useful for read-only inspection of a store at an older format).
1796    pub auto_migrate: bool,
1797
1798    /// Write the valid WAL prefix back over a torn tail on open (default
1799    /// `true`). Truncating a genuinely torn tail is correct crash recovery.
1800    ///
1801    /// Set to `false` for an unattended reader. The valid prefix is still
1802    /// decoded and replayed in memory, but nothing is written: a reader that
1803    /// opens while another process is mid-append would otherwise discard a
1804    /// frame that writer believes durable. `mushroomdb recall`, which runs on
1805    /// every prompt, passes `false` for exactly this reason.
1806    pub repair_wal: bool,
1807
1808    /// Open without ever writing to the store (default `false`).
1809    ///
1810    /// A read-only handle:
1811    /// - returns [`GraphError::ReadOnly`] from every mutation and from
1812    ///   `snapshot()`;
1813    /// - performs no disk write at open — no WAL repair write-back and no
1814    ///   auto-migration rewrite, whatever the other two flags say;
1815    /// - never takes the cross-process write lock, so it opens immediately even
1816    ///   while another process is writing, and never makes a writer wait.
1817    ///
1818    /// [`refresh`](GraphDb::refresh) and [`is_stale`](GraphDb::is_stale) work
1819    /// normally, so a read-only handle can follow another process's commits.
1820    pub read_only: bool,
1821}
1822
1823impl Default for OpenOptions {
1824    fn default() -> Self {
1825        Self {
1826            auto_migrate: true,
1827            repair_wal: true,
1828            read_only: false,
1829        }
1830    }
1831}
1832
1833/// How long a writer polls for the cross-process write lock before giving up
1834/// with [`GraphError::Busy`].
1835///
1836/// Long enough to ride out another process's commit (a batch apply plus one
1837/// fsync), short enough that a stuck peer surfaces as an error rather than a
1838/// hang.
1839pub const WRITE_LOCK_WAIT: std::time::Duration = std::time::Duration::from_secs(2);
1840
1841/// Refusal when a `MERGE` create cannot choose a namespace.
1842///
1843/// A role bound to two or more namespaces cannot have its create arm land in
1844/// `default`, and the statement did not name `ns`. The role must name one.
1845pub const MERGE_CREATE_NEEDS_ONE_NAMESPACE: &str =
1846    "role-bound token: MERGE create requires the role to name one namespace";
1847
1848/// Interval between poll attempts while waiting for the cross-process lock.
1849pub(crate) const LOCK_POLL_INTERVAL: std::time::Duration = std::time::Duration::from_millis(10);
1850
1851/// Why `load_from_disk` is running, which decides whether it may repair.
1852#[derive(Clone, Copy, Debug, PartialEq, Eq)]
1853enum LoadOrigin {
1854    /// A fresh open. Crash recovery is this handle's job: a torn WAL tail is
1855    /// the signature of a crash and truncating it is correct, and archives
1856    /// orphaned by an interrupted prune can be swept.
1857    Open,
1858    /// A reload driven by [`GraphDb::refresh`], because another process
1859    /// replaced the snapshot. Nothing here is crash recovery — the store is
1860    /// live and someone else is writing it — so this origin writes nothing.
1861    Reload,
1862}
1863
1864/// Authorization context carried by `write_batch_authz` / `query_write_authz`.
1865///
1866/// `None` at the call site = full authority (today's zero-cost behavior).
1867/// `Some(WriteAuthz)` = role-scoped: the decision table (plan §"authz decision
1868/// table") is evaluated per-op inside `commit_logged_batch` BEFORE any WAL
1869/// record is built.  A denial returns an error with no WAL frame written.
1870///
1871/// The mask is ALWAYS `Omit`-mode: role-token paths must never acknowledge
1872/// hidden-node existence to callers.
1873#[derive(Clone, Debug)]
1874pub struct WriteAuthz {
1875    pub role: String,
1876    pub scope: WriteScope,
1877    /// Resolved by `mask_for_role` under the same write guard as the mutation.
1878    /// Always `Omit`-mode — never `Stub`.
1879    pub mask: crate::mask::NodeMask,
1880}
1881
1882/// The error every role surface gives when `roles.json` did not parse at open.
1883///
1884/// One text, so `mask_for_role` and [`GraphDb::roles_checked`] cannot drift
1885/// apart on the same cause.
1886fn roles_poisoned() -> GraphError {
1887    GraphError::Corrupt {
1888        detail: "roles.json was corrupt at open; fix the file and re-open to restore role access"
1889            .into(),
1890    }
1891}
1892
1893/// Write `bytes` to `snapshot.bin.bak` atomically with full fsync.
1894///
1895/// Uses [`RealFs::write_atomic`] which applies `F_FULLFSYNC` on macOS and
1896/// `sync_all` on other platforms, then renames the `.tmp` file into place and
1897/// syncs the directory entry. This is the only correct path for writing the
1898/// `.bak` — plain `std::fs::write + sync_all` misses both `F_FULLFSYNC` and
1899/// the directory sync.
1900pub fn write_snapshot_bak(dir: &std::path::Path, bytes: &[u8]) -> crate::Result<()> {
1901    use core_storage::fs::{FileId, Fs as _};
1902    RealFs::new(dir)
1903        .map_err(core_storage::GraphError::Io)?
1904        .write_atomic(FileId::SnapshotBak, bytes)
1905        .map_err(core_storage::GraphError::Io)
1906}
1907
1908/// Return the on-disk snapshot format version without decoding the full snapshot.
1909///
1910/// Reads only the 6-byte header (magic + version LE). Returns `None` when no
1911/// snapshot file exists (WAL-only store). Returns an error if the header is
1912/// malformed.
1913pub fn snapshot_version_at(dir: &std::path::Path) -> crate::Result<Option<u16>> {
1914    use std::io::Read as _;
1915    let path = dir.join("snapshot.bin");
1916    let mut header = [0u8; 6];
1917    let n = match std::fs::File::open(&path) {
1918        Ok(mut f) => f.read(&mut header).map_err(core_storage::GraphError::Io)?,
1919        Err(e) if e.kind() == std::io::ErrorKind::NotFound => return Ok(None),
1920        Err(e) => return Err(core_storage::GraphError::Io(e)),
1921    };
1922    core_storage::snapshot::peek_version(&header[..n])
1923}
1924
1925/// Options for [`GraphDb::snapshot_with`].
1926#[derive(Debug, Clone, Default)]
1927pub struct SnapshotOptions {
1928    /// When `true`, the WAL is preserved after the snapshot write.
1929    /// Pre-snapshot commits remain reachable via [`GraphDb::open_at`].
1930    /// When `false` (the default), the WAL is truncated to a minimal
1931    /// baseline so cold-start replay stays fast.
1932    pub keep_wal: bool,
1933    /// When `true`, the current WAL is renamed to `wal.<commit_seq>.archive`
1934    /// before a fresh WAL baseline is written (history-preserving snapshot).
1935    ///
1936    /// This is the feature opt-in: `false` (the default) leaves the existing
1937    /// truncation / keep-wal behaviour byte-identical.  `archive_wal` takes
1938    /// precedence over `keep_wal` when both are set.
1939    ///
1940    /// Archives can be scanned by [`GraphDb::node_history`],
1941    /// [`GraphDb::edge_history`], [`GraphDb::was_linked`], and
1942    /// [`GraphDb::open_at`], extending the reachable history horizon across
1943    /// snapshot boundaries.
1944    pub archive_wal: bool,
1945}
1946
1947/// Derive the scan-label sym for the commit-skip fast-path.
1948///
1949/// Walks `ops` to find the plan's leading scan op (`ScanLabel`, `IndexScan`,
1950/// or `IndexIntersect`) with a concrete label string, then interns it.
1951///
1952/// Returns `None` in all cases where skipping is unsafe:
1953/// - Any `Expand` op is present (edge traversal; edges change results regardless
1954///   of node labels).
1955/// - The leading scan has no label (`ScanLabel { label: None }` — full scan).
1956/// - No recognizable leading scan op is found.
1957///
1958/// This is the conservative v0.4.3 boundary. The caller stores the result in
1959/// [`QuerySubEntry::scan_label`] at subscribe time; `None` means always execute.
1960fn extract_scan_label(ops: &[PlanOp], syms: &mut Interner) -> Option<u32> {
1961    // Any Expand → must always re-execute (edges can change join results).
1962    if ops.iter().any(|op| matches!(op, PlanOp::Expand { .. })) {
1963        return None;
1964    }
1965    for op in ops {
1966        match op {
1967            PlanOp::ScanLabel {
1968                label: Some(label), ..
1969            } => return Some(syms.intern(label)),
1970            PlanOp::IndexScan {
1971                label: Some(label), ..
1972            } => return Some(syms.intern(label)),
1973            PlanOp::IndexIntersect {
1974                label: Some(label), ..
1975            } => return Some(syms.intern(label)),
1976            _ => {}
1977        }
1978    }
1979    None
1980}
1981
1982/// How an as-of read is restricted — the argument to
1983/// [`GraphDb::query_at_scoped`].
1984///
1985/// Every variant is resolved against the graph **as it was at the requested
1986/// commit**, not against the current graph.
1987#[derive(Debug, Clone, Copy)]
1988pub enum AsOfScope<'a> {
1989    /// Everything the named role may see. The role *definition* is the current
1990    /// one — `roles.json` is a sidecar and has no past version — but its
1991    /// `keys` and `labels` are resolved against the as-of graph.
1992    Role(&'a str),
1993    /// An explicit node-key allow-list. Keys that did not exist at that commit
1994    /// resolve to nothing.
1995    Keys(&'a [String]),
1996    /// A role intersected with a client-supplied allow-list. The intersection
1997    /// is the never-widen rule: a client mask can only narrow a role.
1998    RoleAndKeys(&'a str, &'a [String]),
1999    /// Every live node in one namespace, as the graph was at that commit.
2000    ///
2001    /// A namespace cannot change — it is set at insert and immutable — so the
2002    /// answer is simply "the nodes that existed then and are in this
2003    /// namespace". A name no node uses resolves to nothing, never to
2004    /// everything.
2005    Namespace(&'a str),
2006}
2007
2008impl GraphDb<RealFs> {
2009    /// Open the database at `dir` with default options.
2010    ///
2011    /// Equivalent to `open_with_options(dir, OpenOptions::default())`.
2012    /// Old-format snapshots (V5, V6) are automatically migrated to the
2013    /// current version on a successful load (see [`OpenOptions::auto_migrate`]).
2014    pub fn open(dir: &std::path::Path) -> Result<Self> {
2015        Self::open_with_options(dir, OpenOptions::default())
2016    }
2017
2018    /// Open the database at `dir` with explicit options.
2019    ///
2020    /// When `opts.auto_migrate` is `true` (the default) and the on-disk
2021    /// snapshot is an older format version, this function:
2022    ///   1. Copies the current `snapshot.bin` to `snapshot.bin.bak` (atomic
2023    ///      + fsynced) before any modification.
2024    ///   2. Rewrites `snapshot.bin` at the current format version via
2025    ///      [`GraphDb::snapshot_with`] with `keep_wal: true` (WAL preserved).
2026    ///
2027    /// If migration fails the error is returned and the original files are
2028    /// intact (the `.bak` was written before the new snapshot was attempted).
2029    ///
2030    /// A clean open that finds the snapshot already at the current version
2031    /// deletes any leftover `.bak` file.
2032    ///
2033    /// WAL-only stores (no snapshot) are never auto-migrated on open.
2034    ///
2035    /// `opts.repair_wal` controls the other write this function can make; see
2036    /// [`OpenOptions::repair_wal`]. With both flags `false` the open touches
2037    /// no file on disk.
2038    pub fn open_with_options(dir: &std::path::Path, opts: OpenOptions) -> Result<Self> {
2039        Self::open_dir(dir, opts, true)
2040    }
2041
2042    /// Open without taking the cross-process write lock for the handle's
2043    /// lifetime.
2044    ///
2045    /// Only [`SharedDb`](crate::SharedDb) uses this: a long-lived server holds
2046    /// its handle open indefinitely, so it takes the lock per write instead of
2047    /// keeping every other process out of the store for as long as it runs.
2048    pub(crate) fn open_unlocked(dir: &std::path::Path) -> Result<Self> {
2049        Self::open_dir(dir, OpenOptions::default(), false)
2050    }
2051
2052    fn open_dir(dir: &std::path::Path, opts: OpenOptions, hold_lock: bool) -> Result<Self> {
2053        // Header-only peek — 6 bytes, no full decode.
2054        let snap_version = snapshot_version_at(dir)?;
2055
2056        // Full load: decode snapshot + replay WAL + rebuild indexes.
2057        let mut db = Self::open_generic(RealFs::new(dir)?, opts, hold_lock)?;
2058
2059        // A read-only handle writes nothing at open, so it never migrates —
2060        // the old-format snapshot is loaded and left exactly as it is.
2061        if opts.auto_migrate && !opts.read_only {
2062            match snap_version {
2063                Some(ver) if ver < core_storage::snapshot::VERSION => {
2064                    let _tm = std::time::Instant::now();
2065                    // Copy the original snapshot to .bak at OS level — no in-memory
2066                    // buffer required for a 2+ GiB file.
2067                    //
2068                    // Crash-safety: snapshot.bin remains intact (write_atomic inside
2069                    // snapshot_with uses a .tmp+rename) until the V8 write succeeds.
2070                    // A torn .bak on crash is acceptable because the original
2071                    // snapshot.bin is the authoritative source until after the rename.
2072                    std::fs::copy(dir.join("snapshot.bin"), dir.join("snapshot.bin.bak"))
2073                        .map_err(core_storage::GraphError::Io)?;
2074                    trace_migrate!("bak copy done", _tm);
2075                    // Rewrite snapshot at current version; keep WAL intact.
2076                    db.snapshot_with(SnapshotOptions {
2077                        keep_wal: true,
2078                        ..SnapshotOptions::default()
2079                    })?;
2080                    trace_migrate!("snapshot_with done", _tm);
2081                }
2082                Some(_) => {
2083                    // Already current version: remove any leftover .bak.
2084                    let bak = dir.join("snapshot.bin.bak");
2085                    if bak.exists() {
2086                        std::fs::remove_file(&bak).map_err(core_storage::GraphError::Io)?;
2087                    }
2088                }
2089                None => {
2090                    // WAL-only store — nothing to migrate on open.
2091                }
2092            }
2093        }
2094
2095        Ok(db)
2096    }
2097
2098    /// Open a read-only view of the database as it existed after `commit`.
2099    ///
2100    /// Commit indices are 0-based over the current WAL: commit 0 is the state
2101    /// after the first WAL frame, commit N-1 is the state after the N-th (most
2102    /// recent) frame.  Call [`GraphDb::open`] to read the full current state.
2103    ///
2104    /// **Replay base.** [`GraphDb::snapshot`] truncates the WAL when it runs,
2105    /// so as-of can only reach commits recorded in the current WAL (those
2106    /// written after the most recent snapshot, or all commits if no snapshot
2107    /// was ever taken).  Commit 0 in `open_at` always refers to the first
2108    /// frame in the WAL that exists on disk, not the first ever write to the
2109    /// database.  When the on-disk snapshot recorded that it truncated the
2110    /// WAL (V7, default `keep_wal: false`), it is loaded as the base state
2111    /// before frame replay, so the as-of view includes all pre-snapshot data.
2112    /// Snapshots written with `keep_wal: true` (and legacy V5/V6 snapshots)
2113    /// are ignored and replay is WAL-only, as before.
2114    ///
2115    /// **Read-only.** Every mutation method and `snapshot()` on the returned
2116    /// instance returns [`GraphError::ReadOnly`].  Queries, `explain()`, and
2117    /// `stats()` work normally.
2118    ///
2119    /// # Errors
2120    /// - [`GraphError::CommitOutOfRange`] if `commit >= wal_commit_count` (including
2121    ///   when the WAL is empty after a snapshot).
2122    pub fn open_at(dir: &std::path::Path, commit: u64) -> Result<Self> {
2123        Self::open_at_with(RealFs::new(dir)?, commit)
2124    }
2125
2126    /// Run a **read-only** Cypher query against the graph as it existed at
2127    /// `commit` — the "time-travel" / agent-replay query. Opens a temporal view
2128    /// of this store's directory at that commit and executes the read there.
2129    ///
2130    /// The current instance is unaffected. Write statements are rejected (the
2131    /// temporal view is read-only). `commit` is a 0-based WAL commit index;
2132    /// `commit == wal_commit_count` (or `open_at`'s range) yields the newest
2133    /// state. Prefer this over holding many historical instances open.
2134    ///
2135    /// # Errors
2136    /// - [`GraphError::CommitOutOfRange`] if `commit` is past the WAL horizon.
2137    /// - A query error for a malformed or write query.
2138    pub fn query_at(
2139        &self,
2140        commit: u64,
2141        cypher: &str,
2142        params: &std::collections::BTreeMap<String, Value>,
2143    ) -> Result<ResultSet> {
2144        let temporal = self.open_at_for_read(commit, cypher)?;
2145        temporal.query(cypher, params)
2146    }
2147
2148    /// Run a **read-only** Cypher query at `commit`, restricted by `scope`.
2149    ///
2150    /// The **graph** is as of `commit`; the **role definition** is as it is
2151    /// now, because `roles.json` is a sidecar and is never a WAL record — it
2152    /// has no past version to read. A role's `keys` and `labels` are resolved
2153    /// against the commit-`commit` graph, so a role that may see a label sees
2154    /// exactly the nodes that carried it then, and an explicit key that did
2155    /// not exist yet resolves to nothing.
2156    ///
2157    /// [`AsOfScope::RoleAndKeys`] intersects the two: a client allow-list can
2158    /// only narrow what a role may see, never widen it.
2159    ///
2160    /// Write statements are rejected, exactly as [`GraphDb::query_at`] rejects
2161    /// them.
2162    ///
2163    /// # Errors
2164    /// - [`GraphError::CommitOutOfRange`] if `commit` is outside the retained
2165    ///   range; the error carries that range.
2166    /// - [`GraphError::KeyNotFound`] with a `role:` prefix for an unknown role,
2167    ///   or [`GraphError::Corrupt`] when `roles.json` was corrupt at open.
2168    /// - A query error for a malformed or write query.
2169    pub fn query_at_scoped(
2170        &self,
2171        commit: u64,
2172        cypher: &str,
2173        params: &std::collections::BTreeMap<String, Value>,
2174        scope: AsOfScope<'_>,
2175    ) -> Result<ResultSet> {
2176        let temporal = self.open_at_for_read(commit, cypher)?;
2177        let mask = temporal.mask_at_scope(scope)?;
2178        temporal.query_masked(cypher, params, &mask)
2179    }
2180
2181    /// As [`GraphDb::query_at_scoped`], with `namespace` intersected into
2182    /// whatever `scope` resolves to.
2183    ///
2184    /// This is what a surface needs when a caller passes `namespace` beside a
2185    /// `role` or a client mask on a time-travel read: [`AsOfScope`] names one
2186    /// restriction, and the namespace is a second one that composes with it
2187    /// rather than replacing it. The intersection is the never-widen rule — a
2188    /// namespace can only narrow what the scope already allows — and both legs
2189    /// are resolved against the graph as it was at `commit`.
2190    ///
2191    /// `AsOfScope::Namespace(ns)` is still the way to ask for a namespace alone.
2192    pub fn query_at_scoped_in_namespace(
2193        &self,
2194        commit: u64,
2195        cypher: &str,
2196        params: &std::collections::BTreeMap<String, Value>,
2197        scope: AsOfScope<'_>,
2198        namespace: &str,
2199    ) -> Result<ResultSet> {
2200        let temporal = self.open_at_for_read(commit, cypher)?;
2201        let mask = temporal
2202            .mask_at_scope(scope)?
2203            .intersect(&temporal.mask_for_namespace(namespace));
2204        temporal.query_masked(cypher, params, &mask)
2205    }
2206
2207    /// Run a **read-only** Cypher query at `commit`, restricted by a
2208    /// [`Scope`](crate::mask::Scope).
2209    ///
2210    /// [`AsOfScope`] names *one* restriction — a role, a key list, a namespace,
2211    /// or a role-and-keys pair. A `Scope` is the general shape a handle carries,
2212    /// and nesting can give it several role or namespace legs at once, so it
2213    /// cannot be spelled as an `AsOfScope`. This is the entry point a scoped
2214    /// handle uses for time travel; `query_at_scoped` stays the way to ask for
2215    /// one named restriction.
2216    ///
2217    /// Both the graph and the scope's key and namespace legs are resolved
2218    /// against `commit`; a role's *definition* is the current one, because
2219    /// `roles.json` is a sidecar with no past version — the same split
2220    /// [`GraphDb::query_at_scoped`] documents.
2221    ///
2222    /// The scope resolves **cold** here: a temporal handle is its own store, so
2223    /// its ids could never be served to a live read, but filling the scope's
2224    /// one-entry key memo from a handle thrown away at the end of this call
2225    /// would evict the live entry for nothing. See
2226    /// [`Scope::resolve_uncached`](crate::mask::Scope::resolve_uncached).
2227    ///
2228    /// # Errors
2229    /// - [`GraphError::CommitOutOfRange`] if `commit` is outside the retained
2230    ///   range; the error carries that range.
2231    /// - [`GraphError::KeyNotFound`] with a `role:` prefix for an unknown role,
2232    ///   or [`GraphError::Corrupt`] when `roles.json` was corrupt at open.
2233    /// - A query error for a malformed or write query.
2234    pub fn query_at_with_scope(
2235        &self,
2236        commit: u64,
2237        cypher: &str,
2238        params: &std::collections::BTreeMap<String, Value>,
2239        scope: &crate::mask::Scope,
2240    ) -> Result<ResultSet> {
2241        let temporal = self.open_at_for_read(commit, cypher)?;
2242        let mask = scope.resolve_uncached(&temporal)?;
2243        temporal.query_masked(cypher, params, &mask)
2244    }
2245
2246    /// Open the temporal view for a time-travel read and refuse write Cypher.
2247    ///
2248    /// Shared by [`GraphDb::query_at`] and [`GraphDb::query_at_scoped`] so both
2249    /// resolve the commit and reject writes identically.
2250    fn open_at_for_read(&self, commit: u64, cypher: &str) -> Result<Self> {
2251        let dir = self.fs.dir().to_path_buf();
2252        let temporal = Self::open_at(&dir, commit)?;
2253        if is_write_tokens(&lex(cypher).map_err(|e| GraphError::QueryError {
2254            detail: format!("lex: {e}"),
2255        })?) {
2256            return Err(GraphError::QueryError {
2257                detail: "query_at is read-only: write statements are not permitted in a \
2258                         time-travel query"
2259                    .into(),
2260            });
2261        }
2262        Ok(temporal)
2263    }
2264}
2265
2266impl<F: Fs> GraphDb<F> {
2267    /// Open over an arbitrary [`Fs`], repairing a torn WAL tail as usual.
2268    pub fn open_with(fs: F) -> Result<Self> {
2269        Self::open_with_repair(fs, true)
2270    }
2271
2272    /// As [`GraphDb::open_with`], but `repair_wal: false` decodes the valid WAL
2273    /// prefix without writing the truncation back. See
2274    /// [`OpenOptions::repair_wal`].
2275    pub fn open_with_repair(fs: F, repair_wal: bool) -> Result<Self> {
2276        Self::open_generic(
2277            fs,
2278            OpenOptions {
2279                repair_wal,
2280                ..OpenOptions::default()
2281            },
2282            true,
2283        )
2284    }
2285
2286    /// Shared open path.
2287    ///
2288    /// `hold_lock` requests the cross-process write lock for the whole handle
2289    /// lifetime — the right behaviour for a plain read-write `GraphDb`, whose
2290    /// owner writes through it directly. [`SharedDb`](crate::SharedDb) passes
2291    /// `false` and takes the lock per write instead, so that a long-lived
2292    /// server does not keep every other process out of the store.
2293    ///
2294    /// A read-only open never takes the lock regardless of `hold_lock`.
2295    fn open_generic(fs: F, opts: OpenOptions, hold_lock: bool) -> Result<Self> {
2296        let mut db = Self::new_empty(fs, opts);
2297        db.read_only = opts.read_only;
2298        if hold_lock && !opts.read_only {
2299            if !db.poll_lock(WRITE_LOCK_WAIT)? {
2300                return Err(GraphError::Busy { holder: None });
2301            }
2302            db.holds_lifetime_lock = true;
2303        }
2304        db.load_from_disk(LoadOrigin::Open)?;
2305        Ok(db)
2306    }
2307
2308    /// A handle with no state loaded: every field at its empty value, the
2309    /// filesystem and options in place. Only [`load_from_disk`] makes it
2310    /// usable.
2311    fn new_empty(fs: F, opts: OpenOptions) -> Self {
2312        Self {
2313            fs,
2314            ids: Arc::new(IdMap::new()),
2315            syms: Arc::new(Interner::new()),
2316            topo: Arc::new(Topology::new()),
2317            props: Arc::new(ColumnStore::new()),
2318            labels: Arc::new(Vec::new()),
2319            ns_names: vec![NS_DEFAULT.to_string()],
2320            node_ns: Vec::new(),
2321            edge_props: Arc::new(EdgeProps::new()),
2322            engine: RuleEngine::new(),
2323            view_store: ViewStore::new(),
2324            fulltext: Arc::new(FulltextIndex::new()),
2325            prop_index: PropertyIndex::new(),
2326            multiplicity: false,
2327            event_sink: None,
2328            fsync: FsyncPolicy::Strict,
2329            commit_seq: 0,
2330            commit_times: core_storage::commit_times::CommitTimes::default(),
2331            commit_times_poisoned: false,
2332            fulltext_rebuild_follows: false,
2333            commit_time_override: None,
2334            roles: Some(vec![]),
2335            role_masks: Arc::new(crate::mask::RoleMaskCache::new()),
2336            store_id: crate::mask::StoreId::next(),
2337            subscriptions: Vec::new(),
2338            query_subscriptions: Vec::new(),
2339            sub_capacity: DEFAULT_SUB_CAPACITY,
2340            read_only: false,
2341            total_wal_commits: 0,
2342            base: None,
2343            fold_overlay: None,
2344            delta_tail: Vec::new(),
2345            commits_since_fold: 0,
2346            defer_events: false,
2347            deferred_events: Vec::new(),
2348            degraded: false,
2349            v8_sections_loaded: std::sync::atomic::AtomicBool::new(false),
2350            v8_sections_mutex: std::sync::Mutex::new(()),
2351            last_change: HashMap::new(),
2352            wal_archive_retention: None,
2353            wal_horizon_floor: 0,
2354            archive_genesis_chain: false,
2355            // Nothing is proven until `load_from_disk` has looked at the store.
2356            snapshot_preserved_history: false,
2357            pending_write_authz: None,
2358            slow_query_threshold_ms: std::env::var("MUSHROOMDB_SLOW_QUERY_MS")
2359                .ok()
2360                .and_then(|v| v.parse().ok())
2361                .unwrap_or(100),
2362            slow_queries: std::sync::Mutex::new(SlowQueryLog {
2363                entries: std::collections::VecDeque::new(),
2364                total: 0,
2365            }),
2366            warned_ambiguous_exactness: std::sync::Mutex::new(HashSet::new()),
2367            started_at: std::time::Instant::now(),
2368            wal_consumed: 0,
2369            wal_frames_written: 0,
2370            snapshot_ident: None,
2371            open_opts: opts,
2372            holds_lifetime_lock: false,
2373            lock_denied: false,
2374            pinned: false,
2375        }
2376    }
2377
2378    /// Return every field describing stored graph state to its empty value,
2379    /// leaving this handle's own identity alone.
2380    ///
2381    /// Preserved on purpose: the filesystem, open options, lock ownership, the
2382    /// event sink and subscriptions, fsync policy, degraded flag, and the
2383    /// slow-query configuration and log. A caller that registered a sink or a
2384    /// subscription keeps it across a reload.
2385    fn reset_for_reload(&mut self) {
2386        self.ids = Arc::new(IdMap::new());
2387        self.syms = Arc::new(Interner::new());
2388        self.topo = Arc::new(Topology::new());
2389        self.props = Arc::new(ColumnStore::new());
2390        self.labels = Arc::new(Vec::new());
2391        self.ns_names = vec![NS_DEFAULT.to_string()];
2392        self.node_ns = Vec::new();
2393        self.edge_props = Arc::new(EdgeProps::new());
2394        self.engine = RuleEngine::new();
2395        self.view_store = ViewStore::new();
2396        self.fulltext = Arc::new(FulltextIndex::new());
2397        self.prop_index = PropertyIndex::new();
2398        // Cleared like every other declaration: a reload replays the store's own
2399        // WAL, and the opt-in comes back from it or not at all.
2400        self.multiplicity = false;
2401        self.commit_seq = 0;
2402        self.commit_times = core_storage::commit_times::CommitTimes::default();
2403        self.commit_times_poisoned = false;
2404        self.fulltext_rebuild_follows = false;
2405        self.commit_time_override = None;
2406        self.roles = Some(vec![]);
2407        // A fresh cache, not a cleared one: any reader snapshot still holding
2408        // the old `Arc` keeps it to itself, so nothing it memoised against the
2409        // pre-reload store can be read back through this handle.
2410        self.role_masks = Arc::new(crate::mask::RoleMaskCache::new());
2411        // The same move for memos this handle does not own. `commit_seq` is
2412        // zeroed just above and reseeded from `max(last_change)`, which a
2413        // delete-only commit leaves where it was — so a reload can land back on
2414        // a sequence a caller's `Scope` already cached a mask at. A new id is
2415        // what makes that entry stop matching.
2416        self.store_id = crate::mask::StoreId::next();
2417        self.total_wal_commits = 0;
2418        self.base = None;
2419        self.fold_overlay = None;
2420        self.delta_tail = Vec::new();
2421        self.commits_since_fold = 0;
2422        self.deferred_events = Vec::new();
2423        self.v8_sections_loaded
2424            .store(false, std::sync::atomic::Ordering::Release);
2425        self.last_change = HashMap::new();
2426        self.wal_horizon_floor = 0;
2427        self.archive_genesis_chain = false;
2428        // Re-derived by `load_from_disk` from the store it is about to read.
2429        self.snapshot_preserved_history = false;
2430        self.pending_write_authz = None;
2431        self.wal_consumed = 0;
2432        self.wal_frames_written = 0;
2433        self.snapshot_ident = None;
2434    }
2435
2436    /// Load the snapshot base and replay the WAL into an empty handle — the
2437    /// whole of what opening a store does after the struct exists.
2438    ///
2439    /// Split out of the open path so that [`refresh`](GraphDb::refresh) can
2440    /// rebuild a handle in place, without ownership of `F`, when another
2441    /// process replaces the snapshot underneath it.
2442    ///
2443    /// `origin` decides whether the two repair writes this function can make
2444    /// are appropriate; see [`LoadOrigin`].
2445    fn load_from_disk(&mut self, origin: LoadOrigin) -> Result<usize> {
2446        // Both writes below are crash recovery, and only an open is entitled to
2447        // perform them. A read-only handle promises to touch nothing, and a
2448        // reload driven by `refresh` is looking at a store another process is
2449        // actively writing: what looks like a torn tail there is a peer
2450        // mid-append, and what looks like an orphaned archive may be one that
2451        // peer is about to reference.
2452        let may_repair = origin == LoadOrigin::Open && !self.open_opts.read_only;
2453        let repair_wal = self.open_opts.repair_wal && may_repair;
2454        let db = self;
2455        db.wal_horizon_floor = db.fs.read_horizon_floor()?;
2456        db.archive_genesis_chain = db.fs.has_genesis_marker();
2457        // Opening cleanup: remove orphaned archives — archives whose frames all
2458        // fall below the horizon floor.  Orphans arise when a crash interrupted
2459        // the retention-prune sequence after the floor was written but before
2460        // all surplus archives were deleted.  Safe to delete: floor already
2461        // accounts for their frames.
2462        if may_repair {
2463            db.cleanup_orphaned_archives()?;
2464        }
2465        let _t0 = std::time::Instant::now();
2466        // Peek 6 bytes to determine snapshot version without reading the full
2467        // file. For RealFs this is a true partial read (O(1)); for SimFs the
2468        // default impl reads all bytes and truncates (still correct).
2469        let snap_header = db.fs.read_prefix(FileId::Snapshot, 6)?;
2470        // V8, V9 and V10 share the mmap-able container; V9 only adds section 12
2471        // and V10 adds nothing but its version stamp. A version outside that set
2472        // falls through to the full-read path below, where `snapshot::decode`
2473        // either handles it (V5–V7) or refuses it by name — which is what stops
2474        // an older binary before it reaches the WAL.
2475        let is_v8 = snap_header.len() >= 6
2476            && &snap_header[0..4] == b"GDB1"
2477            && core_storage::snapshot::is_mmap_container(u16::from_le_bytes([
2478                snap_header[4],
2479                snap_header[5],
2480            ]));
2481        // The version this store is stamped with, or `None` when it has never
2482        // been snapshotted. Read from the same six bytes, with no second read.
2483        let snapshot_version = if snap_header.len() >= 6 && &snap_header[0..4] == b"GDB1" {
2484            Some(u16::from_le_bytes([snap_header[4], snap_header[5]]))
2485        } else {
2486            None
2487        };
2488        // No snapshot means no snapshot has ever truncated the WAL, so this
2489        // handle can prove the history is whole. Once a snapshot exists that
2490        // this handle did not take, it cannot: see `snapshot_preserved_history`.
2491        db.snapshot_preserved_history = snap_header.is_empty();
2492        if is_v8 {
2493            // V8: map the file zero-copy (RealFs) or read full bytes (SimFs).
2494            // No 2.4GB heap Vec is allocated on RealFs.
2495            let mapped = Arc::new(
2496                if let Some(snap_path) = db.fs.snapshot_path() {
2497                    core_storage::v8::MappedBase::map(&snap_path)
2498                } else {
2499                    let snap_bytes = db.fs.read(FileId::Snapshot)?;
2500                    core_storage::v8::MappedBase::from_bytes(snap_bytes)
2501                }
2502                .map_err(|e| GraphError::Corrupt {
2503                    detail: format!("v8: mmap open: {e:?}"),
2504                })?,
2505            );
2506            db.restore_v8_base(Arc::clone(&mapped))?;
2507            trace_open!("restore_v8_base", _t0);
2508            db.base = Some(mapped);
2509            trace_open!("base assigned", _t0);
2510        } else if !snap_header.is_empty() {
2511            // Legacy V5-V7: full read required for decode.
2512            let snap_bytes = db.fs.read(FileId::Snapshot)?;
2513            if let Some(state) = core_storage::snapshot::decode(&snap_bytes)? {
2514                db.restore_snapshot_state(state)?;
2515            }
2516        }
2517        // else: snap_header is empty = no snapshot file, fresh store.
2518        //
2519        // Seed commit_seq from the highest seq persisted in last_change so that
2520        // WAL-replay frames (which start at commit_seq+1) always exceed any seq
2521        // already stored in the snapshot.  Without this, a db with one snapshot
2522        // commit would save last_change["a"]=1, then on reopen the first WAL
2523        // frame would replay at seq=1 again — colliding and making WAL-tail
2524        // mutations indistinguishable from the snapshot baseline.
2525        //
2526        // Safety invariant (seq-recycling):
2527        //   Recycled seqs (those below the seeded baseline) were NEVER stored in
2528        //   last_change because they belonged to a previous db lifetime — a new
2529        //   db starts at commit_seq=0 with an empty last_change.  Therefore no
2530        //   CAS precondition can carry a recycled seq as its `expected` value
2531        //   and accidentally match a live node's last_change entry.
2532        //
2533        // `expected:0` on a deleted-then-reinserted node:
2534        //   After deletion, last_changed() returns None; callers that call
2535        //   last_changed() and then use NodeUnchangedSince get None.unwrap_or(0)
2536        //   = 0.  The reinserted node gets seq > 0, so a subsequent CAS with
2537        //   expected=0 correctly conflicts.  The only way to observe actual=0 in
2538        //   a CasConflict would be a caller that invented expected=0 without ever
2539        //   calling last_changed() — unreachable via the documented API contract.
2540        if let Some(&max_seq) = db.last_change.values().max() {
2541            db.commit_seq = db.commit_seq.max(max_seq);
2542        }
2543        let bytes = db.fs.read(FileId::Wal)?;
2544        let (records, valid_len) = decode_all(&bytes);
2545        // The valid prefix is replayed either way; `repair_wal` only decides
2546        // whether the truncation is written back. A reader that races a live
2547        // appender must not persist a truncation the writer never asked for.
2548        if valid_len < bytes.len() && repair_wal {
2549            db.fs.write_atomic(FileId::Wal, &bytes[..valid_len])?;
2550        }
2551        // WAL-present path: build indexes eagerly BEFORE replay so that the
2552        // first replayed record does not trigger the lazy-init guard (which
2553        // would call reindex_all_load_state on an empty graph, defeating the
2554        // point of restoring IVF/HNSW blobs from the snapshot).
2555        if !records.is_empty() {
2556            db.ensure_v8_base_sections_loaded();
2557            trace_open!("lazy sections loaded (WAL path)", _t0);
2558        }
2559        // Scoped to this call: `refresh()` also replays frames and is *not*
2560        // followed by a rebuild, so its backfill must still run.
2561        db.fulltext_rebuild_follows = true;
2562        // Every decoded frame occupies an index, replayed or not: markers
2563        // are state no-ops but they are not index no-ops.
2564        let decoded_frames = records.len() as u64;
2565        let replayed = db.apply_frames(records);
2566        db.fulltext_rebuild_follows = false;
2567        let replayed = replayed?;
2568        // ── The multiplicity declaration, recovered from the stamp ───────────
2569        //
2570        // The opt-in is re-emitted into every baseline WAL a snapshot writes, so
2571        // ordinarily the replay above has already found it. But
2572        // `snapshot_with(archive_wal)` renames the live WAL away and writes its
2573        // replacement afterwards, and between those two points the store holds
2574        // no live declaration at all. A crash there — or a single `Err` from any
2575        // call in between — used to opt the store back out on the next open
2576        // (defect #22): it would stop counting and write a **V9** snapshot while
2577        // the archives still carried discriminant 23, which is the exact state
2578        // the V10 stamp exists to prevent.
2579        //
2580        // The V10 stamp is what carries the conclusion. The archive clause is a
2581        // scope restriction, not a second proof — an earlier version of this
2582        // comment, and defect #22, claimed otherwise, and defect #33 corrects
2583        // it. Taking the two in order:
2584        //
2585        // **The stamp.** `snapshot_with` stamps the snapshot from
2586        // `self.multiplicity` *before* it touches the WAL, and nothing rewrites
2587        // a V10 snapshot at V9 while the store believes it is opted in. So a
2588        // V10 stamp says this store reached `enable_multiplicity` far enough to
2589        // write the snapshot — and, decisively, that every older binary already
2590        // refuses this store by name. Opting in here can cost such a reader
2591        // nothing it was not already being told.
2592        //
2593        // **What the archive clause does not prove.** It is *not* evidence that
2594        // the archive was taken while the store was opted in. A store can
2595        // archive at V9 and opt in afterwards, leaving a V10 snapshot standing
2596        // beside an archive whose WAL carries no declaration at all — see
2597        // `a_failed_opt_in_beside_an_archive_comes_back_opted_in`. The inference
2598        // held in the success case by coincidence, not by construction.
2599        //
2600        // **What it does buy: scope.** Without it the recovery would also fire
2601        // on a store that reached the V10 snapshot write and then failed with no
2602        // archive in sight. That store must stay opted out, and can: no WAL was
2603        // renamed away, nothing carries discriminant 23, and its next snapshot
2604        // rewrites at V9, which puts it back within reach of every older reader.
2605        // An archive is the marker for the one state that is not recoverable
2606        // that way — a WAL renamed away that may hold the only copy of the
2607        // declaration. `no_crash_leaves_discriminant_23_unguarded` pins that
2608        // line: it sweeps a workload with no archives at all and refuses a
2609        // V10-implies-enabled rule.
2610        //
2611        // **The invariant, whichever way the clause goes:** the recovery never
2612        // opts in a store whose snapshot is not V10. A V9 store has made no
2613        // promise to an older reader, so opting it in would start writing
2614        // discriminant 23 behind a stamp that does not guard it. Pinned by
2615        // `the_recovery_never_opts_in_a_store_whose_snapshot_is_not_v10` and
2616        // `the_recovery_does_not_opt_a_store_in_by_itself`.
2617        //
2618        // What this recovery cannot do is make the opt-in atomic; it is not,
2619        // and `enable_multiplicity` says so. See defects #32-#34.
2620        if !db.multiplicity
2621            && snapshot_version == Some(core_storage::snapshot::VERSION_10)
2622            && !db.fs.list_archives()?.is_empty()
2623        {
2624            db.multiplicity = true;
2625        }
2626        // The cursor sits at the end of the valid prefix, not the end of the
2627        // file: a torn or still-being-written tail is unconsumed by definition
2628        // and stays visible to `is_stale` until it decodes.
2629        db.wal_consumed = valid_len as u64;
2630        // The frame cursor counts the same sequence `all_frames` returns:
2631        // surviving archives first, then the live WAL, offset by the floor.
2632        // Counting the archives separately rather than calling
2633        // `wal_total_commits` keeps the live WAL from being decoded twice on
2634        // every open, and costs nothing on a store that has never archived.
2635        db.wal_frames_written = db.wal_horizon_floor + db.archive_frame_count()? + decoded_frames;
2636        db.snapshot_ident = db.fs.snapshot_ident().map_err(GraphError::Io)?;
2637        trace_open!("wal replay done", _t0);
2638        // Rebuild view values after WAL replay only when there is no V8 base.
2639        // With a V8 base, view values are correct in the snapshot and are updated
2640        // incrementally during WAL replay (on_edge_changed / on_prop_changed).
2641        // A full rebuild would read overlay-only props (empty after restore_v8_base)
2642        // and overwrite correct base values with wrong results (e.g. NeighborAgg
2643        // Sum reads no "score" in overlay → writes 0.0, shadowing the correct
2644        // base value).
2645        if db.base.is_none() {
2646            let topo_view = TopologyView::owned(&db.topo);
2647            db.view_store.rebuild_all(
2648                Arc::make_mut(&mut db.props),
2649                &topo_view,
2650                &db.ids,
2651                &db.syms,
2652                &db.labels,
2653            );
2654        }
2655        // Rebuild full-text index after WAL replay.  Corrects drift from
2656        // per-record incremental apply during replay.
2657        Arc::make_mut(&mut db.fulltext).rebuild_all(
2658            &db.ids,
2659            &db.labels,
2660            &db.syms,
2661            build_props_view(&db.props, &db.base),
2662        );
2663        db.prop_index.rebuild_all(
2664            &db.ids,
2665            &db.labels,
2666            &db.syms,
2667            build_props_view(&db.props, &db.base),
2668        );
2669        // Namespaces: one pass over the `ns` column, after the snapshot is
2670        // restored and the WAL replayed. Replay maintains `node_ns` record by
2671        // record as well; this pass is what makes a snapshot-only open right,
2672        // and it reads nothing on a store with no `ns` column.
2673        db.rebuild_node_ns();
2674        // A mid-build snapshot's HNSW blob carries `complete == false`.
2675        // Register it so `serve`'s ticker sees work without waiting for a write.
2676        db.register_outstanding_index_builds();
2677        // Load roles sidecar. Missing file = no roles (Some(vec![])).
2678        // Corrupt/unparseable = poisoned (None); mask_for_role will fail-loud.
2679        db.roles = Self::load_roles_from_fs(&db.fs)?;
2680        // The time sidecar. Absent is the normal case for any store written
2681        // before v0.6.11 and is not an error; unreadable is recorded so date
2682        // queries can say "damaged" rather than "none recorded".
2683        db.load_commit_times_from_fs();
2684        // Capture the initial MVCC fold so reader() is ready immediately.
2685        db.fold_now();
2686        trace_open!("open_with complete", _t0);
2687        Ok(replayed)
2688    }
2689
2690    /// Apply decoded WAL frames to in-memory state, exactly as the open-path
2691    /// replay does — same `apply` calls, same per-frame delta drain, same
2692    /// commit-seq and last-change bookkeeping. Rules therefore fire and derived
2693    /// edges appear identically whether a frame arrives at open, from a local
2694    /// commit, or from another process by way of [`refresh`](GraphDb::refresh).
2695    ///
2696    /// Returns the number of frames applied.
2697    ///
2698    /// Deltas are drained and discarded per frame: replayed frames are already
2699    /// reflected on disk, so they are not news to a subscriber, and draining
2700    /// inside the loop keeps `pending_deltas` O(1) over a large WAL (I-2).
2701    fn apply_frames(&mut self, records: Vec<WalRecord>) -> Result<usize> {
2702        if records.is_empty() {
2703            return Ok(0);
2704        }
2705        // Materialize any state retained in the mmap base before the first
2706        // frame lands, so a replayed record cannot trip the lazy-init guard and
2707        // rebuild indexes from an empty graph. Both calls are idempotent.
2708        self.ensure_v8_base_sections_loaded();
2709        self.engine.consume_retained_state_eager(
2710            &self.ids,
2711            &self.syms,
2712            &self.labels,
2713            build_props_view(&self.props, &self.base),
2714        );
2715        let applied = records.len();
2716        for rec in records {
2717            self.apply(&rec)?;
2718            let _ = self.engine.drain_deltas();
2719            // Track commit_seq during replay so last_change entries are
2720            // consistent with the seqs assigned by log_then_apply_with on
2721            // subsequent live commits.  After N replayed frames, commit_seq=N;
2722            // live commits begin at N+1.
2723            self.commit_seq += 1;
2724            let replay_seq = self.commit_seq;
2725            self.update_last_change_from_rec(&rec, replay_seq);
2726        }
2727        // Enforce I-2: if the per-frame drain above is ever removed or skipped,
2728        // this assert catches the regression in debug builds immediately.
2729        debug_assert_eq!(
2730            self.engine.pending_delta_count(),
2731            0,
2732            "pending_deltas non-empty after replay — \
2733             per-frame drain must run inside the loop to keep memory O(1)"
2734        );
2735        // T2 note: the per-frame drain IS the suppression seam for replay.
2736        // Any future as-of replay path (Plan-15 T2) must drain here to feed
2737        // replaying subscribers; the mechanism is already in place.
2738        let _ = self.engine.drain_deltas(); // belt-and-braces no-op after loop drain
2739        Ok(applied)
2740    }
2741
2742    // ── Multi-process safety: cross-process write lock + WAL tailing ──────────
2743    //
2744    // mushroomdb is many-readers / one-writer across processes. Writers take an
2745    // advisory exclusive lock on the store's `LOCK` file; readers never do.
2746    // Every handle tracks how much of the WAL it has consumed, so it can pick
2747    // up another process's commits by decoding only the new tail rather than
2748    // reopening. See `docs/site/concurrency.md`.
2749
2750    /// Whether the store on disk has moved ahead of (or out from under) this
2751    /// handle's in-memory state.
2752    ///
2753    /// True when the WAL's length differs from this handle's cursor — another
2754    /// process committed, or is mid-append — or when the snapshot file's
2755    /// identity changed. Costs two metadata lookups and reads no file contents,
2756    /// so it is cheap enough for a read path to call.
2757    ///
2758    /// Always false for an as-of view from [`GraphDb::open_at`]: such a view is
2759    /// pinned to one commit and later commits are deliberately invisible to it.
2760    pub fn is_stale(&self) -> Result<bool> {
2761        if self.pinned {
2762            return Ok(false);
2763        }
2764        if self.fs.wal_len().map_err(GraphError::Io)? != self.wal_consumed {
2765            return Ok(true);
2766        }
2767        Ok(self.fs.snapshot_ident().map_err(GraphError::Io)? != self.snapshot_ident)
2768    }
2769
2770    /// Bring this handle up to date with every commit other processes have made,
2771    /// and return how many frames were applied.
2772    ///
2773    /// The WAL tail is decoded from this handle's cursor and applied through the
2774    /// same path the open replay uses, so rules fire and derived edges appear
2775    /// exactly as they would on a fresh open. Interners, id maps and indexes
2776    /// stay valid for the same reason.
2777    ///
2778    /// A frame another process is still writing is left alone: a trailing
2779    /// partial frame is a wait, not a corruption, and the handle stays stale
2780    /// until that frame is complete. Nothing is written to disk, so a read-only
2781    /// handle can refresh freely.
2782    ///
2783    /// When the snapshot file's identity changed, or the WAL is shorter than
2784    /// this handle's cursor, the WAL no longer continues our state — another
2785    /// process snapshotted or archived. The handle is then rebuilt from disk
2786    /// with the options it was opened with, and the return value is the number
2787    /// of frames in the new WAL.
2788    ///
2789    /// Returns 0 for an as-of view, which never follows later commits.
2790    ///
2791    /// # Errors
2792    ///
2793    /// An error here leaves the handle **degraded**: it got partway through
2794    /// applying the tail, or partway through a reload, so its in-memory state
2795    /// no longer matches any point on disk. Further mutations are refused and
2796    /// the handle must be reopened. Nothing on disk was damaged — the store
2797    /// itself is fine, and a fresh open recovers it.
2798    pub fn refresh(&mut self) -> Result<u64> {
2799        if self.pinned {
2800            return Ok(0);
2801        }
2802        let disk_ident = self.fs.snapshot_ident().map_err(GraphError::Io)?;
2803        let wal_len = self.fs.wal_len().map_err(GraphError::Io)?;
2804        if disk_ident != self.snapshot_ident || wal_len < self.wal_consumed {
2805            // The WAL no longer continues our state: rebuild from disk. State
2806            // is cleared first, so a failed load leaves an empty handle — mark
2807            // it degraded rather than let a caller read an empty graph as if
2808            // it were the store's contents.
2809            self.reset_for_reload();
2810            return match self.load_from_disk(LoadOrigin::Reload) {
2811                Ok(frames) => Ok(frames as u64),
2812                Err(e) => {
2813                    self.degraded = true;
2814                    Err(e)
2815                }
2816            };
2817        }
2818        if wal_len == self.wal_consumed {
2819            return Ok(0);
2820        }
2821        let tail = self
2822            .fs
2823            .read_range(FileId::Wal, self.wal_consumed)
2824            .map_err(GraphError::Io)?;
2825        let (records, valid_len) = decode_all(&tail);
2826        let decoded_frames = records.len() as u64;
2827        let applied = match self.apply_frames(records) {
2828            Ok(n) => n,
2829            Err(e) => {
2830                // Some frames landed and some did not, and the cursor cannot
2831                // say how many. Advancing it would skip the rest; leaving it
2832                // would replay what already applied. Neither is recoverable in
2833                // place, so refuse further writes and require a reopen.
2834                self.degraded = true;
2835                return Err(e);
2836            }
2837        };
2838        // Advance by the bytes actually decoded, never by the file length: an
2839        // incomplete trailing frame stays unconsumed for the next refresh.
2840        self.wal_consumed += valid_len as u64;
2841        self.wal_frames_written += decoded_frames;
2842        // The peer that wrote those frames also stamped them. Absorbing the
2843        // frames without the stamps leaves this handle resolving dates from a
2844        // prefix of the store's history, and — while our own map is still
2845        // empty — one commit away from rewriting the peer's file out of
2846        // existence (`first` below decides on the map, and the map is what we
2847        // just brought up to date).
2848        self.load_commit_times_from_fs();
2849        if applied > 0 {
2850            // Peer commits must reach `reader()` snapshots taken from here on.
2851            // A full fold is what open does; refresh does not build per-commit
2852            // deltas, so there is nothing cheaper that stays correct.
2853            self.fold_now();
2854        }
2855        Ok(applied as u64)
2856    }
2857
2858    /// Byte offset of the WAL prefix this handle has applied.
2859    ///
2860    /// Exposed for tests that assert the cursor tracks appended bytes exactly.
2861    #[doc(hidden)]
2862    pub fn wal_consumed(&self) -> u64 {
2863        self.wal_consumed
2864    }
2865
2866    /// Rewind the WAL cursor after the group-commit drain thread truncated a
2867    /// failed group off the tail, so the cursor still describes the file.
2868    pub(crate) fn set_wal_consumed(&mut self, len: u64) {
2869        self.wal_consumed = len;
2870    }
2871
2872    /// One non-blocking attempt at the cross-process write lock.
2873    ///
2874    /// Takes `&self` so a caller can poll for the lock *before* it acquires the
2875    /// in-process write guard. That ordering is what keeps a busy peer in
2876    /// another process from stalling this process's readers.
2877    ///
2878    /// A handle that owns the lock for its lifetime always succeeds.
2879    pub(crate) fn try_cross_process_lock(&self) -> Result<bool> {
2880        if self.holds_lifetime_lock {
2881            return Ok(true);
2882        }
2883        self.fs.try_lock_exclusive().map_err(GraphError::Io)
2884    }
2885
2886    /// Poll for the cross-process write lock until `wait` elapses.
2887    ///
2888    /// One attempt is always made, so a zero wait is a single try. Returns
2889    /// `false` when the lock is still held elsewhere at the deadline; nothing
2890    /// has been written and retrying later is safe.
2891    ///
2892    /// Only the plain-`GraphDb` open path uses this, where the caller owns the
2893    /// handle outright. [`SharedDb`](crate::SharedDb) polls
2894    /// [`try_cross_process_lock`](GraphDb::try_cross_process_lock) itself so
2895    /// that it holds no in-process guard while it waits.
2896    fn poll_lock(&self, wait: std::time::Duration) -> Result<bool> {
2897        let deadline = std::time::Instant::now() + wait;
2898        loop {
2899            if self.try_cross_process_lock()? {
2900                return Ok(true);
2901            }
2902            let now = std::time::Instant::now();
2903            if now >= deadline {
2904                return Ok(false);
2905            }
2906            std::thread::sleep(LOCK_POLL_INTERVAL.min(deadline.saturating_duration_since(now)));
2907        }
2908    }
2909
2910    /// Open a cross-process write scope, given the outcome of an already-made
2911    /// lock attempt.
2912    ///
2913    /// The caller polls for the lock first — outside any in-process guard — and
2914    /// passes what it got. On success this refreshes, so the writes about to
2915    /// happen land on top of every other process's commits. On failure the
2916    /// handle refuses WAL-appending mutations and `snapshot()` with
2917    /// [`GraphError::Busy`] until [`end_write_lock`](GraphDb::end_write_lock)
2918    /// closes the scope, so a caller holding a guard cannot write behind
2919    /// another process's back.
2920    ///
2921    /// A handle that already owns the lock for its lifetime skips the refresh:
2922    /// no other process can have written, so there is nothing to pick up.
2923    pub(crate) fn enter_write_scope(&mut self, acquired: bool) -> Result<()> {
2924        self.lock_denied = !acquired;
2925        if !acquired || self.holds_lifetime_lock {
2926            return Ok(());
2927        }
2928        if let Err(e) = self.refresh() {
2929            // Do not hold a lock we cannot use: release it and let the caller
2930            // see the underlying failure.
2931            let _ = self.fs.unlock();
2932            self.lock_denied = true;
2933            return Err(e);
2934        }
2935        Ok(())
2936    }
2937
2938    /// Close a cross-process write scope opened by
2939    /// [`enter_write_scope`](GraphDb::enter_write_scope): release the lock and
2940    /// clear the Busy latch. Safe to call when the lock was never taken.
2941    pub(crate) fn end_write_lock(&mut self) {
2942        self.lock_denied = false;
2943        if !self.holds_lifetime_lock {
2944            // Releasing a lock we do not hold is a no-op; a failure to release
2945            // is reported by the OS closing the descriptor at handle drop.
2946            let _ = self.fs.unlock();
2947        }
2948    }
2949
2950    /// As-of replay for [`GraphDb::open_at`]: snapshot base (only when the
2951    /// snapshot truncated the WAL) plus the first `commit + 1` WAL frames;
2952    /// see [`GraphDb::open_at`] for the semantics.  The per-frame drain
2953    /// mirrors `open_with` exactly so pending_delta_count is 0 on exit.
2954    /// Restore all persisted state from a decoded snapshot. Shared by
2955    /// `open_with` and (when the snapshot truncated the WAL) `open_at_with`.
2956    fn restore_snapshot_state(
2957        &mut self,
2958        state: core_storage::snapshot::SnapshotState,
2959    ) -> Result<()> {
2960        self.ids = Arc::new(state.ids);
2961        self.syms = Arc::new(state.syms);
2962        self.topo = Arc::new(state.topo);
2963        self.props = Arc::new(state.props);
2964        self.labels = Arc::new(state.labels);
2965        self.edge_props = Arc::new(state.edge_props);
2966        // Cross-section label integrity for V5/V7 snapshots: same invariants as
2967        // restore_v8_base.  A crafted bincode snapshot with a short `labels` vec,
2968        // out-of-range sym ids, or a sentinel label on a live node would otherwise
2969        // open successfully and panic later in `NodeRef::label()` or
2970        // `neighborhood_masked()`.  Catching it here turns those into typed
2971        // `GraphError::Corrupt` at open time.
2972        {
2973            let ids_len = self.ids.len();
2974            if self.labels.len() != ids_len {
2975                return Err(GraphError::Corrupt {
2976                    detail: format!(
2977                        "snapshot: labels vec has {} entries but id table has {} total slots",
2978                        self.labels.len(),
2979                        ids_len,
2980                    ),
2981                });
2982            }
2983            let syms_len = self.syms.len() as u32;
2984            for (i, &sym) in self.labels.iter().enumerate() {
2985                let is_tombstoned = self.ids.is_tombstoned(i as u32);
2986                if sym == u32::MAX {
2987                    if !is_tombstoned {
2988                        return Err(GraphError::Corrupt {
2989                            detail: format!(
2990                                "snapshot: live node at id slot {i} has sentinel label (u32::MAX)"
2991                            ),
2992                        });
2993                    }
2994                } else if sym >= syms_len {
2995                    return Err(GraphError::Corrupt {
2996                        detail: format!(
2997                            "snapshot: label at id slot {i} references sym {sym} \
2998                             which is out of interner range ({syms_len})"
2999                        ),
3000                    });
3001                }
3002            }
3003        }
3004        let defs: Vec<RuleDef> = state
3005            .rule_defs
3006            .iter()
3007            .map(|b| {
3008                decode_rule_def(b).map_err(|e| GraphError::Corrupt {
3009                    detail: format!("snapshot rule_def deserialize: {e}"),
3010                })
3011            })
3012            .collect::<Result<Vec<_>>>()?;
3013        self.engine =
3014            RuleEngine::from_persist(defs, state.provenance, state.rule_tripped, state.rule_fires);
3015        // Candidate indexes are rebuilt lazily on the first mutation (see
3016        // RuleEngine::on_node_changed).  HNSW blobs and IVF centroids from the
3017        // snapshot are retained without deserializing so that:
3018        //   - clean-open (empty WAL): indexes stay empty; blobs load on first
3019        //     ANN query via ensure_hnsw_loaded, or on first mutation via the
3020        //     lazy-init guard which calls reindex_all_load_state (the scan
3021        //     skips the HNSW build for every side the blob supplies).
3022        //   - WAL-present: open_with calls consume_retained_state_eager before
3023        //     replay so HNSW/IVF are live before any record fires the hooks.
3024        let ivf_bytes = if state.ivf_state.is_empty() {
3025            Vec::new()
3026        } else {
3027            bincode::serialize(&state.ivf_state).expect("IVF state serialize cannot fail")
3028        };
3029        // Store blobs without eagerly deserializing them.
3030        // `self.ids` is the snapshot's id table at this point — WAL replay has
3031        // not run — so its length is the line an interrupted build is detected
3032        // against.
3033        let snapshot_ids = self.ids.len() as u32;
3034        self.engine
3035            .store_snapshot_state(state.hnsw_state, ivf_bytes, snapshot_ids);
3036        // Restore view defs from snapshot (V5).
3037        // The ColumnStore already contains view values from the snapshot;
3038        // use restore_view (no collision check, no backfill) so the store
3039        // is aware of the definitions.  rebuild_all runs after WAL replay.
3040        for def_bytes in &state.view_defs {
3041            let def: ViewDef =
3042                bincode::deserialize(def_bytes).map_err(|e| GraphError::Corrupt {
3043                    detail: format!("snapshot view_def deserialize: {e}"),
3044                })?;
3045            self.view_store
3046                .restore_view(def)
3047                .map_err(|e| GraphError::Corrupt {
3048                    detail: format!("snapshot view restore: {e}"),
3049                })?;
3050        }
3051        Ok(())
3052    }
3053
3054    /// Restore all persisted state from a V8 `MappedBase` snapshot, **except**
3055    /// topology (`self.topo` stays empty and serves as the WAL-replay overlay).
3056    ///
3057    /// `self.props` IS fully materialised from the base so that HNSW/IVF blob
3058    /// deserialization and view rebuild have access to all column data.
3059    fn restore_v8_base(&mut self, mapped: Arc<core_storage::v8::MappedBase>) -> Result<()> {
3060        self.ids = Arc::new(archived_to_idmap(mapped.ids().map_err(|e| {
3061            GraphError::Corrupt {
3062                detail: format!("v8: ids section: {e:?}"),
3063            }
3064        })?));
3065        self.syms = Arc::new(archived_to_interner(mapped.syms().map_err(|e| {
3066            GraphError::Corrupt {
3067                detail: format!("v8: syms section: {e:?}"),
3068            }
3069        })?));
3070
3071        // C1: self.props is left as an empty overlay. Column reads go through
3072        // props_view() (ColumnsView::with_base), which consults the archived base
3073        // section zero-copy. This avoids the O(columns) heap copy at every open.
3074
3075        // self.topo deliberately left as Topology::new() — overlay path.
3076
3077        let meta = decode_meta(mapped.meta_bytes().map_err(|e| GraphError::Corrupt {
3078            detail: format!("v8: meta section: {e:?}"),
3079        })?)
3080        .map_err(|e| GraphError::Corrupt {
3081            detail: format!("v8: meta decode: {e:?}"),
3082        })?;
3083        self.labels = Arc::new(meta.labels);
3084        // Cross-section label integrity: labels must cover every id slot (live
3085        // and tombstoned), every non-sentinel sym must be within the interner's
3086        // bound, and no live (non-tombstoned) node may carry the u32::MAX
3087        // sentinel label.  Without this check, a crafted snapshot where the META
3088        // section (small, CRC-validated) holds a short `labels` vec, out-of-range
3089        // sym ids, or a sentinel label on a live node, would open successfully
3090        // and then panic in `NodeRef::label()`, `neighborhood_masked()`, and
3091        // related read paths.  Catching the inconsistency here converts those
3092        // panics into typed `GraphError::Corrupt` at open time.
3093        {
3094            let ids_len = self.ids.len();
3095            if self.labels.len() != ids_len {
3096                return Err(GraphError::Corrupt {
3097                    detail: format!(
3098                        "v8: labels section has {} entries but id table has {} total slots",
3099                        self.labels.len(),
3100                        ids_len,
3101                    ),
3102                });
3103            }
3104            let syms_len = self.syms.len() as u32;
3105            for (i, &sym) in self.labels.iter().enumerate() {
3106                let is_tombstoned = self.ids.is_tombstoned(i as u32);
3107                if sym == u32::MAX {
3108                    // Sentinel is only valid for tombstoned slots.
3109                    if !is_tombstoned {
3110                        return Err(GraphError::Corrupt {
3111                            detail: format!(
3112                                "v8: live node at id slot {i} has sentinel label (u32::MAX)"
3113                            ),
3114                        });
3115                    }
3116                } else if sym >= syms_len {
3117                    return Err(GraphError::Corrupt {
3118                        detail: format!(
3119                            "v8: label at id slot {i} references sym {sym} \
3120                             which is out of interner range ({syms_len})"
3121                        ),
3122                    });
3123                }
3124            }
3125        }
3126        // C3: self.edge_props stays as an empty overlay.  Reads go through
3127        // edge_props_view() which consults the mmap'd base section zero-copy
3128        // via EdgePropsView::with_base.  No heap decode at open time.
3129
3130        // Restore rule engine.
3131        let (rule_def_bytes, rule_tripped, rule_fires) =
3132            archived_rules_meta_to_owned(mapped.rules_meta_section().map_err(|e| {
3133                GraphError::Corrupt {
3134                    detail: format!("v8: rules_meta section: {e:?}"),
3135                }
3136            })?);
3137        let defs: Vec<RuleDef> = rule_def_bytes
3138            .iter()
3139            .map(|b| {
3140                decode_rule_def(b).map_err(|e| GraphError::Corrupt {
3141                    detail: format!("v8: rule_def deserialize: {e}"),
3142                })
3143            })
3144            .collect::<Result<Vec<_>>>()?;
3145        self.engine = RuleEngine::from_persist(defs, BTreeMap::new(), rule_tripped, rule_fires);
3146        // C4+C5: provenance, HNSW, and IVF sections are NOT read here.
3147        // `ensure_v8_base_sections_loaded` reads them on first use from
3148        // `self.base` (set by the caller immediately after this returns).
3149        // A clean open touches only: header + IDS + SYMS + META + RULES_META.
3150
3151        // Restore view definitions.
3152        let view_defs =
3153            archived_views_to_owned(mapped.views_section().map_err(|e| GraphError::Corrupt {
3154                detail: format!("v8: views section: {e:?}"),
3155            })?);
3156        for def_bytes in &view_defs {
3157            let def: ViewDef =
3158                bincode::deserialize(def_bytes).map_err(|e| GraphError::Corrupt {
3159                    detail: format!("v8: view_def deserialize: {e}"),
3160                })?;
3161            self.view_store
3162                .restore_view(def)
3163                .map_err(|e| GraphError::Corrupt {
3164                    detail: format!("v8: view restore: {e}"),
3165                })?;
3166        }
3167        // Load the last-change map from section 11 (small section; load eagerly).
3168        // Pre-Task-3 snapshots lack this section; `last_change_bytes` returns &[]
3169        // in that case and `decode_last_change_bytes` returns an empty map.
3170        let last_change_raw = mapped
3171            .last_change_bytes()
3172            .map_err(|e| GraphError::Corrupt {
3173                detail: format!("v8: last_change section: {e:?}"),
3174            })?;
3175        self.last_change = decode_last_change_bytes(last_change_raw);
3176
3177        // Validate that all deferred sections (provenance, HNSW, IVF) fit within
3178        // the file.  Pure bounds check — no bytes read, no page faults triggered.
3179        // Catches truncated snapshots at open time before the lazy deferred reads.
3180        mapped.validate_section_bounds().map_err(|e| match e {
3181            GraphError::Corrupt { detail } => GraphError::Corrupt {
3182                detail: format!("v8: section bounds: {detail}"),
3183            },
3184            other => other,
3185        })?;
3186        Ok(())
3187    }
3188
3189    /// Read provenance, HNSW, and IVF sections from the mmap base into the
3190    /// engine's retained fields on first call.  Subsequent calls are a no-op
3191    /// (AtomicBool fast-path).
3192    ///
3193    /// Must be called before any code path that reads or mutates engine
3194    /// provenance, HNSW, or IVF state:
3195    /// - WAL replay (before `consume_retained_state_eager`)
3196    /// - First mutation (`log_then_apply_with`)
3197    /// - Read-only paths (`stats`, `explain`, `node_edges`)
3198    /// - Snapshot (`snapshot_with`)
3199    ///
3200    /// No-op for fresh stores and V5-V7 opens (`self.base` is `None`).
3201    fn ensure_v8_base_sections_loaded(&self) {
3202        use std::sync::atomic::Ordering;
3203        if self.v8_sections_loaded.load(Ordering::Acquire) {
3204            return;
3205        }
3206        let _guard = self
3207            .v8_sections_mutex
3208            .lock()
3209            .expect("v8 sections mutex poisoned");
3210        if self.v8_sections_loaded.load(Ordering::Acquire) {
3211            return; // another caller populated while we waited
3212        }
3213        let _t = std::time::Instant::now();
3214        if let Some(base) = &self.base {
3215            // Provenance: raw rkyv bytes; CRC validated inside section_bytes.
3216            // Bounds are already validated at open time (restore_v8_base →
3217            // validate_section_bounds) — unreachable post-validate_section_bounds;
3218            // unwrap_or_default is a safety belt against impossible errors.
3219            let prov_bytes = base
3220                .provenance_raw_bytes()
3221                .map(|b| b.to_vec())
3222                .unwrap_or_default();
3223            self.engine.store_provenance_bytes(prov_bytes);
3224            // HNSW: decode rkyv blobs into owned map.
3225            let hnsw_state = base
3226                .hnsw_section()
3227                .map(archived_hnsw_to_owned)
3228                .unwrap_or_default();
3229            // IVF: raw bincode bytes; deserialized on first mutation/query.
3230            let ivf_bytes = base.ivf_bytes().map(|b| b.to_vec()).unwrap_or_default();
3231            // Called before WAL replay on a WAL-present open (`open_with`) and
3232            // before any write on a clean one, so this is the snapshot's count.
3233            let snapshot_ids = self.ids.len() as u32;
3234            self.engine
3235                .store_snapshot_state(hnsw_state, ivf_bytes, snapshot_ids);
3236        }
3237        self.v8_sections_loaded.store(true, Ordering::Release);
3238        if std::env::var("MUSHROOMDB_TRACE_OPEN").is_ok() {
3239            eprintln!(
3240                "[MUSHROOMDB_TRACE_OPEN] ensure_v8_base_sections_loaded: {:>9.3?}",
3241                _t.elapsed()
3242            );
3243        }
3244    }
3245
3246    /// Return a `TopologyView` that merges the mmap'd base (when present) with
3247    /// the in-memory WAL overlay.  Used by all read paths in db.rs that need
3248    /// the full merged topology without going through `self.view()`.
3249    fn topo_view(&self) -> TopologyView<'_> {
3250        match self.base {
3251            None => TopologyView::owned(&self.topo),
3252            Some(ref base) => {
3253                // SAFETY: base lives as long as self; section bounds validated at open.
3254                // topology() uses access_unchecked; all field reads are bounds-checked in seam.rs.
3255                let archived = base
3256                    .topology()
3257                    .expect("base topology section bounds validated at open");
3258                TopologyView::with_base(&self.topo, archived)
3259            }
3260        }
3261    }
3262
3263    /// Return a `ColumnsView` that merges the mmap'd base columns (when a V8
3264    /// snapshot is open) with the in-memory WAL overlay.  Reads consult the
3265    /// overlay first, then fall through to the archived base section zero-copy.
3266    fn props_view(&self) -> core_storage::v8::seam::ColumnsView<'_> {
3267        match self.base {
3268            None => core_storage::v8::seam::ColumnsView::owned(&self.props),
3269            Some(ref base) => {
3270                // columns() uses access_unchecked; field reads are bounds-checked in seam.rs.
3271                let archived = base
3272                    .columns()
3273                    .expect("base columns section bounds validated at open");
3274                core_storage::v8::seam::ColumnsView::with_base_cached(
3275                    &self.props,
3276                    archived,
3277                    base.mixed_cache(),
3278                )
3279                .with_shared_strings(base_string_table(base))
3280            }
3281        }
3282    }
3283
3284    /// Return an `EdgePropsView` that merges the mmap'd base edge-props section
3285    /// (when a V8 snapshot is open) with the in-memory WAL overlay.
3286    ///
3287    /// Reads consult the overlay first (for post-snapshot mutations), then fall
3288    /// through to the archived base section zero-copy.  Tombstones in the
3289    /// overlay mask deleted-from-base entries.
3290    fn edge_props_view(&self) -> EdgePropsView<'_> {
3291        match self.base {
3292            None => EdgePropsView::owned(&self.edge_props),
3293            Some(ref base) => {
3294                // edge_props_section() uses access_unchecked; field reads bounds-checked in seam.rs.
3295                let archived = base
3296                    .edge_props_section()
3297                    .expect("base edge_props section bounds validated at open");
3298                EdgePropsView::with_base(&self.edge_props, archived)
3299            }
3300        }
3301    }
3302
3303    fn open_at_with(fs: F, commit: u64) -> Result<Self> {
3304        // An as-of view never writes and is pinned to one commit: it takes no
3305        // cross-process lock and does not follow later commits.
3306        let mut db = Self::new_empty(
3307            fs,
3308            OpenOptions {
3309                repair_wal: false,
3310                auto_migrate: false,
3311                read_only: true,
3312            },
3313        );
3314        db.pinned = true; // read_only is set after replay, but pinning is immediate
3315        db.wal_horizon_floor = db.fs.read_horizon_floor()?;
3316        db.archive_genesis_chain = db.fs.has_genesis_marker();
3317        // Same orphaned-archive cleanup as open_with: floor was written first
3318        // during pruning, so a crash may have left stale archives below floor.
3319        db.cleanup_orphaned_archives()?;
3320        // Collect archive frames (oldest-first) and live WAL frames.
3321        // Archives represent pre-snapshot history; the snapshot captures the
3322        // cumulative state at the time of archiving.  Crash-window guarantee:
3323        //   A: crash before rename → WAL intact, no archive. Reopen: normal.
3324        //   B: crash after rename, before new WAL → archive present, WAL
3325        //      absent. Reopen: snapshot loaded (full state), no WAL replay.
3326        //   C: crash after new baseline WAL written → normal post-archive.
3327        let archive_ns = db.fs.list_archives()?;
3328        let mut archive_frames_all: Vec<WalRecord> = Vec::new();
3329        for n in &archive_ns {
3330            let arc_bytes = db.fs.read_archive(*n)?;
3331            let (arc_frames, _) = decode_all(&arc_bytes);
3332            archive_frames_all.extend(arc_frames);
3333        }
3334        let total_archive_frames = archive_frames_all.len() as u64;
3335
3336        let live_bytes = db.fs.read(FileId::Wal)?;
3337        let (live_records, _valid_len) = decode_all(&live_bytes);
3338        let total_surviving = total_archive_frames + live_records.len() as u64;
3339        // Global total including any pruned history below the horizon floor.
3340        let total = db.wal_horizon_floor + total_surviving;
3341
3342        // Horizon and range check.
3343        if commit < db.wal_horizon_floor {
3344            return Err(GraphError::CommitOutOfRange {
3345                commit,
3346                total,
3347                floor: db.wal_horizon_floor,
3348            });
3349        }
3350        if commit >= total {
3351            return Err(GraphError::CommitOutOfRange {
3352                commit,
3353                total,
3354                floor: db.wal_horizon_floor,
3355            });
3356        }
3357
3358        // Local index into surviving frames (0 = first frame of oldest archive).
3359        let local = commit - db.wal_horizon_floor;
3360
3361        if local < total_archive_frames {
3362            // Target commit is in an archive.  Correct replay from empty state
3363            // is only possible when the archive chain is an uninterrupted
3364            // genesis chain (first archive taken from a fresh store, no prior
3365            // WAL truncation) and no archives have been pruned (floor == 0).
3366            //
3367            // If either condition is violated the prefix needed to reconstruct
3368            // the requested state is gone; refuse rather than return wrong data.
3369            if db.wal_horizon_floor > 0 || !db.archive_genesis_chain {
3370                return Err(GraphError::CommitOutOfRange {
3371                    commit,
3372                    total,
3373                    floor: db.wal_horizon_floor,
3374                });
3375            }
3376            // Replay all archive frames up to and including the target commit
3377            // from an empty database state.  Archives must be replayed in order
3378            // so that dense-id intern tables are built up correctly.
3379            for rec in archive_frames_all.into_iter().take((local + 1) as usize) {
3380                db.apply(&rec)?;
3381                let _ = db.engine.drain_deltas();
3382            }
3383        } else {
3384            // Target commit is in the live WAL: load snapshot as base, then
3385            // replay the needed live WAL prefix.
3386            //
3387            // Base state: a truncating snapshot (wal_truncated=true) compacts
3388            // all pre-truncation / pre-archive commits.  Dense-id records in
3389            // the live WAL reference ids/interns that the snapshot provides.
3390            // Peek 6 bytes (same pattern as open_with).
3391            let snap_header = db.fs.read_prefix(FileId::Snapshot, 6)?;
3392            let is_v8 = snap_header.len() >= 6
3393                && &snap_header[0..4] == b"GDB1"
3394                && core_storage::snapshot::is_mmap_container(u16::from_le_bytes([
3395                    snap_header[4],
3396                    snap_header[5],
3397                ]));
3398            if is_v8 {
3399                let state = if let Some(snap_path) = db.fs.snapshot_path() {
3400                    let mapped = core_storage::v8::MappedBase::map(&snap_path).map_err(|e| {
3401                        GraphError::Corrupt {
3402                            detail: format!("v8: open_at mmap: {e:?}"),
3403                        }
3404                    })?;
3405                    core_storage::snapshot::decode_v8_from_mapped(&mapped)?
3406                } else {
3407                    let snap_bytes = db.fs.read(FileId::Snapshot)?;
3408                    core_storage::snapshot::decode(&snap_bytes)?
3409                };
3410                if let Some(state) = state {
3411                    if state.wal_truncated {
3412                        db.restore_snapshot_state(state)?;
3413                    }
3414                }
3415            } else if !snap_header.is_empty() {
3416                let snap_bytes = db.fs.read(FileId::Snapshot)?;
3417                if let Some(state) = core_storage::snapshot::decode(&snap_bytes)? {
3418                    if state.wal_truncated {
3419                        db.restore_snapshot_state(state)?;
3420                    }
3421                }
3422            }
3423            // else: snap_header empty = no snapshot file.
3424            let live_local = local - total_archive_frames;
3425            for rec in live_records.into_iter().take((live_local + 1) as usize) {
3426                db.apply(&rec)?;
3427                let _ = db.engine.drain_deltas();
3428            }
3429        }
3430        // Pin: pending_delta_count must be 0 after as-of replay, mirroring T1's
3431        // post-loop assert in open_with.
3432        debug_assert_eq!(
3433            db.engine.pending_delta_count(),
3434            0,
3435            "pending_deltas non-empty after open_at replay — \
3436             per-frame drain must run inside the loop to keep memory O(1)"
3437        );
3438        let _ = db.engine.drain_deltas(); // belt-and-braces no-op
3439                                          // Rebuild view values after WAL replay so derived-edge-driven views
3440                                          // reflect the as-of state.  open_at always uses the legacy path (no V8
3441                                          // base), so topo_view is always owned.
3442        {
3443            let topo_view = TopologyView::owned(&db.topo);
3444            db.view_store.rebuild_all(
3445                Arc::make_mut(&mut db.props),
3446                &topo_view,
3447                &db.ids,
3448                &db.syms,
3449                &db.labels,
3450            );
3451        }
3452        // Rebuild full-text index for as-of view (mirrors open_with pattern).
3453        Arc::make_mut(&mut db.fulltext).rebuild_all(
3454            &db.ids,
3455            &db.labels,
3456            &db.syms,
3457            build_props_view(&db.props, &db.base),
3458        );
3459        db.prop_index.rebuild_all(
3460            &db.ids,
3461            &db.labels,
3462            &db.syms,
3463            build_props_view(&db.props, &db.base),
3464        );
3465        // Namespaces on the temporal handle, built by the same pass the live
3466        // open uses, so an as-of mask narrows by the namespaces of that commit.
3467        db.rebuild_node_ns();
3468        // Load roles sidecar (current roles, not point-in-time).
3469        db.roles = Self::load_roles_from_fs(&db.fs)?;
3470        db.read_only = true;
3471        db.total_wal_commits = total;
3472        // Capture initial fold so reader() is immediately usable.
3473        db.fold_now();
3474        Ok(db)
3475    }
3476
3477    /// Whether this instance is a read-only as-of view.
3478    pub fn is_read_only(&self) -> bool {
3479        self.read_only
3480    }
3481
3482    // ── MVCC epoch reader ─────────────────────────────────────────────────────
3483
3484    /// Clone the current overlay state into a new `FrozenOverlay` and reset
3485    /// the delta tail. Called automatically every `FOLD_EVERY_K` commits and at
3486    /// the end of `open_with` / `open_at_with` to prime the reader.
3487    fn fold_now(&mut self) {
3488        // Eight `Arc::clone`s — refcount bumps, O(1). This used to deep-copy the
3489        // whole overlay: `IdMap` alone is a `HashMap<String, u32>` plus a
3490        // `Vec<String>`, so every node key was copied twice, on every open,
3491        // after every snapshot, every FOLD_EVERY_K commits on the write path,
3492        // and — with no commit threshold — on every `refresh()` that applied a
3493        // peer commit. A refreshing reader now pays nothing for a fold.
3494        //
3495        // `props` and `topo` were already cheap for a different reason: on a
3496        // snapshotted store `restore_v8_base` leaves them as empty overlays over
3497        // the zero-copy mmap. `ids`, `syms` and `fulltext` were not, and that
3498        // inconsistency was the defect.
3499        let frozen = crate::reader::FrozenOverlay {
3500            ids: std::sync::Arc::clone(&self.ids),
3501            syms: std::sync::Arc::clone(&self.syms),
3502            topo: std::sync::Arc::clone(&self.topo),
3503            props: std::sync::Arc::clone(&self.props),
3504            labels: std::sync::Arc::clone(&self.labels),
3505            edge_props: std::sync::Arc::clone(&self.edge_props),
3506            roles: self.roles.clone().map(std::sync::Arc::new),
3507            fulltext: std::sync::Arc::clone(&self.fulltext),
3508        };
3509        self.fold_overlay = Some(Arc::new(frozen));
3510        self.delta_tail.clear();
3511        self.commits_since_fold = 0;
3512    }
3513
3514    /// Capture a lock-free reader snapshot of the current db state.
3515    ///
3516    /// The read lock is held only for the duration of this call (to clone a
3517    /// handful of `Arc` handles). Subsequent query operations run without any
3518    /// lock.
3519    pub fn reader(&self) -> crate::reader::ReaderSnapshot {
3520        crate::reader::ReaderSnapshot::new(
3521            self.fold_overlay
3522                .clone()
3523                .expect("fold_overlay is always Some after open_with; call reader() after open"),
3524            self.base.clone(),
3525            self.delta_tail.clone(),
3526            // The snapshot's effective state is exactly this handle's state at
3527            // this commit, so it shares the memo and its version key.
3528            self.commit_seq,
3529            Arc::clone(&self.role_masks),
3530        )
3531    }
3532
3533    /// Append a delta the reader cannot apply, so that a corrupt overlay is
3534    /// reachable from a test.
3535    ///
3536    /// Compiled only under `test-hooks`, which the server's dev-dependency on
3537    /// this crate turns on. One call permanently corrupts every
3538    /// [`ReaderSnapshot`](crate::reader::ReaderSnapshot) taken from the handle,
3539    /// so it must not be in the published surface: `#[doc(hidden)]` hides it
3540    /// from rustdoc and from nothing else. The feature gate — not
3541    /// `#[cfg(test)]` — because its only callers are in `crates/server/tests`,
3542    /// a different crate, exactly as `core_rules`'s index counters are.
3543    ///
3544    /// [`ReaderSnapshot::effective`](crate::reader::ReaderSnapshot) folds the
3545    /// delta tail into a clone of the frozen overlay and answers
3546    /// [`GraphError::Corrupt`] when a record will not apply. Nothing a caller
3547    /// can do produces that state — `apply_one`'s failures are disagreements
3548    /// between the tail and the fold it is applied to, which the write path
3549    /// cannot create — so the `Corrupt` arm of every scoped reader method was
3550    /// reachable only by inspection until this hook existed. An `Intern` record
3551    /// claiming an id the frozen interner will not hand back is the smallest
3552    /// such disagreement.
3553    ///
3554    /// Only the tail is touched. This handle's own state is untouched and
3555    /// `commit_seq` does not move, so a role mask already memoised at this
3556    /// version stays memoised — which is exactly the state in which the HTTP
3557    /// role branches reach a scoped read with a corrupt overlay under them.
3558    #[cfg(any(test, feature = "test-hooks"))]
3559    #[doc(hidden)]
3560    pub fn push_unapplyable_delta_for_test(&mut self) {
3561        self.delta_tail.push(Arc::new(crate::reader::CommitDelta {
3562            records: vec![WalRecord::Intern {
3563                id: u32::MAX,
3564                text: "delta-tail-corruption".into(),
3565            }],
3566            derived_inserts: Vec::new(),
3567            derived_deletes: Vec::new(),
3568        }));
3569    }
3570
3571    /// Total number of WAL commits at the time [`open_at`] was called.
3572    /// Returns 0 for normal (non-as-of) instances.
3573    pub fn total_wal_commits(&self) -> u64 {
3574        self.total_wal_commits
3575    }
3576
3577    /// Apply a record to in-memory state. Used by both live writes and replay,
3578    /// so replay is definitionally identical to the original execution.
3579    fn apply(&mut self, rec: &WalRecord) -> Result<()> {
3580        // Before the record mutates anything: a store restored from a snapshot
3581        // defers building its candidate indexes until the first write, and that
3582        // build is a full node scan. Left where it used to fire — inside the
3583        // engine hook, after `props.set` and the label assignment — the scan
3584        // read the half-applied record and took the in-flight node's vector for
3585        // one the snapshot should have carried, which read as an interrupted
3586        // vector-index build and cost a full `RebuildRule` on the first
3587        // embedded write after every reopen. Hoisted here the scan sees exactly
3588        // the persisted state; the record's own hook then files its vector
3589        // through the ordinary insert path a line later.
3590        self.populate_indexes_before_write();
3591        match rec {
3592            WalRecord::InsertNode { label, key, props } => {
3593                let id = Arc::make_mut(&mut self.ids).try_insert(key)?;
3594                let sym = Arc::make_mut(&mut self.syms).intern(label);
3595                if self.labels.len() <= id as usize {
3596                    // gap slots are sentinels, never valid label symbols
3597                    Arc::make_mut(&mut self.labels).resize(id as usize + 1, u32::MAX);
3598                }
3599                Arc::make_mut(&mut self.labels)[id as usize] = sym;
3600                let mut ns_name = NS_DEFAULT.to_string();
3601                for (field, value) in props {
3602                    if field == NS_PROP {
3603                        ns_name = namespace_of_value(Some(value)).to_string();
3604                    }
3605                    Arc::make_mut(&mut self.props).set(id, field, value.clone());
3606                }
3607                self.set_node_ns(id, &ns_name);
3608                // Initialize view values for the new node before the engine runs so
3609                // delta-based increments start from a known zero baseline.
3610                self.view_store.init_node_views(
3611                    id,
3612                    Arc::make_mut(&mut self.props),
3613                    &self.syms,
3614                    &self.labels,
3615                );
3616                // Fire rules for the newly inserted node.
3617                let cursor = self.engine.pending_delta_count();
3618                let mut eng = std::mem::take(&mut self.engine);
3619                {
3620                    let mut gm = make_graph_mut(
3621                        &self.ids,
3622                        Arc::make_mut(&mut self.syms),
3623                        &self.labels,
3624                        build_props_view(&self.props, &self.base),
3625                        Arc::make_mut(&mut self.topo),
3626                        &self.base,
3627                        Arc::make_mut(&mut self.edge_props),
3628                    );
3629                    eng.on_node_changed(id, None, &mut gm);
3630                }
3631                self.engine = eng;
3632                // Process derived-edge deltas for view maintenance.
3633                // Fast path: skip the O(delta_count) allocation when no views exist.
3634                if !self.view_store.is_empty() {
3635                    #[cfg(test)]
3636                    DELTA_COPY_COUNT.with(|c| c.set(c.get() + 1));
3637                    let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
3638                    for d in &new_deltas {
3639                        self.view_store.on_edge_changed(
3640                            d.etype_sym,
3641                            d.src_id,
3642                            d.dst_id,
3643                            d.fired,
3644                            Arc::make_mut(&mut self.props),
3645                            &build_topo_view(&self.topo, &self.base),
3646                            &self.ids,
3647                            &self.syms,
3648                            &self.labels,
3649                            base_columns(&self.base),
3650                        );
3651                    }
3652                }
3653                // Full-text index maintenance: index enabled fields for this label.
3654                if self.fulltext.has_label(label) {
3655                    for (field, value) in props {
3656                        if self.fulltext.is_enabled(label, field) {
3657                            Arc::make_mut(&mut self.fulltext).add_tokens(id, field, value);
3658                        }
3659                    }
3660                }
3661                // Property (equality) index maintenance.
3662                if self.prop_index.has_label(label) {
3663                    for (field, value) in props {
3664                        self.prop_index.set(label, field, id, value);
3665                    }
3666                }
3667            }
3668            WalRecord::InsertEdge {
3669                edge_type,
3670                src_key,
3671                dst_key,
3672            } => {
3673                let src = self.ids.get(src_key).ok_or_else(|| GraphError::Corrupt {
3674                    detail: format!("wal replay references unknown key {src_key}"),
3675                })?;
3676                let dst = self.ids.get(dst_key).ok_or_else(|| GraphError::Corrupt {
3677                    detail: format!("wal replay references unknown key {dst_key}"),
3678                })?;
3679                let etype = Arc::make_mut(&mut self.syms).intern(edge_type);
3680                // Skip if the edge is already visible in the merged base+overlay
3681                // view.  This keeps WAL replay idempotent when the WAL contains
3682                // pre-snapshot records that are already encoded in a V8 base
3683                // (keep_wal=true opens and crash-before-truncation scenarios).
3684                if self.base.is_some()
3685                    && self
3686                        .topo_view()
3687                        .neighbors(etype, Direction::Out, src)
3688                        .contains(&dst)
3689                {
3690                    return Ok(());
3691                }
3692                Arc::make_mut(&mut self.topo).add_edge(etype, src, dst);
3693                // View maintenance for manual edge insert.
3694                self.view_store.on_edge_changed(
3695                    etype,
3696                    src,
3697                    dst,
3698                    true,
3699                    Arc::make_mut(&mut self.props),
3700                    &build_topo_view(&self.topo, &self.base),
3701                    &self.ids,
3702                    &self.syms,
3703                    &self.labels,
3704                    base_columns(&self.base),
3705                );
3706                // Rule engine: via-hop rules must update when user edges change.
3707                let cursor = self.engine.pending_delta_count();
3708                let mut eng = std::mem::take(&mut self.engine);
3709                {
3710                    let mut gm = make_graph_mut(
3711                        &self.ids,
3712                        Arc::make_mut(&mut self.syms),
3713                        &self.labels,
3714                        build_props_view(&self.props, &self.base),
3715                        Arc::make_mut(&mut self.topo),
3716                        &self.base,
3717                        Arc::make_mut(&mut self.edge_props),
3718                    );
3719                    eng.on_edge_changed(edge_type, src, dst, &mut gm);
3720                }
3721                self.engine = eng;
3722                if !self.view_store.is_empty() {
3723                    let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
3724                    for d in &new_deltas {
3725                        self.view_store.on_edge_changed(
3726                            d.etype_sym,
3727                            d.src_id,
3728                            d.dst_id,
3729                            d.fired,
3730                            Arc::make_mut(&mut self.props),
3731                            &build_topo_view(&self.topo, &self.base),
3732                            &self.ids,
3733                            &self.syms,
3734                            &self.labels,
3735                            base_columns(&self.base),
3736                        );
3737                    }
3738                }
3739            }
3740            WalRecord::SetProp { key, field, value } => {
3741                let id = self.ids.get(key).ok_or_else(|| GraphError::Corrupt {
3742                    detail: format!("wal replay references unknown key {key}"),
3743                })?;
3744                let old_value = build_props_view(&self.props, &self.base)
3745                    .get(id, field)
3746                    .map(|vr| vr.into_value());
3747                Arc::make_mut(&mut self.props).set(id, field, value.clone());
3748                // Fire rules for the changed field.
3749                let cursor = self.engine.pending_delta_count();
3750                let mut eng = std::mem::take(&mut self.engine);
3751                {
3752                    let mut gm = make_graph_mut(
3753                        &self.ids,
3754                        Arc::make_mut(&mut self.syms),
3755                        &self.labels,
3756                        build_props_view(&self.props, &self.base),
3757                        Arc::make_mut(&mut self.topo),
3758                        &self.base,
3759                        Arc::make_mut(&mut self.edge_props),
3760                    );
3761                    eng.on_node_changed(id, Some((field, old_value)), &mut gm);
3762                }
3763                self.engine = eng;
3764                // Derived-edge deltas → view updates.
3765                if !self.view_store.is_empty() {
3766                    #[cfg(test)]
3767                    DELTA_COPY_COUNT.with(|c| c.set(c.get() + 1));
3768                    let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
3769                    for d in &new_deltas {
3770                        self.view_store.on_edge_changed(
3771                            d.etype_sym,
3772                            d.src_id,
3773                            d.dst_id,
3774                            d.fired,
3775                            Arc::make_mut(&mut self.props),
3776                            &build_topo_view(&self.topo, &self.base),
3777                            &self.ids,
3778                            &self.syms,
3779                            &self.labels,
3780                            base_columns(&self.base),
3781                        );
3782                    }
3783                }
3784                // Neighbor-aggregate views that read `field` must also update.
3785                self.view_store.on_prop_changed(
3786                    id,
3787                    field,
3788                    Arc::make_mut(&mut self.props),
3789                    &build_topo_view(&self.topo, &self.base),
3790                    &self.ids,
3791                    &self.syms,
3792                    &self.labels,
3793                    base_columns(&self.base),
3794                );
3795                // Full-text index maintenance: update tokens for this field if indexed.
3796                if self.fulltext.field_indexed(field) {
3797                    let label_opt = self.labels.get(id as usize).and_then(|&sym| {
3798                        if sym == u32::MAX {
3799                            None
3800                        } else {
3801                            self.syms.resolve(sym)
3802                        }
3803                    });
3804                    if let Some(label) = label_opt {
3805                        if self.fulltext.is_enabled(label, field) {
3806                            Arc::make_mut(&mut self.fulltext).remove_node_field(id, field);
3807                            Arc::make_mut(&mut self.fulltext).add_tokens(id, field, value);
3808                        }
3809                    }
3810                }
3811                // Property (equality) index maintenance: re-key this node's value.
3812                if self.prop_index.field_indexed(field) {
3813                    let label_opt = self.labels.get(id as usize).and_then(|&sym| {
3814                        if sym == u32::MAX {
3815                            None
3816                        } else {
3817                            self.syms.resolve(sym)
3818                        }
3819                    });
3820                    if let Some(label) = label_opt {
3821                        self.prop_index.set(label, field, id, value);
3822                    }
3823                }
3824            }
3825            WalRecord::Intern { id, text } => {
3826                if let Some(existing) = self.syms.get(text) {
3827                    if existing != *id {
3828                        return Err(GraphError::Corrupt {
3829                            detail: format!(
3830                                "wal intern mismatch for {text:?}: have {existing}, record {id}"
3831                            ),
3832                        });
3833                    }
3834                } else {
3835                    let got = Arc::make_mut(&mut self.syms).intern(text);
3836                    if got != *id {
3837                        return Err(GraphError::Corrupt {
3838                            detail: format!(
3839                                "wal intern assigned {got} for {text:?}, record wanted {id}"
3840                            ),
3841                        });
3842                    }
3843                }
3844            }
3845            WalRecord::InsertNodeId { label, key, props } => {
3846                let id = Arc::make_mut(&mut self.ids).try_insert(key)?;
3847                if self.labels.len() <= id as usize {
3848                    Arc::make_mut(&mut self.labels).resize(id as usize + 1, u32::MAX);
3849                }
3850                Arc::make_mut(&mut self.labels)[id as usize] = *label;
3851                let label_str = self
3852                    .syms
3853                    .resolve(*label)
3854                    .ok_or_else(|| GraphError::Corrupt {
3855                        detail: format!("wal InsertNodeId unknown label intern {label}"),
3856                    })?
3857                    .to_string();
3858                let mut ns_name = NS_DEFAULT.to_string();
3859                for (field_sym, value) in props {
3860                    let field =
3861                        self.syms
3862                            .resolve(*field_sym)
3863                            .ok_or_else(|| GraphError::Corrupt {
3864                                detail: format!(
3865                                    "wal InsertNodeId unknown field intern {field_sym}"
3866                                ),
3867                            })?;
3868                    if field == NS_PROP {
3869                        ns_name = namespace_of_value(Some(value)).to_string();
3870                    }
3871                    Arc::make_mut(&mut self.props).set(id, field, value.clone());
3872                }
3873                self.set_node_ns(id, &ns_name);
3874                self.view_store.init_node_views(
3875                    id,
3876                    Arc::make_mut(&mut self.props),
3877                    &self.syms,
3878                    &self.labels,
3879                );
3880                let cursor = self.engine.pending_delta_count();
3881                let mut eng = std::mem::take(&mut self.engine);
3882                {
3883                    let mut gm = make_graph_mut(
3884                        &self.ids,
3885                        Arc::make_mut(&mut self.syms),
3886                        &self.labels,
3887                        build_props_view(&self.props, &self.base),
3888                        Arc::make_mut(&mut self.topo),
3889                        &self.base,
3890                        Arc::make_mut(&mut self.edge_props),
3891                    );
3892                    eng.on_node_changed(id, None, &mut gm);
3893                }
3894                self.engine = eng;
3895                if !self.view_store.is_empty() {
3896                    #[cfg(test)]
3897                    DELTA_COPY_COUNT.with(|c| c.set(c.get() + 1));
3898                    let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
3899                    for d in &new_deltas {
3900                        self.view_store.on_edge_changed(
3901                            d.etype_sym,
3902                            d.src_id,
3903                            d.dst_id,
3904                            d.fired,
3905                            Arc::make_mut(&mut self.props),
3906                            &build_topo_view(&self.topo, &self.base),
3907                            &self.ids,
3908                            &self.syms,
3909                            &self.labels,
3910                            base_columns(&self.base),
3911                        );
3912                    }
3913                }
3914                if self.fulltext.has_label(&label_str) {
3915                    for (field_sym, value) in props {
3916                        let Some(field) = self.syms.resolve(*field_sym) else {
3917                            continue;
3918                        };
3919                        if self.fulltext.is_enabled(&label_str, field) {
3920                            Arc::make_mut(&mut self.fulltext).add_tokens(id, field, value);
3921                        }
3922                    }
3923                }
3924                if self.prop_index.has_label(&label_str) {
3925                    for (field_sym, value) in props {
3926                        let Some(field) = self.syms.resolve(*field_sym) else {
3927                            continue;
3928                        };
3929                        self.prop_index.set(&label_str, field, id, value);
3930                    }
3931                }
3932            }
3933            WalRecord::InsertEdgeId { etype, src, dst } => {
3934                // Replay-over-snapshot: dense ids in the pre-snapshot WAL may
3935                // already be tombstoned. Skip rather than attaching edges to
3936                // dead ids (DeleteNode keys the live re-insert, not the old id).
3937                if self.ids.is_tombstoned(*src)
3938                    || self.ids.is_tombstoned(*dst)
3939                    || self.ids.key_of(*src).is_none()
3940                    || self.ids.key_of(*dst).is_none()
3941                {
3942                    return Ok(());
3943                }
3944                // Skip if already visible in the merged view (same idempotency
3945                // guard as InsertEdge above: prevents double-counting when
3946                // pre-snapshot WAL records are replayed over a V8 base).
3947                if self.base.is_some()
3948                    && self
3949                        .topo_view()
3950                        .neighbors(*etype, Direction::Out, *src)
3951                        .contains(dst)
3952                {
3953                    return Ok(());
3954                }
3955                Arc::make_mut(&mut self.topo).add_edge(*etype, *src, *dst);
3956                self.view_store.on_edge_changed(
3957                    *etype,
3958                    *src,
3959                    *dst,
3960                    true,
3961                    Arc::make_mut(&mut self.props),
3962                    &build_topo_view(&self.topo, &self.base),
3963                    &self.ids,
3964                    &self.syms,
3965                    &self.labels,
3966                    base_columns(&self.base),
3967                );
3968                // Rule engine: via-hop rules fire when user via-edges are inserted.
3969                // Resolve etype back to string so on_edge_changed can match rules by name.
3970                if let Some(etype_str) = self.syms.resolve(*etype).map(|s| s.to_string()) {
3971                    let cursor = self.engine.pending_delta_count();
3972                    let mut eng = std::mem::take(&mut self.engine);
3973                    {
3974                        let mut gm = make_graph_mut(
3975                            &self.ids,
3976                            Arc::make_mut(&mut self.syms),
3977                            &self.labels,
3978                            build_props_view(&self.props, &self.base),
3979                            Arc::make_mut(&mut self.topo),
3980                            &self.base,
3981                            Arc::make_mut(&mut self.edge_props),
3982                        );
3983                        eng.on_edge_changed(&etype_str, *src, *dst, &mut gm);
3984                    }
3985                    self.engine = eng;
3986                    if !self.view_store.is_empty() {
3987                        let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
3988                        for d in &new_deltas {
3989                            self.view_store.on_edge_changed(
3990                                d.etype_sym,
3991                                d.src_id,
3992                                d.dst_id,
3993                                d.fired,
3994                                Arc::make_mut(&mut self.props),
3995                                &build_topo_view(&self.topo, &self.base),
3996                                &self.ids,
3997                                &self.syms,
3998                                &self.labels,
3999                                base_columns(&self.base),
4000                            );
4001                        }
4002                    }
4003                }
4004            }
4005            WalRecord::SetPropId { id, field, value } => {
4006                if self.ids.is_tombstoned(*id) || self.ids.key_of(*id).is_none() {
4007                    return Ok(());
4008                }
4009                let field_str = self
4010                    .syms
4011                    .resolve(*field)
4012                    .ok_or_else(|| GraphError::Corrupt {
4013                        detail: format!("wal SetPropId unknown field intern {field}"),
4014                    })?
4015                    .to_string();
4016                let old_value = build_props_view(&self.props, &self.base)
4017                    .get(*id, &field_str)
4018                    .map(|vr| vr.into_value());
4019                Arc::make_mut(&mut self.props).set(*id, &field_str, value.clone());
4020                let cursor = self.engine.pending_delta_count();
4021                let mut eng = std::mem::take(&mut self.engine);
4022                {
4023                    let mut gm = make_graph_mut(
4024                        &self.ids,
4025                        Arc::make_mut(&mut self.syms),
4026                        &self.labels,
4027                        build_props_view(&self.props, &self.base),
4028                        Arc::make_mut(&mut self.topo),
4029                        &self.base,
4030                        Arc::make_mut(&mut self.edge_props),
4031                    );
4032                    eng.on_node_changed(*id, Some((field_str.as_str(), old_value)), &mut gm);
4033                }
4034                self.engine = eng;
4035                if !self.view_store.is_empty() {
4036                    #[cfg(test)]
4037                    DELTA_COPY_COUNT.with(|c| c.set(c.get() + 1));
4038                    let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
4039                    for d in &new_deltas {
4040                        self.view_store.on_edge_changed(
4041                            d.etype_sym,
4042                            d.src_id,
4043                            d.dst_id,
4044                            d.fired,
4045                            Arc::make_mut(&mut self.props),
4046                            &build_topo_view(&self.topo, &self.base),
4047                            &self.ids,
4048                            &self.syms,
4049                            &self.labels,
4050                            base_columns(&self.base),
4051                        );
4052                    }
4053                }
4054                self.view_store.on_prop_changed(
4055                    *id,
4056                    &field_str,
4057                    Arc::make_mut(&mut self.props),
4058                    &build_topo_view(&self.topo, &self.base),
4059                    &self.ids,
4060                    &self.syms,
4061                    &self.labels,
4062                    base_columns(&self.base),
4063                );
4064                if self.fulltext.field_indexed(&field_str) {
4065                    let label_opt = self.labels.get(*id as usize).and_then(|&sym| {
4066                        if sym == u32::MAX {
4067                            None
4068                        } else {
4069                            self.syms.resolve(sym)
4070                        }
4071                    });
4072                    if let Some(label) = label_opt {
4073                        if self.fulltext.is_enabled(label, &field_str) {
4074                            Arc::make_mut(&mut self.fulltext).remove_node_field(*id, &field_str);
4075                            Arc::make_mut(&mut self.fulltext).add_tokens(*id, &field_str, value);
4076                        }
4077                    }
4078                }
4079                if self.prop_index.field_indexed(&field_str) {
4080                    let label_opt = self.labels.get(*id as usize).and_then(|&sym| {
4081                        if sym == u32::MAX {
4082                            None
4083                        } else {
4084                            self.syms.resolve(sym)
4085                        }
4086                    });
4087                    if let Some(label) = label_opt {
4088                        self.prop_index.set(label, &field_str, *id, value);
4089                    }
4090                }
4091            }
4092            WalRecord::CreateRule { def_bytes } => {
4093                let def: RuleDef = decode_rule_def(def_bytes).map_err(|e| GraphError::Corrupt {
4094                    detail: format!("CreateRule def_bytes deserialize failed: {e}"),
4095                })?;
4096                // Replay-over-snapshot idempotency: the rule was captured in the snapshot
4097                // so the engine already has it; silently skip to avoid a spurious
4098                // RuleInvalid error in the crash window between snapshot write and WAL
4099                // truncation.
4100                if self.engine.rules().any(|r| r.name == def.name) {
4101                    return Ok(());
4102                }
4103                let cursor = self.engine.pending_delta_count();
4104                let mut eng = std::mem::take(&mut self.engine);
4105                let result = {
4106                    let mut gm = make_graph_mut(
4107                        &self.ids,
4108                        Arc::make_mut(&mut self.syms),
4109                        &self.labels,
4110                        build_props_view(&self.props, &self.base),
4111                        Arc::make_mut(&mut self.topo),
4112                        &self.base,
4113                        Arc::make_mut(&mut self.edge_props),
4114                    );
4115                    eng.create_rule(def, &mut gm)
4116                };
4117                self.engine = eng;
4118                result.map_err(|e| GraphError::RuleInvalid { detail: e })?;
4119                // Derived-edge fires from backfill → view updates.
4120                // Fast path: skip O(edge_count) allocation when no views exist.
4121                if !self.view_store.is_empty() {
4122                    #[cfg(test)]
4123                    DELTA_COPY_COUNT.with(|c| c.set(c.get() + 1));
4124                    let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
4125                    for d in &new_deltas {
4126                        self.view_store.on_edge_changed(
4127                            d.etype_sym,
4128                            d.src_id,
4129                            d.dst_id,
4130                            d.fired,
4131                            Arc::make_mut(&mut self.props),
4132                            &build_topo_view(&self.topo, &self.base),
4133                            &self.ids,
4134                            &self.syms,
4135                            &self.labels,
4136                            base_columns(&self.base),
4137                        );
4138                    }
4139                }
4140            }
4141            WalRecord::DeleteRule { name } => {
4142                // Replay-over-snapshot idempotency: the snapshot already captured the
4143                // post-delete state so the rule is absent; silently skip to avoid a
4144                // spurious RuleNotFound error in the crash window between snapshot write
4145                // and WAL truncation.
4146                if !self.engine.rules().any(|r| r.name == *name) {
4147                    return Ok(());
4148                }
4149                let cursor = self.engine.pending_delta_count();
4150                let mut eng = std::mem::take(&mut self.engine);
4151                let result = {
4152                    let mut gm = make_graph_mut(
4153                        &self.ids,
4154                        Arc::make_mut(&mut self.syms),
4155                        &self.labels,
4156                        build_props_view(&self.props, &self.base),
4157                        Arc::make_mut(&mut self.topo),
4158                        &self.base,
4159                        Arc::make_mut(&mut self.edge_props),
4160                    );
4161                    eng.delete_rule(name, &mut gm)
4162                };
4163                self.engine = eng;
4164                result.map_err(|_| GraphError::RuleNotFound { name: name.clone() })?;
4165                // Derived-edge retractions → view updates.
4166                if !self.view_store.is_empty() {
4167                    #[cfg(test)]
4168                    DELTA_COPY_COUNT.with(|c| c.set(c.get() + 1));
4169                    let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
4170                    for d in &new_deltas {
4171                        self.view_store.on_edge_changed(
4172                            d.etype_sym,
4173                            d.src_id,
4174                            d.dst_id,
4175                            d.fired,
4176                            Arc::make_mut(&mut self.props),
4177                            &build_topo_view(&self.topo, &self.base),
4178                            &self.ids,
4179                            &self.syms,
4180                            &self.labels,
4181                            base_columns(&self.base),
4182                        );
4183                    }
4184                }
4185            }
4186            WalRecord::RemoveProp { key, field } => {
4187                // Recovery-safe: unknown key or already-absent field is a
4188                // clean no-op. Crash-window replay over a snapshot that
4189                // already applied this record must not Err.
4190                let Some(id) = self.ids.get(key) else {
4191                    return Ok(());
4192                };
4193                // Read old value through the seam for rule retraction.
4194                let old = build_props_view(&self.props, &self.base)
4195                    .get(id, field)
4196                    .map(|vr| vr.into_value());
4197                Arc::make_mut(&mut self.props).remove(id, field);
4198                // If the base still supplies the value after the overlay removal,
4199                // record a tombstone so ColumnsView::get does not resurrect it.
4200                // This covers both the base-only case AND the both-resident case:
4201                //   base-only (in_overlay=false): old prop was only in base, remove
4202                //     is a no-op on overlay, base still visible → tombstone needed.
4203                //   both-resident (in_overlay=true): overlay had v2, base has v1;
4204                //     removing overlay uncovers v1 → tombstone needed.
4205                // Idempotent on double-replay: second pass sees the tombstone →
4206                // get() returns None → condition is false → no duplicate tombstone.
4207                if build_props_view(&self.props, &self.base)
4208                    .get(id, field)
4209                    .is_some()
4210                {
4211                    Arc::make_mut(&mut self.props).record_prop_tombstone(id, field);
4212                }
4213                let cursor = self.engine.pending_delta_count();
4214                let mut eng = std::mem::take(&mut self.engine);
4215                {
4216                    let mut gm = make_graph_mut(
4217                        &self.ids,
4218                        Arc::make_mut(&mut self.syms),
4219                        &self.labels,
4220                        build_props_view(&self.props, &self.base),
4221                        Arc::make_mut(&mut self.topo),
4222                        &self.base,
4223                        Arc::make_mut(&mut self.edge_props),
4224                    );
4225                    eng.on_node_changed(id, Some((field, old)), &mut gm);
4226                }
4227                self.engine = eng;
4228                // Derived-edge deltas → view updates.
4229                if !self.view_store.is_empty() {
4230                    #[cfg(test)]
4231                    DELTA_COPY_COUNT.with(|c| c.set(c.get() + 1));
4232                    let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
4233                    for d in &new_deltas {
4234                        self.view_store.on_edge_changed(
4235                            d.etype_sym,
4236                            d.src_id,
4237                            d.dst_id,
4238                            d.fired,
4239                            Arc::make_mut(&mut self.props),
4240                            &build_topo_view(&self.topo, &self.base),
4241                            &self.ids,
4242                            &self.syms,
4243                            &self.labels,
4244                            base_columns(&self.base),
4245                        );
4246                    }
4247                }
4248                // Neighbor-aggregate views that read `field` must also update.
4249                self.view_store.on_prop_changed(
4250                    id,
4251                    field,
4252                    Arc::make_mut(&mut self.props),
4253                    &build_topo_view(&self.topo, &self.base),
4254                    &self.ids,
4255                    &self.syms,
4256                    &self.labels,
4257                    base_columns(&self.base),
4258                );
4259                // Full-text index maintenance: remove tokens for this field.
4260                if self.fulltext.field_indexed(field) {
4261                    Arc::make_mut(&mut self.fulltext).remove_node_field(id, field);
4262                }
4263                // Property (equality) index maintenance: drop this node's entry.
4264                if self.prop_index.field_indexed(field) {
4265                    if let Some(label) = self.labels.get(id as usize).and_then(|&sym| {
4266                        (sym != u32::MAX).then(|| self.syms.resolve(sym)).flatten()
4267                    }) {
4268                        self.prop_index.remove_node(label, field, id);
4269                    }
4270                }
4271            }
4272            WalRecord::DeleteEdge {
4273                edge_type,
4274                src_key,
4275                dst_key,
4276            } => {
4277                // Recovery-safe: unknown keys, unknown etype, or already-
4278                // absent edge is a clean no-op (remove_edge returns false).
4279                let Some(src) = self.ids.get(src_key) else {
4280                    return Ok(());
4281                };
4282                let Some(dst) = self.ids.get(dst_key) else {
4283                    return Ok(());
4284                };
4285                let Some(etype) = self.syms.get(edge_type) else {
4286                    return Ok(());
4287                };
4288                // I3: phantom-tombstone guard.  When a V8 base is present, a
4289                // DeleteEdge WAL record for an edge that was already absorbed into
4290                // the new base (i.e. neither in overlay nor in base) must be skipped.
4291                // Without this guard, remove_edge records a tombstone for an edge
4292                // that no longer exists, incorrectly understating edge_count.
4293                if self.base.is_some()
4294                    && !self
4295                        .topo_view()
4296                        .neighbors(etype, core_storage::topology::Direction::Out, src)
4297                        .contains(&dst)
4298                {
4299                    return Ok(());
4300                }
4301                Arc::make_mut(&mut self.topo).remove_edge(etype, src, dst);
4302                Arc::make_mut(&mut self.edge_props).remove_edge(etype, src, dst);
4303                // View maintenance for manual edge delete (topo already updated above).
4304                self.view_store.on_edge_changed(
4305                    etype,
4306                    src,
4307                    dst,
4308                    false,
4309                    Arc::make_mut(&mut self.props),
4310                    &build_topo_view(&self.topo, &self.base),
4311                    &self.ids,
4312                    &self.syms,
4313                    &self.labels,
4314                    base_columns(&self.base),
4315                );
4316                // Rule engine: via-hop rules must retract when user via-edges are deleted.
4317                let cursor = self.engine.pending_delta_count();
4318                let mut eng = std::mem::take(&mut self.engine);
4319                {
4320                    let mut gm = make_graph_mut(
4321                        &self.ids,
4322                        Arc::make_mut(&mut self.syms),
4323                        &self.labels,
4324                        build_props_view(&self.props, &self.base),
4325                        Arc::make_mut(&mut self.topo),
4326                        &self.base,
4327                        Arc::make_mut(&mut self.edge_props),
4328                    );
4329                    eng.on_edge_changed(edge_type, src, dst, &mut gm);
4330                }
4331                self.engine = eng;
4332                if !self.view_store.is_empty() {
4333                    let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
4334                    for d in &new_deltas {
4335                        self.view_store.on_edge_changed(
4336                            d.etype_sym,
4337                            d.src_id,
4338                            d.dst_id,
4339                            d.fired,
4340                            Arc::make_mut(&mut self.props),
4341                            &build_topo_view(&self.topo, &self.base),
4342                            &self.ids,
4343                            &self.syms,
4344                            &self.labels,
4345                            base_columns(&self.base),
4346                        );
4347                    }
4348                }
4349            }
4350            WalRecord::DeleteNode { key } => {
4351                // Recovery-safe: already-tombstoned / unknown key is a clean
4352                // no-op. Crash-window replay over a snapshot that already
4353                // applied this record cannot recover the retired id from the
4354                // key (`IdMap::get` is None), so every subsequent step is
4355                // skipped. Each step is independently idempotent if invoked
4356                // twice on a still-live id: retraction is a no-op on empty
4357                // provenance, `remove_edge` returns false, `remove_all` is a
4358                // no-op, `ids.delete` returns None, label sentinel is sticky.
4359                let Some(n) = self.ids.get(key) else {
4360                    return Ok(());
4361                };
4362
4363                // (1) Retract derived edges + de-index while props/labels live.
4364                let cursor = self.engine.pending_delta_count();
4365                let mut eng = std::mem::take(&mut self.engine);
4366                {
4367                    let mut gm = make_graph_mut(
4368                        &self.ids,
4369                        Arc::make_mut(&mut self.syms),
4370                        &self.labels,
4371                        build_props_view(&self.props, &self.base),
4372                        Arc::make_mut(&mut self.topo),
4373                        &self.base,
4374                        Arc::make_mut(&mut self.edge_props),
4375                    );
4376                    eng.on_node_removed(n, &mut gm);
4377                }
4378                self.engine = eng;
4379                // Derived-edge retractions → view updates for neighbors.
4380                if !self.view_store.is_empty() {
4381                    #[cfg(test)]
4382                    DELTA_COPY_COUNT.with(|c| c.set(c.get() + 1));
4383                    let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
4384                    for d in &new_deltas {
4385                        self.view_store.on_edge_changed(
4386                            d.etype_sym,
4387                            d.src_id,
4388                            d.dst_id,
4389                            d.fired,
4390                            Arc::make_mut(&mut self.props),
4391                            &build_topo_view(&self.topo, &self.base),
4392                            &self.ids,
4393                            &self.syms,
4394                            &self.labels,
4395                            base_columns(&self.base),
4396                        );
4397                    }
4398                }
4399
4400                // (2) Sweep ALL remaining edges incident to n, both directions,
4401                // every etype. This cascade is intentionally mask-independent:
4402                // topology integrity requires removing every edge touching the
4403                // deleted node regardless of the caller's visibility scope.
4404                // (The mask limits which nodes a role's read phase can return;
4405                // the WAL delete always executes with full storage authority.)
4406                // Collect then remove so neighbor slices stay valid during
4407                // iteration. Remove from topo first, then call view maintenance
4408                // so Avg/Min/Max recompute sees the correct (reduced) neighbor set.
4409                let etypes: Vec<u32> = self.topo.etypes().collect();
4410                let mut doomed = Vec::new();
4411                for et in &etypes {
4412                    for &dst in self.topo.neighbors(*et, Direction::Out, n).as_ref() {
4413                        doomed.push((*et, n, dst));
4414                    }
4415                    for &src in self.topo.neighbors(*et, Direction::In, n).as_ref() {
4416                        doomed.push((*et, src, n));
4417                    }
4418                }
4419                for (et, s, d) in doomed {
4420                    Arc::make_mut(&mut self.topo).remove_edge(et, s, d);
4421                    Arc::make_mut(&mut self.edge_props).remove_edge(et, s, d);
4422                    // View maintenance: n's own view values will be cleared by
4423                    // remove_all below; only update surviving neighbors.
4424                    self.view_store.on_edge_changed(
4425                        et,
4426                        s,
4427                        d,
4428                        false,
4429                        Arc::make_mut(&mut self.props),
4430                        &build_topo_view(&self.topo, &self.base),
4431                        &self.ids,
4432                        &self.syms,
4433                        &self.labels,
4434                        base_columns(&self.base),
4435                    );
4436                }
4437
4438                // (3) Drop every remaining prop (`ColumnStore::remove_all`).
4439                Arc::make_mut(&mut self.props).remove_all(n);
4440                // Full-text index maintenance: remove all tokens for this node.
4441                Arc::make_mut(&mut self.fulltext).remove_node(n);
4442                // Property (equality) index maintenance: drop all entries for n.
4443                self.prop_index.remove_node_all(n);
4444
4445                // (4) Retire the dense id and stamp the label sentinel.
4446                Arc::make_mut(&mut self.ids).delete(key);
4447                if let Some(slot) = Arc::make_mut(&mut self.labels).get_mut(n as usize) {
4448                    *slot = u32::MAX;
4449                }
4450            }
4451            WalRecord::Batch(inner) => {
4452                // Apply each inner record in order through the same apply path.
4453                // Inner records are validated free of nested Batch by encode_record.
4454                for rec in inner {
4455                    self.apply(rec)?;
4456                }
4457            }
4458            WalRecord::RebuildRule { name } => {
4459                // Replay-over-snapshot idempotency: the snapshot may already
4460                // reflect a later delete_rule, so the rule is absent; skip.
4461                if !self.engine.rules().any(|r| r.name == *name) {
4462                    return Ok(());
4463                }
4464                let cursor = self.engine.pending_delta_count();
4465                let mut eng = std::mem::take(&mut self.engine);
4466                let result = {
4467                    let mut gm = make_graph_mut(
4468                        &self.ids,
4469                        Arc::make_mut(&mut self.syms),
4470                        &self.labels,
4471                        build_props_view(&self.props, &self.base),
4472                        Arc::make_mut(&mut self.topo),
4473                        &self.base,
4474                        Arc::make_mut(&mut self.edge_props),
4475                    );
4476                    eng.rebuild(name, &mut gm)
4477                };
4478                self.engine = eng;
4479                result.map_err(|_| GraphError::RuleNotFound { name: name.clone() })?;
4480                // Derived-edge delta changes → view updates.
4481                if !self.view_store.is_empty() {
4482                    #[cfg(test)]
4483                    DELTA_COPY_COUNT.with(|c| c.set(c.get() + 1));
4484                    let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
4485                    for d in &new_deltas {
4486                        self.view_store.on_edge_changed(
4487                            d.etype_sym,
4488                            d.src_id,
4489                            d.dst_id,
4490                            d.fired,
4491                            Arc::make_mut(&mut self.props),
4492                            &build_topo_view(&self.topo, &self.base),
4493                            &self.ids,
4494                            &self.syms,
4495                            &self.labels,
4496                            base_columns(&self.base),
4497                        );
4498                    }
4499                }
4500            }
4501            WalRecord::CreateView { def_bytes } => {
4502                let def: ViewDef =
4503                    bincode::deserialize(def_bytes).map_err(|e| GraphError::Corrupt {
4504                        detail: format!("CreateView def_bytes deserialize failed: {e}"),
4505                    })?;
4506                // Replay-over-snapshot idempotency: view already present → skip.
4507                if self.view_store.has_view(&def.name) {
4508                    return Ok(());
4509                }
4510                self.view_store
4511                    .create_view(
4512                        def,
4513                        Arc::make_mut(&mut self.props),
4514                        &build_topo_view(&self.topo, &self.base),
4515                        &self.ids,
4516                        &self.syms,
4517                        &self.labels,
4518                    )
4519                    .map_err(|e| GraphError::RuleInvalid { detail: e })?;
4520            }
4521            WalRecord::DeleteView { name } => {
4522                // Replay-over-snapshot idempotency: view already absent → skip.
4523                if !self.view_store.has_view(name) {
4524                    return Ok(());
4525                }
4526                self.view_store
4527                    .delete_view(
4528                        name,
4529                        Arc::make_mut(&mut self.props),
4530                        &self.ids,
4531                        &self.labels,
4532                        &self.syms,
4533                    )
4534                    .map_err(|_| GraphError::RuleNotFound { name: name.clone() })?;
4535            }
4536            WalRecord::EnableFulltext { label, field } => {
4537                // Replay-over-snapshot idempotency: already enabled → skip.
4538                if self.fulltext.is_enabled(label, field) {
4539                    return Ok(());
4540                }
4541                Arc::make_mut(&mut self.fulltext).enable(label, field);
4542                if self.fulltext_rebuild_follows {
4543                    // The open path rebuilds the whole index after replay, which
4544                    // clears every posting this scan would write. Doing it twice
4545                    // costs a full tokenise-and-stem pass over the corpus per
4546                    // enabled pair: measured at 456 ms against 3.8 ms for the
4547                    // same 30,000-entity store with no pair enabled.
4548                    return Ok(());
4549                }
4550                // Backfill: index all live nodes of this label that have the field.
4551                let n = self.ids.len() as u32;
4552                for id in 0..n {
4553                    let Some(&sym) = self.labels.get(id as usize) else {
4554                        continue;
4555                    };
4556                    if sym == u32::MAX {
4557                        continue; // tombstoned
4558                    }
4559                    let Some(lbl) = self.syms.resolve(sym) else {
4560                        continue;
4561                    };
4562                    if lbl != label {
4563                        continue;
4564                    }
4565                    if let Some(value) = build_props_view(&self.props, &self.base)
4566                        .get(id, field)
4567                        .map(|vr| vr.into_value())
4568                    {
4569                        Arc::make_mut(&mut self.fulltext).add_tokens(id, field, &value);
4570                    }
4571                }
4572            }
4573            WalRecord::DisableFulltext { label, field } => {
4574                // Replay-over-snapshot idempotency: already disabled → skip.
4575                if !self.fulltext.is_enabled(label, field) {
4576                    return Ok(());
4577                }
4578                // If another label still indexes this field, the postings column
4579                // is kept — but it must not contain node_ids from the now-disabled
4580                // label.  Remove them before calling disable() so the field_indexed
4581                // guard inside disable() sees the correct post-removal state.
4582                if self.fulltext.field_indexed_by_other(label, field) {
4583                    if let Some(label_sym) = self.syms.get(label) {
4584                        for (node_id, &lsym) in self.labels.iter().enumerate() {
4585                            if lsym == label_sym {
4586                                Arc::make_mut(&mut self.fulltext)
4587                                    .remove_node_field(node_id as u32, field);
4588                            }
4589                        }
4590                    }
4591                }
4592                Arc::make_mut(&mut self.fulltext).disable(label, field);
4593            }
4594            WalRecord::EnableIndex { label, field } => {
4595                // Replay-over-snapshot idempotency: already enabled → skip.
4596                if self.prop_index.is_enabled(label, field) {
4597                    return Ok(());
4598                }
4599                self.prop_index.enable(label, field);
4600                // Backfill: index all live nodes of this label that have the field.
4601                let n = self.ids.len() as u32;
4602                for id in 0..n {
4603                    let Some(&sym) = self.labels.get(id as usize) else {
4604                        continue;
4605                    };
4606                    if sym == u32::MAX {
4607                        continue; // tombstoned
4608                    }
4609                    let Some(lbl) = self.syms.resolve(sym) else {
4610                        continue;
4611                    };
4612                    if lbl != label {
4613                        continue;
4614                    }
4615                    if let Some(value) = build_props_view(&self.props, &self.base)
4616                        .get(id, field)
4617                        .map(|vr| vr.into_value())
4618                    {
4619                        self.prop_index.set(label, field, id, &value);
4620                    }
4621                }
4622            }
4623            WalRecord::DisableIndex { label, field } => {
4624                self.prop_index.disable(label, field);
4625            }
4626            // ── insert-count multiplicity (§5.13) ────────────────────────────
4627            //
4628            // Two shapes, told apart by `count`: the opt-in declaration, and an
4629            // absolute count for one triple. Absolute is what makes this
4630            // idempotent over a snapshot base — a pre-snapshot frame replayed
4631            // over a base that already folded it in lands on the same number
4632            // rather than adding to it, which is the failure a delta (or a count
4633            // derived from `InsertEdgeId` records) would have.
4634            WalRecord::SetEdgeCount {
4635                etype,
4636                src,
4637                dst,
4638                count,
4639            } => {
4640                if rec.is_multiplicity_decl() {
4641                    self.multiplicity = true;
4642                } else {
4643                    Arc::make_mut(&mut self.edge_props).set(
4644                        *etype,
4645                        *src,
4646                        *dst,
4647                        EDGE_COUNT_PROP,
4648                        Value::Int(*count as i64),
4649                    );
4650                }
4651            }
4652            // History markers carry no replay state — rules re-derive edges
4653            // deterministically on open/replay. Skip unconditionally.
4654            WalRecord::DerivedEdgeAdded { .. } | WalRecord::DerivedEdgeRetracted { .. } => {}
4655            // ── rename_node ──────────────────────────────────────────────────
4656            WalRecord::RenameNode { old_key, new_key } => {
4657                // Recovery-safe: if old_key is already gone (key was renamed
4658                // by a snapshot or a prior replay frame), skip cleanly.
4659                if self.ids.get(old_key).is_none() {
4660                    return Ok(());
4661                }
4662                // The rename only updates the key-table; the dense id, all
4663                // topo edges, props, labels, and rule state are id-indexed and
4664                // require no change.
4665                Arc::make_mut(&mut self.ids)
4666                    .rename(old_key, new_key)
4667                    .map_err(|e| GraphError::Corrupt {
4668                        detail: format!("wal replay RenameNode {old_key}→{new_key}: {e}"),
4669                    })?;
4670            }
4671        }
4672        Ok(())
4673    }
4674
4675    /// Intern `s` in `syms` and emit a WAL `Intern` record so `*Id` records
4676    /// replay on WAL-only `open_at` (no snapshot intern table). Apply is
4677    /// idempotent when the string is already bound. Always emit: after
4678    /// `snapshot()` the WAL is truncated and live intern is not on disk.
4679    fn intern_wal(&mut self, s: &str) -> (u32, WalRecord) {
4680        let id = if let Some(id) = self.syms.get(s) {
4681            id
4682        } else {
4683            Arc::make_mut(&mut self.syms).intern(s)
4684        };
4685        (
4686            id,
4687            WalRecord::Intern {
4688                id,
4689                text: s.to_string(),
4690            },
4691        )
4692    }
4693
4694    /// Rewrite user-facing records into dense-id records. On `Err`, no live
4695    /// state is left mutated: speculative interns made while building the
4696    /// output are rolled back, so a later successful mutation cannot log an
4697    /// `Intern` record whose id replay would never reproduce.
4698    fn rewrite_wal_dense(&mut self, recs: Vec<WalRecord>) -> Result<Vec<WalRecord>> {
4699        self.rewrite_wal_dense_planned(recs.into_iter().map(PlannedRec::Rec).collect())
4700    }
4701
4702    /// [`rewrite_wal_dense`](Self::rewrite_wal_dense) for a frame that still
4703    /// carries [`PlannedRec::DuplicateCount`] entries — the shape a batch
4704    /// produces, where a duplicate's count can only be named once this pass has
4705    /// assigned the frame's own ids.
4706    fn rewrite_wal_dense_planned(&mut self, recs: Vec<PlannedRec>) -> Result<Vec<WalRecord>> {
4707        let syms_checkpoint = self.syms.len();
4708        let result = self.rewrite_wal_dense_inner(recs);
4709        if result.is_err() {
4710            Arc::make_mut(&mut self.syms).truncate(syms_checkpoint);
4711        }
4712        result
4713    }
4714
4715    fn rewrite_wal_dense_inner(&mut self, recs: Vec<PlannedRec>) -> Result<Vec<WalRecord>> {
4716        let mut out = Vec::with_capacity(recs.len());
4717        // Node ids allocated by later apply(InsertNodeId) in this same batch.
4718        let mut pending: std::collections::HashMap<String, u32> = std::collections::HashMap::new();
4719        // Namespace of each node inserted earlier in this same frame, so a SET
4720        // on a node this frame created is measured against the namespace it was
4721        // created in rather than against the store, where it does not exist yet.
4722        let mut pending_ns: std::collections::HashMap<String, String> =
4723            std::collections::HashMap::new();
4724        let mut interned = std::collections::HashSet::<u32>::new();
4725        let mut next = u32::try_from(self.ids.len()).map_err(|_| GraphError::Corrupt {
4726            detail: "id space exhausted".into(),
4727        })?;
4728        // Insert counts this frame has already raised. `edge_insert_count`
4729        // reads committed state, which cannot see a count queued earlier in
4730        // this same frame, so N duplicates of one pair would otherwise all
4731        // compute `committed + 1` and the last would win.
4732        let mut pending_counts: HashMap<(u32, u32, u32), u64> = HashMap::new();
4733        let lookup = |ids: &IdMap,
4734                      pending: &std::collections::HashMap<String, u32>,
4735                      key: &str|
4736         -> Option<u32> { ids.get(key).or_else(|| pending.get(key).copied()) };
4737        for rec in recs {
4738            // A duplicate insert's count, resolved here and nowhere else.
4739            //
4740            // This is the only pass that knows the frame's own ids: a node
4741            // created earlier in the same frame has no dense id until the
4742            // `InsertNodeId` above allocates one, and an edge type first used in
4743            // this frame is not in `syms` until `intern_wal` puts it there.
4744            // Resolving the count in the batch's validate pass instead — where
4745            // it used to live — meant that a duplicate whose endpoints or type
4746            // were created in the same frame silently produced no count at all,
4747            // which is exactly the shape a mirror rebuild writes (defect #24).
4748            let rec = match rec {
4749                PlannedRec::Rec(rec) => rec,
4750                PlannedRec::DuplicateCount {
4751                    edge_type,
4752                    src_key,
4753                    dst_key,
4754                } => {
4755                    let (etype, intern) = self.intern_wal(&edge_type);
4756                    if interned.insert(etype) {
4757                        out.push(intern);
4758                    }
4759                    let src = lookup(&self.ids, &pending, &src_key).ok_or_else(|| {
4760                        GraphError::Corrupt {
4761                            detail: format!("dense WAL rewrite missing src {src_key}"),
4762                        }
4763                    })?;
4764                    let dst = lookup(&self.ids, &pending, &dst_key).ok_or_else(|| {
4765                        GraphError::Corrupt {
4766                            detail: format!("dense WAL rewrite missing dst {dst_key}"),
4767                        }
4768                    })?;
4769                    let count = pending_counts
4770                        .get(&(etype, src, dst))
4771                        .copied()
4772                        .unwrap_or_else(|| self.edge_insert_count(etype, src, dst))
4773                        .saturating_add(1);
4774                    pending_counts.insert((etype, src, dst), count);
4775                    out.push(WalRecord::SetEdgeCount {
4776                        etype,
4777                        src,
4778                        dst,
4779                        count,
4780                    });
4781                    continue;
4782                }
4783            };
4784            match rec {
4785                WalRecord::InsertNode { label, key, props } => {
4786                    // Namespace validation and normalisation, on the one seam
4787                    // every user-visible node insert passes through: insert_node,
4788                    // a batch, ingest, Cypher CREATE and MERGE all arrive here
4789                    // before the WAL append, and replay never does.
4790                    let (props, ns_name) = Self::normalise_insert_ns(&key, props)?;
4791                    pending_ns.insert(key.clone(), ns_name);
4792                    let (label_id, intern) = self.intern_wal(&label);
4793                    if interned.insert(label_id) {
4794                        out.push(intern);
4795                    }
4796                    let mut props_id = Vec::with_capacity(props.len());
4797                    for (field, value) in props {
4798                        let (field_id, intern) = self.intern_wal(&field);
4799                        if interned.insert(field_id) {
4800                            out.push(intern);
4801                        }
4802                        props_id.push((field_id, value));
4803                    }
4804                    if lookup(&self.ids, &pending, &key).is_none() {
4805                        pending.insert(key.clone(), next);
4806                        next = next.checked_add(1).ok_or_else(|| GraphError::Corrupt {
4807                            detail: "id space exhausted".into(),
4808                        })?;
4809                    }
4810                    out.push(WalRecord::InsertNodeId {
4811                        label: label_id,
4812                        key,
4813                        props: props_id,
4814                    });
4815                }
4816                WalRecord::SetProp { key, field, value } => {
4817                    // A namespace is set at insert and fixed after: the write is
4818                    // refused when it would move the node, and dropped when it
4819                    // names the namespace the node is already in. Checked here
4820                    // so set_prop, a batch, Cypher SET/MERGE and every upsert
4821                    // that merges props get the same answer.
4822                    if field == NS_PROP {
4823                        let Value::Str(ref to) = value else {
4824                            return Err(GraphError::RuleInvalid {
4825                                detail: format!(
4826                                    "node {key}: {NS_PROP} must be a string naming a namespace, \
4827                                     got {value:?}"
4828                                ),
4829                            });
4830                        };
4831                        let from = pending_ns
4832                            .get(&key)
4833                            .cloned()
4834                            .or_else(|| self.namespace_of(&key))
4835                            .unwrap_or_else(|| NS_DEFAULT.to_string());
4836                        let to = to.clone();
4837                        if to != from {
4838                            return Err(GraphError::NamespaceImmutable {
4839                                key: key.clone(),
4840                                from,
4841                                to,
4842                            });
4843                        }
4844                        continue;
4845                    }
4846                    let id =
4847                        lookup(&self.ids, &pending, &key).ok_or_else(|| GraphError::Corrupt {
4848                            detail: format!("dense WAL rewrite missing key {key}"),
4849                        })?;
4850                    let (field_id, intern) = self.intern_wal(&field);
4851                    if interned.insert(field_id) {
4852                        out.push(intern);
4853                    }
4854                    out.push(WalRecord::SetPropId {
4855                        id,
4856                        field: field_id,
4857                        value,
4858                    });
4859                }
4860                WalRecord::InsertEdge {
4861                    edge_type,
4862                    src_key,
4863                    dst_key,
4864                } => {
4865                    let (etype, intern) = self.intern_wal(&edge_type);
4866                    if interned.insert(etype) {
4867                        out.push(intern);
4868                    }
4869                    let src = lookup(&self.ids, &pending, &src_key).ok_or_else(|| {
4870                        GraphError::Corrupt {
4871                            detail: format!("dense WAL rewrite missing src {src_key}"),
4872                        }
4873                    })?;
4874                    let dst = lookup(&self.ids, &pending, &dst_key).ok_or_else(|| {
4875                        GraphError::Corrupt {
4876                            detail: format!("dense WAL rewrite missing dst {dst_key}"),
4877                        }
4878                    })?;
4879                    out.push(WalRecord::InsertEdgeId { etype, src, dst });
4880                }
4881                WalRecord::RenameNode {
4882                    ref old_key,
4883                    ref new_key,
4884                } => {
4885                    // Track the rename in `pending` so subsequent InsertEdge /
4886                    // SetProp records in this batch can resolve the new key.
4887                    let id = lookup(&self.ids, &pending, old_key).ok_or_else(|| {
4888                        GraphError::Corrupt {
4889                            detail: format!(
4890                                "dense WAL rewrite: RenameNode old key {old_key} not found"
4891                            ),
4892                        }
4893                    })?;
4894                    pending.remove(old_key.as_str());
4895                    pending.insert(new_key.clone(), id);
4896                    out.push(rec);
4897                }
4898                // # Symbol-order invariant (load-bearing)
4899                //
4900                // Write-time and replay-time symbol assignment must agree: every
4901                // symbol in a `Batch` frame has to receive the same dense id when
4902                // the frame's records are replayed in order as it received when
4903                // the frame was written.
4904                //
4905                // A rule's backfill interns its `edge_type` lazily
4906                // (`core_rules::engine`, every `g.syms.intern(&def.edge_type)`
4907                // site), and that backfill runs from `apply` — during the
4908                // `CreateRule` record itself, and again from any later
4909                // `InsertNodeId` in the same frame that makes the rule fire. At
4910                // write time the whole batch is rewritten before any of it is
4911                // applied, so a later `InsertEdge` in the same batch would win the
4912                // lower id for its edge type; on replay the rule's lazy intern
4913                // gets there first and steals it, and the `Intern` record fails at
4914                // the `wal intern assigned …` check in `apply`.
4915                //
4916                // Pre-interning the rule's `edge_type` here, and emitting its
4917                // `Intern` record ahead of the `CreateRule` record, makes both
4918                // orders identical. `weight_prop` needs no pre-intern:
4919                // `EdgeProps::set` keys props by `String`, never through the
4920                // interner. `via_edge` needs none either: via-hop rules resolve it
4921                // with `syms.get` and skip when it is absent.
4922                //
4923                // `RebuildRule` and `DeleteRule` need no such handling here:
4924                // `RebuildRule` has no `BatchOp` variant, so it never appears
4925                // inside a `Batch` today — it is only ever issued as its own
4926                // standalone commit (`rebuild_rule`, or the auto-rebuild path
4927                // that logs it as a second commit after the triggering op).
4928                // `DeleteRule` does have a `BatchOp` variant and can appear
4929                // inside a `Batch`, but it carries only a rule `name` — no
4930                // `edge_type` or other symbol that needs pre-interning — so
4931                // only `CreateRule` needs this arm.
4932                WalRecord::CreateRule { ref def_bytes } => {
4933                    let def = decode_rule_def(def_bytes).map_err(|e| GraphError::Corrupt {
4934                        detail: format!("CreateRule def_bytes deserialize failed: {e}"),
4935                    })?;
4936                    let (etype, intern) = self.intern_wal(&def.edge_type);
4937                    if interned.insert(etype) {
4938                        out.push(intern);
4939                    }
4940                    out.push(rec);
4941                }
4942                other => out.push(other),
4943            }
4944        }
4945        Ok(out)
4946    }
4947
4948    fn log_dense(&mut self, recs: Vec<WalRecord>) -> Result<()> {
4949        let recs = self.rewrite_wal_dense(recs)?;
4950        match recs.len() {
4951            0 => Ok(()),
4952            1 => self.log_then_apply(recs.into_iter().next().unwrap()),
4953            _ => self.log_then_apply(WalRecord::Batch(recs)),
4954        }
4955    }
4956
4957    /// Durable write, then notify the event sink. Replay (`apply` during
4958    /// `open`) never enters this function, so it is the replay-silent seam.
4959    /// Record that the commit occupying `frame_index` happened now, and append
4960    /// those 16 bytes to the sidecar.
4961    ///
4962    /// `frame_index` is the **global 0-based WAL frame index** of the commit's
4963    /// own record — the space every history surface addresses — taken from
4964    /// [`wal_frames_written`](GraphDb::wal_frames_written) before the append
4965    /// that puts the record there.
4966    ///
4967    /// **It is deliberately not derived from `commit_seq`.** A commit is not a
4968    /// frame: one whose rules fire appends a second frame for the derived-edge
4969    /// history marker, so `commit_seq - 1` falls one frame further behind per
4970    /// rule-firing commit and every date resolves to an ever-earlier graph.
4971    /// That was the shipped behaviour through v0.6.11 and it failed silently,
4972    /// because an older graph is a plausible answer rather than an error.
4973    ///
4974    /// Called from exactly one place — `log_then_apply_with`, immediately after
4975    /// `commit_seq` is incremented. Every write path in the engine funnels
4976    /// through that function, and replay deliberately does not: `apply_frames`
4977    /// re-applies commits that already happened, so stamping there would record
4978    /// replay time as commit time.
4979    ///
4980    /// **This is the engine's only wall-clock read.** Everything else uses
4981    /// `Instant`, which is monotonic and not a date.
4982    ///
4983    /// Failure is swallowed on purpose. The sidecar is not part of the WAL or
4984    /// the snapshot, so a failed append must not fail a commit that is already
4985    /// durable — it costs a date, not data. The map is marked poisoned so the
4986    /// gap is reported rather than resolved across.
4987    fn stamp_commit_time(&mut self, frame_index: u64) {
4988        if self.commit_times_poisoned {
4989            return;
4990        }
4991        let unix_ms = match self.commit_time_override {
4992            Some(ms) => ms,
4993            None => {
4994                let Ok(now) = std::time::SystemTime::now().duration_since(std::time::UNIX_EPOCH)
4995                else {
4996                    // A clock before 1970. Refuse to invent a timestamp.
4997                    self.commit_times_poisoned = true;
4998                    return;
4999                };
5000                now.as_millis() as i64
5001            }
5002        };
5003        let first = self.commit_times.is_empty();
5004        self.commit_times.push(frame_index, unix_ms);
5005
5006        let wrote = if first {
5007            self.fs.write_atomic(
5008                FileId::CommitTimes,
5009                &core_storage::commit_times::encode(&self.commit_times),
5010            )
5011        } else {
5012            self.fs.append(
5013                FileId::CommitTimes,
5014                &core_storage::commit_times::encode_entry(frame_index, unix_ms),
5015            )
5016        };
5017        if wrote.is_err() {
5018            self.commit_times_poisoned = true;
5019        }
5020    }
5021
5022    /// Read the time sidecar from disk into this handle.
5023    ///
5024    /// Absent is the normal case for any store written before v0.6.11 and is
5025    /// not an error; unreadable is recorded so date queries can say "damaged"
5026    /// rather than "none recorded".
5027    ///
5028    /// Called at open **and** by `refresh` when a peer's frames are absorbed.
5029    /// Both, because the map is a file another process appends to: a handle
5030    /// that raises its frame cursor to include a peer's commits while holding a
5031    /// stale map would answer dates from a prefix of the truth — and, if its
5032    /// own map were still empty, would rewrite the whole file with one entry
5033    /// and destroy the peer's.
5034    fn load_commit_times_from_fs(&mut self) {
5035        match self.fs.read(FileId::CommitTimes) {
5036            Ok(bytes) if bytes.is_empty() => {}
5037            Ok(bytes) => {
5038                // A map from an older format version is discarded, not
5039                // reported as damage and not reinterpreted. Its entries were
5040                // written correctly against a different meaning of the number
5041                // — see `COMMIT_TIMES_VERSION` — and reading them in this
5042                // build's space would resolve dates onto unrelated commits.
5043                // Leaving the map empty makes the store answer
5044                // `NoRecordedTime`, which is the truth: it records no times
5045                // this build can use, and the next commit starts a usable map.
5046                if core_storage::commit_times::superseded_version(&bytes).is_some() {
5047                    return;
5048                }
5049                match core_storage::commit_times::decode(&bytes) {
5050                    Ok(t) => self.commit_times = t,
5051                    Err(_) => self.commit_times_poisoned = true,
5052                }
5053            }
5054            Err(_) => {}
5055        }
5056    }
5057
5058    /// Rewrite the sidecar from memory. Used after truncation, which is the one
5059    /// operation that cannot be expressed as an append.
5060    fn rewrite_commit_times(&mut self) {
5061        if self.commit_times_poisoned {
5062            return;
5063        }
5064        if self
5065            .fs
5066            .write_atomic(
5067                FileId::CommitTimes,
5068                &core_storage::commit_times::encode(&self.commit_times),
5069            )
5070            .is_err()
5071        {
5072            self.commit_times_poisoned = true;
5073        }
5074    }
5075
5076    /// The greatest commit whose recorded time is at or before `unix_ms`.
5077    ///
5078    /// Errors name what they can answer instead of guessing a commit:
5079    /// `Corrupt` when the sidecar would not decode, `NoRecordedTime` when the
5080    /// store records none, `TimeBeforeFloor` when the instant predates the
5081    /// oldest entry, and `CommitOutOfRange` when the answer falls below the WAL
5082    /// horizon and so cannot be replayed.
5083    ///
5084    /// The answer is a **0-based frame index**, ready to hand to `edges_at` or
5085    /// `was_linked` without adjustment.
5086    pub fn resolve_instant(&self, unix_ms: i64) -> Result<u64> {
5087        if self.commit_times_poisoned {
5088            return Err(GraphError::Corrupt {
5089                detail: "commit_times.bin will not decode; date queries are \
5090                         unavailable on this store"
5091                    .into(),
5092            });
5093        }
5094        let at = self
5095            .commit_times
5096            .resolve_instant(unix_ms, self.wal_horizon_floor)?;
5097        // The map outlives the history it describes. A truncating snapshot folds
5098        // the WAL and discards it, so entries can name commits the engine can no
5099        // longer replay — the floor check above catches pruning, and this catches
5100        // discarding. Returning an index the caller's next call will reject is a
5101        // two-step error where one will do, and `resolve_date` is public: it
5102        // either hands back a usable index or refuses.
5103        let total = self.wal_total_commits()?;
5104        if at >= total {
5105            return Err(GraphError::CommitOutOfRange {
5106                commit: at,
5107                total,
5108                floor: self.wal_horizon_floor,
5109            });
5110        }
5111        Ok(at)
5112    }
5113
5114    /// Record subsequent commits as having happened at `unix_ms`, or pass
5115    /// `None` to go back to the system clock.
5116    ///
5117    /// For **backfilled history**: a mirror importing rows that already carry
5118    /// their own timestamps, or a replay of events that happened months ago.
5119    /// Without this every such commit is stamped "now", so a store holding a
5120    /// year of imported history answers every date question with
5121    /// `TimeBeforeFloor` — the data is there and no date reaches it.
5122    ///
5123    /// Sticky until changed or cleared, because a day of backfilled rows
5124    /// genuinely shares one instant.
5125    ///
5126    /// **Import in chronological order.** A supplied instant earlier than
5127    /// anything already recorded is refused with
5128    /// [`GraphError::CommitTimeNotMonotonic`], because resolution walks commit
5129    /// order: a later commit carrying an earlier time would silently widen every
5130    /// answer after it. Equal is allowed — that is what a shared day means. The
5131    /// live clock is never held to this, so an NTP step backwards still commits.
5132    ///
5133    /// Deliberately **not** exposed over HTTP or MCP: asserting when a commit
5134    /// happened rewrites the store's apparent history, which is not something a
5135    /// role token models. It is an embedding-caller's operation.
5136    pub fn record_commits_at(&mut self, unix_ms: Option<i64>) -> Result<()> {
5137        if self.read_only {
5138            return Err(GraphError::ReadOnly);
5139        }
5140        if let Some(ms) = unix_ms {
5141            if self.commit_times_poisoned {
5142                return Err(GraphError::Corrupt {
5143                    detail: "commit_times.bin will not decode; this store cannot \
5144                             record an asserted commit time"
5145                        .into(),
5146                });
5147            }
5148            if let Some(newest) = self.commit_times.max_ms() {
5149                if ms < newest {
5150                    return Err(GraphError::CommitTimeNotMonotonic {
5151                        supplied_ms: ms,
5152                        newest_ms: newest,
5153                    });
5154                }
5155            }
5156        }
5157        self.commit_time_override = unix_ms;
5158        Ok(())
5159    }
5160
5161    /// The instant subsequent commits are being recorded at, when one is set.
5162    pub fn commit_time_override(&self) -> Option<i64> {
5163        self.commit_time_override
5164    }
5165
5166    /// [`Self::edges_at`] addressed by an instant rather than a commit index.
5167    ///
5168    /// Resolves through [`Self::resolve_instant`] — the last commit at or
5169    /// before the instant — then answers exactly as the commit-indexed call
5170    /// does. A store that records no times refuses by name; it never guesses.
5171    pub fn edges_at_instant(&self, key: &str, unix_ms: i64) -> Result<Vec<EdgeAt>> {
5172        let commit = self.resolve_instant(unix_ms)?;
5173        self.edges_at(key, commit)
5174    }
5175
5176    /// [`Self::was_linked`] addressed by an instant rather than a commit index.
5177    pub fn was_linked_at_instant(
5178        &self,
5179        a: &str,
5180        b: &str,
5181        edge_type: &str,
5182        unix_ms: i64,
5183    ) -> Result<bool> {
5184        let commit = self.resolve_instant(unix_ms)?;
5185        self.was_linked(a, b, edge_type, commit)
5186    }
5187
5188    /// Parse an RFC 3339 instant (or a bare `YYYY-MM-DD`) and resolve it.
5189    ///
5190    /// The one place every caller-facing surface converts a date string, so
5191    /// HTTP, MCP, Python and the CLI cannot drift in what they accept.
5192    pub fn resolve_date(&self, s: &str) -> Result<u64> {
5193        // The **end** of what the string denotes. A bare date is a day, so it
5194        // resolves to the last commit at or before that day's end — resolving to
5195        // the midnight that starts it would exclude everything that happened on
5196        // the date the caller asked about.
5197        let ms = core_storage::commit_times::parse_rfc3339_end_ms(s).ok_or_else(|| {
5198            GraphError::QueryError {
5199                detail: format!(
5200                    "could not parse {s:?} as a date; expected RFC 3339 \
5201                     (2026-06-19, or 2026-06-19T12:00:00Z)"
5202                ),
5203            }
5204        })?;
5205        self.resolve_instant(ms)
5206    }
5207
5208    /// The recorded wall-clock time of `commit`, when the sidecar holds one.
5209    ///
5210    /// `commit` is a **0-based frame index** — the space `edges_at`,
5211    /// `was_linked` and the history events use, not the 1-based `commit_seq`.
5212    pub fn commit_time_ms(&self, commit: u64) -> Option<i64> {
5213        if self.commit_times_poisoned {
5214            return None;
5215        }
5216        self.commit_times.time_of(commit)
5217    }
5218
5219    fn log_then_apply(&mut self, rec: WalRecord) -> Result<()> {
5220        self.log_then_apply_with(rec, None, self.fsync)
5221    }
5222
5223    /// Whether this frame must fsync under `policy`.
5224    ///
5225    /// Batched contract: user-visible batches (>1 mutation) fsync; single
5226    /// mutations do not. The dense rewrite wraps a single mutation in a
5227    /// `Batch([Intern.., <one *Id record>])`, so `Intern` records are excluded
5228    /// from the count — removing that filter would make every single-op write
5229    /// fsync under Batched (or, if the threshold were raised instead, skip a
5230    /// needed fsync for real two-op batches).
5231    fn wal_needs_sync(policy: FsyncPolicy, rec: &WalRecord) -> bool {
5232        match policy {
5233            FsyncPolicy::Relaxed => false,
5234            FsyncPolicy::Strict => true,
5235            FsyncPolicy::Batched => match rec {
5236                // Intern + one mutation is the single-op rewrite, not a user batch.
5237                WalRecord::Batch(inner) => {
5238                    inner
5239                        .iter()
5240                        .filter(|r| !matches!(r, WalRecord::Intern { .. }))
5241                        .count()
5242                        > 1
5243                }
5244                _ => false,
5245            },
5246        }
5247    }
5248
5249    /// # Apply-infallibility invariant (load-bearing)
5250    ///
5251    /// The ordering is: WAL append → fsync → apply. If `apply` returned `Err`
5252    /// for a `Batch` frame after a successful WAL write, the WAL would contain
5253    /// the full frame while in-memory state would reflect only the ops before
5254    /// the failure. On reopen, WAL replay would then apply the entire batch —
5255    /// diverging permanently from what the pre-crash process had in memory.
5256    ///
5257    /// For `Batch` frames this situation cannot arise because:
5258    /// - All validation runs via `commit_logged_batch`/`MutPreview` **before**
5259    ///   the WAL write. `MutPreview` uses the same `&mut self` that apply will
5260    ///   use, with no concurrent mutation between validation exit and apply entry.
5261    /// - Every `apply` arm for a validated op is either infallible by construction
5262    ///   (`InsertNode`, `RemoveProp`, `DeleteEdge`, `DeleteNode`), has idempotency
5263    ///   guards that return `Ok(())` (`CreateRule`, `DeleteRule`), or is
5264    ///   guaranteed-present by validation (`InsertEdge`/`SetProp` key lookups).
5265    /// - `on_node_changed` and `on_node_removed` return `()` — never `Err`.
5266    ///
5267    /// A `debug_assert!` below fires in debug builds if `apply` ever returns
5268    /// `Err` for a `Batch` frame, making any future regression immediately visible
5269    /// in tests rather than silently diverging crash-recovery behaviour.
5270    fn log_then_apply_with(
5271        &mut self,
5272        rec: WalRecord,
5273        ingest: Option<(String, usize)>,
5274        policy: FsyncPolicy,
5275    ) -> Result<()> {
5276        // Read-only guard: as-of instances must never write the WAL.
5277        if self.read_only {
5278            return Err(GraphError::ReadOnly);
5279        }
5280        // Degraded guard: fsync failure left WAL truncated, or a refresh failed
5281        // partway; in-memory state is ahead of (or out of step with) the
5282        // on-disk WAL, so further mutations would deepen the divergence.
5283        // Reopen the database to recover.  Checked before the lock guard: this
5284        // is the more serious condition and the more useful error.
5285        if self.degraded {
5286            return Err(GraphError::Io(std::io::Error::other(
5287                "database degraded after group-commit fsync failure; reopen required",
5288            )));
5289        }
5290        // Cross-process guard: this write scope asked for the store's write
5291        // lock and did not get it. Writing anyway would append frames on top of
5292        // a WAL another process is extending, so refuse instead.
5293        if self.lock_denied {
5294            return Err(GraphError::Busy { holder: None });
5295        }
5296        // Ensure retained provenance bytes are decoded into the live mutable
5297        // fields before any mutation touches self.engine.provenance.  This is a
5298        // no-op if provenance was never stored (fresh store) or has already been
5299        // consumed (subsequent mutations).  WAL replay calls apply() directly
5300        // and is covered by consume_retained_state_eager before replay.
5301        self.ensure_v8_base_sections_loaded();
5302        self.engine.ensure_provenance_loaded_mut();
5303        // Invariant (I-1): no stale deltas may enter from a previous apply.
5304        // If any engine method ever accumulates deltas before erroring, they would
5305        // contaminate the *next* commit's event stream. This assert fires in debug
5306        // builds, making any future regression visible at the earliest point.
5307        debug_assert_eq!(
5308            self.engine.pending_delta_count(),
5309            0,
5310            "stale engine deltas at log_then_apply_with entry — \
5311             a previous apply arm may have accumulated deltas before erroring; \
5312             the caller must drain_deltas() on any error path before returning"
5313        );
5314        let frame = encode_record(&rec);
5315        self.fs.append(FileId::Wal, &frame)?;
5316        // The cursor advances by exactly the bytes appended: these frames are
5317        // ours and already applied, so a later refresh must not replay them.
5318        self.wal_consumed += frame.len() as u64;
5319        self.wal_frames_written += 1;
5320        if Self::wal_needs_sync(policy, &rec) {
5321            self.fs.sync(FileId::Wal)?;
5322        }
5323        // Marker writing always needs the engine deltas, but the engine only
5324        // accumulates them when emit_deltas is true (normally gated on subscribers
5325        // or views being present).  Enable emission for this apply if it is
5326        // currently off, then restore the original state unconditionally via an
5327        // RAII guard — this prevents a panic in apply() from leaking the flag.
5328        // The same guard resets the engine's transient chaining state. A panic
5329        // unwinding out of a rule hook would otherwise leave `chain_depth`
5330        // non-zero, which makes every later `begin_chain` decide chaining is
5331        // already running and silently switch it off for good.
5332        struct RestoreEmitDeltas(*mut RuleEngine, bool);
5333        impl Drop for RestoreEmitDeltas {
5334            fn drop(&mut self) {
5335                // SAFETY: pointer into self (GraphDb); guard is dropped within
5336                // this frame before log_then_apply_with returns.
5337                unsafe {
5338                    (*self.0).set_emit_deltas(self.1);
5339                    (*self.0).reset_chain_state();
5340                }
5341            }
5342        }
5343        let original_emit = self.engine.emit_deltas();
5344        if !original_emit {
5345            self.engine.set_emit_deltas(true);
5346        }
5347        // SAFETY: raw pointer into self; guard dropped within this frame.
5348        let _emit_guard = RestoreEmitDeltas(&mut self.engine as *mut _, original_emit);
5349
5350        let apply_result = self.apply(&rec);
5351        // For Batch frames, post-validation apply must be infallible (see above).
5352        // A debug_assert here catches any future change that makes apply fallible
5353        // before the caller notices via silent WAL/memory divergence.
5354        if matches!(&rec, WalRecord::Batch(_)) {
5355            debug_assert!(
5356                apply_result.is_ok(),
5357                "Batch apply returned Err after successful WAL write — \
5358                 the validate-then-apply invariant has been violated; \
5359                 see log_then_apply_with invariant doc"
5360            );
5361        }
5362        if apply_result.is_err() {
5363            // Discard any partial deltas accumulated by the failed apply.
5364            // They must not ride the next commit's event stream (I-1).
5365            // _emit_guard restores emit_deltas on drop automatically.
5366            let _ = self.engine.drain_deltas();
5367            let _ = self.engine.take_rebuild_needed();
5368            apply_result?;
5369        }
5370        self.commit_seq += 1;
5371        let seq = self.commit_seq;
5372        // Update per-node last-change map for the committed record.
5373        // Must happen after commit_seq is incremented so the seq is correct.
5374        self.update_last_change_from_rec(&rec, seq);
5375        // Drain engine deltas and distribute to subscribers before the existing
5376        // MutationEvent sink fires — both happen post-fsync, post-apply.
5377        // _emit_guard restores emit_deltas after this line when it drops.
5378        let engine_deltas = self.engine.drain_deltas();
5379
5380        // Append history-marker WAL records for any derived-edge changes so
5381        // that `edge_history` and `was_linked` can surface rule-attributed
5382        // events. Markers are STATE NO-OPS during replay; they are written
5383        // without an additional fsync (the triggering commit's sync already
5384        // happened; the next commit's sync covers these lazily).
5385        if !engine_deltas.is_empty() {
5386            let markers: Vec<WalRecord> = engine_deltas
5387                .iter()
5388                .map(|d| {
5389                    if d.fired {
5390                        WalRecord::DerivedEdgeAdded {
5391                            rule: d.rule.clone(),
5392                            edge_type: d.edge_type.clone(),
5393                            src_key: d.src_key.clone(),
5394                            dst_key: d.dst_key.clone(),
5395                        }
5396                    } else {
5397                        WalRecord::DerivedEdgeRetracted {
5398                            rule: d.rule.clone(),
5399                            edge_type: d.edge_type.clone(),
5400                            src_key: d.src_key.clone(),
5401                            dst_key: d.dst_key.clone(),
5402                        }
5403                    }
5404                })
5405                .collect();
5406            let marker_frame = if markers.len() == 1 {
5407                markers.into_iter().next().unwrap()
5408            } else {
5409                WalRecord::Batch(markers)
5410            };
5411            // Ignore append errors: markers are best-effort history
5412            // annotations. Losing them does not affect state correctness.
5413            // The cursor only advances when the bytes actually landed.
5414            let marker_bytes = encode_record(&marker_frame);
5415            if self.fs.append(FileId::Wal, &marker_bytes).is_ok() {
5416                self.wal_consumed += marker_bytes.len() as u64;
5417                // A marker is a state no-op during replay but it is not an
5418                // index no-op: it occupies a frame that every history surface
5419                // counts. Missing this increment is the whole of defect 1.
5420                self.wal_frames_written += 1;
5421            }
5422        }
5423
5424        // Stamp the commit against the **last** frame it wrote.
5425        //
5426        // `edges_at` and `was_linked` reconstruct derived edges by reading the
5427        // history markers out of the WAL, not by re-running rules over a
5428        // prefix. A commit's complete state — its record *and* the edges its
5429        // rules derived — is therefore only reached at its marker frame, so
5430        // that is the frame a date naming this commit must resolve to. Stamping
5431        // the record's own frame would answer every date with the graph as it
5432        // was one derivation short.
5433        //
5434        // This runs after the marker append for that reason, and it is still
5435        // the single stamping site: one commit, one entry.
5436        self.stamp_commit_time(self.wal_frames_written - 1);
5437
5438        // Record MVCC CommitDelta for the epoch reader.  The WAL record is
5439        // stored as-is (including any nested Batch / Intern records); the
5440        // ReaderSnapshot's apply_one function handles all variants.
5441        {
5442            let derived_inserts = engine_deltas
5443                .iter()
5444                .filter(|d| d.fired)
5445                .map(|d| (d.etype_sym, d.src_id, d.dst_id))
5446                .collect();
5447            let derived_deletes = engine_deltas
5448                .iter()
5449                .filter(|d| !d.fired)
5450                .map(|d| (d.etype_sym, d.src_id, d.dst_id))
5451                .collect();
5452            let delta = Arc::new(crate::reader::CommitDelta {
5453                records: vec![rec.clone()],
5454                derived_inserts,
5455                derived_deletes,
5456            });
5457            self.delta_tail.push(delta);
5458            self.commits_since_fold += 1;
5459            if self.commits_since_fold >= crate::reader::FOLD_EVERY_K {
5460                self.fold_now();
5461            }
5462        }
5463
5464        if self.defer_events {
5465            // Group-commit drain thread: hold events until after the group
5466            // fsync so subscribers only observe durable data (R2).
5467            self.deferred_events.push(DeferredEvent {
5468                rec: rec.clone(),
5469                engine_deltas,
5470                seq,
5471                ingest,
5472            });
5473        } else {
5474            self.distribute_events(&rec, &engine_deltas, seq);
5475            self.emit_committed(&rec, ingest);
5476        }
5477        // Drift is only known after apply, so auto-rebuild cannot join the
5478        // triggering op's WAL frame. Issue RebuildRule as a second commit.
5479        // Skip when `rec` is itself RebuildRule: rebuild resets drift, so a
5480        // retrigger loop is impossible if the fit succeeded, but we still
5481        // drain the flag so a leftover cannot re-enter.
5482        // One slice of any outstanding vector-index build rides here too, so a
5483        // store that is being written to finishes its build without anyone
5484        // calling `pump_index_build`. A rule that becomes whole joins the same
5485        // RebuildRule loop below.
5486        let mut rebuilds = self.engine.take_rebuild_needed();
5487        if !matches!(&rec, WalRecord::RebuildRule { .. }) {
5488            // Not after `CreateRule`: that record's own apply already did the
5489            // rule's first slice, and pumping again here would make one
5490            // `create_rule` call do two slices' work under one lock.
5491            // Nothing pending is the overwhelmingly common case and must cost
5492            // a map lookup, not an engine swap: a store being written to has
5493            // long since populated its indexes, so the `pump_index_build`
5494            // entry point owns the not-yet-populated case on its own.
5495            if !matches!(&rec, WalRecord::CreateRule { .. })
5496                && !self.engine.builds_in_progress().is_empty()
5497            {
5498                rebuilds.extend(self.pump_one_slice().into_iter().map(|b| b.rule));
5499            }
5500            let mut failed = Vec::new();
5501            for name in rebuilds {
5502                if self.engine.rules().any(|r| r.name == name) {
5503                    // User op is already durable. A failed second commit must
5504                    // not surface as the caller's error.
5505                    if let Err(e) =
5506                        self.log_then_apply(WalRecord::RebuildRule { name: name.clone() })
5507                    {
5508                        eprintln!(
5509                            "auto-rebuild of rule {name:?} failed after durable user commit: {e}"
5510                        );
5511                        failed.push(name);
5512                    }
5513                }
5514            }
5515            for name in failed {
5516                self.engine.queue_rebuild_needed(name);
5517            }
5518        }
5519        Ok(())
5520    }
5521
5522    /// Install a post-commit hook. Replaces any previous sink.
5523    ///
5524    /// The sink runs inside `log_then_apply` after a successful
5525    /// durable commit, while the caller still holds `&mut self`. When this
5526    /// database is behind a [`crate::SharedDb`], that means the **write
5527    /// guard is held**. The sink must never call `read` / `write` (or any
5528    /// other method) on the same `SharedDb` — the `RwLock` is not
5529    /// re-entrant and doing so deadlocks. The sink is `Send + Sync`;
5530    /// `std::sync::mpsc::Sender` is not `Sync` and will not type-check.
5531    /// Intended examples: `std::sync::mpsc::SyncSender`,
5532    /// `tokio::sync::mpsc::Sender`, `tokio::sync::broadcast::Sender`
5533    /// (non-blocking `send`), or `Arc<Mutex<Vec<MutationEvent>>>`.
5534    pub fn set_event_sink(&mut self, sink: Box<dyn Fn(MutationEvent) + Send + Sync>) {
5535        self.event_sink = Some(sink);
5536    }
5537
5538    /// Whether a post-commit event sink is currently installed.
5539    pub fn has_event_sink(&self) -> bool {
5540        self.event_sink.is_some()
5541    }
5542
5543    /// Set WAL fsync cadence. Default [`FsyncPolicy::Strict`].
5544    pub fn set_fsync_policy(&mut self, p: FsyncPolicy) {
5545        self.fsync = p;
5546    }
5547
5548    /// Return the current WAL fsync cadence.
5549    pub fn fsync_policy(&self) -> FsyncPolicy {
5550        self.fsync
5551    }
5552
5553    // ── Group-commit event deferral ───────────────────────────────────────────
5554
5555    /// Enable or disable deferred event mode.
5556    ///
5557    /// When `true`, event notifications (subscription `DbEvent`s and legacy
5558    /// `MutationEvent` sink calls) are buffered rather than fired immediately.
5559    /// Call [`flush_deferred_events`] after the group fsync to deliver them,
5560    /// or [`discard_deferred_events`] if the fsync failed and the group must
5561    /// be treated as lost.
5562    pub fn set_deferred_events_mode(&mut self, defer: bool) {
5563        self.defer_events = defer;
5564    }
5565
5566    /// Fire all buffered events accumulated since [`set_deferred_events_mode`]
5567    /// was set to true.  Clears the buffer.
5568    ///
5569    /// Called by the drain thread AFTER a successful group fsync, so
5570    /// subscribers observe only data that is durably on disk.
5571    pub fn flush_deferred_events(&mut self) {
5572        let events = std::mem::take(&mut self.deferred_events);
5573        for de in events {
5574            self.distribute_events(&de.rec, &de.engine_deltas, de.seq);
5575            self.emit_committed(&de.rec, de.ingest);
5576        }
5577    }
5578
5579    /// Discard all buffered events without firing them.
5580    ///
5581    /// Called by the drain thread when a group fsync fails: the WAL has been
5582    /// truncated back to the pre-group offset, so the committed-but-unsynced
5583    /// ops must not be observable to subscribers.
5584    pub fn discard_deferred_events(&mut self) {
5585        self.deferred_events.clear();
5586    }
5587
5588    // ── Degraded state ────────────────────────────────────────────────────────
5589
5590    /// Mark this database as degraded.
5591    ///
5592    /// Called by the group-commit drain thread after a group fsync failure and
5593    /// WAL truncation: the in-memory state is now ahead of the on-disk WAL, so
5594    /// further mutations would deepen the divergence.  All subsequent calls to
5595    /// [`log_then_apply_with`] return `Err` until the database is reopened.
5596    pub fn set_degraded(&mut self) {
5597        self.degraded = true;
5598    }
5599
5600    fn emit(&self, ev: MutationEvent) {
5601        if let Some(sink) = &self.event_sink {
5602            sink(ev);
5603        }
5604    }
5605
5606    fn emit_committed(&self, rec: &WalRecord, ingest: Option<(String, usize)>) {
5607        match rec {
5608            WalRecord::Batch(inner) => {
5609                for r in inner {
5610                    if let Some(ev) = event_from_record(r, &self.syms, &self.ids) {
5611                        self.emit(ev);
5612                    }
5613                }
5614                match ingest {
5615                    Some((label, inserted)) => {
5616                        self.emit(MutationEvent::Ingested { label, inserted })
5617                    }
5618                    None => {
5619                        let ops = inner
5620                            .iter()
5621                            .filter(|r| !matches!(r, WalRecord::Intern { .. }))
5622                            .count();
5623                        if ops > 1 {
5624                            self.emit(MutationEvent::BatchApplied { ops });
5625                        }
5626                    }
5627                }
5628            }
5629            other => {
5630                if let Some(ev) = event_from_record(other, &self.syms, &self.ids) {
5631                    self.emit(ev);
5632                }
5633            }
5634        }
5635    }
5636
5637    // -----------------------------------------------------------------------
5638    // Subscription API
5639    // -----------------------------------------------------------------------
5640
5641    /// Distribute post-commit events to all live subscribers.
5642    ///
5643    /// Build a row-key → row-data map from a [`ResultSet`].
5644    ///
5645    /// Each row is serialized to JSON to form its key; a debug fallback is used
5646    /// if serialization fails. Used by both the initial-seed path in
5647    /// [`Self::subscribe_query`] and the per-commit diff path in
5648    /// [`Self::distribute_events`] to keep the two in sync.
5649    fn result_to_row_map(
5650        result: &core_query::ResultSet,
5651    ) -> std::collections::HashMap<String, Vec<Option<Value>>> {
5652        (0..result.len())
5653            .map(|i| {
5654                let row = result.row(i).to_vec();
5655                let key = serde_json::to_string(&row).unwrap_or_else(|_| format!("{row:?}"));
5656                (key, row)
5657            })
5658            .collect()
5659    }
5660
5661    /// Collect the set of label syms touched by a WAL record.
5662    ///
5663    /// Returns `Some(set)` when every record in this commit can be attributed to
5664    /// a known label sym. Returns `None` when the commit must not be skipped:
5665    /// edge records, unresolvable key→label lookups, or any record type not in
5666    /// the explicit handled set.
5667    ///
5668    /// Handled record types and their actions:
5669    /// - `InsertNode`   → look up label in interner (fails → None)
5670    /// - `InsertNodeId` → label sym is carried directly
5671    /// - `SetProp`      → resolve key→id→label (fails → None)
5672    /// - `DeleteNode`   → resolve key→id→label (fails → None)
5673    /// - `Batch`        → recurse into every inner record
5674    /// - `InsertEdge`, `DeleteEdge`, `InsertEdgeId` → always None (edge records)
5675    /// - everything else → None (conservative)
5676    fn commit_touched_labels(
5677        rec: &WalRecord,
5678        syms: &Interner,
5679        ids: &IdMap,
5680        labels: &[u32],
5681    ) -> Option<BTreeSet<u32>> {
5682        let mut out = BTreeSet::new();
5683        if Self::collect_touched_labels(rec, syms, ids, labels, &mut out) {
5684            Some(out)
5685        } else {
5686            None
5687        }
5688    }
5689
5690    fn collect_touched_labels(
5691        rec: &WalRecord,
5692        syms: &Interner,
5693        ids: &IdMap,
5694        labels: &[u32],
5695        out: &mut BTreeSet<u32>,
5696    ) -> bool {
5697        match rec {
5698            // String-key insert: the dense rewrite converts this to
5699            // [Intern, InsertNodeId], so this arm fires only for legacy WAL
5700            // records written before the dense path was added.
5701            WalRecord::InsertNode { label, .. } => {
5702                if let Some(sym) = syms.get(label) {
5703                    out.insert(sym);
5704                    true
5705                } else {
5706                    false
5707                }
5708            }
5709            // Dense-id insert (produced by rewrite_wal_dense for every
5710            // insert_node call in the current codebase).
5711            WalRecord::InsertNodeId { label, .. } => {
5712                out.insert(*label);
5713                true
5714            }
5715            // String-key prop set: dense path converts to [Intern, SetPropId].
5716            WalRecord::SetProp { key, .. } => {
5717                if let Some(sym) = Self::resolve_key_label_sym(key, ids, labels) {
5718                    out.insert(sym);
5719                    true
5720                } else {
5721                    false
5722                }
5723            }
5724            // Dense-id prop set (produced by rewrite_wal_dense for set_prop).
5725            WalRecord::SetPropId { id, .. } => {
5726                if let Some(sym) = labels.get(*id as usize).copied().filter(|&s| s != u32::MAX) {
5727                    out.insert(sym);
5728                    true
5729                } else {
5730                    false
5731                }
5732            }
5733            WalRecord::DeleteNode { key } => {
5734                if let Some(sym) = Self::resolve_key_label_sym(key, ids, labels) {
5735                    out.insert(sym);
5736                    true
5737                } else {
5738                    false
5739                }
5740            }
5741            WalRecord::Batch(inner) => inner
5742                .iter()
5743                .all(|r| Self::collect_touched_labels(r, syms, ids, labels, out)),
5744            // Intern is a pure metadata record — it does not touch any node's
5745            // label and is safe to skip for the label-skip predicate.
5746            WalRecord::Intern { .. } => true,
5747            // Edge records: always re-execute (edges can change join results).
5748            WalRecord::InsertEdge { .. }
5749            | WalRecord::DeleteEdge { .. }
5750            | WalRecord::InsertEdgeId { .. } => false,
5751            _ => false,
5752        }
5753    }
5754
5755    /// Resolve a node key to its label sym via the dense id table.
5756    /// Returns `None` if the key is unknown or the label is a tombstone sentinel.
5757    fn resolve_key_label_sym(key: &str, ids: &IdMap, labels: &[u32]) -> Option<u32> {
5758        let id = ids.get(key)?;
5759        let sym = labels.get(id as usize).copied()?;
5760        (sym != u32::MAX).then_some(sym)
5761    }
5762
5763    /// Distribute post-commit events to all live subscribers.
5764    ///
5765    /// Called from `log_then_apply_with` after apply + fsync, before the
5766    /// legacy MutationEvent sink. Prunes dead `Weak` entries in-place.
5767    ///
5768    /// Query subscriptions (subscribe_query) re-execute their plan on every
5769    /// call and diff the result against the previous run. Zero overhead when
5770    /// no query subscriptions are active.
5771    fn distribute_events(&mut self, rec: &WalRecord, engine_deltas: &[EngineEdgeDelta], seq: u64) {
5772        if self.subscriptions.is_empty() && self.query_subscriptions.is_empty() {
5773            return;
5774        }
5775
5776        if !self.subscriptions.is_empty() {
5777            // Build write events from the WAL record.
5778            let write_events: Vec<DbEvent> =
5779                Self::write_events_from_record(rec, seq, &self.syms, &self.ids);
5780
5781            // Build edge events from engine deltas.  Weight is looked up from
5782            // edge_props at distribution time (after apply), so it's always fresh.
5783            let edge_events: Vec<DbEvent> = engine_deltas
5784                .iter()
5785                .map(|d| {
5786                    if d.fired {
5787                        // The score lives under the rule's declared weight_prop,
5788                        // which is not always the literal "weight".
5789                        let prop = self
5790                            .engine
5791                            .rules()
5792                            .find(|r| r.name == d.rule)
5793                            .and_then(|r| r.weight_prop.as_deref());
5794                        let weight = prop.and_then(|p| {
5795                            self.edge_props
5796                                .get(d.etype_sym, d.src_id, d.dst_id, p)
5797                                .and_then(|v| {
5798                                    if let core_storage::Value::Float(f) = v {
5799                                        Some(*f)
5800                                    } else {
5801                                        None
5802                                    }
5803                                })
5804                        });
5805                        DbEvent::EdgeFired {
5806                            rule: d.rule.clone(),
5807                            src_key: d.src_key.clone(),
5808                            dst_key: d.dst_key.clone(),
5809                            edge_type: d.edge_type.clone(),
5810                            weight,
5811                            commit_seq: seq,
5812                        }
5813                    } else {
5814                        DbEvent::EdgeRetracted {
5815                            rule: d.rule.clone(),
5816                            src_key: d.src_key.clone(),
5817                            dst_key: d.dst_key.clone(),
5818                            edge_type: d.edge_type.clone(),
5819                            commit_seq: seq,
5820                        }
5821                    }
5822                })
5823                .collect();
5824
5825            // Prune dead entries; push matching events to live ones.
5826            self.subscriptions.retain(|entry| {
5827                let Some(inner) = entry.inner.upgrade() else {
5828                    return false;
5829                };
5830                for ev in &write_events {
5831                    if event_matches(ev, &entry.filter) {
5832                        inner.push(ev.clone());
5833                    }
5834                }
5835                for ev in &edge_events {
5836                    if event_matches(ev, &entry.filter) {
5837                        inner.push(ev.clone());
5838                    }
5839                }
5840                true
5841            });
5842
5843            // Turn off delta accumulation if all subscribers dropped and no views remain.
5844            if self.subscriptions.is_empty() && self.view_store.is_empty() {
5845                self.engine.set_emit_deltas(false);
5846            }
5847        }
5848
5849        // Query subscriptions: full re-run per commit, then diff rows.
5850        // IMPORTANT: full re-execution on every commit — use LIMIT to bound cost.
5851        // Differential evaluation is roadmap / Phase 5.
5852        if !self.query_subscriptions.is_empty() {
5853            // Take the list out so we can call self.view() without borrow conflict.
5854            let mut query_subs = std::mem::take(&mut self.query_subscriptions);
5855            let empty_params = BTreeMap::new();
5856            query_subs.retain_mut(|entry| {
5857                let Some(inner) = entry.inner.upgrade() else {
5858                    return false; // subscriber dropped — prune
5859                };
5860                // Label-skip: if the plan has a known scan label and this commit
5861                // can be proven to touch only different labels (and no rule-derived
5862                // edge deltas fired), the result set cannot have changed — skip.
5863                if let Some(scan_sym) = entry.scan_label {
5864                    if engine_deltas.is_empty() {
5865                        let touched =
5866                            Self::commit_touched_labels(rec, &self.syms, &self.ids, &self.labels);
5867                        if touched.map(|t| !t.contains(&scan_sym)).unwrap_or(false) {
5868                            return true; // safe to skip — result set unchanged
5869                        }
5870                    }
5871                }
5872                QUERY_SUB_EXECS_TL.with(|c| c.set(c.get() + 1));
5873                let result = match execute(&self.view(), &entry.ops, &Params(&empty_params)) {
5874                    Ok(r) => r,
5875                    Err(e) => {
5876                        // Keep the subscription alive; skip the diff for this commit.
5877                        // Re-run errors are transient (e.g., planner change) and
5878                        // self-heal when the next commit succeeds.
5879                        eprintln!("[mushroomdb] subscribe_query re-run failed: {e}");
5880                        return true;
5881                    }
5882                };
5883                // Build new row map: serialized-key → row data.
5884                let new_row_map = Self::result_to_row_map(&result);
5885                // Removed rows: in prev but not in new.
5886                for (key, row) in &entry.prev_row_map {
5887                    if !new_row_map.contains_key(key) {
5888                        inner.push(DbEvent::QueryRowRemoved {
5889                            columns: entry.columns.clone(),
5890                            row: row.clone(),
5891                        });
5892                    }
5893                }
5894                // Added rows: in new but not in prev.
5895                for (key, row) in &new_row_map {
5896                    if !entry.prev_row_map.contains_key(key) {
5897                        inner.push(DbEvent::QueryRowAdded {
5898                            columns: entry.columns.clone(),
5899                            row: row.clone(),
5900                        });
5901                    }
5902                }
5903                entry.prev_row_map = new_row_map;
5904                true
5905            });
5906            self.query_subscriptions = query_subs;
5907        }
5908    }
5909
5910    /// Returns `true` if any live subscriber or view definition requires delta
5911    /// accumulation. Used to set `engine.emit_deltas` on subscribe/view DDL.
5912    fn needs_emit_deltas(&self) -> bool {
5913        !self.view_store.is_empty()
5914            || self
5915                .subscriptions
5916                .iter()
5917                .any(|e| e.inner.upgrade().is_some())
5918    }
5919
5920    /// Convert a WAL record into `DbEvent` write events with the given seq.
5921    fn write_events_from_record(
5922        rec: &WalRecord,
5923        seq: u64,
5924        intern: &Interner,
5925        ids: &IdMap,
5926    ) -> Vec<DbEvent> {
5927        match rec {
5928            WalRecord::InsertNode { label, key, .. } => vec![DbEvent::NodeInserted {
5929                label: label.clone(),
5930                key: key.clone(),
5931                commit_seq: seq,
5932            }],
5933            // *Id arms run after a successful apply, so resolution can only
5934            // fail on a programming error. Skip the event rather than emit a
5935            // fabricated "" that clients can't tell from a real empty value
5936            // (mirrors event_from_record returning None).
5937            WalRecord::InsertNodeId { label, key, .. } => intern
5938                .resolve(*label)
5939                .map(|label| DbEvent::NodeInserted {
5940                    label: label.to_string(),
5941                    key: key.clone(),
5942                    commit_seq: seq,
5943                })
5944                .into_iter()
5945                .collect(),
5946            WalRecord::SetProp { key, field, .. } => vec![DbEvent::PropSet {
5947                key: key.clone(),
5948                field: field.clone(),
5949                commit_seq: seq,
5950            }],
5951            WalRecord::SetPropId { id, field, .. } => ids
5952                .key_of(*id)
5953                .zip(intern.resolve(*field))
5954                .map(|(key, field)| DbEvent::PropSet {
5955                    key: key.to_string(),
5956                    field: field.to_string(),
5957                    commit_seq: seq,
5958                })
5959                .into_iter()
5960                .collect(),
5961            WalRecord::RemoveProp { key, field } => vec![DbEvent::PropRemoved {
5962                key: key.clone(),
5963                field: field.clone(),
5964                commit_seq: seq,
5965            }],
5966            WalRecord::InsertEdge {
5967                edge_type,
5968                src_key,
5969                dst_key,
5970            } => vec![DbEvent::EdgeInserted {
5971                edge_type: edge_type.clone(),
5972                src: src_key.clone(),
5973                dst: dst_key.clone(),
5974                commit_seq: seq,
5975            }],
5976            WalRecord::InsertEdgeId { etype, src, dst } => (|| {
5977                Some(DbEvent::EdgeInserted {
5978                    edge_type: intern.resolve(*etype)?.to_string(),
5979                    src: ids.key_of(*src)?.to_string(),
5980                    dst: ids.key_of(*dst)?.to_string(),
5981                    commit_seq: seq,
5982                })
5983            })()
5984            .into_iter()
5985            .collect(),
5986            WalRecord::DeleteEdge {
5987                edge_type,
5988                src_key,
5989                dst_key,
5990            } => vec![DbEvent::EdgeDeleted {
5991                edge_type: edge_type.clone(),
5992                src: src_key.clone(),
5993                dst: dst_key.clone(),
5994                commit_seq: seq,
5995            }],
5996            WalRecord::DeleteNode { key } => vec![DbEvent::NodeDeleted {
5997                key: key.clone(),
5998                commit_seq: seq,
5999            }],
6000            WalRecord::Batch(inner) => inner
6001                .iter()
6002                .flat_map(|r| Self::write_events_from_record(r, seq, intern, ids))
6003                .collect(),
6004            WalRecord::CreateRule { .. }
6005            | WalRecord::DeleteRule { .. }
6006            | WalRecord::RebuildRule { .. }
6007            | WalRecord::CreateView { .. }
6008            | WalRecord::DeleteView { .. }
6009            | WalRecord::EnableFulltext { .. }
6010            | WalRecord::DisableFulltext { .. }
6011            | WalRecord::EnableIndex { .. }
6012            | WalRecord::DisableIndex { .. }
6013            | WalRecord::Intern { .. }
6014            // History markers produce no DbEvent — the engine delta already
6015            // fired the EdgeFired/EdgeRetracted subscription events.
6016            | WalRecord::DerivedEdgeAdded { .. }
6017            | WalRecord::DerivedEdgeRetracted { .. }
6018            // A count is not an edge event: the pair it counts already fired one
6019            // when it was first inserted.
6020            | WalRecord::SetEdgeCount { .. }
6021            | WalRecord::RenameNode { .. } => vec![],
6022        }
6023    }
6024
6025    /// Subscribe to edge-fire and edge-retract events for one named rule.
6026    ///
6027    /// Returns `Err(GraphError::RuleNotFound)` if `rule_name` is not
6028    /// currently registered. Dropping the returned [`Subscription`] handle
6029    /// unregisters the subscriber — no further events are queued, no
6030    /// resources leak.
6031    pub fn subscribe_rule(&mut self, rule_name: &str) -> core_storage::Result<Subscription> {
6032        if self.read_only {
6033            return Err(core_storage::GraphError::ReadOnly);
6034        }
6035        if !self.engine.rules().any(|r| r.name == rule_name) {
6036            return Err(core_storage::GraphError::RuleNotFound {
6037                name: rule_name.to_string(),
6038            });
6039        }
6040        let inner = SubInner::new(self.sub_capacity());
6041        self.subscriptions.push(SubEntry {
6042            filter: SubFilter::Rule(rule_name.to_string()),
6043            inner: std::sync::Arc::downgrade(&inner),
6044        });
6045        self.engine.set_emit_deltas(true);
6046        Ok(Subscription(inner))
6047    }
6048
6049    /// Subscribe to edge-fire and edge-retract events for **all** rules.
6050    ///
6051    /// Returns `Err(GraphError::ReadOnly)` if called on an as-of instance —
6052    /// as-of instances never commit, so `distribute_events` never runs and the
6053    /// subscription would never deliver events.
6054    pub fn subscribe_all_rules(&mut self) -> core_storage::Result<Subscription> {
6055        if self.read_only {
6056            return Err(core_storage::GraphError::ReadOnly);
6057        }
6058        let inner = SubInner::new(self.sub_capacity());
6059        self.subscriptions.push(SubEntry {
6060            filter: SubFilter::AllRules,
6061            inner: std::sync::Arc::downgrade(&inner),
6062        });
6063        self.engine.set_emit_deltas(true);
6064        Ok(Subscription(inner))
6065    }
6066
6067    /// Subscribe to write events: node insert/delete, prop set/remove.
6068    ///
6069    /// Does not include edge-fire / edge-retract (rule-derived edge events).
6070    ///
6071    /// Returns `Err(GraphError::ReadOnly)` if called on an as-of instance —
6072    /// as-of instances never commit, so `distribute_events` never runs and the
6073    /// subscription would never deliver events.
6074    pub fn subscribe_writes(&mut self) -> core_storage::Result<Subscription> {
6075        if self.read_only {
6076            return Err(core_storage::GraphError::ReadOnly);
6077        }
6078        let inner = SubInner::new(self.sub_capacity());
6079        self.subscriptions.push(SubEntry {
6080            filter: SubFilter::Writes,
6081            inner: std::sync::Arc::downgrade(&inner),
6082        });
6083        self.engine.set_emit_deltas(true);
6084        Ok(Subscription(inner))
6085    }
6086
6087    /// Subscribe to incremental Cypher query results.
6088    ///
6089    /// Parses and plans `cypher`; rejects the query if the plan is not in the
6090    /// allowlisted subset (see [`core_query::cypher::is_subscribable`]):
6091    ///   - `MATCH (n:Label) WHERE … RETURN … [LIMIT n]`
6092    ///   - `MATCH (a)-[r:TYPE]->(b) RETURN … [LIMIT n]`  (exactly one hop)
6093    ///
6094    /// SKIP is not supported — it shifts the result window on every commit,
6095    /// causing spurious Added/Removed churn for rows whose data never changed.
6096    /// Multi-hop Expand chains are not supported; each additional MATCH clause
6097    /// widens scope beyond the documented single-scan / single-hop subset.
6098    ///
6099    /// After each successful commit, the plan is **fully re-executed** and the
6100    /// result is diffed against the previous run. Added rows produce
6101    /// [`DbEvent::QueryRowAdded`]; removed rows produce
6102    /// [`DbEvent::QueryRowRemoved`].
6103    ///
6104    /// **Full re-run per commit; use LIMIT to bound execution cost.**
6105    /// The existing 1 M intermediate-row cap applies. Differential evaluation
6106    /// is roadmap / Phase 5.
6107    ///
6108    /// Returns `Err(GraphError::ReadOnly)` if called on an as-of instance —
6109    /// as-of instances never commit, so `distribute_events` never runs and the
6110    /// subscription would never deliver events.
6111    ///
6112    /// Returns `Err(GraphError::QueryError)` if the query fails to parse, plan,
6113    /// or if the plan shape is not in the allowlist.
6114    pub fn subscribe_query(&mut self, cypher: &str) -> Result<Subscription> {
6115        if self.read_only {
6116            return Err(GraphError::ReadOnly);
6117        }
6118        let tokens = lex(cypher).map_err(|e| GraphError::QueryError {
6119            detail: format!("lex: {e}"),
6120        })?;
6121        let ast = parse(&tokens).map_err(|e| GraphError::QueryError {
6122            detail: format!("parse: {e}"),
6123        })?;
6124        let ops = plan(&ast).map_err(|e| GraphError::QueryError {
6125            detail: format!("plan: {e}"),
6126        })?;
6127        if !is_subscribable(&ops) {
6128            return Err(GraphError::QueryError {
6129                detail: "subscribe_query only supports allowlisted plan shapes: \
6130                         MATCH (n:Label) WHERE … RETURN … [LIMIT n] or \
6131                         MATCH (a)-[r:TYPE]->(b) RETURN … [LIMIT n] (exactly one hop). \
6132                         Not supported: multi-hop Expand chains, SKIP (creates \
6133                         unstable offset windows), ORDER BY, DISTINCT, aggregates, \
6134                         variable-length paths, OPTIONAL MATCH, WITH, UNWIND. \
6135                         Use LIMIT to bound re-execution cost."
6136                    .to_string(),
6137            });
6138        }
6139        // Execute once to capture initial state (initial rows are not emitted as
6140        // events — the subscriber learns the baseline via the first query call).
6141        let empty_params = BTreeMap::new();
6142        let initial = execute(&self.view(), &ops, &Params(&empty_params)).map_err(|e| {
6143            GraphError::QueryError {
6144                detail: format!("execute: {e}"),
6145            }
6146        })?;
6147        let columns = initial.columns().to_vec();
6148        let prev_row_map = Self::result_to_row_map(&initial);
6149        let inner = SubInner::new(self.sub_capacity());
6150        // Derive the scan-label sym for the commit-skip fast-path.  Any Expand op
6151        // or unrecognized leading scan → None (always re-execute).
6152        let scan_label = extract_scan_label(&ops, Arc::make_mut(&mut self.syms));
6153        self.query_subscriptions.push(QuerySubEntry {
6154            ops,
6155            columns,
6156            prev_row_map,
6157            inner: std::sync::Arc::downgrade(&inner),
6158            scan_label,
6159        });
6160        Ok(Subscription(inner))
6161    }
6162
6163    /// Queue capacity used for new subscriptions.
6164    fn sub_capacity(&self) -> usize {
6165        self.sub_capacity
6166    }
6167
6168    /// Override per-subscriber queue capacity for subsequently created
6169    /// subscriptions on this db instance.
6170    ///
6171    /// Default is [`DEFAULT_SUB_CAPACITY`] (65,536 events). Use a smaller
6172    /// value in tests to exercise the [`DbEvent::Lagged`] path without
6173    /// generating tens of thousands of events.
6174    ///
6175    /// This is a test-support escape hatch. Calling it in production reduces
6176    /// subscriber reliability (more Lagged events). It is hidden from rustdoc
6177    /// to discourage accidental production use.
6178    #[doc(hidden)]
6179    pub fn set_sub_capacity(&mut self, capacity: usize) {
6180        self.sub_capacity = capacity;
6181    }
6182
6183    // -----------------------------------------------------------------------
6184
6185    /// Start an atomic batch.
6186    ///
6187    /// The returned [`BatchBuilder`] borrows `self` mutably until
6188    /// [`BatchBuilder::commit`]. Builder methods queue ops only — no
6189    /// validation, no WAL I/O. `commit` validates every queued op against
6190    /// live state plus preceding ops in this batch (duplicate key inside
6191    /// the batch is `Err`; an edge between two nodes created earlier in
6192    /// the batch is valid; `delete_node` then insert of the same key is a
6193    /// fresh identity). Validation never mutates the database. Any failure
6194    /// leaves WAL bytes and in-memory state identical to before `commit`.
6195    /// On success, one `WalRecord::Batch` frame is appended (one fsync)
6196    /// and each inner record is applied in order so rules fire per record.
6197    /// An empty batch, or a batch of only no-ops, writes zero WAL bytes.
6198    ///
6199    /// **Rule-window limitation:** batch validation cannot see edges that a
6200    /// rule created earlier in the *same* batch will derive at apply time, so
6201    /// a `delete_edge` / `insert_edge` in that window is silently no-oped
6202    /// where sequential calls would return `Err(RuleOwned)`. State integrity
6203    /// is unaffected (idempotent apply, provenance intact). Create rules in
6204    /// their own batch, or sequentially, when later ops may touch derived
6205    /// edges.
6206    pub fn batch(&mut self) -> BatchBuilder<'_, F> {
6207        BatchBuilder {
6208            db: self,
6209            ops: Vec::new(),
6210        }
6211    }
6212
6213    /// Closure-style atomic write batch.
6214    ///
6215    /// Equivalent to calling [`GraphDb::batch`], invoking `build` to queue ops,
6216    /// then committing. All ops queued inside `build` are validated in order and
6217    /// committed as a single `WalRecord::Batch` frame (one fsync). Rules fire
6218    /// once per inner record, in order, after commit — semantically identical to
6219    /// sequential single-op writes.
6220    ///
6221    /// **Error semantics — validate-then-apply.** `build` queues ops without
6222    /// touching the database. [`BatchBuilder::commit`] validates every op against
6223    /// live state plus earlier ops in this batch before writing anything. If op N
6224    /// fails validation (duplicate key, unknown key, rule-owned edge, …) the
6225    /// entire batch is rejected: no WAL bytes are written and no in-memory state
6226    /// changes. The database is identical to its state before `write_batch` was
6227    /// called.
6228    ///
6229    /// **Atomicity is crash-level, NOT isolation-level.** On replay after a crash,
6230    /// a partial (torn) `Batch` frame applies NONE of its ops — the frame is
6231    /// either fully applied or not at all. However, while applying a committed
6232    /// batch, concurrent readers may observe intermediate states as ops are applied
6233    /// sequentially in memory. There is no interactive transaction isolation in v1.
6234    /// This is documented as "crash-atomic write batches; no interactive
6235    /// transactions or read isolation."
6236    ///
6237    /// **Returns** `(nodes_inserted, edges_inserted)`. An empty or all-noop batch
6238    /// writes zero WAL bytes and returns `(0, 0)`.
6239    ///
6240    /// # Example
6241    ///
6242    /// ```rust,ignore
6243    /// let (nodes, edges) = db.write_batch(|b| {
6244    ///     b.insert_node("Person", "alice", vec![("age".into(), Value::Int(30))]);
6245    ///     b.insert_node("Person", "bob", vec![]);
6246    ///     b.insert_edge("KNOWS", "alice", "bob");
6247    ///     b.set_prop("alice", "role", Value::Str("admin".into()));
6248    ///     b.delete_node("old_key");
6249    /// })?;
6250    /// // One fsync; on crash replay: all five ops land or none do.
6251    /// ```
6252    pub fn write_batch<C>(&mut self, build: C) -> Result<(usize, usize)>
6253    where
6254        C: FnOnce(&mut BatchBuilder<'_, F>),
6255    {
6256        let mut b = self.batch();
6257        build(&mut b);
6258        b.commit()
6259    }
6260
6261    /// Insert `rows` as nodes of `label`. One call is one atomic batch:
6262    /// auto-declared KeyMatch rules (if any) first, then the accepted node
6263    /// inserts, so incremental fire sees the new rules. Per-row key problems
6264    /// are collected in [`IngestReport::row_errors`] and skipped; a commit
6265    /// `Err` means nothing was applied.
6266    ///
6267    /// Auto-FK rule names are `auto_fk_<src_label_lowercase>_<field>` so
6268    /// distinct source labels sharing an FK field each get their own rule.
6269    pub fn ingest(
6270        &mut self,
6271        label: &str,
6272        rows: Vec<BTreeMap<String, Value>>,
6273        opts: &IngestOptions,
6274    ) -> Result<IngestReport> {
6275        self.ingest_with_edges(label, rows, opts, &[])
6276    }
6277
6278    /// [`ingest`] plus user edges in the **same** previewed WAL batch.
6279    /// A failing edge rejects the whole request; nothing is applied.
6280    pub fn ingest_with_edges(
6281        &mut self,
6282        label: &str,
6283        rows: Vec<BTreeMap<String, Value>>,
6284        opts: &IngestOptions,
6285        edges: &[(String, String, String)],
6286    ) -> Result<IngestReport> {
6287        crate::ingest::run(self, label, rows, opts, edges)
6288    }
6289
6290    /// Parse `json` as an array of objects and ingest via [`GraphDb::ingest`].
6291    ///
6292    /// JSON `null` fields are silently omitted (not stored, not a row error).
6293    /// Nested objects and arrays-of-objects are a per-row error (row skipped).
6294    /// Parse failures and a top-level value that is not an array of objects
6295    /// return [`GraphError::IngestError`].
6296    pub fn ingest_json(
6297        &mut self,
6298        label: &str,
6299        json: &str,
6300        opts: &IngestOptions,
6301    ) -> Result<IngestReport> {
6302        crate::ingest::run_json(self, label, json, opts)
6303    }
6304
6305    fn commit_logged_batch(
6306        &mut self,
6307        ops: Vec<BatchOp>,
6308        ingest: Option<(String, usize)>,
6309        // Two-source rule: write_batch_authz threads authz here directly (never
6310        // touches pending_write_authz); query_write_authz sets the field instead
6311        // and passes None.  Only one source is non-None per call.
6312        param_authz: Option<WriteAuthz>,
6313    ) -> Result<BatchOutcome> {
6314        // Read-only guard: catches empty-batch calls before the early-return
6315        // that skips log_then_apply_with, ensuring all mutation entry points fail.
6316        if self.read_only {
6317            return Err(GraphError::ReadOnly);
6318        }
6319        // Ensure provenance is decoded before MutPreview accesses it
6320        // (note_delete_rule / is_rule_owned may call engine.provenance()).
6321        self.engine.ensure_provenance_loaded_mut();
6322
6323        // ── Authz pre-check ──────────────────────────────────────────────────
6324        // Evaluate the decision table per-op BEFORE MutPreview so that a denial
6325        // produces no WAL frame (all-or-nothing at the authz boundary extends
6326        // the existing validate-then-apply contract to role-scope checks).
6327        //
6328        // `batch_created` tracks key→label for nodes created by earlier ops in
6329        // THIS batch, so InsertEdgeUpsert can count same-batch placeholder nodes
6330        // as visible without needing to call `self.ids.get` on not-yet-committed
6331        // keys (they won't be there yet).
6332        //
6333        // Two-source rule: param_authz (write_batch_authz path) takes precedence;
6334        // fall back to self.pending_write_authz (query_write_authz/Cypher path).
6335        // Cloning the field copy avoids a simultaneous borrow of self.ids below.
6336        let authz_opt = param_authz.or_else(|| self.pending_write_authz.clone());
6337        if let Some(ref authz) = authz_opt {
6338            let mut batch_created: BTreeMap<String, String> = BTreeMap::new();
6339            for op in &ops {
6340                self.check_single_op_authz(authz, op, &batch_created)?;
6341                // Update batch_created after a passing authz check so that
6342                // subsequent ops in this batch see the nodes as "about to exist".
6343                match op {
6344                    BatchOp::InsertNode { label, key, .. } => {
6345                        // Only track genuinely new nodes (absent from the
6346                        // snapshot at authz-check time). A pre-existing visible
6347                        // key would be a DuplicateKey — not a real creation —
6348                        // so MutPreview handles it. Letting it into batch_created
6349                        // would allow a later SetProp to bypass update_labels
6350                        // via the "batch-created → always updatable" ruling
6351                        // (delete+recreate exploit, fix for I1 review round 2).
6352                        //
6353                        // Accepted edge: for a delete+recreate-with-different-
6354                        // label batch, node_status resolves the pre-delete
6355                        // (store) label for any subsequent update checks. This
6356                        // grants no net-new capability — a role that can delete+
6357                        // create can already place arbitrary props via
6358                        // InsertNode's own props field.
6359                        if self.ids.get(key.as_str()).is_none() {
6360                            batch_created.insert(key.clone(), label.clone());
6361                        }
6362                    }
6363                    BatchOp::InsertEdgeUpsert {
6364                        placeholder_label,
6365                        src_key,
6366                        dst_key,
6367                        ..
6368                    } => {
6369                        // Both endpoints will be created if not already in store.
6370                        for ep_key in [src_key, dst_key] {
6371                            if self.ids.get(ep_key.as_str()).is_none()
6372                                && !batch_created.contains_key(ep_key.as_str())
6373                            {
6374                                batch_created.insert(ep_key.clone(), placeholder_label.clone());
6375                            }
6376                        }
6377                    }
6378                    _ => {}
6379                }
6380            }
6381        }
6382
6383        let mut outcome = BatchOutcome::default();
6384        let recs = {
6385            let mut preview = MutPreview::new(self);
6386            let mut recs = Vec::with_capacity(ops.len());
6387            // Which node row we are on, counted over the node-insert ops only.
6388            // A caller that queues its rows in order reads this straight back
6389            // as the index into its own list.
6390            let mut node_row = 0usize;
6391            // Every field name the store knows, which a `Replace` needs to work
6392            // out what it removes. Resolved on the first `Replace` in the frame
6393            // and reused, so N replaces read the field list once, not N times.
6394            let mut store_fields: Option<Vec<String>> = None;
6395            // Duplicate inserts this frame has to count, each paired with the
6396            // position in `recs` it belongs at. The count itself is named in the
6397            // dense rewrite and not here: a duplicate's endpoints and edge type
6398            // may all be created by earlier ops in this same frame, and nothing
6399            // in the frame has a dense id yet. See [`PlannedRec`].
6400            let mut deferred_counts: Vec<(usize, String, String, String)> = Vec::new();
6401            for op in ops {
6402                match op {
6403                    BatchOp::InsertNode { label, key, props } => {
6404                        node_row += 1;
6405                        preview.check_insert_node(&key, &props)?;
6406                        preview.note_insert_node(&label, &key, &props);
6407                        recs.push(WalRecord::InsertNode { label, key, props });
6408                    }
6409                    BatchOp::InsertNodeOnConflict {
6410                        label,
6411                        key,
6412                        props,
6413                        on_conflict,
6414                    } => {
6415                        let row = node_row;
6416                        node_row += 1;
6417                        if !preview.has_key(&key) {
6418                            // No conflict: an ordinary insert on any policy —
6419                            // except that a supplied view-owned field is the
6420                            // same mistake here as on a taken key, and gets the
6421                            // same row error rather than a frame error. Without
6422                            // this, one op answered one request two ways
6423                            // depending on whether the store already had the
6424                            // key (defect #19).
6425                            if let Some(why) = preview.supplied_view_owned_prop(&key, &props) {
6426                                outcome.row_errors.push((row, why));
6427                                continue;
6428                            }
6429                            preview.note_insert_node(&label, &key, &props);
6430                            recs.push(WalRecord::InsertNode { label, key, props });
6431                            continue;
6432                        }
6433                        match on_conflict {
6434                            OnConflict::Error => {
6435                                return Err(GraphError::DuplicateKey { key });
6436                            }
6437                            OnConflict::Skip => outcome.skipped += 1,
6438                            OnConflict::Replace => {
6439                                if store_fields.is_none() {
6440                                    store_fields = Some(preview.db.props_view().field_names());
6441                                }
6442                                let fields = store_fields.as_deref().unwrap_or_default();
6443                                match preview.plan_replace(&label, &key, &props, fields) {
6444                                    Ok((writes, kept_view_owned)) => {
6445                                        outcome.kept_view_owned += kept_view_owned;
6446                                        for (field, value) in writes {
6447                                            match value {
6448                                                Some(value) => {
6449                                                    preview.note_set_prop(&key, &field, &value);
6450                                                    recs.push(WalRecord::SetProp {
6451                                                        key: key.clone(),
6452                                                        field,
6453                                                        value,
6454                                                    });
6455                                                }
6456                                                None => {
6457                                                    preview.note_remove_prop(&key, &field);
6458                                                    recs.push(WalRecord::RemoveProp {
6459                                                        key: key.clone(),
6460                                                        field,
6461                                                    });
6462                                                }
6463                                            }
6464                                        }
6465                                        outcome.replaced += 1;
6466                                    }
6467                                    Err(why) => outcome.row_errors.push((row, why)),
6468                                }
6469                            }
6470                        }
6471                    }
6472                    BatchOp::InsertEdge {
6473                        edge_type,
6474                        src_key,
6475                        dst_key,
6476                    } => {
6477                        if preview.prepare_insert_edge(&edge_type, &src_key, &dst_key)? {
6478                            preview.note_insert_edge(&edge_type, &src_key, &dst_key);
6479                            recs.push(WalRecord::InsertEdge {
6480                                edge_type,
6481                                src_key,
6482                                dst_key,
6483                            });
6484                        } else if preview.db.multiplicity {
6485                            // A duplicate inside a batch counts the way a
6486                            // duplicate through `insert_edge` does: `ingest` and
6487                            // Cypher `CREATE` reach this choke-point and not
6488                            // that one, and a count only one entry point keeps
6489                            // would be worse than no count at all.
6490                            //
6491                            // This is the one gate on discriminant 23 from the
6492                            // batch path: a store that never opted in queues
6493                            // nothing here and so writes no such record.
6494                            deferred_counts.push((recs.len(), edge_type, src_key, dst_key));
6495                        }
6496                    }
6497                    BatchOp::SetProp { key, field, value } => {
6498                        if let Some(view_name) = preview.db.view_store.view_for_prop(&field) {
6499                            return Err(GraphError::ViewPropReadOnly {
6500                                view_name: view_name.to_string(),
6501                            });
6502                        }
6503                        preview.check_live_key(&key)?;
6504                        preview.note_set_prop(&key, &field, &value);
6505                        recs.push(WalRecord::SetProp { key, field, value });
6506                    }
6507                    BatchOp::RemoveProp { key, field } => {
6508                        if preview.prepare_remove_prop(&key, &field)? {
6509                            preview.note_remove_prop(&key, &field);
6510                            recs.push(WalRecord::RemoveProp { key, field });
6511                        }
6512                    }
6513                    BatchOp::DeleteEdge {
6514                        edge_type,
6515                        src_key,
6516                        dst_key,
6517                    } => {
6518                        if preview.prepare_delete_edge(&edge_type, &src_key, &dst_key)? {
6519                            preview.note_delete_edge(&edge_type, &src_key, &dst_key);
6520                            recs.push(WalRecord::DeleteEdge {
6521                                edge_type,
6522                                src_key,
6523                                dst_key,
6524                            });
6525                        }
6526                    }
6527                    BatchOp::DeleteNode { key } => {
6528                        preview.check_live_key(&key)?;
6529                        preview.note_delete_node(&key);
6530                        recs.push(WalRecord::DeleteNode { key });
6531                    }
6532                    BatchOp::CreateRule(def) => {
6533                        preview.check_create_rule(&def)?;
6534                        let def_bytes =
6535                            bincode::serialize(&def).map_err(|e| GraphError::Corrupt {
6536                                detail: format!("serialize rule: {e}"),
6537                            })?;
6538                        preview.note_create_rule(&def);
6539                        recs.push(WalRecord::CreateRule { def_bytes });
6540                    }
6541                    BatchOp::DeleteRule { name } => {
6542                        preview.check_delete_rule(&name)?;
6543                        preview.note_delete_rule(&name);
6544                        recs.push(WalRecord::DeleteRule { name });
6545                    }
6546                    BatchOp::RenameNode { old_key, new_key } => {
6547                        preview.check_rename_node(&old_key, &new_key)?;
6548                        preview.note_rename_node(&old_key, &new_key);
6549                        recs.push(WalRecord::RenameNode { old_key, new_key });
6550                    }
6551                    BatchOp::InsertEdgeUpsert {
6552                        edge_type,
6553                        src_key,
6554                        dst_key,
6555                        placeholder_label,
6556                    } => {
6557                        // Auto-create any missing endpoints as plain InsertNode ops.
6558                        // Rules fire and last-change is updated for each created node.
6559                        for key in [&src_key, &dst_key] {
6560                            if !preview.has_key(key) {
6561                                // A placeholder endpoint carries no props, so
6562                                // the view-owned check has nothing to refuse.
6563                                preview.check_insert_node(key, &[])?;
6564                                preview.note_insert_node(&placeholder_label, key, &[]);
6565                                recs.push(WalRecord::InsertNode {
6566                                    label: placeholder_label.clone(),
6567                                    key: key.clone(),
6568                                    props: vec![],
6569                                });
6570                            }
6571                        }
6572                        if preview.prepare_insert_edge(&edge_type, &src_key, &dst_key)? {
6573                            preview.note_insert_edge(&edge_type, &src_key, &dst_key);
6574                            recs.push(WalRecord::InsertEdge {
6575                                edge_type,
6576                                src_key,
6577                                dst_key,
6578                            });
6579                        } else if preview.db.multiplicity {
6580                            // Same choke-point, same gate as `BatchOp::InsertEdge`
6581                            // above: an upsert that finds the pair already there
6582                            // is a duplicate insert and counts as one.
6583                            deferred_counts.push((recs.len(), edge_type, src_key, dst_key));
6584                        }
6585                    }
6586                }
6587            }
6588            // Splice the deferred counts back into the positions they were
6589            // raised at, so a count still sits exactly where the duplicate did
6590            // — before any later op in the frame that deletes the pair.
6591            let mut planned: Vec<PlannedRec> =
6592                Vec::with_capacity(recs.len() + deferred_counts.len());
6593            let mut deferred = deferred_counts.into_iter().peekable();
6594            for (i, rec) in recs.into_iter().enumerate() {
6595                while deferred.peek().is_some_and(|(at, ..)| *at == i) {
6596                    let (_, edge_type, src_key, dst_key) = deferred.next().expect("just peeked");
6597                    planned.push(PlannedRec::DuplicateCount {
6598                        edge_type,
6599                        src_key,
6600                        dst_key,
6601                    });
6602                }
6603                planned.push(PlannedRec::Rec(rec));
6604            }
6605            for (_, edge_type, src_key, dst_key) in deferred {
6606                planned.push(PlannedRec::DuplicateCount {
6607                    edge_type,
6608                    src_key,
6609                    dst_key,
6610                });
6611            }
6612            planned
6613        };
6614        // A frame that is nothing but skips or refused rows writes no WAL, but
6615        // it still has counts to report, so the early returns carry `outcome`
6616        // rather than zeros.
6617        if recs.is_empty() {
6618            return Ok(outcome);
6619        }
6620        // rewrite_wal_dense converts every InsertNode/InsertEdge into its
6621        // *Id form, so only the dense variants can appear in `recs` here.
6622        let recs = self.rewrite_wal_dense_planned(recs)?;
6623        // The rewrite can empty a non-empty batch: a `SET n.ns` naming the
6624        // namespace the node is already in is a no-op and is dropped there. An
6625        // empty `Batch` frame would still take a commit sequence and a WAL
6626        // record, so a batch that turns out to be nothing writes nothing.
6627        if recs.is_empty() {
6628            return Ok(outcome);
6629        }
6630        outcome.nodes_inserted = recs
6631            .iter()
6632            .filter(|r| matches!(r, WalRecord::InsertNodeId { .. }))
6633            .count();
6634        outcome.edges_inserted = recs
6635            .iter()
6636            .filter(|r| matches!(r, WalRecord::InsertEdgeId { .. }))
6637            .count();
6638        // Ingest / write_batch / query_write: one Batch frame, one fsync per call
6639        // under Strict.  Pass self.fsync directly so Strict stays Strict —
6640        // wal_needs_sync(Strict, _) always returns true regardless of op count.
6641        // Mapping Strict → Batched (the prior bug) caused wal_needs_sync to
6642        // short-circuit on single-op batches and silently skip the fsync.
6643        // Batched fsyncs only for multi-op batches; Relaxed always skips.
6644        self.log_then_apply_with(WalRecord::Batch(recs), ingest, self.fsync)?;
6645        Ok(outcome)
6646    }
6647
6648    fn commit_batch(&mut self, ops: Vec<BatchOp>) -> Result<(usize, usize)> {
6649        self.commit_logged_batch(ops, None, None).map(inserted_pair)
6650    }
6651
6652    /// Commit one submission WITHOUT an fsync — for use inside `commit_group`
6653    /// and the group-commit drain thread, which do a single group fsync later.
6654    fn commit_batch_nosync(&mut self, ops: Vec<BatchOp>) -> Result<(usize, usize)> {
6655        // Restore fsync policy even on panic via a raw-pointer drop guard.
6656        // A panic here would poison the RwLock anyway, but the correct policy
6657        // must be in place if the guard is ever unwrapped.
6658        struct RestoreFsync(*mut FsyncPolicy, FsyncPolicy);
6659        impl Drop for RestoreFsync {
6660            fn drop(&mut self) {
6661                // SAFETY: the pointer is valid for the full duration of
6662                // commit_batch_nosync; the guard is dropped before the frame
6663                // returns, and GraphDb outlives this frame.
6664                unsafe {
6665                    *self.0 = self.1;
6666                }
6667            }
6668        }
6669        let saved = self.fsync;
6670        // SAFETY: raw pointer into self; guard dropped within this frame.
6671        let _g = RestoreFsync(&mut self.fsync as *mut FsyncPolicy, saved);
6672        self.fsync = FsyncPolicy::Relaxed;
6673        self.commit_logged_batch(ops, None, None).map(inserted_pair)
6674    }
6675
6676    /// Commit multiple op-batches as a **group**: each submission gets its own
6677    /// WAL `Batch` frame, but there is exactly **one** `Fs::sync` for the whole
6678    /// group (under `Strict` / `Batched` policy; `Relaxed` skips all syncs).
6679    ///
6680    /// # Durability semantics
6681    ///
6682    /// A crash before the group fsync may lose **all** submissions in the group.
6683    /// A crash after the group fsync preserves all of them.  No submission is
6684    /// ever torn: each WAL frame is either fully applied on replay or dropped
6685    /// in its entirety (CRC-protected frame boundaries).
6686    ///
6687    /// Events and subscription notifications fire per-submission immediately
6688    /// after apply, which may be before the group fsync.  From a subscriber's
6689    /// perspective this is equivalent to the `Relaxed` durability window.
6690    /// Submitters using [`SharedDb::submit_batch`] only unblock after the group
6691    /// fsync, so from their perspective durability is fully guaranteed.
6692    ///
6693    /// # MVCC interplay
6694    ///
6695    /// Each submission records its own `CommitDelta`; the fold-every-K counter
6696    /// increments per submission (not per group), preserving existing reader
6697    /// snapshot semantics.
6698    ///
6699    /// # Returns
6700    ///
6701    /// One `Result<(nodes_inserted, edges_inserted)>` per input group element,
6702    /// in order.  Failures are per-submission (validation errors); the group
6703    /// fsync error (if any) is returned as the second tuple element.
6704    pub fn commit_group(
6705        &mut self,
6706        groups: Vec<Vec<BatchOp>>,
6707    ) -> (Vec<Result<(usize, usize)>>, Option<GraphError>) {
6708        let mut results = Vec::with_capacity(groups.len());
6709        for ops in groups {
6710            results.push(self.commit_batch_nosync(ops));
6711        }
6712        let any_ok = results.iter().any(|r| r.is_ok());
6713        let sync_err = if self.fsync != FsyncPolicy::Relaxed && any_ok {
6714            self.fs
6715                .sync(core_storage::fs::FileId::Wal)
6716                .map_err(GraphError::Io)
6717                .err()
6718        } else {
6719            None
6720        };
6721        (results, sync_err)
6722    }
6723
6724    /// Like [`commit_group`] but skips the group fsync entirely.
6725    ///
6726    /// Used by the drain thread to apply submissions under the write lock and
6727    /// then perform the single fsync OUTSIDE the lock (via
6728    /// `core_storage::sync_wal_at`), reducing the write-lock hold time visible
6729    /// to concurrent readers.
6730    pub fn commit_group_nosync(
6731        &mut self,
6732        groups: Vec<Vec<BatchOp>>,
6733    ) -> Vec<Result<(usize, usize)>> {
6734        let mut results = Vec::with_capacity(groups.len());
6735        for ops in groups {
6736            results.push(self.commit_batch_nosync(ops));
6737        }
6738        results
6739    }
6740
6741    pub fn insert_node(
6742        &mut self,
6743        label: &str,
6744        key: &str,
6745        props: Vec<(String, Value)>,
6746    ) -> Result<()> {
6747        if self.read_only {
6748            return Err(GraphError::ReadOnly);
6749        }
6750        MutPreview::new(self).check_insert_node(key, &props)?;
6751        self.log_dense(vec![WalRecord::InsertNode {
6752            label: label.into(),
6753            key: key.into(),
6754            props,
6755        }])
6756    }
6757
6758    /// Insert a user edge. `Ok(true)` when the pair was new, `Ok(false)` when it
6759    /// was already there — the question is "was this pair new", and a duplicate
6760    /// does not make it so.
6761    ///
6762    /// On a store that called [`enable_multiplicity`](Self::enable_multiplicity)
6763    /// a duplicate is no longer a total no-op: it raises the pair's insert count
6764    /// (§5.13). Adjacency is still a set, so [`degree`](Self::degree) is
6765    /// unchanged and the return value is still `Ok(false)`; the count is visible
6766    /// only through [`degree_multiplicity`](Self::degree_multiplicity) and the
6767    /// reserved [`EDGE_COUNT_PROP`]. On every other store a duplicate writes
6768    /// nothing at all, as it always has.
6769    pub fn insert_edge(&mut self, edge_type: &str, src_key: &str, dst_key: &str) -> Result<bool> {
6770        if self.read_only {
6771            return Err(GraphError::ReadOnly);
6772        }
6773        if !MutPreview::new(self).prepare_insert_edge(edge_type, src_key, dst_key)? {
6774            // The pair exists. The only thing left to record is that it was
6775            // asked for again, and only where the store asked to be told.
6776            if let Some(rec) = self.edge_count_record(edge_type, src_key, dst_key) {
6777                self.log_then_apply(rec)?;
6778            }
6779            return Ok(false);
6780        }
6781        self.log_dense(vec![WalRecord::InsertEdge {
6782            edge_type: edge_type.into(),
6783            src_key: src_key.into(),
6784            dst_key: dst_key.into(),
6785        }])?;
6786        Ok(true)
6787    }
6788
6789    pub fn set_prop(&mut self, key: &str, field: &str, value: Value) -> Result<()> {
6790        if self.read_only {
6791            return Err(GraphError::ReadOnly);
6792        }
6793        if let Some(view_name) = self.view_store.view_for_prop(field) {
6794            return Err(GraphError::ViewPropReadOnly {
6795                view_name: view_name.to_string(),
6796            });
6797        }
6798        MutPreview::new(self).check_live_key(key)?;
6799        self.log_dense(vec![WalRecord::SetProp {
6800            key: key.into(),
6801            field: field.into(),
6802            value,
6803        }])
6804    }
6805
6806    /// Set several properties on one live node in a single WAL commit.
6807    ///
6808    /// Every per-property check [`set_prop`](Self::set_prop) runs — view-owned
6809    /// names, live key, the `ns` immutability rule and its type — is evaluated
6810    /// for the whole list before any record is logged. The first refusal
6811    /// returns and the node is unchanged. An empty list writes nothing.
6812    pub fn set_props(&mut self, key: &str, props: Vec<(String, Value)>) -> Result<()> {
6813        if self.read_only {
6814            return Err(GraphError::ReadOnly);
6815        }
6816        MutPreview::new(self).check_live_key(key)?;
6817        for (field, _) in &props {
6818            if let Some(view_name) = self.view_store.view_for_prop(field) {
6819                return Err(GraphError::ViewPropReadOnly {
6820                    view_name: view_name.to_string(),
6821                });
6822            }
6823        }
6824        if props.is_empty() {
6825            return Ok(());
6826        }
6827        self.write_batch(|b| {
6828            for (field, value) in props {
6829                b.set_prop(key, &field, value);
6830            }
6831        })
6832        .map(|_| ())
6833    }
6834
6835    /// Remove a property. Returns `Ok(false)` (and does not log) if the field
6836    /// is already absent. Unknown or tombstoned keys are `Err(KeyNotFound)`.
6837    /// A field a view owns is `Err(ViewPropReadOnly)` — stated once, in
6838    /// [`MutPreview::prepare_remove_prop`], so that the batch ops reaching that
6839    /// same choke-point cannot miss it.
6840    pub fn remove_prop(&mut self, key: &str, field: &str) -> Result<bool> {
6841        if self.read_only {
6842            return Err(GraphError::ReadOnly);
6843        }
6844        if !MutPreview::new(self).prepare_remove_prop(key, field)? {
6845            return Ok(false);
6846        }
6847        self.log_then_apply(WalRecord::RemoveProp {
6848            key: key.into(),
6849            field: field.into(),
6850        })?;
6851        Ok(true)
6852    }
6853
6854    /// Delete a user edge. Returns `Ok(false)` (and does not log) if the edge
6855    /// is absent. Unknown keys are `Err(KeyNotFound)`. Rule-owned edges — in
6856    /// provenance, or a pair a live rule would derive — are `Err(RuleOwned)`
6857    /// (the rule would just put the edge back; delete or change the rule).
6858    pub fn delete_edge(&mut self, edge_type: &str, src_key: &str, dst_key: &str) -> Result<bool> {
6859        if self.read_only {
6860            return Err(GraphError::ReadOnly);
6861        }
6862        if !MutPreview::new(self).prepare_delete_edge(edge_type, src_key, dst_key)? {
6863            return Ok(false);
6864        }
6865        self.log_then_apply(WalRecord::DeleteEdge {
6866            edge_type: edge_type.into(),
6867            src_key: src_key.into(),
6868            dst_key: dst_key.into(),
6869        })?;
6870        Ok(true)
6871    }
6872
6873    /// Delete a live node. Unknown or already-tombstoned keys are
6874    /// `Err(KeyNotFound)` and are not logged. Validation runs before the WAL
6875    /// write; `apply` of a logged `DeleteNode` for an already-tombstoned key
6876    /// (crash window) is a clean no-op.
6877    ///
6878    /// Returns a [`DeleteReport`] with counts of manual and derived edges
6879    /// removed (computed from live state before the deletion is applied).
6880    pub fn delete_node(&mut self, key: &str) -> Result<DeleteReport> {
6881        if self.read_only {
6882            return Err(GraphError::ReadOnly);
6883        }
6884        // Provenance must be loaded before we query provenance_touching.
6885        self.engine.ensure_provenance_loaded_mut();
6886        let id = self
6887            .ids
6888            .get(key)
6889            .ok_or_else(|| GraphError::KeyNotFound { key: key.into() })?;
6890
6891        // Count edges before the delete is applied so we can report counts.
6892        let derived_set: BTreeSet<(u32, u32, u32)> = self
6893            .engine
6894            .provenance_touching(id)
6895            .map(|(_, etype, src, dst)| (etype, src, dst))
6896            .collect();
6897        let derived_edges = derived_set.len() as u64;
6898
6899        let mut total_topo = 0u64;
6900        let tv = self.topo_view();
6901        for et in tv.etypes() {
6902            total_topo += tv.neighbors(et, Direction::Out, id).len() as u64
6903                + tv.neighbors(et, Direction::In, id).len() as u64;
6904        }
6905        // For symmetric rules (e.g. Overlap), a→b and b→a are two separate directed
6906        // triples in both the topo scan (Out and In from id) and in provenance_touching.
6907        // The subtraction remains correct because both counts include both directions.
6908        let manual_edges = total_topo.saturating_sub(derived_edges);
6909
6910        self.log_then_apply(WalRecord::DeleteNode { key: key.into() })?;
6911        Ok(DeleteReport {
6912            manual_edges,
6913            derived_edges,
6914        })
6915    }
6916
6917    /// Rename a live node's key.  The dense id (and therefore all edges,
6918    /// props, history, and last-change tracking) is unaffected.
6919    ///
6920    /// Returns `Err(KeyNotFound)` if `old` is not a live key.
6921    /// Returns `Err(DuplicateKey)` if `new` is already live.
6922    pub fn rename_node(&mut self, old: &str, new: &str) -> Result<()> {
6923        if self.read_only {
6924            return Err(GraphError::ReadOnly);
6925        }
6926        MutPreview::new(self).check_rename_node(old, new)?;
6927        self.log_then_apply(WalRecord::RenameNode {
6928            old_key: old.into(),
6929            new_key: new.into(),
6930        })
6931    }
6932
6933    /// Return the IVF drift counter for the dst-side candidate index of `rule`.
6934    /// `None` if the rule does not exist or is not approximate.
6935    ///
6936    /// The drift counter increments on IVF insert/remove after the last fit.
6937    /// When dst-side drift exceeds [`core_rules::IVF_DRIFT_REBUILD`], apply
6938    /// WAL-logs `RebuildRule` as a second commit (rebuild resets the counter).
6939    pub fn ivf_dst_drift(&self, rule: &str) -> Option<u64> {
6940        // SideIvfExport = (centroids, node→cluster, drift)
6941        self.engine
6942            .export_ivf_state()
6943            .remove(rule)
6944            .map(|(_src, dst)| dst.2)
6945    }
6946
6947    /// Validate and WAL-log a new rule, then backfill derived edges inside apply.
6948    /// Validation and duplicate-name check run before logging so invalid rules
6949    /// never enter the WAL.
6950    pub fn create_rule(&mut self, def: RuleDef) -> Result<()> {
6951        if self.read_only {
6952            return Err(GraphError::ReadOnly);
6953        }
6954        MutPreview::new(self).check_create_rule(&def)?;
6955        let def_bytes = bincode::serialize(&def).map_err(|e| GraphError::Corrupt {
6956            detail: format!("serialize rule: {e}"),
6957        })?;
6958        self.log_then_apply(WalRecord::CreateRule { def_bytes })
6959    }
6960
6961    /// Override this handle's HNSW build-slice size, or `None` to restore
6962    /// [`core_rules::HNSW_BUILD_BATCH`].
6963    ///
6964    /// Exposed for tests that need a small slice without a large corpus; not
6965    /// part of the stable surface.
6966    #[doc(hidden)]
6967    pub fn set_hnsw_build_batch(&mut self, batch: Option<usize>) {
6968        self.engine.set_hnsw_build_batch(batch);
6969    }
6970
6971    /// Rules whose vector index is still being built, in name order.
6972    ///
6973    /// The same list [`GraphDb::stats`] reports per rule in `building`.
6974    /// After a clean open this includes a build a snapshot cut short, so
6975    /// `serve`'s ticker can pump it without a write.
6976    pub fn builds_in_progress(&self) -> Vec<BuildProgress> {
6977        self.engine.builds_in_progress()
6978    }
6979
6980    /// Advance any vector index still building and backfill each rule that
6981    /// finishes. Returns what is still outstanding.
6982    ///
6983    /// A map lookup when nothing is pending, so it is cheap to call on a timer.
6984    /// One write lock and at most [`core_rules::HNSW_BUILD_BATCH`] vector
6985    /// inserts per pending rule per call, so a caller can drive a large build
6986    /// to completion without ever holding the lock for more than a slice.
6987    ///
6988    /// A rule that finishes here is backfilled through the same
6989    /// `WalRecord::RebuildRule` second commit that IVF drift already uses, so
6990    /// its derived edges are produced by [`GraphDb::rebuild_rule`]'s code path
6991    /// and appear all at once.
6992    ///
6993    /// Every ordinary write pumps one slice on its own (see the post-commit
6994    /// hook in `log_then_apply_with`), so this is for quiescent stores and for
6995    /// operators who want the build finished before traffic arrives.
6996    pub fn pump_index_build(&mut self) -> Result<Vec<BuildProgress>> {
6997        Ok(self.pump_index_build_reporting()?.1)
6998    }
6999
7000    /// [`GraphDb::pump_index_build`], also reporting the builds that **this**
7001    /// call finished, so a progress display can say so.
7002    ///
7003    /// A build can be registered and completed inside a single call — that is
7004    /// what a mid-build snapshot looks like on reopen, where the index scan
7005    /// finishes the graph and only the backfill is outstanding — and the
7006    /// outstanding list alone cannot show that anything happened.
7007    pub fn pump_index_build_reporting(
7008        &mut self,
7009    ) -> Result<(Vec<BuildProgress>, Vec<BuildProgress>)> {
7010        // A read-only handle cannot issue the `RebuildRule` a finished build
7011        // needs, so it would advance the index and then silently fail to
7012        // produce the edges. Refusing is the honest answer.
7013        if self.read_only {
7014            return Err(GraphError::ReadOnly);
7015        }
7016        let finished = self.pump_one_slice();
7017        for done in &finished {
7018            // The index is whole but the rule still owns no edges. A failed
7019            // second commit must leave the rule re-pumpable rather than
7020            // silently edge-less, so the error is surfaced here — unlike the
7021            // post-commit hook, this call is not riding someone else's commit.
7022            self.log_then_apply(WalRecord::RebuildRule {
7023                name: done.rule.clone(),
7024            })?;
7025        }
7026        Ok((finished, self.engine.builds_in_progress()))
7027    }
7028
7029    /// Run the deferred candidate-index build, if it is still owed, against the
7030    /// graph as it stands *now* — before the caller applies anything.
7031    ///
7032    /// A no-op bool test once the indexes are populated, which is after the
7033    /// first write of the handle's life, and for a store with no rules at all.
7034    fn populate_indexes_before_write(&mut self) {
7035        if !self.engine.needs_index_population() {
7036            return;
7037        }
7038        // The retained snapshot blobs arrive with the V8 base sections; without
7039        // them the scan would rebuild every graph the snapshot already holds.
7040        self.ensure_v8_base_sections_loaded();
7041        if !self.engine.needs_index_population() {
7042            return;
7043        }
7044        let mut eng = std::mem::take(&mut self.engine);
7045        {
7046            let gm = make_graph_mut(
7047                &self.ids,
7048                Arc::make_mut(&mut self.syms),
7049                &self.labels,
7050                build_props_view(&self.props, &self.base),
7051                Arc::make_mut(&mut self.topo),
7052                &self.base,
7053                Arc::make_mut(&mut self.edge_props),
7054            );
7055            eng.populate_indexes(&gm);
7056        }
7057        self.engine = eng;
7058    }
7059
7060    /// One slice of build work for every pending rule. Returns the rules whose
7061    /// index just became whole, which the caller must `RebuildRule`.
7062    ///
7063    /// Goes through the engine even with nothing pending when the indexes have
7064    /// not been populated yet: that call adopts the persisted graphs and, for
7065    /// an incomplete blob already registered at open, leaves the remainder to
7066    /// this slice rather than inserting it inline.
7067    fn pump_one_slice(&mut self) -> Vec<BuildProgress> {
7068        // The retained snapshot blobs — and the id count an interrupted build
7069        // is recognised against — arrive with the V8 base sections, which a
7070        // clean open reads lazily. Without this a freshly opened handle pumps
7071        // against empty retained state and concludes there is nothing to do,
7072        // which is precisely the store `build-index` exists for.
7073        self.ensure_v8_base_sections_loaded();
7074        let mut eng = std::mem::take(&mut self.engine);
7075        let finished = {
7076            let mut gm = make_graph_mut(
7077                &self.ids,
7078                Arc::make_mut(&mut self.syms),
7079                &self.labels,
7080                build_props_view(&self.props, &self.base),
7081                Arc::make_mut(&mut self.topo),
7082                &self.base,
7083                Arc::make_mut(&mut self.edge_props),
7084            );
7085            eng.pump_index_build(&mut gm)
7086        };
7087        self.engine = eng;
7088        finished
7089    }
7090
7091    /// Register a sliced build a snapshot cut short, from blobs with
7092    /// `complete == false`.
7093    ///
7094    /// Peeks the V8 mmap for incomplete entries without copying complete
7095    /// graphs. V5–V7 already hold the blobs in the engine from restore.
7096    fn register_outstanding_index_builds(&mut self) {
7097        if self.engine.indexes_populated() {
7098            return;
7099        }
7100        let extra = self.collect_incomplete_hnsw_blobs();
7101        let mut eng = std::mem::take(&mut self.engine);
7102        {
7103            let gm = make_graph_mut(
7104                &self.ids,
7105                Arc::make_mut(&mut self.syms),
7106                &self.labels,
7107                build_props_view(&self.props, &self.base),
7108                Arc::make_mut(&mut self.topo),
7109                &self.base,
7110                Arc::make_mut(&mut self.edge_props),
7111            );
7112            eng.register_incomplete_hnsw_builds(&extra, &gm);
7113        }
7114        self.engine = eng;
7115    }
7116
7117    /// Incomplete `(src, dst)` HNSW blobs from the V8 mmap, copied only when
7118    /// `complete` is false. Empty when there is no mmap base (V5–V7 uses the
7119    /// engine's retained map instead).
7120    fn collect_incomplete_hnsw_blobs(&self) -> BTreeMap<String, (Vec<u8>, Vec<u8>)> {
7121        let Some(base) = &self.base else {
7122            return BTreeMap::new();
7123        };
7124        let Ok(archived) = base.hnsw_section() else {
7125            return BTreeMap::new();
7126        };
7127        archived
7128            .rules
7129            .iter()
7130            .filter_map(|e| {
7131                let src = e.src_blob.as_slice();
7132                let dst = e.dst_blob.as_slice();
7133                if core_rules::hnsw::hnsw_blob_complete(src) == Some(false)
7134                    || core_rules::hnsw::hnsw_blob_complete(dst) == Some(false)
7135                {
7136                    Some((e.name.as_str().to_string(), (src.to_vec(), dst.to_vec())))
7137                } else {
7138                    None
7139                }
7140            })
7141            .collect()
7142    }
7143
7144    /// WAL-log rule deletion. Returns RuleNotFound if the rule does not exist.
7145    pub fn delete_rule(&mut self, name: &str) -> Result<()> {
7146        if self.read_only {
7147            return Err(GraphError::ReadOnly);
7148        }
7149        MutPreview::new(self).check_delete_rule(name)?;
7150        self.log_then_apply(WalRecord::DeleteRule { name: name.into() })
7151    }
7152
7153    /// Return a snapshot of all registered rules.
7154    pub fn rules(&self) -> Vec<RuleDef> {
7155        self.engine.rules().cloned().collect()
7156    }
7157
7158    // -----------------------------------------------------------------------
7159    // Rule suggestion API
7160    // -----------------------------------------------------------------------
7161
7162    /// Profile the database and suggest linking rules with previewed edge counts.
7163    ///
7164    /// Uses the default seed ([`core_rules::SUGGEST_DEFAULT_SEED`]) for deterministic
7165    /// sampling. Suggestions are sorted by estimated edge count (descending).
7166    /// **NO auto-accept** — call [`GraphDb::create_rule`] explicitly to apply.
7167    pub fn suggest_rules(&self) -> Vec<core_rules::RuleSuggestion> {
7168        self.suggest_rules_seeded(core_rules::SUGGEST_DEFAULT_SEED)
7169    }
7170
7171    /// Like [`suggest_rules`] but with a caller-supplied RNG seed for
7172    /// reproducibility. Same seed + same data = identical output.
7173    pub fn suggest_rules_seeded(&self, seed: u64) -> Vec<core_rules::RuleSuggestion> {
7174        self.suggest_rules_with_config(&core_rules::suggest::SuggestConfig::default(), seed)
7175            .suggestions
7176    }
7177
7178    /// [`suggest_rules_seeded`] with a fully custom [`SuggestConfig`].
7179    ///
7180    /// Returns a [`core_rules::SuggestReport`] that includes both the candidate list
7181    /// and a `truncated` flag indicating whether the global budget fired before all
7182    /// candidates were evaluated.
7183    pub fn suggest_rules_with_config(
7184        &self,
7185        config: &core_rules::suggest::SuggestConfig,
7186        seed: u64,
7187    ) -> core_rules::SuggestReport {
7188        use std::collections::BTreeMap;
7189
7190        // Collect (node_id, key) pairs per label, skipping tombstoned nodes.
7191        let mut label_nodes: BTreeMap<String, Vec<(u32, String)>> = BTreeMap::new();
7192        for id in 0..self.ids.len() as u32 {
7193            let Some(key) = self.ids.key_of(id) else {
7194                continue;
7195            };
7196            let Some(&sym) = self.labels.get(id as usize) else {
7197                continue;
7198            };
7199            if sym == u32::MAX {
7200                continue; // tombstoned
7201            }
7202            let Some(label) = self.syms.resolve(sym) else {
7203                continue;
7204            };
7205            label_nodes
7206                .entry(label.to_string())
7207                .or_default()
7208                .push((id, key.to_string()));
7209        }
7210
7211        let existing = self.rules();
7212        let pv = build_props_view(&self.props, &self.base);
7213        let all_fields: Vec<String> = pv.field_names();
7214
7215        core_rules::suggest::suggest_rules(
7216            &label_nodes,
7217            &|id, field| pv.get(id, field).map(|vr| vr.into_value()),
7218            &all_fields,
7219            &existing,
7220            config,
7221            seed,
7222        )
7223    }
7224
7225    /// Recompute a rule's derived edges from scratch. WAL-logged so un-trip
7226    /// plus later mutations replay identically (rebuild is a pure function
7227    /// of state).
7228    ///
7229    /// Only exit from the tripped latch: if the full desired set fits the
7230    /// budget, it is applied completely and `tripped` clears; if it still
7231    /// exceeds the budget, provenance is left untouched and `tripped` stays
7232    /// true. Counts as a fire evaluation (see [`RuleStats::fires`]).
7233    /// Unknown rule → `RuleNotFound`, nothing logged.
7234    pub fn rebuild_rule(&mut self, name: &str) -> Result<()> {
7235        if self.read_only {
7236            return Err(GraphError::ReadOnly);
7237        }
7238        if !self.engine.rules().any(|r| r.name == name) {
7239            return Err(GraphError::RuleNotFound { name: name.into() });
7240        }
7241        self.log_then_apply(WalRecord::RebuildRule { name: name.into() })
7242    }
7243
7244    // -----------------------------------------------------------------------
7245    // Materialized view API
7246    // -----------------------------------------------------------------------
7247
7248    /// Register a new materialized property view, backfill its values for all
7249    /// existing nodes, and WAL-log the definition.
7250    ///
7251    /// # Errors
7252    /// - `ReadOnly`: called on an as-of instance.
7253    /// - `RuleInvalid`: name collision, view_prop collision, or invalid def.
7254    pub fn create_view(&mut self, def: ViewDef) -> Result<()> {
7255        if self.read_only {
7256            return Err(GraphError::ReadOnly);
7257        }
7258        // Pre-validate before WAL write.
7259        def.validate()
7260            .map_err(|e| GraphError::RuleInvalid { detail: e })?;
7261        if self.view_store.has_view(&def.name) {
7262            return Err(GraphError::RuleInvalid {
7263                detail: format!("view {:?} already exists", def.name),
7264            });
7265        }
7266        if let Some(existing) = self.view_store.view_for_prop(&def.view_prop) {
7267            return Err(GraphError::RuleInvalid {
7268                detail: format!(
7269                    "view_prop {:?} is already used by view {:?}",
7270                    def.view_prop, existing
7271                ),
7272            });
7273        }
7274        let def_bytes = bincode::serialize(&def).map_err(|e| GraphError::Corrupt {
7275            detail: format!("serialize view: {e}"),
7276        })?;
7277        // Enable delta accumulation before the view is registered so subsequent
7278        // incremental edge events reach view maintenance from this point onward.
7279        // (The backfill inside create_view reads topo directly; it does not rely
7280        // on pending deltas.)
7281        self.engine.set_emit_deltas(true);
7282        self.log_then_apply(WalRecord::CreateView { def_bytes })
7283    }
7284
7285    /// Remove a named view and delete its values from every node.
7286    ///
7287    /// # Errors
7288    /// - `ReadOnly`: called on an as-of instance.
7289    /// - `RuleNotFound`: view does not exist.
7290    pub fn delete_view(&mut self, name: &str) -> Result<()> {
7291        if self.read_only {
7292            return Err(GraphError::ReadOnly);
7293        }
7294        if !self.view_store.has_view(name) {
7295            return Err(GraphError::RuleNotFound { name: name.into() });
7296        }
7297        let result = self.log_then_apply(WalRecord::DeleteView { name: name.into() });
7298        // After deletion, disable accumulation if no listeners remain.
7299        if !self.needs_emit_deltas() {
7300            self.engine.set_emit_deltas(false);
7301        }
7302        result
7303    }
7304
7305    /// Snapshot of all registered view definitions.
7306    pub fn views(&self) -> Vec<ViewDef> {
7307        self.view_store.views().cloned().collect()
7308    }
7309
7310    // -----------------------------------------------------------------------
7311    // Full-text-lite API
7312    // -----------------------------------------------------------------------
7313
7314    /// Enable full-text indexing for all nodes of `label` on property `field`.
7315    ///
7316    /// After this call, every subsequent write to `(label, field)` is reflected
7317    /// in the index incrementally.  Existing nodes are backfilled immediately.
7318    /// The declaration is persisted as a WAL record; the index itself is rebuilt
7319    /// from scratch on re-open (no snapshot format changes).
7320    ///
7321    /// # Errors
7322    /// - [`GraphError::ReadOnly`]: called on an as-of instance.
7323    /// - [`GraphError::RuleInvalid`]: `(label, field)` is already indexed.
7324    pub fn enable_fulltext(&mut self, label: &str, field: &str) -> Result<()> {
7325        if self.read_only {
7326            return Err(GraphError::ReadOnly);
7327        }
7328        if self.fulltext.is_enabled(label, field) {
7329            return Err(GraphError::RuleInvalid {
7330                detail: format!("full-text index for ({label:?}, {field:?}) already enabled"),
7331            });
7332        }
7333        self.log_then_apply(WalRecord::EnableFulltext {
7334            label: label.into(),
7335            field: field.into(),
7336        })
7337    }
7338
7339    /// Disable full-text indexing for `(label, field)` and drop its postings.
7340    ///
7341    /// # Errors
7342    /// - [`GraphError::ReadOnly`]: called on an as-of instance.
7343    /// - [`GraphError::RuleNotFound`]: `(label, field)` is not currently indexed.
7344    pub fn disable_fulltext(&mut self, label: &str, field: &str) -> Result<()> {
7345        if self.read_only {
7346            return Err(GraphError::ReadOnly);
7347        }
7348        if !self.fulltext.is_enabled(label, field) {
7349            return Err(GraphError::RuleNotFound {
7350                name: format!("fulltext({label},{field})"),
7351            });
7352        }
7353        self.log_then_apply(WalRecord::DisableFulltext {
7354            label: label.into(),
7355            field: field.into(),
7356        })
7357    }
7358
7359    /// Whether `(label, field)` is currently indexed for full-text search.
7360    pub fn is_fulltext_enabled(&self, label: &str, field: &str) -> bool {
7361        self.fulltext.is_enabled(label, field)
7362    }
7363
7364    /// Every `(label, field)` pair with a live full-text index, sorted.
7365    ///
7366    /// Note that [`GraphDb::search`] is keyed by field alone — a pair only
7367    /// declares which nodes are *indexed*, so callers that want to search
7368    /// everything indexed should query each distinct field once.
7369    pub fn fulltext_pairs(&self) -> Vec<(String, String)> {
7370        let mut v: Vec<(String, String)> = self.fulltext.enabled_pairs().cloned().collect();
7371        v.sort();
7372        v
7373    }
7374
7375    /// Enable an equality index for all nodes of `label` on scalar property
7376    /// `field`. Subsequent `WHERE n.field = value` lookups become O(matches)
7377    /// instead of an O(N_label) scan. Existing nodes are backfilled; the
7378    /// declaration persists via WAL and the postings rebuild on re-open.
7379    ///
7380    /// # Errors
7381    /// - [`GraphError::ReadOnly`]: called on an as-of instance.
7382    /// - [`GraphError::RuleInvalid`]: `(label, field)` is already indexed.
7383    pub fn enable_index(&mut self, label: &str, field: &str) -> Result<()> {
7384        if self.read_only {
7385            return Err(GraphError::ReadOnly);
7386        }
7387        if self.prop_index.is_enabled(label, field) {
7388            return Err(GraphError::RuleInvalid {
7389                detail: format!("property index for ({label:?}, {field:?}) already enabled"),
7390            });
7391        }
7392        self.log_then_apply(WalRecord::EnableIndex {
7393            label: label.into(),
7394            field: field.into(),
7395        })
7396    }
7397
7398    /// Disable the equality index for `(label, field)` and drop its postings.
7399    ///
7400    /// # Errors
7401    /// - [`GraphError::ReadOnly`]: called on an as-of instance.
7402    /// - [`GraphError::RuleNotFound`]: `(label, field)` is not currently indexed.
7403    pub fn disable_index(&mut self, label: &str, field: &str) -> Result<()> {
7404        if self.read_only {
7405            return Err(GraphError::ReadOnly);
7406        }
7407        if !self.prop_index.is_enabled(label, field) {
7408            return Err(GraphError::RuleNotFound {
7409                name: format!("index({label},{field})"),
7410            });
7411        }
7412        self.log_then_apply(WalRecord::DisableIndex {
7413            label: label.into(),
7414            field: field.into(),
7415        })
7416    }
7417
7418    /// Whether `(label, field)` currently has an equality index.
7419    pub fn is_index_enabled(&self, label: &str, field: &str) -> bool {
7420        self.prop_index.is_enabled(label, field)
7421    }
7422
7423    /// Start recording insert-count multiplicity on this store (§5.13).
7424    ///
7425    /// Adjacency stays a set and nothing about an existing read changes: a
7426    /// duplicate [`insert_edge`](Self::insert_edge) still returns `Ok(false)`
7427    /// and still leaves [`degree`](Self::degree) alone. What it gains is that
7428    /// the duplicate is *counted*, as the reserved edge property
7429    /// [`EDGE_COUNT_PROP`], readable through
7430    /// [`degree_multiplicity`](Self::degree_multiplicity).
7431    ///
7432    /// # This is a one-way step, and that is why it is a call
7433    ///
7434    /// The count is durable, so it is written to the WAL — as discriminant 23,
7435    /// which no release before v0.6.10 knows. A reader meeting an unknown WAL
7436    /// discriminant cannot know what the record would have changed, so it
7437    /// cannot degrade the way an unreadable index blob can. **After this call
7438    /// the store can no longer be read by an older binary, and there is no call
7439    /// that undoes it.** Gating the record behind this method is what keeps
7440    /// that step a decision an operator makes when they want the feature,
7441    /// rather than one everybody takes by upgrading.
7442    ///
7443    /// # It fails loudly, and that costs a snapshot
7444    ///
7445    /// An older binary does not refuse discriminant 23 — it truncates the WAL
7446    /// at it and, with `repair_wal`, persists the truncation. So this call also
7447    /// writes a **V10 snapshot**, a version no earlier release knows, and it
7448    /// writes it *first*: the snapshot is read before the WAL, so an older
7449    /// binary stops at `snapshot: unsupported version 10` with the WAL
7450    /// untouched. Taking the snapshot before appending the record is what makes
7451    /// the guard unconditional — the store is never, at any interruption point,
7452    /// carrying the record without the stamp that announces it.
7453    ///
7454    /// The snapshot keeps the WAL (`keep_wal: true`): opting in is not a
7455    /// compaction, and history reachable by [`open_at`](Self::open_at) stays
7456    /// reachable. On a large store the call therefore costs one full snapshot
7457    /// write.
7458    ///
7459    /// # What it costs a store that archives
7460    ///
7461    /// Writing `snapshot.bin` is also how the archive path decides whether the
7462    /// store may have a *genesis chain* — whether `open_at` can replay
7463    /// archive-resident commits from empty state. The rule is conservative: a
7464    /// snapshot that was already on disk might have been a truncating one, and
7465    /// once the handle that took it is gone this binary cannot tell. A
7466    /// `keep_wal` snapshot taken by **this** handle is the case where it can, so
7467    /// opting in and then archiving **in the same session** keeps the chain.
7468    ///
7469    /// Opting in, closing the store, and archiving in a *later* session does
7470    /// not — but that is the answer any store with a prior snapshot gets, not
7471    /// something this call causes. A store that wants the chain should take its
7472    /// first archive in the session that opted in.
7473    ///
7474    /// Calling it on a store that has already opted in writes nothing and
7475    /// returns `Ok(())`: an operator should not have to ask first.
7476    ///
7477    /// # This call is not atomic, and an `Err` does not undo it
7478    ///
7479    /// There is no rollback here, and there never was one. An `Err` means this
7480    /// handle stopped believing the store is opted in — `self.multiplicity` is
7481    /// reset, so this handle reports `false` from then on — and nothing more. It
7482    /// says nothing about what reached disk. Two reachable failures leave the
7483    /// opt-in standing:
7484    ///
7485    /// * **The declaration landed and only its fsync failed.** `log_then_apply`
7486    ///   appends, then syncs; a failed barrier leaves `MULTIPLICITY_ENABLED`
7487    ///   already in `wal.bin`. The next open replays it and the store is opted
7488    ///   in. No archive is involved — this one predates the recovery below.
7489    /// * **The declaration never landed, but the V10 snapshot did, on a store
7490    ///   that already had an archive.** The open-time recovery in
7491    ///   `load_from_disk` reads V10-beside-an-archive as an interrupted archive
7492    ///   sequence and opts the store in.
7493    ///
7494    /// So a failed call may leave the opt-in on disk immediately (the first
7495    /// case) or conjure it at the next open (the second), and nothing puts the
7496    /// store back out. Treat `Err` as "the outcome is unknown", not as "nothing
7497    /// happened".
7498    ///
7499    /// **This is safe, and the ordering is the reason.** The V10 stamp is
7500    /// written *before* the declaration, so every one of these intermediate
7501    /// states is one an older binary refuses by name rather than truncates at.
7502    /// The failure direction costs a refusal, never a commit. That ordering is
7503    /// the property worth protecting, not the atomicity this call never had.
7504    ///
7505    /// **To know where the store stands, ask the store.** Reopen it and call
7506    /// [`is_multiplicity_enabled`](Self::is_multiplicity_enabled); that is the
7507    /// only answer that accounts for what reached disk.
7508    ///
7509    /// The one case that really does leave the store opted out is a failure with
7510    /// no archive present and no record written: a stray V10 snapshot remains,
7511    /// costing an older reader a refusal it did not strictly need, and *that*
7512    /// store's next snapshot rewrites at V9.
7513    ///
7514    /// # Errors
7515    /// - [`GraphError::ReadOnly`]: called on an as-of instance.
7516    /// - Anything [`snapshot_with`](Self::snapshot_with) can return, and
7517    ///   anything the WAL append or its fsync can return. See the atomicity
7518    ///   section above for what the store is left holding.
7519    pub fn enable_multiplicity(&mut self) -> Result<()> {
7520        if self.read_only {
7521            return Err(GraphError::ReadOnly);
7522        }
7523        if self.multiplicity {
7524            return Ok(());
7525        }
7526        // The snapshot goes first, and the order is the guard.
7527        //
7528        // An older reader refuses a V10 snapshot by name and stops; it does not
7529        // refuse discriminant 23, it truncates the WAL at it. So the store must
7530        // never hold the record without the snapshot that announces it — not
7531        // even for the width of one fsync. Writing the snapshot before the
7532        // record makes the only reachable intermediate state "V10 snapshot, no
7533        // record", which is merely conservative: this binary reads it as a
7534        // store that has not opted in, and an older one refuses it.
7535        //
7536        // `keep_wal: true` because opting in is not a compaction: an operator
7537        // asking for multiplicity has not asked to lose the history `open_at`
7538        // can reach.
7539        self.multiplicity = true;
7540        let forced = self
7541            .snapshot_with(SnapshotOptions {
7542                keep_wal: true,
7543                ..SnapshotOptions::default()
7544            })
7545            .and_then(|()| self.log_then_apply(core_storage::wal::MULTIPLICITY_ENABLED));
7546        if forced.is_err() {
7547            // This handle stops believing it is opted in. That is all this line
7548            // does — it is not a rollback, and cannot be one: the declaration
7549            // may already be in `wal.bin` (the append succeeded and only the
7550            // fsync failed), and even when it is not, the V10 snapshot beside an
7551            // existing archive is enough for the open-time recovery to opt the
7552            // store in. See the "not atomic" section on this method.
7553            //
7554            // It fails in the safe direction either way: the V10 stamp reached
7555            // disk before anything a v0.6.9 reader would truncate at, so the
7556            // worst an interruption costs that reader is a refusal by name.
7557            self.multiplicity = false;
7558        }
7559        forced
7560    }
7561
7562    /// Whether this store records insert-count multiplicity.
7563    ///
7564    /// `false` on every store that has not called
7565    /// [`enable_multiplicity`](Self::enable_multiplicity) — which is every
7566    /// store that did not ask for it, including one upgraded from an earlier
7567    /// release.
7568    pub fn is_multiplicity_enabled(&self) -> bool {
7569        self.multiplicity
7570    }
7571
7572    /// How many times `(etype, src, dst)` has been inserted: the reserved
7573    /// `count` edge property, or 1 when it is absent.
7574    ///
7575    /// Answers 1 for a pair on a store that never opted in, which is the truth
7576    /// available there — the pair was inserted at least once, and the store
7577    /// kept no record of any second insert.
7578    fn edge_insert_count(&self, etype: u32, src: u32, dst: u32) -> u64 {
7579        match self.edge_props_view().get(etype, src, dst, EDGE_COUNT_PROP) {
7580            Some(Value::Int(n)) if n > 0 => n as u64,
7581            _ => 1,
7582        }
7583    }
7584
7585    /// The `SetEdgeCount` record a duplicate insert of `(edge_type, src_key,
7586    /// dst_key)` should log, or `None` when nothing should be written.
7587    ///
7588    /// `None` when the store has not opted in, so **no discriminant-23 record
7589    /// is written at all** — the gate the whole feature rests on.
7590    ///
7591    /// The other two `None`s are unreachable from the one caller. This is the
7592    /// single-mutation path, where `prepare_insert_edge` has already refused a
7593    /// missing endpoint and an existing pair's edge type is necessarily
7594    /// interned. A batch is the case where a pair's endpoints and type can all
7595    /// be created by the same frame, and it does not come through here: it
7596    /// queues a [`PlannedRec::DuplicateCount`] and names the count in the dense
7597    /// rewrite, which is the only pass that knows the frame's own ids.
7598    fn edge_count_record(
7599        &self,
7600        edge_type: &str,
7601        src_key: &str,
7602        dst_key: &str,
7603    ) -> Option<WalRecord> {
7604        if !self.multiplicity {
7605            return None;
7606        }
7607        let etype = self.syms.get(edge_type)?;
7608        let src = self.ids.get(src_key)?;
7609        let dst = self.ids.get(dst_key)?;
7610        Some(WalRecord::SetEdgeCount {
7611            etype,
7612            src,
7613            dst,
7614            count: self.edge_insert_count(etype, src, dst).saturating_add(1),
7615        })
7616    }
7617
7618    /// Search a full-text-indexed field.
7619    ///
7620    /// Returns `(node_key, match_count)` pairs sorted by match_count descending,
7621    /// ties broken by key (lexicographic).  Tombstoned nodes are excluded.
7622    ///
7623    /// **Query syntax:**
7624    /// - Space-separated terms are AND'd: `"foo bar"` requires both.
7625    /// - `OR` between terms forms disjunction: `"foo OR bar"` matches either.
7626    /// - Trailing `*` on a term is a prefix match: `"rust*"` matches `rustlang`, `rusty`.
7627    /// - `AND` keyword is accepted explicitly and is the default.
7628    /// - Tokenization is unicode-alphanumeric (same as index time); case-insensitive.
7629    ///
7630    /// **Unindexed field:** returns `Ok(vec![])` if `field` is not indexed.
7631    /// Pin: this is the documented, tested, stable behavior for v1.
7632    ///
7633    /// **Memory / performance:** O(postings) lookup; no scan.  The index is
7634    /// in-memory and proportional to total indexed text across all enabled fields.
7635    ///
7636    /// **v2 grammar:** supports `"phrase"`, `-negation`, `prefix*`, `OR`, `AND`.
7637    /// Results are BM25-scored (k1=1.2, b=0.75) and sorted by score descending,
7638    /// key ascending for deterministic tiebreaking.
7639    pub fn search(&self, field: &str, query: &str) -> Vec<(String, f64)> {
7640        // Resolve node_ids to keys (excluding tombstones) then re-sort by
7641        // (score DESC, key ASC) to give a deterministic, key-lexicographic
7642        // tiebreak.  FulltextIndex::search sorts by (score DESC, node_id ASC)
7643        // which diverges from key order when nodes were not inserted in key-lex order.
7644        let mut results: Vec<(String, f64)> = self
7645            .fulltext
7646            .search(field, query, 0)
7647            .into_iter()
7648            .filter_map(|(id, score)| self.ids.key_of(id).map(|key| (key.to_string(), score)))
7649            .collect();
7650        results.sort_by(|a, b| {
7651            b.1.partial_cmp(&a.1)
7652                .unwrap_or(std::cmp::Ordering::Equal)
7653                .then(a.0.cmp(&b.0))
7654        });
7655        results
7656    }
7657
7658    /// [`search`](Self::search), stopping at the `k` best hits.
7659    ///
7660    /// Same ranking and the same deterministic tiebreak, but the index drops
7661    /// everything past `k` before any key is resolved, so a caller that wants
7662    /// the top few out of a field that matched thousands does not pay to
7663    /// materialise and re-sort the tail. `k == 0` means no limit, exactly as
7664    /// [`search`](Self::search) behaves.
7665    ///
7666    /// The BM25 scoring itself is not bounded by `k` — every candidate is
7667    /// scored either way — so this trims the resolve and the sort, not the
7668    /// search.
7669    pub fn search_top(&self, field: &str, query: &str, k: usize) -> Vec<(String, f64)> {
7670        // A tombstoned id resolves to nothing, so asking the index for exactly
7671        // `k` could return fewer. Over-fetching a little and truncating after
7672        // the filter keeps the count right without unbounding the call.
7673        let want = if k == 0 { 0 } else { k.saturating_mul(2) };
7674        let mut results: Vec<(String, f64)> = self
7675            .fulltext
7676            .search(field, query, want)
7677            .into_iter()
7678            .filter_map(|(id, score)| self.ids.key_of(id).map(|key| (key.to_string(), score)))
7679            .collect();
7680        results.sort_by(|a, b| {
7681            b.1.partial_cmp(&a.1)
7682                .unwrap_or(std::cmp::Ordering::Equal)
7683                .then(a.0.cmp(&b.0))
7684        });
7685        if k > 0 {
7686            results.truncate(k);
7687        }
7688        results
7689    }
7690
7691    /// Hybrid search: Reciprocal Rank Fusion (RRF) over fulltext + vector results.
7692    ///
7693    /// Takes up to `4*k` fulltext hits for `(text_field, query_text)` and up to
7694    /// `4*k` vector hits for `(vector_field, query_vec, min=0.0)`, then fuses
7695    /// them with RRF using a fixed constant of 60.
7696    ///
7697    /// ```text
7698    /// score(d) = Σ  1 / (60 + rank_i(d))    (rank 1-based per list)
7699    /// ```
7700    ///
7701    /// Returns the top `k` nodes by fused score, ties broken by node key
7702    /// ascending (deterministic).
7703    ///
7704    /// # Vector leg fallback
7705    ///
7706    /// When `query_vec` is empty the vector leg is skipped entirely and
7707    /// results are ranked by the text list alone through the same RRF path
7708    /// (each text result scores `1/(60 + rank)` from that single list).
7709    ///
7710    /// When `label` is `None`, the vector leg **always** returns empty results.
7711    /// Internally `label` is mapped to `""`, which does not match any rule-created
7712    /// HNSW index (all such indexes are keyed to a specific non-empty label), and
7713    /// the brute-force fallback finds no nodes with an empty label.  The fused
7714    /// ranking is therefore text-only in this case.
7715    pub fn search_hybrid(
7716        &self,
7717        text_field: &str,
7718        query_text: &str,
7719        vector_field: &str,
7720        query_vec: &[f64],
7721        label: Option<&str>,
7722        k: usize,
7723    ) -> Vec<(String, f64)> {
7724        self.search_hybrid_inner(
7725            text_field,
7726            query_text,
7727            vector_field,
7728            query_vec,
7729            label,
7730            k,
7731            None,
7732        )
7733    }
7734
7735    /// [`search_hybrid`](Self::search_hybrid) with **each leg** filtered to the
7736    /// mask before the fusion.
7737    ///
7738    /// Filtering the fused list afterwards would quietly return fewer than `k`.
7739    /// Each leg over-fetches `4*k` candidates, so when the visible nodes rank
7740    /// below `4*k` hidden ones neither leg carries them into the fusion at all
7741    /// and the post-filter has nothing left to keep. Filtering first spends the
7742    /// `4*k` on **visible** hits, so a scoped call is as long as the corpus it
7743    /// can see allows.
7744    ///
7745    /// The ranks that enter RRF are therefore the ranks of the visible corpus,
7746    /// not the visible entries of the store-wide ranking. The constant stays 60
7747    /// and the tiebreak stays key-ascending.
7748    #[allow(clippy::too_many_arguments)]
7749    pub fn search_hybrid_scoped(
7750        &self,
7751        text_field: &str,
7752        query_text: &str,
7753        vector_field: &str,
7754        query_vec: &[f64],
7755        label: Option<&str>,
7756        k: usize,
7757        mask: &crate::mask::NodeMask,
7758    ) -> Vec<(String, f64)> {
7759        self.search_hybrid_inner(
7760            text_field,
7761            query_text,
7762            vector_field,
7763            query_vec,
7764            label,
7765            k,
7766            Some(mask),
7767        )
7768    }
7769
7770    /// The body shared by [`search_hybrid`](Self::search_hybrid) and
7771    /// [`search_hybrid_scoped`](Self::search_hybrid_scoped). `mask = None` is
7772    /// the unscoped contract unchanged: the filter below is then a no-op and
7773    /// the vector leg is the same unmasked call it has always been.
7774    #[allow(clippy::too_many_arguments)]
7775    fn search_hybrid_inner(
7776        &self,
7777        text_field: &str,
7778        query_text: &str,
7779        vector_field: &str,
7780        query_vec: &[f64],
7781        label: Option<&str>,
7782        k: usize,
7783        mask: Option<&crate::mask::NodeMask>,
7784    ) -> Vec<(String, f64)> {
7785        use std::collections::HashMap;
7786
7787        const RRF_K: f64 = 60.0;
7788        let pool = 4 * k;
7789
7790        // Accumulate per-node RRF scores.
7791        let mut scores: HashMap<String, f64> = HashMap::new();
7792
7793        // Text leg. The mask bites on the candidates, before `take(pool)`, so
7794        // the over-fetch is a budget of visible hits rather than one a hidden
7795        // prefix can exhaust.
7796        let text_hits = self.search(text_field, query_text);
7797        let visible_text = text_hits
7798            .into_iter()
7799            .filter(|(key, _count)| mask.is_none_or(|m| m.contains_node(self, key)));
7800        for (rank0, (key, _count)) in visible_text.take(pool).enumerate() {
7801            let rank = (rank0 + 1) as f64;
7802            *scores.entry(key).or_insert(0.0) += 1.0 / (RRF_K + rank);
7803        }
7804
7805        // Vector leg (skipped when query_vec is empty). The masked variant
7806        // applies the mask before its own k-truncation, for the same reason.
7807        if !query_vec.is_empty() {
7808            // `ExactnessCaller::Hybrid`: the leg is the same one
7809            // `find_similar_vector_masked` runs, but the advice its warning
7810            // gives has to fit *this* signature, which has no `exact`.
7811            let vec_hits = self
7812                .find_similar_vector_as(
7813                    vector_field,
7814                    label,
7815                    query_vec,
7816                    pool,
7817                    0.0,
7818                    mask,
7819                    None,
7820                    false,
7821                    ExactnessCaller::Hybrid,
7822                )
7823                .expect("find_similar_vector_as is infallible without where_");
7824            for (rank0, (key, _sim)) in vec_hits.into_iter().enumerate() {
7825                let rank = (rank0 + 1) as f64;
7826                *scores.entry(key).or_insert(0.0) += 1.0 / (RRF_K + rank);
7827            }
7828        }
7829
7830        // Sort: score DESC, then key ASC for deterministic tie-breaking.
7831        let mut ranked: Vec<(String, f64)> = scores.into_iter().collect();
7832        ranked.sort_by(|a, b| {
7833            b.1.partial_cmp(&a.1)
7834                .unwrap_or(std::cmp::Ordering::Equal)
7835                .then(a.0.cmp(&b.0))
7836        });
7837        ranked.truncate(k);
7838        ranked
7839    }
7840
7841    /// For DST/testing: scratch BM25 search over live nodes without the index.
7842    /// Walks every live node, re-stems field tokens, computes corpus stats, and
7843    /// returns BM25-ranked results.
7844    ///
7845    /// The oracle: the ordered key list of `search(field, q)` must equal that of
7846    /// `scratch_search(field, q)` at every quiescent state.
7847    #[doc(hidden)]
7848    pub fn scratch_search(&self, field: &str, query: &str) -> Vec<(String, f64)> {
7849        use core_storage::fulltext::{parse_query, value_tokens_stemmed_with_positions};
7850        use std::collections::BTreeMap;
7851
7852        let groups = parse_query(query);
7853        if groups.is_empty() {
7854            return vec![];
7855        }
7856
7857        // --- Pass 1: collect all live indexed nodes with stemmed token data ---
7858        struct NodeData {
7859            key: String,
7860            /// stemmed_token → positions (sorted)
7861            tokens: BTreeMap<String, Vec<u32>>,
7862            dl: u32,
7863        }
7864
7865        let mut nodes: Vec<NodeData> = Vec::new();
7866        for id in 0..self.ids.len() as u32 {
7867            let Some(key) = self.ids.key_of(id) else {
7868                continue;
7869            };
7870            let Some(&sym) = self.labels.get(id as usize) else {
7871                continue;
7872            };
7873            if sym == u32::MAX {
7874                continue;
7875            }
7876            let label = match self.syms.resolve(sym) {
7877                Some(l) => l,
7878                None => continue,
7879            };
7880            if !self.fulltext.is_enabled(label, field) {
7881                continue;
7882            }
7883            let Some(value) = self.props_view().get(id, field).map(|vr| vr.into_value()) else {
7884                continue;
7885            };
7886            // Use value_tokens_stemmed_with_positions so list elements are
7887            // separated by POSITION_GAP — identical to the index path, which
7888            // prevents phrase queries from matching across element boundaries.
7889            let stemmed_with_pos = match &value {
7890                Value::Str(_) | Value::List(_) => value_tokens_stemmed_with_positions(&value),
7891                _ => continue,
7892            };
7893            let dl = stemmed_with_pos.len() as u32;
7894            let mut tok_map: BTreeMap<String, Vec<u32>> = BTreeMap::new();
7895            for (tok, pos) in stemmed_with_pos {
7896                tok_map.entry(tok).or_default().push(pos);
7897            }
7898            nodes.push(NodeData {
7899                key: key.to_string(),
7900                tokens: tok_map,
7901                dl,
7902            });
7903        }
7904
7905        if nodes.is_empty() {
7906            return vec![];
7907        }
7908
7909        // --- BM25 corpus stats ---
7910        let n = nodes.len() as f64;
7911        let avg_dl: f64 = nodes.iter().map(|nd| nd.dl as f64).sum::<f64>() / n;
7912        // df per stemmed token across all live indexed nodes.
7913        let mut df_map: BTreeMap<&str, f64> = BTreeMap::new();
7914        for nd in &nodes {
7915            for tok in nd.tokens.keys() {
7916                *df_map.entry(tok.as_str()).or_insert(0.0) += 1.0;
7917            }
7918        }
7919
7920        const K1: f64 = 1.2;
7921        const B: f64 = 0.75;
7922
7923        // --- Pass 2: score each node against each OR-group ---
7924        let mut results: Vec<(String, f64)> = Vec::new();
7925        for nd in &nodes {
7926            let dl = nd.dl as f64;
7927            let mut total_score = 0.0f64;
7928
7929            'group: for group in &groups {
7930                let mut group_score = 0.0f64;
7931
7932                for term in group {
7933                    if term.negated {
7934                        // Negated: if doc has this stemmed token → group fails.
7935                        let present = if term.prefix {
7936                            nd.tokens.keys().any(|t| t.starts_with(term.token.as_str()))
7937                        } else {
7938                            nd.tokens.contains_key(term.token.as_str())
7939                        };
7940                        if present {
7941                            continue 'group;
7942                        }
7943                        continue;
7944                    }
7945                    if term.prefix {
7946                        // Prefix: sum BM25 for all matching stemmed tokens.
7947                        let mut prefix_matched = false;
7948                        for (tok, positions) in &nd.tokens {
7949                            if tok.starts_with(term.token.as_str()) {
7950                                let tf = positions.len() as f64;
7951                                let df = df_map.get(tok.as_str()).copied().unwrap_or(1.0);
7952                                let idf = ((n - df + 0.5) / (df + 0.5) + 1.0).ln();
7953                                let tf_norm =
7954                                    tf * (K1 + 1.0) / (tf + K1 * (1.0 - B + B * dl / avg_dl));
7955                                group_score += idf * tf_norm;
7956                                prefix_matched = true;
7957                            }
7958                        }
7959                        if !prefix_matched {
7960                            continue 'group;
7961                        }
7962                    } else {
7963                        // term.token is already stemmed by parse_query; use directly.
7964                        match nd.tokens.get(term.token.as_str()) {
7965                            None => continue 'group,
7966                            Some(positions) => {
7967                                let tf = positions.len() as f64;
7968                                let df = df_map.get(term.token.as_str()).copied().unwrap_or(1.0);
7969                                let idf = ((n - df + 0.5) / (df + 0.5) + 1.0).ln();
7970                                let tf_norm =
7971                                    tf * (K1 + 1.0) / (tf + K1 * (1.0 - B + B * dl / avg_dl));
7972                                group_score += idf * tf_norm;
7973                            }
7974                        }
7975                    }
7976                }
7977
7978                if group_score > 0.0 {
7979                    total_score += group_score;
7980                }
7981            }
7982
7983            if total_score > 0.0 {
7984                results.push((nd.key.clone(), total_score));
7985            }
7986        }
7987
7988        results.sort_by(|a, b| {
7989            b.1.partial_cmp(&a.1)
7990                .unwrap_or(std::cmp::Ordering::Equal)
7991                .then(a.0.cmp(&b.0))
7992        });
7993        results
7994    }
7995
7996    /// Return the current view-maintained value of `view_prop` for node `key`.
7997    /// Equivalent to `get_prop` but documents that it reads a view-managed column.
7998    pub fn get_view_prop(&self, key: &str, view_prop: &str) -> Option<Value> {
7999        let id = self.ids.get(key)?;
8000        self.props_view()
8001            .get(id, view_prop)
8002            .map(|vr| vr.into_value())
8003    }
8004
8005    /// For testing / DST oracle: scratch recompute of a view value for one node.
8006    ///
8007    /// Returns `None` if the node does not exist, the view does not exist, or
8008    /// the view has no result for the node (e.g. Avg with no qualifying neighbors).
8009    #[doc(hidden)]
8010    pub fn scratch_view_value(&self, key: &str, view_name: &str) -> Option<Value> {
8011        let node = self.ids.get(key)?;
8012        let def = self.view_store.views().find(|v| v.name == view_name)?;
8013        // Use TopologyView so that NeighborAgg sees base + overlay edges
8014        // without materialising a temporary Topology (I1).
8015        let topo_view = self.topo_view();
8016        core_rules::views::compute_view_value(
8017            def,
8018            node,
8019            self.props_view(),
8020            &topo_view,
8021            &self.ids,
8022            &self.syms,
8023            &self.labels,
8024        )
8025    }
8026
8027    // -----------------------------------------------------------------------
8028    // Graph algorithm API
8029    // -----------------------------------------------------------------------
8030
8031    /// Run PageRank over the unified topology (manual + derived edges).
8032    ///
8033    /// Returns a [`PageRankReport`] with scores sorted descending (ties: key
8034    /// ascending).  Set `config.edge_type` to restrict to one edge type.
8035    /// `config.converged` is `true` only when the power iteration converged
8036    /// within `config.max_iters` and within any time budget.
8037    pub fn pagerank(&self, config: &crate::algo::PageRankConfig) -> crate::algo::PageRankReport {
8038        let topo = build_topo_view(&self.topo, &self.base);
8039        let edge_props = self.edge_props_view();
8040        crate::algo::pagerank(
8041            &topo,
8042            &self.ids,
8043            &self.syms,
8044            &self.labels,
8045            &edge_props,
8046            config,
8047        )
8048    }
8049
8050    /// Weakly-connected components over the unified topology (treated as
8051    /// undirected regardless of how edges were inserted).
8052    ///
8053    /// Component IDs are the key of the smallest member in the component
8054    /// (deterministic).  Result sorted by (component_id, key).
8055    pub fn connected_components(&self, config: &crate::algo::WccConfig) -> crate::algo::WccReport {
8056        let topo = build_topo_view(&self.topo, &self.base);
8057        let edge_props = self.edge_props_view();
8058        crate::algo::wcc(
8059            &topo,
8060            &self.ids,
8061            &self.syms,
8062            &self.labels,
8063            &edge_props,
8064            config,
8065        )
8066    }
8067
8068    /// Degree centrality for every live node.
8069    ///
8070    /// `direction`: `AlgoDir::Out` = out-degree, `AlgoDir::In` = in-degree,
8071    /// `AlgoDir::Both` = out + in (total directed degree).
8072    ///
8073    /// For one-shot ranking use this; for a live property updated on every
8074    /// write, create a Degree materialized view instead (see `docs/site/algorithms.md`).
8075    pub fn degree_centrality(
8076        &self,
8077        config: &crate::algo::DegreeConfig,
8078    ) -> crate::algo::DegreeReport {
8079        let topo = build_topo_view(&self.topo, &self.base);
8080        let edge_props = self.edge_props_view();
8081        crate::algo::degree_centrality(
8082            &topo,
8083            &self.ids,
8084            &self.syms,
8085            &self.labels,
8086            &edge_props,
8087            config,
8088        )
8089    }
8090
8091    /// Louvain community detection over the unified topology (undirected).
8092    ///
8093    /// See [`crate::algo::LouvainConfig`] for edge-type/weight/label
8094    /// restriction and [`crate::algo::CommunityReport`] for the shape of the
8095    /// result (communities sorted size-desc, then smallest member key asc).
8096    pub fn communities(&self, config: &crate::algo::LouvainConfig) -> crate::algo::CommunityReport {
8097        let topo = build_topo_view(&self.topo, &self.base);
8098        let edge_props = self.edge_props_view();
8099        crate::algo::louvain(
8100            &topo,
8101            &self.ids,
8102            &self.syms,
8103            &self.labels,
8104            &edge_props,
8105            config,
8106        )
8107    }
8108
8109    /// Write a vector of `(node_key, score)` pairs as `prop_name` on each node,
8110    /// atomically via a single write-batch (one WAL frame, one fsync).
8111    ///
8112    /// # Errors
8113    /// - [`GraphError::ReadOnly`]: called on an as-of instance.
8114    /// - [`GraphError::RuleInvalid`]: `prop_name` is managed by an existing view
8115    ///   (collision check mirrors `create_view`).
8116    /// - [`GraphError::KeyNotFound`]: a key in `scores` does not exist as a live node.
8117    pub fn write_scores(&mut self, prop_name: &str, scores: &[(String, f64)]) -> Result<()> {
8118        if self.read_only {
8119            return Err(GraphError::ReadOnly);
8120        }
8121        // Collision check: refuse if prop_name is view-managed.
8122        if let Some(view_name) = self.view_store.view_for_prop(prop_name) {
8123            return Err(GraphError::RuleInvalid {
8124                detail: format!(
8125                    "prop {:?} is managed by view {:?} and cannot be written as scores",
8126                    prop_name, view_name
8127                ),
8128            });
8129        }
8130        // Refuse if prop_name is a view name itself (confusing namespace collision).
8131        if self.view_store.has_view(prop_name) {
8132            return Err(GraphError::RuleInvalid {
8133                detail: format!(
8134                    "prop_name {:?} collides with an existing view name",
8135                    prop_name
8136                ),
8137            });
8138        }
8139        // Write all scores in a single crash-atomic batch.
8140        self.write_batch(|b| {
8141            for (key, score) in scores {
8142                b.set_prop(key, prop_name, Value::Float(*score));
8143            }
8144        })?;
8145        Ok(())
8146    }
8147
8148    /// Return the value of `field` for the node with key `key`, or `None` if
8149    /// the node or field is absent.  Reads through the overlay-over-base
8150    /// `ColumnsView`, materialising base values on demand (zero heap cost for
8151    /// overlay hits; one clone per base hit).
8152    pub fn get_prop(&self, key: &str, field: &str) -> Option<Value> {
8153        let id = self.ids.get(key)?;
8154        self.props_view().get(id, field).map(|vr| vr.into_value())
8155    }
8156
8157    pub fn has_node(&self, key: &str) -> bool {
8158        self.ids.get(key).is_some()
8159    }
8160
8161    /// Borrow the raw id map. Used by `NodeMask::from_keys` to resolve keys.
8162    pub(crate) fn ids(&self) -> &IdMap {
8163        &self.ids
8164    }
8165
8166    // -----------------------------------------------------------------------
8167    // Namespaces
8168    // -----------------------------------------------------------------------
8169
8170    /// The index `name` already has in `ns_names`, if any.
8171    fn ns_index_of(&self, name: &str) -> Option<u32> {
8172        self.ns_names
8173            .iter()
8174            .position(|n| n == name)
8175            .map(|i| i as u32)
8176    }
8177
8178    /// The index for `name`, appending it to `ns_names` when it is new.
8179    ///
8180    /// The table holds one entry per distinct namespace in the store — a
8181    /// tenant count, not a node count — so the linear scan is cheaper than a
8182    /// map and keeps `namespaces()` allocation-free of a second index.
8183    fn ns_index_for(&mut self, name: &str) -> u32 {
8184        match self.ns_index_of(name) {
8185            Some(i) => i,
8186            None => {
8187                self.ns_names.push(name.to_string());
8188                (self.ns_names.len() - 1) as u32
8189            }
8190        }
8191    }
8192
8193    /// The namespace name at `idx`, or [`NS_DEFAULT`] for an index this handle
8194    /// does not know (unreachable; the default is the narrowing answer).
8195    fn ns_name(&self, idx: u32) -> &str {
8196        self.ns_names
8197            .get(idx as usize)
8198            .map(String::as_str)
8199            .unwrap_or(NS_DEFAULT)
8200    }
8201
8202    /// The namespace index of dense node `id`, defaulting for an id with no
8203    /// entry (a node inserted before this handle rebuilt the array cannot
8204    /// exist: every insert path maintains it).
8205    fn node_ns_idx(&self, id: u32) -> u32 {
8206        self.node_ns
8207            .get(id as usize)
8208            .copied()
8209            .unwrap_or(NS_DEFAULT_IDX)
8210    }
8211
8212    /// File node `id` under namespace `name`, growing `node_ns` as `labels`
8213    /// grows. Called from `apply` for every node insert, live and replayed.
8214    fn set_node_ns(&mut self, id: u32, name: &str) {
8215        let idx = if name == NS_DEFAULT {
8216            NS_DEFAULT_IDX
8217        } else {
8218            self.ns_index_for(name)
8219        };
8220        if self.node_ns.len() <= id as usize {
8221            self.node_ns.resize(id as usize + 1, NS_DEFAULT_IDX);
8222        }
8223        self.node_ns[id as usize] = idx;
8224    }
8225
8226    /// Rebuild `node_ns` from the `ns` column — one pass, at the end of an
8227    /// open or a reload, after the snapshot is restored and the WAL replayed.
8228    ///
8229    /// A store with no `ns` column reads nothing: the column-name check fails
8230    /// and the vector is filled with one constant.
8231    fn rebuild_node_ns(&mut self) {
8232        let total = self.ids.len();
8233        self.ns_names.truncate(1);
8234        self.node_ns.clear();
8235        self.node_ns.resize(total, NS_DEFAULT_IDX);
8236        let has_ns_column = {
8237            let cv = self.props_view();
8238            cv.field_names().iter().any(|f| f == NS_PROP)
8239        };
8240        if !has_ns_column {
8241            return;
8242        }
8243        // Collected first so the props view is released before `ns_index_for`
8244        // takes `&mut self`.
8245        let named: Vec<(u32, String)> = {
8246            let cv = self.props_view();
8247            (0..total as u32)
8248                .filter_map(|id| match cv.get(id, NS_PROP).map(|vr| vr.into_value()) {
8249                    Some(Value::Str(s)) if s != NS_DEFAULT => Some((id, s)),
8250                    _ => None,
8251                })
8252                .collect()
8253        };
8254        for (id, name) in named {
8255            let idx = self.ns_index_for(&name);
8256            self.node_ns[id as usize] = idx;
8257        }
8258    }
8259
8260    /// Every namespace with at least one live node, in name order.
8261    ///
8262    /// `["default"]` on any store that has never named a namespace, including
8263    /// an empty one: a store is always at least its default namespace.
8264    pub fn namespaces(&self) -> Vec<String> {
8265        let mut out: BTreeSet<&str> = BTreeSet::new();
8266        out.insert(NS_DEFAULT);
8267        for (id, &idx) in self.node_ns.iter().enumerate() {
8268            if idx == NS_DEFAULT_IDX || !self.is_live_node(id as u32) {
8269                continue;
8270            }
8271            out.insert(self.ns_name(idx));
8272        }
8273        out.into_iter().map(str::to_string).collect()
8274    }
8275
8276    /// The namespace of `key`, or `None` when the key names no live node.
8277    pub fn namespace_of(&self, key: &str) -> Option<String> {
8278        let id = self.ids.get(key)?;
8279        if !self.is_live_node(id) {
8280            return None;
8281        }
8282        Some(self.ns_name(self.node_ns_idx(id)).to_string())
8283    }
8284
8285    /// Every live node in `namespace`, as a visibility mask.
8286    ///
8287    /// Built off `node_ns` on whichever handle this is, so on a temporal handle
8288    /// it is the namespace's membership at that commit. A name no node uses
8289    /// gives an empty mask — a namespace scope never widens.
8290    pub fn mask_for_namespace(&self, namespace: &str) -> crate::mask::NodeMask {
8291        let Some(idx) = self.ns_index_of(namespace) else {
8292            return crate::mask::NodeMask::from_ids(std::collections::HashSet::new());
8293        };
8294        let visible: std::collections::HashSet<u32> = (0..self.ids.len() as u32)
8295            .filter(|&id| self.node_ns_idx(id) == idx && self.is_live_node(id))
8296            .collect();
8297        crate::mask::NodeMask::from_ids(visible)
8298    }
8299
8300    /// Live-node test used by the namespace accessors: a deleted node keeps its
8301    /// dense id and its `node_ns` slot, and the label sentinel is what marks it
8302    /// gone — the same test `mask_for_role`'s label leg applies implicitly.
8303    fn is_live_node(&self, id: u32) -> bool {
8304        self.labels
8305            .get(id as usize)
8306            .is_some_and(|&sym| sym != u32::MAX)
8307            && self.ids.key_of(id).is_some()
8308    }
8309
8310    /// Per-namespace live node counts for [`Stats`], in name order.
8311    fn namespace_stats(&self) -> Vec<NamespaceStats> {
8312        let mut counts: BTreeMap<&str, usize> = BTreeMap::new();
8313        counts.insert(NS_DEFAULT, 0);
8314        for id in 0..self.ids.len() as u32 {
8315            if !self.is_live_node(id) {
8316                continue;
8317            }
8318            *counts
8319                .entry(self.ns_name(self.node_ns_idx(id)))
8320                .or_insert(0) += 1;
8321        }
8322        counts
8323            .into_iter()
8324            .filter(|&(name, n)| n > 0 || name == NS_DEFAULT)
8325            .map(|(name, nodes_live)| NamespaceStats {
8326                name: name.to_string(),
8327                nodes_live,
8328            })
8329            .collect()
8330    }
8331
8332    /// The namespace a create-class op would put its node in: the `ns` entry of
8333    /// the props it carries, normalised, with absent meaning [`NS_DEFAULT`].
8334    fn created_namespace<'a>(key: &str, props: &'a [(String, Value)]) -> Result<&'a str> {
8335        Ok(namespace_of_value(Self::sole_ns_entry(key, props)?))
8336    }
8337
8338    /// The one `ns` entry in a node's props, or `None` when it carries none.
8339    ///
8340    /// A props list naming `ns` twice is refused. Without that refusal the
8341    /// write path and the authorisation path can read the same list
8342    /// differently — one taking the first entry, the other the last — and
8343    /// `CREATE (n:L {ns: 'mine', ns: 'theirs'})` lands a node in a namespace
8344    /// the role was checked against the other of. One entry is the only shape
8345    /// where "the node's namespace" is a single fact, so it is the only shape
8346    /// accepted, and every reader of it agrees by construction.
8347    fn sole_ns_entry<'a>(key: &str, props: &'a [(String, Value)]) -> Result<Option<&'a Value>> {
8348        let mut found: Option<&'a Value> = None;
8349        for (field, value) in props {
8350            if field != NS_PROP {
8351                continue;
8352            }
8353            if found.is_some() {
8354                return Err(GraphError::RuleInvalid {
8355                    detail: format!(
8356                        "node {key}: {NS_PROP} is given more than once; a node has exactly \
8357                         one namespace"
8358                    ),
8359                });
8360            }
8361            found = Some(value);
8362        }
8363        Ok(found)
8364    }
8365
8366    /// The definition of the role a write authorisation names.
8367    ///
8368    /// `None` when `roles.json` was corrupt at open or the role has since been
8369    /// removed — neither can reach a write, because the authorisation carries a
8370    /// mask `mask_for_role` already resolved for that name.
8371    fn role_def_for(&self, role: &str) -> Option<&RoleDef> {
8372        self.roles.as_ref()?.iter().find(|r| r.name == role)
8373    }
8374
8375    /// Validate the `ns` entry of a node's props and drop an explicit default.
8376    ///
8377    /// Runs on the write path only (see `rewrite_wal_dense`), never on replay:
8378    /// a record that reached the WAL was already accepted here.
8379    fn normalise_insert_ns(
8380        key: &str,
8381        props: Vec<(String, Value)>,
8382    ) -> Result<(Vec<(String, Value)>, String)> {
8383        // One `ns` or none: this is where that is enforced, so every later
8384        // reader of the list — the authorisation gate, the two `apply` arms,
8385        // `node_ns` — is looking at a single entry and cannot disagree about
8386        // which one counts.
8387        Self::sole_ns_entry(key, &props)?;
8388        let mut name = NS_DEFAULT.to_string();
8389        let mut out = Vec::with_capacity(props.len());
8390        for (field, value) in props {
8391            if field != NS_PROP {
8392                out.push((field, value));
8393                continue;
8394            }
8395            let Value::Str(ref s) = value else {
8396                return Err(GraphError::RuleInvalid {
8397                    detail: format!(
8398                        "node {key}: {NS_PROP} must be a string naming a namespace, \
8399                         got {value:?}"
8400                    ),
8401                });
8402            };
8403            if !valid_namespace(s) {
8404                return Err(GraphError::RuleInvalid {
8405                    detail: format!(
8406                        "node {key}: {s:?} is not a valid namespace name — 1 to {NS_MAX_LEN} \
8407                         characters of [A-Za-z0-9_.-]"
8408                    ),
8409                });
8410            }
8411            name = s.clone();
8412            // An explicit default stores nothing, so a single-tenant store
8413            // never grows an `ns` column.
8414            if name != NS_DEFAULT {
8415                out.push((field, value));
8416            }
8417        }
8418        Ok((out, name))
8419    }
8420
8421    // -----------------------------------------------------------------------
8422    // RBAC role resolution
8423    // -----------------------------------------------------------------------
8424
8425    /// Parse `roles.json` bytes from `fs`.
8426    ///
8427    /// Return values:
8428    ///   `Ok(Some(roles))` — file absent (returns `vec![]`) **or** file present
8429    ///                       and valid; in both cases `mask_for_role` uses the
8430    ///                       list normally (an absent file means no roles defined).
8431    ///   `Ok(None)`        — file present but corrupt or unrecognised version
8432    ///                       → poisoned state; `mask_for_role` returns `Err` for
8433    ///                       any role name until the file is fixed and the DB
8434    ///                       re-opened (or `apply_schema` is called to repair it).
8435    ///
8436    /// Note: `None` signals corruption, not absence — the opposite of what an
8437    /// optional "file missing" convention would suggest.  The open path stores
8438    /// this result on `db.roles` directly.
8439    fn load_roles_from_fs(fs: &F) -> Result<Option<Vec<RoleDef>>> {
8440        let bytes = fs.read(FileId::Roles).map_err(GraphError::Io)?;
8441        if bytes.is_empty() {
8442            // Empty bytes means either the file is absent or zero-byte — both
8443            // are treated identically as "no roles defined".  A zero-byte
8444            // roles.json does NOT widen access: an absent file and a zero-byte
8445            // file both resolve to an empty role list (sees nothing by default).
8446            return Ok(Some(vec![]));
8447        }
8448        match serde_json::from_slice::<RolesFile>(&bytes) {
8449            Ok(f) if matches!(f.version, 1..=4) => Ok(Some(f.roles)),
8450            // Corrupt or unrecognised version (>4): poison the roles state.
8451            // Never widen: a version this binary does not know may carry a
8452            // narrowing this binary would not apply.
8453            _ => Ok(None),
8454        }
8455    }
8456
8457    /// Resolve a role to a node-visibility mask against the current graph state.
8458    ///
8459    /// Returns `Err` when:
8460    /// - `roles.json` was present but corrupt at open (poisoned state), or
8461    /// - `role` does not match any defined role name.
8462    ///
8463    /// The mask union is: explicit `keys` (unknown keys silently ignored) plus
8464    /// all live nodes carrying any label in `labels` that also pass the role's
8465    /// [`visible_where`](crate::roles::RoleDef::visible_where) predicate, if it
8466    /// has one.  Label resolution is live — new nodes of an allowed label are
8467    /// visible without re-applying the schema, and a property edited out of the
8468    /// predicate takes its node out of the mask on the next read.  An empty
8469    /// union = empty mask = sees nothing.
8470    ///
8471    /// This is the one resolver every read path calls, live and as-of alike, so
8472    /// the predicate applies everywhere at once.  On an as-of handle the role
8473    /// *definition* is the current one and the graph is the historical one: the
8474    /// predicate is evaluated against the property values at the commit being
8475    /// read.
8476    ///
8477    /// The result is memoised per `(role, commit_seq)`, so a scoped reader
8478    /// between two writes resolves the role once.  See
8479    /// [`RoleMaskCache`](crate::mask::RoleMaskCache) for why that cannot go
8480    /// stale.
8481    pub fn mask_for_role(&self, role: &str) -> Result<crate::mask::NodeMask> {
8482        self.role_masks
8483            .get_or_build(role, self.commit_seq, || self.build_mask_for_role(role))
8484            .map(|m| (*m).clone())
8485    }
8486
8487    /// The mask an [`AsOfScope`] names, resolved against this handle.
8488    ///
8489    /// Shared by [`GraphDb::query_at_scoped`] and
8490    /// [`GraphDb::query_at_scoped_in_namespace`] so one scope resolves one way
8491    /// however the namespace leg is added.
8492    fn mask_at_scope(&self, scope: AsOfScope<'_>) -> Result<crate::mask::NodeMask> {
8493        // One resolver answers "what may this role see" — `mask_for_role` — and
8494        // it runs against this handle, so on a temporal one the answer is the
8495        // as-of one.
8496        Ok(match scope {
8497            AsOfScope::Role(role) => self.mask_for_role(role)?,
8498            AsOfScope::Keys(keys) => {
8499                crate::mask::NodeMask::from_keys(self, keys.iter().map(String::as_str))
8500            }
8501            AsOfScope::RoleAndKeys(role, keys) => {
8502                self.mask_for_role(role)?
8503                    .intersect(&crate::mask::NodeMask::from_keys(
8504                        self,
8505                        keys.iter().map(String::as_str),
8506                    ))
8507            }
8508            AsOfScope::Namespace(namespace) => self.mask_for_namespace(namespace),
8509        })
8510    }
8511
8512    /// Resolve `role` against the current graph, ignoring the memo.
8513    fn build_mask_for_role(&self, role: &str) -> Result<crate::mask::NodeMask> {
8514        let roles = self.roles.as_ref().ok_or_else(roles_poisoned)?;
8515        let def = roles
8516            .iter()
8517            .find(|r| r.name == role)
8518            .ok_or_else(|| GraphError::KeyNotFound {
8519                key: format!("role:{role}"),
8520            })?;
8521
8522        let mut visible = std::collections::HashSet::new();
8523
8524        // Key leg: resolve explicit keys to dense ids (unknown keys ignored).
8525        // An administrative grant, never narrowed by the predicate.
8526        for key in &def.keys {
8527            if let Some(id) = self.ids.get(key) {
8528                visible.insert(id);
8529            }
8530        }
8531
8532        // Label leg: live scan — iterate labels vec for matching symbol, and
8533        // when the role carries a predicate, test the property as well.  The
8534        // property comes from the store's own merged view (overlay over the
8535        // mmap'd base), so an as-of handle reads the values of its own commit.
8536        let props = def.visible_where.as_ref().map(|_| self.props_view());
8537        for label_name in &def.labels {
8538            if let Some(sym) = self.syms.get(label_name) {
8539                for (i, &s) in self.labels.iter().enumerate() {
8540                    if s != sym {
8541                        continue;
8542                    }
8543                    let id = i as u32;
8544                    match (&def.visible_where, &props) {
8545                        (Some(pred), Some(view)) => {
8546                            let value = view.get(id, &pred.field).map(|vr| vr.into_value());
8547                            if pred.holds(value.as_ref()) {
8548                                visible.insert(id);
8549                            }
8550                        }
8551                        _ => {
8552                            visible.insert(id);
8553                        }
8554                    }
8555                }
8556            }
8557        }
8558
8559        // Namespace leg: an intersection over the whole union, the key leg
8560        // included. A namespace is a tenancy boundary, so a key naming a node in
8561        // another tenant's namespace is not an administrative grant — and
8562        // `apply_schema` has already refused that role, so this only has to be
8563        // right about the node that moved into existence afterwards.
8564        if def.namespaces.is_some() {
8565            visible.retain(|&id| def.sees_namespace(self.ns_name(self.node_ns_idx(id))));
8566        }
8567
8568        Ok(crate::mask::NodeMask::from_ids(visible))
8569    }
8570
8571    /// Return the current list of role definitions.
8572    ///
8573    /// Returns an empty list when no roles are defined or when `roles.json`
8574    /// was corrupt at open (check [`mask_for_role`](Self::mask_for_role) for
8575    /// the fail-loud error in that case, or call
8576    /// [`roles_checked`](Self::roles_checked), which is this readout with that
8577    /// error in it).
8578    pub fn roles(&self) -> Vec<RoleDef> {
8579        self.roles.as_deref().unwrap_or(&[]).to_vec()
8580    }
8581
8582    /// The role definitions, or the poison error when `roles.json` was corrupt
8583    /// at open.
8584    ///
8585    /// [`roles`](Self::roles) answers `[]` both for a store that defines no
8586    /// roles and for one whose sidecar did not parse, and a caller validating a
8587    /// role name at boot cannot tell those apart. The wrong reading of the pair
8588    /// is the dangerous one: a store with no roles at all is an unrestricted
8589    /// store, so a poisoned file would read as "nothing is restricted here".
8590    ///
8591    /// This is the same answer, for the same cause, that
8592    /// [`mask_for_role`](Self::mask_for_role) gives on the first read.
8593    pub fn roles_checked(&self) -> Result<Vec<RoleDef>> {
8594        match self.roles.as_deref() {
8595            Some(roles) => Ok(roles.to_vec()),
8596            None => Err(roles_poisoned()),
8597        }
8598    }
8599
8600    // ── Role-scoped write authz ───────────────────────────────────────────────
8601
8602    /// Execute `ops` with optional role-scoped write authorization.
8603    ///
8604    /// - `None` → full authority, identical to [`write_batch`](Self::write_batch)
8605    ///   (zero-cost bypass of all authz checks).
8606    /// - `Some(authz)` → the decision table is evaluated per-op BEFORE any WAL
8607    ///   record is built.  A denial returns an error with no WAL frame written
8608    ///   (all-or-nothing at the authz boundary, then at the MutPreview boundary).
8609    ///
8610    /// See the plan's "authz decision table" section for the full semantics.
8611    pub fn write_batch_authz(
8612        &mut self,
8613        authz: Option<&WriteAuthz>,
8614        ops: Vec<BatchOp>,
8615    ) -> Result<(usize, usize)> {
8616        // Thread authz as a direct parameter — never touches pending_write_authz.
8617        self.commit_logged_batch(ops, None, authz.cloned())
8618            .map(inserted_pair)
8619    }
8620
8621    /// Execute a Cypher write statement with role-scoped write authorization.
8622    ///
8623    /// Resolves scope + mask from `self.roles` inside the call (same write-guard
8624    /// lifetime as execution, satisfying §5 lock discipline).  The resolved
8625    /// `WriteAuthz` is stored as `pending_write_authz` for the duration of the
8626    /// call so that all inner `batch.commit()` calls are authz-checked.
8627    ///
8628    /// MERGE is handled specially: the MERGE scope precondition (§3.3) is
8629    /// checked in `exec_merge` BEFORE `has_node` to close the §6.2
8630    /// timing-oracle item (hidden ≡ absent for unscoped roles).
8631    ///
8632    /// Roles with `write: None` (v1 behavior) → `RoleWriteDenied` with
8633    /// "this endpoint is not permitted".
8634    pub fn query_write_authz(
8635        &mut self,
8636        role: &str,
8637        cypher: &str,
8638        params: &BTreeMap<String, Value>,
8639    ) -> Result<ResultSet> {
8640        // Resolve scope (fails fast if role has no write scope).
8641        // write:None → byte-identical v1 blanket-403 body (plan §v1-sidecar mandate).
8642        let scope =
8643            {
8644                let roles = self.roles.as_deref().ok_or_else(|| GraphError::Corrupt {
8645                    detail: "roles.json was corrupt at open; re-open to restore role access".into(),
8646                })?;
8647                let def = roles.iter().find(|r| r.name == role).ok_or_else(|| {
8648                    GraphError::KeyNotFound {
8649                        key: format!("role:{role}"),
8650                    }
8651                })?;
8652                def.write
8653                    .clone()
8654                    .ok_or_else(|| GraphError::RoleWriteDenied {
8655                        reason: "role-bound token: writes are not permitted".into(),
8656                    })?
8657            };
8658        // Resolve mask inside the call (same guard, §5 coherence).
8659        let mask = self.mask_for_role(role)?;
8660        self.pending_write_authz = Some(WriteAuthz {
8661            role: role.into(),
8662            scope,
8663            mask,
8664        });
8665        // RAII guard: always clears pending_write_authz on scope exit, including
8666        // on panic or early-return, mirroring the RestoreEmitDeltas precedent.
8667        struct ClearPendingAuthzOnDrop(*mut Option<WriteAuthz>);
8668        impl Drop for ClearPendingAuthzOnDrop {
8669            fn drop(&mut self) {
8670                // SAFETY: pointer into the owning GraphDb; guard is dropped
8671                // within this function's frame before it returns.
8672                unsafe { *self.0 = None };
8673            }
8674        }
8675        // SAFETY: raw pointer into self; guard dropped before this fn returns.
8676        let _authz_guard = ClearPendingAuthzOnDrop(&mut self.pending_write_authz as *mut _);
8677        let tokens = lex(cypher).map_err(|e| GraphError::QueryError {
8678            detail: format!("lex: {e}"),
8679        })?;
8680        let stmt = parse_write(&tokens).map_err(|e| GraphError::QueryError {
8681            detail: format!("parse: {e}"),
8682        })?;
8683        self.exec_write_stmt(stmt, params)
8684    }
8685
8686    /// Execute `ops` with optional role-scoped write authorization, suppressing
8687    /// fsync (for use inside the group-commit drain thread, which performs one
8688    /// group fsync after releasing the write lock).
8689    ///
8690    /// Identical to [`write_batch_authz`] except the fsync policy is temporarily
8691    /// forced to `Relaxed` for the duration of the call, matching the drain-thread
8692    /// contract established by [`commit_batch_nosync`].
8693    pub(crate) fn write_batch_authz_nosync(
8694        &mut self,
8695        authz: Option<&WriteAuthz>,
8696        ops: Vec<BatchOp>,
8697    ) -> Result<(usize, usize)> {
8698        let saved = self.fsync;
8699        struct RestoreFsync(*mut FsyncPolicy, FsyncPolicy);
8700        impl Drop for RestoreFsync {
8701            fn drop(&mut self) {
8702                // SAFETY: pointer into the owning GraphDb; guard is dropped
8703                // within the enclosing function's frame before it returns.
8704                unsafe { *self.0 = self.1 };
8705            }
8706        }
8707        // SAFETY: raw pointer into self; guard dropped before this fn returns.
8708        let _g = RestoreFsync(&mut self.fsync as *mut FsyncPolicy, saved);
8709        self.fsync = FsyncPolicy::Relaxed;
8710        self.commit_logged_batch(ops, None, authz.cloned())
8711            .map(inserted_pair)
8712    }
8713
8714    /// Execute a `/ingest` request with role-scoped write authorization.
8715    ///
8716    /// Resolves the role's `WriteScope` and `NodeMask` inside this call (same
8717    /// write-guard lifetime as the mutation, satisfying §5 lock discipline).
8718    /// Sets `pending_write_authz` for the duration of the call so that the
8719    /// `commit_ingest` → `commit_logged_batch` path picks up the authz context
8720    /// and evaluates the decision table per-op before any WAL write.
8721    ///
8722    /// §7.3: roles with empty `create_labels` will see every `InsertNode` op
8723    /// denied by the decision table with the appropriate §4.3 scope reason;
8724    /// no special HTTP-layer check is needed.
8725    ///
8726    /// Roles with `write: None` return `RoleWriteDenied` with
8727    /// "writes are not permitted" (byte-identical to v1 blanket 403).
8728    pub fn ingest_with_edges_authz(
8729        &mut self,
8730        role: &str,
8731        label: &str,
8732        rows: Vec<std::collections::BTreeMap<String, Value>>,
8733        opts: &crate::ingest::IngestOptions,
8734        edges: &[(String, String, String)],
8735    ) -> Result<crate::ingest::IngestReport> {
8736        // Resolve scope (fails fast if role has no write scope).
8737        // write:None → byte-identical v1 blanket-403 body (plan §v1-sidecar mandate).
8738        let scope =
8739            {
8740                let roles = self.roles.as_deref().ok_or_else(|| GraphError::Corrupt {
8741                    detail: "roles.json was corrupt at open; re-open to restore role access".into(),
8742                })?;
8743                let def = roles.iter().find(|r| r.name == role).ok_or_else(|| {
8744                    GraphError::KeyNotFound {
8745                        key: format!("role:{role}"),
8746                    }
8747                })?;
8748                def.write
8749                    .clone()
8750                    .ok_or_else(|| GraphError::RoleWriteDenied {
8751                        reason: "role-bound token: writes are not permitted".into(),
8752                    })?
8753            };
8754        let mask = self.mask_for_role(role)?;
8755        self.pending_write_authz = Some(WriteAuthz {
8756            role: role.into(),
8757            scope,
8758            mask,
8759        });
8760        // RAII guard: always clears pending_write_authz on scope exit, including
8761        // on panic or early-return, mirroring the RestoreEmitDeltas precedent.
8762        struct ClearPendingAuthzOnDrop(*mut Option<WriteAuthz>);
8763        impl Drop for ClearPendingAuthzOnDrop {
8764            fn drop(&mut self) {
8765                // SAFETY: pointer into the owning GraphDb; guard is dropped
8766                // within this function's frame before it returns.
8767                unsafe { *self.0 = None };
8768            }
8769        }
8770        // SAFETY: raw pointer into self; guard dropped before this fn returns.
8771        let _authz_guard = ClearPendingAuthzOnDrop(&mut self.pending_write_authz as *mut _);
8772        self.ingest_with_edges(label, rows, opts, edges)
8773    }
8774
8775    /// Evaluate the write-authz decision table for one `BatchOp`.
8776    ///
8777    /// Called by `commit_logged_batch` for each op when `pending_write_authz`
8778    /// is `Some`, BEFORE MutPreview.  A denial returns an error immediately;
8779    /// the remaining ops are not evaluated and no WAL frame is written.
8780    ///
8781    /// `batch_created` carries the key→label pairs of nodes that earlier ops in
8782    /// THIS batch will create.  Used by `InsertEdgeUpsert` to count same-batch
8783    /// placeholder nodes as visible (spec: "a placeholder endpoint the SAME
8784    /// batch creates counts as visible if its label passed the create-class gate").
8785    fn check_single_op_authz(
8786        &self,
8787        authz: &WriteAuthz,
8788        op: &BatchOp,
8789        batch_created: &BTreeMap<String, String>,
8790    ) -> Result<()> {
8791        // Helper: 3-way node status under the authz mask.
8792        //
8793        // Batch-created nodes (from earlier InsertNode in THIS batch) are treated
8794        // as Visible with their recorded label — their create gate already passed
8795        // and they are not yet in self.ids (not committed).  This fixes the
8796        // MERGE+ON CREATE SET case where InsertNode + SetProp arrive together:
8797        // the SetProp must not see the node as Absent.
8798        let node_status = |key: &str| -> NodeAuthzStatus {
8799            if let Some(label) = batch_created.get(key) {
8800                return NodeAuthzStatus::Visible(label.clone());
8801            }
8802            match self.ids.get(key) {
8803                None => NodeAuthzStatus::Absent,
8804                Some(id) if !authz.mask.contains_id(id) => NodeAuthzStatus::Hidden,
8805                Some(id) => {
8806                    let label = self
8807                        .labels
8808                        .get(id as usize)
8809                        .and_then(|&sym| {
8810                            if sym == u32::MAX {
8811                                None
8812                            } else {
8813                                self.syms.resolve(sym).map(str::to_string)
8814                            }
8815                        })
8816                        .unwrap_or_default();
8817                    NodeAuthzStatus::Visible(label)
8818                }
8819            }
8820        };
8821
8822        // Helper: is an InsertEdgeUpsert endpoint visible?
8823        // A same-batch placeholder counts as visible if its label passed
8824        // the create-class gate (spec "upsert placeholder-counts-as-visible").
8825        let upsert_ep_visible = |ep_key: &str, placeholder_label: &str| -> bool {
8826            // In store and visible?
8827            if let Some(id) = self.ids.get(ep_key) {
8828                return authz.mask.contains_id(id);
8829            }
8830            // Created by an earlier op in this batch?
8831            if let Some(created_label) = batch_created.get(ep_key) {
8832                return authz.scope.create_labels.contains(created_label);
8833            }
8834            // Will be created by THIS InsertEdgeUpsert: placeholder_label
8835            // must pass the create-class gate.
8836            authz
8837                .scope
8838                .create_labels
8839                .contains(&placeholder_label.to_string())
8840        };
8841
8842        match op {
8843            // RenameNode / CreateRule / DeleteRule: defense-in-depth gate.
8844            // These ops are never routed to role-scoped paths by the HTTP layer,
8845            // but we 403 them here to close any future bypass route.
8846            //
8847            // InsertNodeOnConflict joins them: it is reachable only from the
8848            // embedded Python binding, which has no role token, and `Replace`
8849            // is a create and an update at once. Rather than split the decision
8850            // table for an op no role-scoped path constructs, refuse it — a
8851            // role-scoped caller writes through the ops that are already in the
8852            // table.
8853            BatchOp::RenameNode { .. }
8854            | BatchOp::CreateRule(_)
8855            | BatchOp::DeleteRule { .. }
8856            | BatchOp::InsertNodeOnConflict { .. } => {
8857                return Err(GraphError::RoleWriteDenied {
8858                    reason: "role-bound token: this endpoint is not permitted".into(),
8859                });
8860            }
8861
8862            // ── CREATE-class: InsertNode ─────────────────────────────────────
8863            //
8864            // Decision table row 1 (scope-before-lookup): check label in
8865            // create_labels BEFORE any key lookup.  This is the structural
8866            // closure of the §6.2 timing-oracle item — the denial fires even
8867            // when the store is EMPTY (see test_create_scope_denied_empty_store).
8868            BatchOp::InsertNode { label, key, props } => {
8869                if !authz.scope.create_labels.contains(label) {
8870                    return Err(GraphError::RoleWriteDenied {
8871                        reason: format!(
8872                            "role-bound token: label '{}' not in write scope (create_labels)",
8873                            label
8874                        ),
8875                    });
8876                }
8877                // A role bound to namespaces may only create inside them. The
8878                // never-widen rule is about what a write makes visible to *any*
8879                // party, not only to the writer: a node this role could never
8880                // read back is a write into somebody else's tenancy. Also a
8881                // scope check, so it runs before the key lookup — it discloses
8882                // nothing about the store. Covers Cypher `CREATE` and the node
8883                // `MERGE` creates, both of which arrive as this op.
8884                // Resolved before the role lookup so a props list naming `ns`
8885                // twice is refused for every role, scoped or not: it is the same
8886                // malformed write the seam refuses, and leaving it to the seam
8887                // would mean the gate had already read one of the two.
8888                let target = Self::created_namespace(key, props)?;
8889                if let Some(def) = self.role_def_for(&authz.role) {
8890                    if !def.sees_namespace(target) {
8891                        return Err(GraphError::RoleWriteDenied {
8892                            reason: format!(
8893                                "role-bound token: namespace '{target}' not in the role's \
8894                                 namespaces"
8895                            ),
8896                        });
8897                    }
8898                }
8899                // Row 2/3: key lookup.
8900                match self.ids.get(key.as_str()) {
8901                    Some(id) if authz.mask.contains_id(id) => {
8902                        // Visible: DuplicateKey — let MutPreview handle this.
8903                    }
8904                    Some(_) => {
8905                        // Hidden: indistinguishable from absent to the role.
8906                        return Err(GraphError::RoleWriteDenied {
8907                            reason: "role-bound token: target node not visible".into(),
8908                        });
8909                    }
8910                    None => {
8911                        // Absent: proceed (create).
8912                    }
8913                }
8914            }
8915
8916            // ── UPDATE-class: SetProp, RemoveProp ────────────────────────────
8917            BatchOp::SetProp { key, .. } | BatchOp::RemoveProp { key, .. } => {
8918                if batch_created.contains_key(key.as_str()) {
8919                    // Batch-created node: create gate already passed this batch.
8920                    // Updating it in the same batch is always allowed, regardless
8921                    // of update_labels (ruling §3.5: "writer just created it").
8922                } else {
8923                    let label = match node_status(key) {
8924                        NodeAuthzStatus::Visible(lbl) => lbl,
8925                        _ => {
8926                            return Err(GraphError::RoleWriteDenied {
8927                                reason: "role-bound token: target node not visible".into(),
8928                            });
8929                        }
8930                    };
8931                    if !authz.scope.update_labels.contains(&label) {
8932                        return Err(GraphError::RoleWriteDenied {
8933                            reason: format!(
8934                                "role-bound token: label '{}' not in write scope (update_labels)",
8935                                label
8936                            ),
8937                        });
8938                    }
8939                }
8940            }
8941
8942            // ── DELETE-class: DeleteNode ─────────────────────────────────────
8943            BatchOp::DeleteNode { key } => {
8944                let label = match node_status(key) {
8945                    NodeAuthzStatus::Visible(lbl) => lbl,
8946                    _ => {
8947                        return Err(GraphError::RoleWriteDenied {
8948                            reason: "role-bound token: target node not visible".into(),
8949                        });
8950                    }
8951                };
8952                if !authz.scope.delete_labels.contains(&label) {
8953                    return Err(GraphError::RoleWriteDenied {
8954                        reason: format!(
8955                            "role-bound token: label '{}' not in write scope (delete_labels)",
8956                            label
8957                        ),
8958                    });
8959                }
8960            }
8961
8962            // ── DELETE-class: DeleteEdge ─────────────────────────────────────
8963            //
8964            // Derived-edge rejection runs BEFORE the delete_edge_types scope
8965            // check (spec §3.5: "existing derived-edge rejection precedes
8966            // delete_edge_types check").
8967            BatchOp::DeleteEdge {
8968                edge_type,
8969                src_key,
8970                dst_key,
8971            } => {
8972                // Check provenance ownership BEFORE scope (spec §3.5 ordering).
8973                if let (Some(src_id), Some(dst_id), Some(et_sym)) = (
8974                    self.ids.get(src_key.as_str()),
8975                    self.ids.get(dst_key.as_str()),
8976                    self.syms.get(edge_type.as_str()),
8977                ) {
8978                    if self.engine.is_owned(et_sym, src_id, dst_id) {
8979                        return Err(GraphError::RuleOwned {
8980                            detail: format!(
8981                                "edge {edge_type} {src_key}→{dst_key} is rule-owned; \
8982                                 delete or change the owning rule"
8983                            ),
8984                        });
8985                    }
8986                    // Also check would_derive via MutPreview (empty overlay, pre-batch).
8987                    let preview = MutPreview::new(self);
8988                    if preview.would_derive(edge_type, src_key, dst_key) {
8989                        return Err(GraphError::RuleOwned {
8990                            detail: format!(
8991                                "edge {edge_type} {src_key}→{dst_key} is rule-owned; \
8992                                 delete or change the owning rule, or a live rule would \
8993                                 re-derive it"
8994                            ),
8995                        });
8996                    }
8997                }
8998                // Scope check (AFTER derived-edge check, BEFORE endpoint visibility).
8999                if !authz.scope.delete_edge_types.contains(edge_type) {
9000                    return Err(GraphError::RoleWriteDenied {
9001                        reason: format!(
9002                            "role-bound token: edge type '{}' not in write scope (delete_edge_types)",
9003                            edge_type
9004                        ),
9005                    });
9006                }
9007                // Both endpoints must be visible.
9008                for ep_key in [src_key.as_str(), dst_key.as_str()] {
9009                    match self.ids.get(ep_key) {
9010                        None => {
9011                            return Err(GraphError::RoleWriteDenied {
9012                                reason: "role-bound token: edge endpoint not visible".into(),
9013                            });
9014                        }
9015                        Some(id) if !authz.mask.contains_id(id) => {
9016                            return Err(GraphError::RoleWriteDenied {
9017                                reason: "role-bound token: edge endpoint not visible".into(),
9018                            });
9019                        }
9020                        _ => {}
9021                    }
9022                }
9023            }
9024
9025            // ── EDGE-CREATE: InsertEdge ──────────────────────────────────────
9026            //
9027            // Scope check BEFORE endpoint lookup (preserves timing symmetry).
9028            BatchOp::InsertEdge {
9029                edge_type,
9030                src_key,
9031                dst_key,
9032            } => {
9033                if !authz.scope.create_edge_types.contains(edge_type) {
9034                    return Err(GraphError::RoleWriteDenied {
9035                        reason: format!(
9036                            "role-bound token: edge type '{}' not in write scope (create_edge_types)",
9037                            edge_type
9038                        ),
9039                    });
9040                }
9041                // Both endpoints must be visible. A node created by an earlier
9042                // InsertNode in the same batch (tracked in batch_created) counts
9043                // as visible if its label passed the create-class gate.
9044                for ep_key in [src_key.as_str(), dst_key.as_str()] {
9045                    if batch_created.contains_key(ep_key) {
9046                        // Created earlier this batch — already scope-checked.
9047                        continue;
9048                    }
9049                    match self.ids.get(ep_key) {
9050                        None => {
9051                            return Err(GraphError::RoleWriteDenied {
9052                                reason: "role-bound token: edge endpoint not visible".into(),
9053                            });
9054                        }
9055                        Some(id) if !authz.mask.contains_id(id) => {
9056                            return Err(GraphError::RoleWriteDenied {
9057                                reason: "role-bound token: edge endpoint not visible".into(),
9058                            });
9059                        }
9060                        _ => {}
9061                    }
9062                }
9063            }
9064
9065            // ── EDGE-CREATE: InsertEdgeUpsert ────────────────────────────────
9066            //
9067            // Scope check first; then endpoint visibility using same-batch
9068            // placeholder awareness (spec: "a placeholder endpoint the SAME
9069            // batch creates counts as visible if its label passed the
9070            // create-class gate").
9071            BatchOp::InsertEdgeUpsert {
9072                edge_type,
9073                src_key,
9074                dst_key,
9075                placeholder_label,
9076            } => {
9077                if !authz.scope.create_edge_types.contains(edge_type) {
9078                    return Err(GraphError::RoleWriteDenied {
9079                        reason: format!(
9080                            "role-bound token: edge type '{}' not in write scope (create_edge_types)",
9081                            edge_type
9082                        ),
9083                    });
9084                }
9085                // Check placeholder label against create_labels (create-class gate).
9086                // This ensures the auto-created endpoints are scope-allowed.
9087                for ep_key in [src_key.as_str(), dst_key.as_str()] {
9088                    if !upsert_ep_visible(ep_key, placeholder_label) {
9089                        return Err(GraphError::RoleWriteDenied {
9090                            reason: "role-bound token: edge endpoint not visible".into(),
9091                        });
9092                    }
9093                }
9094                // A placeholder is created with no props, so it lands in the
9095                // default namespace. A role that cannot read `default` must not
9096                // create one there, for the same reason it may not create a node
9097                // there outright.
9098                //
9099                // The refusal is byte-identical to the hidden-endpoint one above,
9100                // and deliberately so: this arm fires only for an endpoint that
9101                // does **not** exist, and the one above only for an endpoint that
9102                // does. Two different strings would make the pair an existence
9103                // oracle — ask for an upsert and read off whether the key is
9104                // taken. Hidden ≡ absent is the rule everywhere else in this
9105                // table and it holds here too.
9106                if let Some(def) = self.role_def_for(&authz.role) {
9107                    if !def.sees_namespace(NS_DEFAULT) {
9108                        for ep_key in [src_key.as_str(), dst_key.as_str()] {
9109                            if self.ids.get(ep_key).is_none() && !batch_created.contains_key(ep_key)
9110                            {
9111                                return Err(GraphError::RoleWriteDenied {
9112                                    reason: "role-bound token: edge endpoint not visible".into(),
9113                                });
9114                            }
9115                        }
9116                    }
9117                }
9118            }
9119        }
9120        Ok(())
9121    }
9122
9123    /// Write `roles` to `roles.json` atomically and update the in-memory list.
9124    ///
9125    /// Called by `apply_schema` when roles change. Never called on unchanged
9126    /// re-apply — this preserves byte-identical idempotency.
9127    pub(crate) fn commit_roles(&mut self, roles: Vec<RoleDef>) -> Result<()> {
9128        let file = RolesFile::new_versioned(roles.clone());
9129        let bytes = serde_json::to_vec(&file).map_err(|e| GraphError::Corrupt {
9130            detail: format!("roles serialization: {e}"),
9131        })?;
9132        self.fs
9133            .write_atomic(FileId::Roles, &bytes)
9134            .map_err(GraphError::Io)?;
9135        self.roles = Some(roles);
9136        // Rewriting the sidecar is not a commit, so `commit_seq` does not move
9137        // and a memoised mask would still match its version. Install a fresh
9138        // cache instead of clearing the shared one: a reader snapshot frozen
9139        // against the old definitions keeps the old `Arc` to itself and can
9140        // never publish an answer this handle would read back.
9141        self.role_masks = Arc::new(crate::mask::RoleMaskCache::new());
9142        // Refresh the MVCC frozen overlay so that reader() immediately sees the
9143        // updated role definitions without waiting for the next K-commit fold.
9144        self.fold_now();
9145        Ok(())
9146    }
9147
9148    fn view(&self) -> GraphView<'_> {
9149        GraphView {
9150            ids: &self.ids,
9151            syms: &self.syms,
9152            labels: &self.labels,
9153            props: self.props_view(),
9154            topo: self.topo_view(),
9155            edge_props: self.edge_props_view(),
9156            mask: None,
9157            prop_index: Some(&self.prop_index),
9158        }
9159    }
9160
9161    fn view_masked<'a>(&'a self, mask: &'a crate::mask::NodeMask) -> GraphView<'a> {
9162        GraphView {
9163            ids: &self.ids,
9164            syms: &self.syms,
9165            labels: &self.labels,
9166            props: self.props_view(),
9167            topo: self.topo_view(),
9168            edge_props: self.edge_props_view(),
9169            mask: Some(&mask.visible),
9170            prop_index: Some(&self.prop_index),
9171        }
9172    }
9173
9174    /// Execute a read-only Cypher query with a node visibility mask.
9175    ///
9176    /// Only nodes whose key is in `mask` are accessible: label scans, key
9177    /// lookups, and neighbor expansions all respect the mask. Edges where
9178    /// either endpoint is hidden are silently dropped.
9179    ///
9180    /// Returns `Err` with a "masked queries are read-only" message when
9181    /// `cypher` is a write statement (CREATE / MERGE / MATCH…SET / DELETE).
9182    pub fn query_masked(
9183        &self,
9184        cypher: &str,
9185        params: &std::collections::BTreeMap<String, Value>,
9186        mask: &crate::mask::NodeMask,
9187    ) -> Result<ResultSet> {
9188        // Reject write statements up front.
9189        let tokens = lex(cypher).map_err(|e| GraphError::QueryError {
9190            detail: format!("lex: {e}"),
9191        })?;
9192        if is_write_tokens(&tokens) {
9193            return Err(GraphError::MaskedReadOnly);
9194        }
9195        let union = parse_read(&tokens).map_err(|e| GraphError::QueryError {
9196            detail: format!("parse: {e}"),
9197        })?;
9198        // Each UNION part executes against the same masked view, so the mask
9199        // applies uniformly across the chain.
9200        execute_union(&self.view_masked(mask), &union, &Params(params)).map_err(|e| {
9201            GraphError::QueryError {
9202                detail: format!("execute: {e}"),
9203            }
9204        })
9205    }
9206
9207    pub fn node_ref(&self, key: &str) -> Option<NodeRef<'_, F>> {
9208        let id = self.ids.get(key)?;
9209        Some(NodeRef { db: self, id })
9210    }
9211
9212    /// BFS neighborhood expansion restricted to visible nodes in `mask`.
9213    ///
9214    /// Hidden nodes are never used as traversal intermediaries in either
9215    /// [`MaskMode::Omit`] or [`MaskMode::Stub`] — a visible node reachable
9216    /// only through a hidden node will not appear in results.
9217    ///
9218    /// In [`MaskMode::Stub`] mode, hidden nodes that are direct neighbours of
9219    /// a visited visible node are appended to the result as stub rows
9220    /// (`label` column is `null`, same key+depth columns as visible rows).
9221    /// They are NOT added to the BFS frontier.
9222    ///
9223    /// Returns `None` when `key` does not exist (caller should 404).
9224    ///
9225    /// **SECURITY**: role-token callers always pass an Omit-mode mask, so
9226    /// stub rows are never produced on the role path.
9227    pub fn neighborhood_masked(
9228        &self,
9229        key: &str,
9230        depth: u32,
9231        edge_types: Option<&[&str]>,
9232        dir: Dir,
9233        mask: &crate::mask::NodeMask,
9234    ) -> Option<ResultSet> {
9235        let start_id = self.ids.get(key)?;
9236        let view = self.view_masked(mask);
9237        let resolved: Option<Vec<u32>> = edge_types.map(|names| {
9238            names
9239                .iter()
9240                .filter_map(|name| view.syms.get(name))
9241                .collect()
9242        });
9243        let nb = neighborhood(&view, start_id, depth, resolved.as_deref(), dir);
9244        let mut rs = ResultSet::new(vec!["key".into(), "label".into(), "depth".into()]);
9245        // Collect visible BFS results (start_id at depth 0, BFS nodes after).
9246        let mut visited: Vec<(u32, u32)> = Vec::with_capacity(nb.nodes.len() + 1);
9247        visited.push((start_id, 0));
9248        for (nid, d) in &nb.nodes {
9249            let k = view.key_of(*nid);
9250            let label = view
9251                .label_of(*nid)
9252                .expect("real nodes always have a label; u32::MAX sentinel cannot occur");
9253            rs.push_row(vec![
9254                Some(Value::Str(k.to_string())),
9255                Some(Value::Str(label.to_string())),
9256                Some(Value::Int(*d as i64)),
9257            ]);
9258            visited.push((*nid, *d));
9259        }
9260        // Stub mode: add hidden direct neighbours of each visited node as stubs.
9261        // Hidden nodes are edge-endpoints only — they are not added to the BFS
9262        // frontier, so the BFS never expands through them.
9263        if mask.mode() == crate::mask::MaskMode::Stub {
9264            let raw_view = self.view();
9265            let mut seen: std::collections::HashSet<u32> =
9266                visited.iter().map(|(id, _)| *id).collect();
9267            for (node_id, node_depth) in &visited {
9268                if *node_depth >= depth {
9269                    continue;
9270                }
9271                for e in expand(&raw_view, *node_id, resolved.as_deref(), dir) {
9272                    let nbr = if e.src == *node_id { e.dst } else { e.src };
9273                    if !mask.contains_id(nbr) && seen.insert(nbr) {
9274                        if let Some(k) = self.ids.key_of(nbr) {
9275                            rs.push_row(vec![
9276                                Some(Value::Str(k.to_string())),
9277                                None,
9278                                Some(Value::Int((*node_depth + 1) as i64)),
9279                            ]);
9280                        }
9281                    }
9282                }
9283            }
9284        }
9285        Some(rs)
9286    }
9287
9288    /// [`neighborhood_masked`](Self::neighborhood_masked) with the **subject
9289    /// check** a scoped caller needs: a start key the mask hides answers exactly
9290    /// as an absent one does.
9291    ///
9292    /// `neighborhood_masked` expands from any existing key, hidden or not,
9293    /// because a full-token caller supplying a client mask already knows which
9294    /// keys exist. A scoped caller does not, so telling it apart a hidden key
9295    /// from an absent one would be an existence oracle.
9296    ///
9297    /// Expansion itself is unchanged: hidden nodes are neither returned nor used
9298    /// as traversal intermediaries, so a visible node reachable only through a
9299    /// hidden one stays out of the result.
9300    ///
9301    /// Hidden or unknown `key` → [`GraphError::KeyNotFound`].
9302    pub fn neighborhood_scoped(
9303        &self,
9304        key: &str,
9305        depth: u32,
9306        edge_types: Option<&[&str]>,
9307        dir: Dir,
9308        mask: &crate::mask::NodeMask,
9309    ) -> Result<ResultSet> {
9310        if !mask.contains_node(self, key) {
9311            return Err(GraphError::KeyNotFound { key: key.into() });
9312        }
9313        self.neighborhood_masked(key, depth, edge_types, dir, mask)
9314            .ok_or_else(|| GraphError::KeyNotFound { key: key.into() })
9315    }
9316
9317    /// Live node's key, label, and columnar props. Unknown or tombstoned → `None`.
9318    pub fn node_info(&self, key: &str) -> Option<NodeInfo> {
9319        let n = self.node_ref(key)?;
9320        Some(NodeInfo {
9321            key: n.key().to_string(),
9322            label: n.label().to_string(),
9323            props: n.props(),
9324        })
9325    }
9326
9327    /// Look up a node with mask awareness.
9328    ///
9329    /// | Key state         | Omit mode       | Stub mode              |
9330    /// |-------------------|-----------------|------------------------|
9331    /// | does not exist    | `None` (→ 404)  | `None` (→ 404)         |
9332    /// | exists, visible   | `Some(Visible)` | `Some(Visible)`        |
9333    /// | exists, hidden    | `None` (→ 404)  | `Some(Restricted)`     |
9334    ///
9335    /// **SECURITY**: only call from client-mask (full-token) paths.
9336    /// Role-token paths must use [`node_info`] after an explicit visibility check.
9337    pub fn node_info_masked(
9338        &self,
9339        key: &str,
9340        mask: &crate::mask::NodeMask,
9341    ) -> Option<MaskedNodeResult> {
9342        let id = self.ids.get(key)?;
9343        if mask.contains_id(id) {
9344            Some(MaskedNodeResult::Visible(self.node_info(key)?))
9345        } else {
9346            match mask.mode() {
9347                crate::mask::MaskMode::Stub => Some(MaskedNodeResult::Restricted),
9348                crate::mask::MaskMode::Omit => None,
9349            }
9350        }
9351    }
9352
9353    /// Get edges for `key` with mask-aware hidden-endpoint handling.
9354    ///
9355    /// - Omit mode: edges to hidden endpoints are excluded (same as role-path filtering).
9356    /// - Stub mode: edges to hidden endpoints are included; `src_restricted`/`dst_restricted`
9357    ///   is `true` for each hidden endpoint.
9358    ///
9359    /// Unknown key → [`GraphError::KeyNotFound`].
9360    ///
9361    /// **SECURITY**: only call from client-mask (full-token) paths.
9362    pub fn node_edges_masked(
9363        &self,
9364        key: &str,
9365        mask: &crate::mask::NodeMask,
9366    ) -> Result<Vec<MaskedEdge>> {
9367        self.ensure_v8_base_sections_loaded();
9368        let id = self
9369            .ids
9370            .get(key)
9371            .ok_or_else(|| GraphError::KeyNotFound { key: key.into() })?;
9372        let derived: BTreeSet<(u32, u32, u32)> = self
9373            .engine
9374            .provenance_touching(id)
9375            .map(|(_rule, etype, src, dst)| (etype, src, dst))
9376            .collect();
9377        let mut edges = Vec::new();
9378        let tv = self.topo_view();
9379        for etype in tv.etypes() {
9380            // etype comes from the archived CSR (access_unchecked, no eager CRC).
9381            // A bit-flip in the large TOPOLOGY section can produce an etype id
9382            // that is not in the interner.  Return Corrupt rather than panic.
9383            let edge_type = self
9384                .syms
9385                .resolve(etype)
9386                .ok_or_else(|| GraphError::Corrupt {
9387                    detail: format!("v8: topology etype {etype} not in interner"),
9388                })?
9389                .to_string();
9390            for dir in [Direction::Out, Direction::In] {
9391                for &nbr in tv.neighbors(etype, dir, id).as_ref() {
9392                    let nbr_restricted = !mask.contains_id(nbr);
9393                    if nbr_restricted && mask.mode() == crate::mask::MaskMode::Omit {
9394                        continue;
9395                    }
9396                    let nbr_key = self
9397                        .ids
9398                        .key_of(nbr)
9399                        .ok_or_else(|| GraphError::Corrupt {
9400                            detail: format!("topology id {nbr} has no key"),
9401                        })?
9402                        .to_string();
9403                    let (src_id, dst_id, src_key, dst_key, src_restricted, dst_restricted) =
9404                        match dir {
9405                            Direction::Out => {
9406                                (id, nbr, key.to_string(), nbr_key, false, nbr_restricted)
9407                            }
9408                            Direction::In => {
9409                                (nbr, id, nbr_key, key.to_string(), nbr_restricted, false)
9410                            }
9411                        };
9412                    edges.push(MaskedEdge {
9413                        edge_type: edge_type.clone(),
9414                        src_key,
9415                        src_restricted,
9416                        dst_key,
9417                        dst_restricted,
9418                        derived: derived.contains(&(etype, src_id, dst_id)),
9419                    });
9420                }
9421            }
9422        }
9423        edges.sort_by(|a, b| {
9424            a.edge_type
9425                .cmp(&b.edge_type)
9426                .then(a.src_key.cmp(&b.src_key))
9427                .then(a.dst_key.cmp(&b.dst_key))
9428        });
9429        edges.dedup_by(|a, b| {
9430            a.edge_type == b.edge_type && a.src_key == b.src_key && a.dst_key == b.dst_key
9431        });
9432        Ok(edges)
9433    }
9434
9435    /// [`node_edges_masked`](Self::node_edges_masked) with the **subject check**
9436    /// a scoped caller needs, and a plain [`EdgeInfo`] list.
9437    ///
9438    /// `node_edges_masked` raises [`GraphError::KeyNotFound`] only when `key` is
9439    /// unknown; a key that exists but is hidden still yields its (filtered) edge
9440    /// list, which is correct for a full-token client mask and an existence
9441    /// oracle for a scoped one. Here a hidden subject answers exactly as an
9442    /// absent one does.
9443    ///
9444    /// Every edge naming a hidden endpoint is dropped, whatever the mask's
9445    /// [`MaskMode`](crate::mask::MaskMode): a scoped caller never sees a
9446    /// restricted stub, so there is nothing for it to render.
9447    ///
9448    /// Hidden or unknown `key` → [`GraphError::KeyNotFound`].
9449    pub fn node_edges_scoped(
9450        &self,
9451        key: &str,
9452        mask: &crate::mask::NodeMask,
9453    ) -> Result<Vec<EdgeInfo>> {
9454        if !mask.contains_node(self, key) {
9455            return Err(GraphError::KeyNotFound { key: key.into() });
9456        }
9457        Ok(self
9458            .node_edges_masked(key, mask)?
9459            .into_iter()
9460            .filter(|e| !e.src_restricted && !e.dst_restricted)
9461            .map(|e| EdgeInfo {
9462                edge_type: e.edge_type,
9463                src_key: e.src_key,
9464                dst_key: e.dst_key,
9465                derived: e.derived,
9466            })
9467            .collect())
9468    }
9469
9470    /// Every directed edge incident on `key`, both directions, every etype.
9471    ///
9472    /// Walk is `topology.etypes()` × `{Out, In}` × `neighbors()`. `derived` is
9473    /// membership in [`RuleEngine::provenance_touching`] (O(degree) via the
9474    /// Plan-8 `by_node` index). Sorted by `(edge_type, src_key, dst_key)`.
9475    /// Unknown key → [`GraphError::KeyNotFound`].
9476    pub fn node_edges(&self, key: &str) -> Result<Vec<EdgeInfo>> {
9477        self.ensure_v8_base_sections_loaded();
9478        let id = self
9479            .ids
9480            .get(key)
9481            .ok_or_else(|| GraphError::KeyNotFound { key: key.into() })?;
9482        let derived: BTreeSet<(u32, u32, u32)> = self
9483            .engine
9484            .provenance_touching(id)
9485            .map(|(_rule, etype, src, dst)| (etype, src, dst))
9486            .collect();
9487        let mut edges = Vec::new();
9488        let tv = self.topo_view();
9489        for etype in tv.etypes() {
9490            // Same guard as node_edges_masked: etype from unchecked-CRC CSR.
9491            let edge_type = self
9492                .syms
9493                .resolve(etype)
9494                .ok_or_else(|| GraphError::Corrupt {
9495                    detail: format!("v8: topology etype {etype} not in interner"),
9496                })?
9497                .to_string();
9498            for dir in [Direction::Out, Direction::In] {
9499                for &nbr in tv.neighbors(etype, dir, id).as_ref() {
9500                    let (src, dst, src_key, dst_key) = match dir {
9501                        Direction::Out => (
9502                            id,
9503                            nbr,
9504                            key.to_string(),
9505                            self.ids
9506                                .key_of(nbr)
9507                                .ok_or_else(|| GraphError::Corrupt {
9508                                    detail: format!("topology id {nbr} has no key"),
9509                                })?
9510                                .to_string(),
9511                        ),
9512                        Direction::In => (
9513                            nbr,
9514                            id,
9515                            self.ids
9516                                .key_of(nbr)
9517                                .ok_or_else(|| GraphError::Corrupt {
9518                                    detail: format!("topology id {nbr} has no key"),
9519                                })?
9520                                .to_string(),
9521                            key.to_string(),
9522                        ),
9523                    };
9524                    edges.push(EdgeInfo {
9525                        edge_type: edge_type.clone(),
9526                        src_key,
9527                        dst_key,
9528                        derived: derived.contains(&(etype, src, dst)),
9529                    });
9530                }
9531            }
9532        }
9533        edges.sort_by(|a, b| {
9534            a.edge_type
9535                .cmp(&b.edge_type)
9536                .then(a.src_key.cmp(&b.src_key))
9537                .then(a.dst_key.cmp(&b.dst_key))
9538        });
9539        // Self-loops appear in both Out and In; sort makes the pair adjacent
9540        // (sort key matches PartialEq for this case) so one pass drops the dup.
9541        edges.dedup();
9542        Ok(edges)
9543    }
9544
9545    // ── Backup ────────────────────────────────────────────────────────────────
9546
9547    /// Copy this store to `dest` as a consistent, verified snapshot.
9548    ///
9549    /// Copies every durable file in the database directory — `snapshot.bin`,
9550    /// `wal.bin`, all `wal.<N>.archive` files, `wal.floor`, `wal.genesis`, and
9551    /// `roles.json` — into a freshly created `dest` directory using OS-level
9552    /// `copy` calls (no large in-process buffers).
9553    ///
9554    /// # Consistency guarantee
9555    ///
9556    /// The guarantee is **process-local**: the caller holds `&self`, which
9557    /// prevents any concurrent writer in the **same process** from modifying
9558    /// the files during the copy.  Running `mushroomdb backup` against a
9559    /// directory that is **concurrently being written by another process** (e.g.
9560    /// `mushroomdb serve`) is **unsafe** — the copy can be torn.  The post-copy
9561    /// `verified: true` result reduces but does not eliminate the risk of a
9562    /// silent corrupt backup (CRC catches many bit-flips; it cannot catch a
9563    /// consistent mid-write snapshot).
9564    ///
9565    /// **The safe path for a live-served store is `POST /backup` on the HTTP
9566    /// server.** That handler acquires the read lock on the shared database
9567    /// before calling this method, which is the correct cross-process
9568    /// synchronisation point because the server is the single process writing
9569    /// the files.
9570    ///
9571    /// After copying, opens the destination read-only and runs the CRC section
9572    /// verifier (`verify_snapshot`) to confirm byte-for-byte integrity.
9573    /// `BackupReport::verified` reflects whether both checks passed.
9574    ///
9575    /// Returns `Err` when `self` is not backed by a `RealFs` (e.g. `SimFs`).
9576    pub fn backup_to(&self, dest: &std::path::Path) -> Result<BackupReport> {
9577        // Derive source directory from snapshot_path (RealFs only).
9578        let src_dir = match self.fs.snapshot_path() {
9579            Some(p) => p.parent().map(|d| d.to_path_buf()).ok_or_else(|| {
9580                GraphError::Io(std::io::Error::other("snapshot has no parent dir"))
9581            })?,
9582            None => {
9583                return Err(GraphError::Io(std::io::Error::other(
9584                    "backup_to requires a real filesystem (RealFs)",
9585                )))
9586            }
9587        };
9588
9589        std::fs::create_dir_all(dest)?;
9590
9591        let mut files: Vec<String> = Vec::new();
9592        let mut bytes: u64 = 0;
9593
9594        // Helper: copy src_dir/name → dest/name if the file exists.
9595        let mut try_copy = |name: &str| -> std::io::Result<()> {
9596            let src_path = src_dir.join(name);
9597            if src_path.exists() {
9598                let n = std::fs::copy(&src_path, dest.join(name))?;
9599                bytes += n;
9600                files.push(name.to_string());
9601            }
9602            Ok(())
9603        };
9604
9605        try_copy("snapshot.bin")?;
9606        try_copy("snapshot.bin.bak")?;
9607        try_copy("wal.bin")?;
9608        try_copy("wal.floor")?;
9609        try_copy("wal.genesis")?;
9610        try_copy("roles.json")?;
9611
9612        // Copy WAL archives.
9613        let archives = self.fs.list_archives()?;
9614        for n in &archives {
9615            let name = format!("wal.{n}.archive");
9616            let n_bytes = std::fs::copy(src_dir.join(&name), dest.join(&name))?;
9617            bytes += n_bytes;
9618            files.push(name);
9619        }
9620
9621        files.sort();
9622
9623        // Post-copy verification: open dest and run CRC checks.
9624        let snap_in_dest = dest.join("snapshot.bin").exists();
9625        let crc_ok = if snap_in_dest {
9626            crate::verify_snapshot(dest)
9627                .map(|results| results.iter().all(|(_, _, _, r)| r.is_ok()))
9628                .unwrap_or(false)
9629        } else {
9630            true // WAL-only store: nothing to CRC-check in snapshot
9631        };
9632        let opens_ok = GraphDb::<core_storage::fs::RealFs>::open(dest).is_ok();
9633        let verified = crc_ok && opens_ok;
9634
9635        Ok(BackupReport {
9636            files,
9637            bytes,
9638            verified,
9639        })
9640    }
9641
9642    // ── Export helpers ────────────────────────────────────────────────────────
9643
9644    /// All live nodes, sorted by key (deterministic).
9645    ///
9646    /// Reads base + WAL overlay. Tombstoned nodes are excluded.
9647    pub fn all_nodes_for_export(&self) -> Vec<NodeInfo> {
9648        self.ensure_v8_base_sections_loaded();
9649        let pv = self.props_view();
9650        let mut nodes = Vec::new();
9651        for id in 0..self.ids.len() as u32 {
9652            let Some(key) = self.ids.key_of(id) else {
9653                continue;
9654            };
9655            let Some(&sym) = self.labels.get(id as usize) else {
9656                continue;
9657            };
9658            if sym == u32::MAX {
9659                continue; // tombstoned
9660            }
9661            let Some(label) = self.syms.resolve(sym) else {
9662                continue;
9663            };
9664            let mut props = BTreeMap::new();
9665            for field in pv.field_names() {
9666                if let Some(vr) = pv.get(id, &field) {
9667                    props.insert(field, vr.into_value());
9668                }
9669            }
9670            nodes.push(NodeInfo {
9671                key: key.to_string(),
9672                label: label.to_string(),
9673                props,
9674            });
9675        }
9676        nodes.sort_by(|a, b| a.key.cmp(&b.key));
9677        nodes
9678    }
9679
9680    /// All directed edges, sorted by `(edge_type, src, dst)`. Each edge appears once.
9681    ///
9682    /// Derived edges carry `derived: true` and the creating rule's name in `rule`.
9683    /// Manual edges carry `derived: false` and `rule: None`.
9684    /// `weight` is the creating rule's `weight_prop` value read off the edge
9685    /// (numeric only), mirroring the convention used by [`GraphDb::explain`]
9686    /// and [`GraphDb::weighted_edges`]. Deterministic across runs on the same
9687    /// store state.
9688    pub fn all_edges_for_export(&self) -> Vec<ExportEdge> {
9689        self.ensure_v8_base_sections_loaded();
9690
9691        // Build (etype_sym, src_id, dst_id) → rule_name for O(1) derivation lookup.
9692        let mut prov: HashMap<(u32, u32, u32), String> = HashMap::new();
9693        for (rule_name, triples) in self.engine.provenance() {
9694            for &(etype, src, dst) in triples {
9695                prov.insert((etype, src, dst), rule_name.clone());
9696            }
9697        }
9698
9699        // rule_name → weight_prop, for O(1) lookup per derived edge.
9700        let weight_props: HashMap<&str, Option<&str>> = self
9701            .engine
9702            .rules()
9703            .map(|r| (r.name.as_str(), r.weight_prop.as_deref()))
9704            .collect();
9705
9706        let tv = self.topo_view();
9707        let ep = self.edge_props_view();
9708        let mut edges = Vec::new();
9709
9710        for id in 0..self.ids.len() as u32 {
9711            let Some(key) = self.ids.key_of(id) else {
9712                continue;
9713            };
9714            let Some(&lsym) = self.labels.get(id as usize) else {
9715                continue;
9716            };
9717            if lsym == u32::MAX {
9718                continue; // tombstoned
9719            }
9720
9721            for etype_sym in tv.etypes() {
9722                // etype from archived CSR (access_unchecked, no eager CRC).
9723                // Skip edges whose etype is not in the interner; this can only
9724                // occur with a corrupt large TOPOLOGY section (bit-flip on an
9725                // etype field in the archived data).  The function returns Vec,
9726                // not Result, so we continue rather than propagate.
9727                let Some(edge_type) = self.syms.resolve(etype_sym) else {
9728                    continue;
9729                };
9730                let edge_type = edge_type.to_string();
9731                for &nbr in tv.neighbors(etype_sym, Direction::Out, id).as_ref() {
9732                    let Some(dst_key) = self.ids.key_of(nbr) else {
9733                        continue; // skip corrupt entries
9734                    };
9735                    let prov_key = (etype_sym, id, nbr);
9736                    let rule = prov.get(&prov_key).cloned();
9737                    let derived = rule.is_some();
9738                    let weight = rule
9739                        .as_deref()
9740                        .and_then(|rn| weight_props.get(rn).copied().flatten())
9741                        .and_then(|prop| match ep.get(etype_sym, id, nbr, prop) {
9742                            Some(Value::Float(f)) => Some(f),
9743                            Some(Value::Int(i)) => Some(i as f64),
9744                            _ => None,
9745                        });
9746                    edges.push(ExportEdge {
9747                        edge_type: edge_type.clone(),
9748                        src: key.to_string(),
9749                        dst: dst_key.to_string(),
9750                        derived,
9751                        rule,
9752                        weight,
9753                    });
9754                }
9755            }
9756        }
9757
9758        edges.sort_by(|a, b| {
9759            a.edge_type
9760                .cmp(&b.edge_type)
9761                .then(a.src.cmp(&b.src))
9762                .then(a.dst.cmp(&b.dst))
9763        });
9764        edges
9765    }
9766
9767    /// What each edge type *is*, without building one record per edge.
9768    ///
9769    /// [`all_edges_for_export`](Self::all_edges_for_export) answers the same
9770    /// question by materialising every edge — three `String`s apiece, a
9771    /// provenance `HashMap` over every derived edge, and a final sort. That is
9772    /// the right shape for an export, and the wrong one for a summary: on a
9773    /// store with 1.3 M derived edges it allocates hundreds of megabytes to
9774    /// produce nine lines. This walks the topology instead, summing neighbour
9775    /// slice lengths and collecting *label symbols* rather than label strings,
9776    /// so the per-edge cost is an integer add and a set insert on a set with
9777    /// as many members as the store has labels.
9778    ///
9779    /// The rule names come off the rule *definitions*, which each declare the
9780    /// `edge_type` they derive, so naming them costs one pass over the rules
9781    /// rather than one provenance lookup per edge. That is also why `rules`
9782    /// is a list: two rules may derive the same type — the association store
9783    /// derives `INDUSTRY_ALIGNMENT` from both a talent→company and a
9784    /// talent→job rule — and naming only one of them would be a half-truth.
9785    /// A type with no rules is one written by hand.
9786    ///
9787    /// `sample` is the first edge of the type in the store's own id order,
9788    /// which is insertion order: deterministic for a given store, and not the
9789    /// same as key order, which cannot be had without resolving a key per
9790    /// edge. Sorted by `edge_type`.
9791    pub fn edge_type_census(&self) -> Vec<EdgeTypeCensus> {
9792        self.ensure_v8_base_sections_loaded();
9793
9794        let mut rules_by_type: BTreeMap<&str, BTreeSet<&str>> = BTreeMap::new();
9795        for r in self.engine.rules() {
9796            rules_by_type
9797                .entry(r.edge_type.as_str())
9798                .or_default()
9799                .insert(r.name.as_str());
9800        }
9801
9802        let tv = self.topo_view();
9803        let node_count = self.ids.len() as u32;
9804        let mut out = Vec::new();
9805        for etype_sym in tv.etypes() {
9806            // An etype the interner cannot resolve means a corrupt TOPOLOGY
9807            // section; skip it rather than name it, as `all_edges_for_export`
9808            // does for the same reason.
9809            let Some(edge_type) = self.syms.resolve(etype_sym) else {
9810                continue;
9811            };
9812            let mut edges: u64 = 0;
9813            let mut src_syms: BTreeSet<u32> = BTreeSet::new();
9814            let mut dst_syms: BTreeSet<u32> = BTreeSet::new();
9815            let mut sample: Option<(u32, u32)> = None;
9816            for id in 0..node_count {
9817                let Some(&lsym) = self.labels.get(id as usize) else {
9818                    continue;
9819                };
9820                if lsym == u32::MAX {
9821                    continue; // tombstoned
9822                }
9823                let nbrs = tv.neighbors(etype_sym, Direction::Out, id);
9824                let nbrs = nbrs.as_ref();
9825                if nbrs.is_empty() {
9826                    continue;
9827                }
9828                edges += nbrs.len() as u64;
9829                src_syms.insert(lsym);
9830                for &nbr in nbrs {
9831                    if let Some(&dsym) = self.labels.get(nbr as usize) {
9832                        if dsym != u32::MAX {
9833                            dst_syms.insert(dsym);
9834                        }
9835                    }
9836                }
9837                if sample.is_none() {
9838                    sample = Some((id, nbrs[0]));
9839                }
9840            }
9841            let resolve = |syms: &BTreeSet<u32>| -> Vec<String> {
9842                syms.iter()
9843                    .filter_map(|&s| self.syms.resolve(s))
9844                    .map(ToString::to_string)
9845                    .collect()
9846            };
9847            out.push(EdgeTypeCensus {
9848                edge_type: edge_type.to_string(),
9849                edges,
9850                src_labels: resolve(&src_syms),
9851                dst_labels: resolve(&dst_syms),
9852                rules: rules_by_type
9853                    .get(edge_type)
9854                    .map(|rs| rs.iter().map(ToString::to_string).collect())
9855                    .unwrap_or_default(),
9856                sample: sample.and_then(|(s, d)| {
9857                    Some((
9858                        self.ids.key_of(s)?.to_string(),
9859                        self.ids.key_of(d)?.to_string(),
9860                    ))
9861                }),
9862            });
9863        }
9864        out.sort_by(|a, b| a.edge_type.cmp(&b.edge_type));
9865        out
9866    }
9867
9868    /// All directed edges of `edge_type`, with the raw value of `weight_prop`
9869    /// on each edge when given.
9870    ///
9871    /// `weight` is `Some(f)` only when `weight_prop` is set and the edge
9872    /// carries that property with a numeric (`Int`/`Float`) value; otherwise
9873    /// `None` — callers that want a default weight (e.g. `1.0` for missing
9874    /// props) apply it themselves, matching the convention used internally
9875    /// by [`GraphDb::pagerank`], [`GraphDb::connected_components`],
9876    /// [`GraphDb::degree_centrality`], and [`GraphDb::communities`].
9877    ///
9878    /// Sorted by `(src, dst)` for determinism. Reads the unified topology
9879    /// (manual + rule-derived edges).  An unknown `edge_type` returns an
9880    /// empty vec.
9881    pub fn weighted_edges(
9882        &self,
9883        edge_type: &str,
9884        weight_prop: Option<&str>,
9885    ) -> Vec<(String, String, Option<f64>)> {
9886        let Some(etype_sym) = self.syms.get(edge_type) else {
9887            return Vec::new();
9888        };
9889        let tv = self.topo_view();
9890        let ep = self.edge_props_view();
9891        let mut out = Vec::new();
9892        for id in 0..self.ids.len() as u32 {
9893            let Some(key) = self.ids.key_of(id) else {
9894                continue;
9895            };
9896            let Some(&sym) = self.labels.get(id as usize) else {
9897                continue;
9898            };
9899            if sym == u32::MAX {
9900                continue; // tombstoned
9901            }
9902            for &nbr in tv.neighbors(etype_sym, Direction::Out, id).as_ref() {
9903                let Some(dst_key) = self.ids.key_of(nbr) else {
9904                    continue;
9905                };
9906                let weight = weight_prop.and_then(|prop| match ep.get(etype_sym, id, nbr, prop) {
9907                    Some(Value::Float(f)) => Some(f),
9908                    Some(Value::Int(i)) => Some(i as f64),
9909                    _ => None,
9910                });
9911                out.push((key.to_string(), dst_key.to_string(), weight));
9912            }
9913        }
9914        out.sort_by(|a, b| a.0.cmp(&b.0).then(a.1.cmp(&b.1)));
9915        out
9916    }
9917
9918    pub fn nodes_with_label(&self, label: &str) -> Vec<NodeRef<'_, F>> {
9919        self.view()
9920            .nodes_with_label(label)
9921            .into_iter()
9922            .map(|id| NodeRef { db: self, id })
9923            .collect()
9924    }
9925
9926    pub fn find_nodes(&self, label: &str, filter: &Filter) -> Vec<NodeRef<'_, F>> {
9927        let view = self.view();
9928        view.nodes_with_label(label)
9929            .into_iter()
9930            .filter(|&id| {
9931                eval_filter(filter, &|field| {
9932                    view.prop(id, field).map(|vr| vr.into_value())
9933                })
9934            })
9935            .map(|id| NodeRef { db: self, id })
9936            .collect()
9937    }
9938
9939    /// Returns `true` if any approximate (HNSW) VectorSimilar rule covers
9940    /// `field`.  Use as a capability probe: when `true`, `find_similar_vector`
9941    /// with `label = None` will use the native ANN path rather than the O(n)
9942    /// brute-force scan.
9943    pub fn has_vector_rule(&self, field: &str) -> bool {
9944        self.engine.hnsw_has_rule(field)
9945    }
9946
9947    /// How many HNSW graphs this handle has built from scratch since it was
9948    /// opened (one per side of an approximate rule).
9949    ///
9950    /// An open that restored every graph from the snapshot reports `0`.
9951    /// Exposed for tests that assert the open path reuses the persisted index
9952    /// rather than rebuilding it; not part of the stable surface.
9953    #[doc(hidden)]
9954    pub fn hnsw_build_count(&self) -> u64 {
9955        self.engine.hnsw_build_count()
9956    }
9957
9958    /// How many rules this handle still holds a lazily-decoded HNSW graph for.
9959    ///
9960    /// Zero before the first ANN query on a clean open, and again once the
9961    /// live indexes own the graphs. See [`core_rules::RuleEngine::lazy_hnsw_len`].
9962    /// Exposed for tests that assert the lazy copies are released; not part of
9963    /// the stable surface.
9964    #[doc(hidden)]
9965    pub fn lazy_hnsw_len(&self) -> usize {
9966        self.engine.lazy_hnsw_len()
9967    }
9968
9969    /// Find nodes whose `field` vector is most similar to `q` (cosine
9970    /// similarity), returning up to `k` results with similarity ≥ `min`,
9971    /// sorted descending.
9972    ///
9973    /// When `label` is `None` the search spans all labels (via
9974    /// `hnsw_search_any_dst` or a full brute-force scan); when `label` is
9975    /// `Some(lbl)` it restricts to nodes with that label.
9976    ///
9977    /// Uses the HNSW index when one is available (fast path); otherwise falls
9978    /// back to an O(n) brute-force scan.
9979    ///
9980    /// **The index supplies candidates, never scores.** Its own distances are
9981    /// `f32` (accurate to ~1e-6, so an exact duplicate scores 0.9999999), so
9982    /// every candidate is re-scored from the `f64` property vectors by
9983    /// [`exact_vector_similarity`] before `min`, the ordering and the reported
9984    /// score are decided. `k + VECTOR_RESCORE_MARGIN` candidates are fetched so
9985    /// the re-ordering cannot drop a true top-`k` member; see that constant for
9986    /// the rule. The score a caller receives is therefore the same number the
9987    /// brute-force path would have produced, to `f64` precision, and `min = 1.0`
9988    /// finds an exact duplicate.
9989    pub fn find_similar_vector(
9990        &self,
9991        field: &str,
9992        label: Option<&str>,
9993        q: &[f64],
9994        k: usize,
9995        min: f64,
9996    ) -> Vec<(String, f64)> {
9997        self.find_similar_vector_filtered(field, label, q, k, min, None, None, false)
9998            .expect("find_similar_vector_filtered is infallible without where_")
9999    }
10000
10001    /// Like [`find_similar_vector`] but restricts results to nodes visible in
10002    /// `mask`. Hidden nodes never appear in results; the mask is applied
10003    /// **before** k-truncation so a caller still receives up to `k` visible
10004    /// hits.
10005    ///
10006    /// # HNSW path (widening beam)
10007    ///
10008    /// When an HNSW index covers the request, the beam starts at an over-fetch
10009    /// of `k × n / |visible|` (plus the rescore margin) when the mask's
10010    /// selectivity is known from the index length, otherwise at `k` plus that
10011    /// margin. If fewer than `k` visible candidates remain after the mask and
10012    /// `min` filter, the beam doubles — the same ×2 loop exact `VectorSimilar`
10013    /// rules use, capped at `ef_max()` (`EF_MAX` = 4,096). Reaching the cap,
10014    /// or a beam that comes back short of its own width, falls through to the
10015    /// exhaustive masked scan rather than returning a short result.
10016    ///
10017    /// Every surviving candidate is re-scored from the `f64` property vectors,
10018    /// exactly as [`find_similar_vector`] does and for the same reason.
10019    ///
10020    /// # Brute-force path
10021    ///
10022    /// When no HNSW index covers the request, or the beam cannot admit `k`
10023    /// hits, the function builds a masked [`GraphView`] so that `nodes_all` /
10024    /// `nodes_with_label` return only visible nodes, guaranteeing exact `k`
10025    /// results (or all visible nodes if fewer than `k` exist).
10026    pub fn find_similar_vector_masked(
10027        &self,
10028        field: &str,
10029        label: Option<&str>,
10030        q: &[f64],
10031        k: usize,
10032        min: f64,
10033        mask: &crate::mask::NodeMask,
10034    ) -> Vec<(String, f64)> {
10035        self.find_similar_vector_filtered(field, label, q, k, min, Some(mask), None, false)
10036            .expect("find_similar_vector_filtered is infallible without where_")
10037    }
10038
10039    /// Exact or ANN kNN with optional key-list `mask` and property `where_`.
10040    ///
10041    /// `where_` present and failing [`PropPredicate::validate_named`] `"where"`
10042    /// → `QueryError`. `exact=true` or `where_=Some` skip HNSW and GEMM-brute
10043    /// the candidate set (`label ∩ mask ∩ holds(where)`). `mask` alone still
10044    /// uses HNSW when an index covers the field.
10045    #[allow(clippy::too_many_arguments)]
10046    pub fn find_similar_vector_filtered(
10047        &self,
10048        field: &str,
10049        label: Option<&str>,
10050        q: &[f64],
10051        k: usize,
10052        min: f64,
10053        mask: Option<&crate::mask::NodeMask>,
10054        where_: Option<&PropPredicate>,
10055        exact: bool,
10056    ) -> Result<Vec<(String, f64)>> {
10057        self.find_similar_vector_as(
10058            field,
10059            label,
10060            q,
10061            k,
10062            min,
10063            mask,
10064            where_,
10065            exact,
10066            ExactnessCaller::Vector,
10067        )
10068    }
10069
10070    /// [`find_similar_vector_filtered`](Self::find_similar_vector_filtered)
10071    /// with the caller shape named, so the exactness warning can advise the
10072    /// signature that actually reached it. Everything else is identical.
10073    #[allow(clippy::too_many_arguments)]
10074    fn find_similar_vector_as(
10075        &self,
10076        field: &str,
10077        label: Option<&str>,
10078        q: &[f64],
10079        k: usize,
10080        min: f64,
10081        mask: Option<&crate::mask::NodeMask>,
10082        where_: Option<&PropPredicate>,
10083        exact: bool,
10084        caller: ExactnessCaller,
10085    ) -> Result<Vec<(String, f64)>> {
10086        if let Some(pred) = where_ {
10087            pred.validate_named("where")
10088                .map_err(|detail| GraphError::QueryError { detail })?;
10089        }
10090
10091        // Ensure any HNSW blobs retained from the snapshot are deserialized
10092        // before the first ANN query on a clean-open (no-WAL) path.  The
10093        // section read has to come first: on a clean open nothing else has
10094        // called it, so without it `retained_hnsw_blobs` is empty,
10095        // `ensure_hnsw_loaded` caches an empty map in its `OnceLock`, and every
10096        // approximate query on the handle runs brute force — correct results,
10097        // silently off the index.  Both calls are idempotent and cheap once hot.
10098        self.ensure_v8_base_sections_loaded();
10099        self.engine.ensure_hnsw_loaded();
10100        let norm: f64 = q.iter().map(|x| x * x).sum::<f64>().sqrt();
10101        if norm == 0.0 {
10102            return Ok(vec![]);
10103        }
10104        if let Some(m) = mask {
10105            if k == 0 || m.is_empty() {
10106                return Ok(vec![]);
10107            }
10108        }
10109        let q_unit: Vec<f64> = q.iter().map(|x| x / norm).collect();
10110
10111        // `where` implies exact: a predicate must not ride a silent ANN.
10112        let skip_hnsw = exact || where_.is_some();
10113        if !skip_hnsw {
10114            if let Some(mask) = mask {
10115                if let Some(out) =
10116                    self.find_similar_hnsw_masked(field, label, &q_unit, k, min, mask, caller)
10117                {
10118                    return Ok(out);
10119                }
10120            } else if let Some(out) = self.find_similar_hnsw(field, label, &q_unit, k, min) {
10121                return Ok(out);
10122            }
10123        }
10124
10125        let view = match mask {
10126            Some(m) => self.view_masked(m),
10127            None => self.view(),
10128        };
10129        let candidate_ids = Self::vector_candidates(&view, label, where_);
10130        Ok(self.brute_vector_hits(&view, candidate_ids, field, &q_unit, k, min))
10131    }
10132
10133    /// Unmasked HNSW path. `None` when no populated index covers the request.
10134    fn find_similar_hnsw(
10135        &self,
10136        field: &str,
10137        label: Option<&str>,
10138        q_unit: &[f64],
10139        k: usize,
10140        min: f64,
10141    ) -> Option<Vec<(String, f64)>> {
10142        // Try HNSW fast path.
10143        // `None` label searches across all VectorSimilar rules covering `field`
10144        // (merging their results); `Some(lbl)` restricts to rules whose
10145        // dst_label matches.  Returns `None` when no populated HNSW index
10146        // covers the request — the O(n) brute-force fallback handles that case.
10147        let over_k = k.saturating_add(VECTOR_RESCORE_MARGIN);
10148        let hits = match label {
10149            Some(lbl) => self.engine.hnsw_search_dst(field, lbl, q_unit, over_k)?,
10150            None => self.engine.hnsw_search_any_dst(field, q_unit, over_k)?,
10151        };
10152        // Candidates only: the index's `f32` similarity is discarded and
10153        // each hit is re-scored against the `f64` vectors.
10154        let view = self.view();
10155        let mut out: Vec<(String, f64)> = hits
10156            .into_iter()
10157            .filter_map(|(id, _)| {
10158                let sim = exact_vector_similarity(&view, id, field, q_unit)?;
10159                if sim < min {
10160                    return None;
10161                }
10162                Some((self.ids.key_of(id)?.to_string(), sim))
10163            })
10164            .collect();
10165        out.sort_by(|a, b| {
10166            b.1.partial_cmp(&a.1)
10167                .unwrap_or(std::cmp::Ordering::Equal)
10168                .then_with(|| a.0.cmp(&b.0))
10169        });
10170        out.truncate(k);
10171        Some(out)
10172    }
10173
10174    /// Masked HNSW widening beam. `None` when no index covers the request or
10175    /// the beam cannot admit `k` visible hits (caller falls through to brute).
10176    #[allow(clippy::too_many_arguments)]
10177    fn find_similar_hnsw_masked(
10178        &self,
10179        field: &str,
10180        label: Option<&str>,
10181        q_unit: &[f64],
10182        k: usize,
10183        min: f64,
10184        mask: &crate::mask::NodeMask,
10185        caller: ExactnessCaller,
10186    ) -> Option<Vec<(String, f64)>> {
10187        let index_len = match label {
10188            Some(lbl) => self.engine.hnsw_dst_len(field, lbl, q_unit.len()),
10189            None => self.engine.hnsw_any_dst_len(field, q_unit.len()),
10190        };
10191        let n = index_len?;
10192        // The `?` above is the coverage test: past it, an index exists and this
10193        // masked, non-exact call is about to ride it.
10194        self.note_ambiguous_exactness(field, label, caller);
10195        // Same ceiling the exact-rule widening loop in `hnsw_candidates`
10196        // consults — including the `with_ef_max` test hook.
10197        let cap = ef_max();
10198        let visible = mask.len();
10199        let mut ef = k.saturating_add(VECTOR_RESCORE_MARGIN);
10200        if visible > 0 && n > 0 {
10201            let over = k
10202                .saturating_mul(n)
10203                .div_ceil(visible)
10204                .saturating_add(VECTOR_RESCORE_MARGIN);
10205            ef = ef.max(over);
10206        }
10207        loop {
10208            let hits = match label {
10209                Some(lbl) => self
10210                    .engine
10211                    .hnsw_search_dst_with_ef(field, lbl, q_unit, ef, ef),
10212                None => self
10213                    .engine
10214                    .hnsw_search_any_dst_with_ef(field, q_unit, ef, ef),
10215            };
10216            let hits = hits?;
10217            let full = hits.len() == ef;
10218            let mut out = self.score_masked_hnsw_hits(&hits, field, q_unit, min, mask);
10219            if out.len() >= k {
10220                out.truncate(k);
10221                return Some(out);
10222            }
10223            // Short of its width (frontier exhausted) or at the ceiling:
10224            // a wider beam reaches nothing new, so the scan answers.
10225            if !full || ef >= cap {
10226                return None;
10227            }
10228            ef = ef.saturating_mul(2);
10229        }
10230    }
10231
10232    /// Say once, per `(field, label)` index and caller shape, that a masked
10233    /// search is answering approximately.
10234    ///
10235    /// A mask narrows *which nodes may be returned*. It does not choose a
10236    /// kernel — `exact=true` and a `where=` predicate do, and nothing else
10237    /// does. A caller who needed exact answers, passed `mask=` alone, and read
10238    /// the mask as a promise of exhaustiveness gets a correct-looking
10239    /// approximate answer and no signal at all; that is a silent wrong answer,
10240    /// and it has cost an integration team real time.
10241    ///
10242    /// The fix is a question, not a behaviour change. Making a mask imply
10243    /// `exact` would turn every existing masked caller's ANN into an O(n) GEMM
10244    /// without asking them, which is a worse trade than the ambiguity.
10245    ///
10246    /// Printed once per index for the reason the dimension-mismatch skip in
10247    /// `core_rules::hnsw` is: a line on every call is a line callers learn to
10248    /// scroll past.
10249    ///
10250    /// `caller` decides the advice. The same leg is reached from two signatures
10251    /// and only one of them has an `exact` argument to pass; see
10252    /// [`ExactnessCaller`].
10253    fn note_ambiguous_exactness(&self, field: &str, label: Option<&str>, caller: ExactnessCaller) {
10254        let entry = (field.to_string(), label.unwrap_or("").to_string(), caller);
10255        let first = match self.warned_ambiguous_exactness.lock() {
10256            Ok(mut seen) => seen.insert(entry),
10257            Err(poisoned) => poisoned.into_inner().insert(entry),
10258        };
10259        if !first {
10260            return;
10261        }
10262        AMBIGUOUS_EXACTNESS_WARNS.with(|c| c.set(c.get().saturating_add(1)));
10263        let which = match label {
10264            Some(lbl) => format!(" (label `{lbl}`)"),
10265            None => String::new(),
10266        };
10267        let subject = caller.subject();
10268        let advice = caller.advice();
10269        let line = format!(
10270            "mushroomdb: {subject} on field `{field}`{which} is answering \
10271             approximately. A mask narrows which nodes may be returned; it does not \
10272             change which kernel runs, and an index covers this field. For an exact \
10273             answer over the same visible candidate set, {advice} Further masked \
10274             searches of this shape on this index are silent."
10275        );
10276        AMBIGUOUS_EXACTNESS_LAST.with(|c| *c.borrow_mut() = Some(line.clone()));
10277        eprintln!("{line}");
10278    }
10279
10280    /// `label ∩ mask ∩ holds(where)`. Index fast path when `label` is `Some`
10281    /// and `(label, where.field)` is enabled; otherwise scan with `visible()`.
10282    fn vector_candidates(
10283        view: &GraphView<'_>,
10284        label: Option<&str>,
10285        where_: Option<&PropPredicate>,
10286    ) -> Vec<u32> {
10287        if let (Some(lbl), Some(pred)) = (label, where_) {
10288            let indexed = view
10289                .prop_index
10290                .is_some_and(|idx| idx.is_enabled(lbl, &pred.field));
10291            if indexed {
10292                match (&pred.eq, &pred.in_) {
10293                    (Some(eq), None) => {
10294                        if let Some(ids) = view.nodes_with_prop(lbl, &pred.field, eq) {
10295                            return ids;
10296                        }
10297                    }
10298                    (None, Some(allowed)) => {
10299                        let mut seen = HashSet::new();
10300                        let mut out = Vec::new();
10301                        for v in allowed {
10302                            if let Some(ids) = view.nodes_with_prop(lbl, &pred.field, v) {
10303                                for id in ids {
10304                                    if seen.insert(id) {
10305                                        out.push(id);
10306                                    }
10307                                }
10308                            }
10309                        }
10310                        return out;
10311                    }
10312                    _ => {}
10313                }
10314            }
10315        }
10316
10317        let mut ids: Vec<u32> = match label {
10318            Some(lbl) => view
10319                .nodes_with_label(lbl)
10320                .into_iter()
10321                .filter(|&id| view.visible(id))
10322                .collect(),
10323            None => view.nodes_all(),
10324        };
10325        if let Some(pred) = where_ {
10326            ids.retain(|&id| match view.prop(id, &pred.field) {
10327                None => pred.holds(None),
10328                Some(vr) => pred.holds(Some(vr.as_value())),
10329            });
10330        }
10331        ids
10332    }
10333
10334    /// Exact brute kNN: pack candidates at `q_unit`'s dim, GEMV, keep
10335    /// `score >= min`, sort `(sim desc, key asc)`, truncate to `k`.
10336    fn brute_vector_hits(
10337        &self,
10338        view: &GraphView<'_>,
10339        candidate_ids: impl IntoIterator<Item = u32>,
10340        field: &str,
10341        q_unit: &[f64],
10342        k: usize,
10343        min: f64,
10344    ) -> Vec<(String, f64)> {
10345        let rows: Vec<(u32, std::borrow::Cow<'_, [f64]>)> = candidate_ids
10346            .into_iter()
10347            .filter_map(|id| crate::exact_knn::vector_f64(view, id, field).map(|v| (id, v)))
10348            .collect();
10349        let packed =
10350            crate::exact_knn::pack(rows.iter().map(|(id, v)| (*id, v.as_ref())), q_unit.len());
10351        let scores = crate::exact_knn::gemv(&packed, q_unit);
10352        let mut scored: Vec<(String, f64)> = packed
10353            .ids
10354            .iter()
10355            .zip(scores.iter())
10356            .filter_map(|(&id, &sim)| {
10357                if sim < min {
10358                    return None;
10359                }
10360                let key = self.ids.key_of(id)?.to_string();
10361                Some((key, sim))
10362            })
10363            .collect();
10364        scored.sort_by(|a, b| {
10365            b.1.partial_cmp(&a.1)
10366                .unwrap_or(std::cmp::Ordering::Equal)
10367                .then_with(|| a.0.cmp(&b.0))
10368        });
10369        scored.truncate(k);
10370        scored
10371    }
10372
10373    /// Exact cosine top-k for each key in `keys`, scored only against `keys`.
10374    ///
10375    /// `min` is cosine similarity in [-1, 1], inclusive (`score >= min`), the
10376    /// same unit and inequality as `find_similar_vector`. Self-matches are
10377    /// excluded. Unknown keys, keys with no `field`, zero-norm or wrong-dim
10378    /// embeddings are omitted as both query and candidate. Duplicate keys are
10379    /// collapsed, first-seen order. Empty `keys` → empty `Ok(vec![])`. Never
10380    /// uses HNSW. `n > PAIRWISE_MAX_N` → `QueryError`.
10381    #[allow(clippy::type_complexity)]
10382    pub fn pairwise_similar(
10383        &self,
10384        keys: &[&str],
10385        field: &str,
10386        k: usize,
10387        min: f64,
10388    ) -> Result<Vec<(String, Vec<(String, f64)>)>> {
10389        let mut seen = HashSet::new();
10390        let mut unique_ids = Vec::new();
10391        for key in keys {
10392            let Some(id) = self.ids.get(key) else {
10393                continue;
10394            };
10395            if seen.insert(id) {
10396                unique_ids.push(id);
10397            }
10398        }
10399        let max_n = crate::exact_knn::pairwise_max_n();
10400        if unique_ids.len() > max_n {
10401            return Err(GraphError::QueryError {
10402                detail: format!(
10403                    "pairwise_similar: n={} exceeds PAIRWISE_MAX_N ({max_n})",
10404                    unique_ids.len()
10405                ),
10406            });
10407        }
10408        if unique_ids.is_empty() {
10409            return Ok(Vec::new());
10410        }
10411
10412        let view = self.view();
10413        let mut rows: Vec<(u32, std::borrow::Cow<'_, [f64]>)> = Vec::new();
10414        let mut counts: HashMap<usize, usize> = HashMap::new();
10415        for id in unique_ids {
10416            let Some(v) = crate::exact_knn::vector_f64(&view, id, field) else {
10417                continue;
10418            };
10419            let norm: f64 = v.iter().map(|x| x * x).sum::<f64>().sqrt();
10420            if norm == 0.0 {
10421                continue;
10422            }
10423            *counts.entry(v.len()).or_default() += 1;
10424            rows.push((id, v));
10425        }
10426        if rows.is_empty() {
10427            return Ok(Vec::new());
10428        }
10429        let dim = counts
10430            .into_iter()
10431            .max_by_key(|&(d, c)| (c, d))
10432            .map(|(d, _)| d)
10433            .expect("rows non-empty");
10434        let packed = crate::exact_knn::pack(rows.iter().map(|(id, v)| (*id, v.as_ref())), dim);
10435        let n = packed.ids.len();
10436        if n == 0 {
10437            return Ok(Vec::new());
10438        }
10439        let src_keys: Vec<String> = packed
10440            .ids
10441            .iter()
10442            .map(|&id| self.ids.key_of(id).unwrap_or("").to_string())
10443            .collect();
10444
10445        let mut out = Vec::with_capacity(n);
10446        if n <= crate::exact_knn::pairwise_gram_max() {
10447            let sims = crate::exact_knn::gram(&packed);
10448            for i in 0..n {
10449                out.push(Self::topk_from_row(
10450                    &src_keys,
10451                    i,
10452                    &sims[i * n..(i + 1) * n],
10453                    k,
10454                    min,
10455                ));
10456            }
10457        } else {
10458            for i in 0..n {
10459                let row = &packed.data[i * packed.dim..(i + 1) * packed.dim];
10460                let scores = crate::exact_knn::gemv(&packed, row);
10461                out.push(Self::topk_from_row(&src_keys, i, &scores, k, min));
10462            }
10463        }
10464        Ok(out)
10465    }
10466
10467    /// [`pairwise_similar`](Self::pairwise_similar) over the keys the mask
10468    /// admits — intersected **before** the matmul, never filtered after it.
10469    ///
10470    /// A hidden vector packed into the Gram is a row every visible key is
10471    /// scored against. It can take a visible neighbour's place in the top-`k`,
10472    /// and because the packed dimension is a majority vote over the candidate
10473    /// rows it can decide whether a visible pair is scored at all. Dropping
10474    /// hidden names from the finished answer leaves both effects standing, so
10475    /// the intersection happens first and the answer is byte-for-byte the one
10476    /// `pairwise_similar` gives for the visible keys alone.
10477    ///
10478    /// The caps therefore measure the **post-filter** count: a key set over
10479    /// [`PAIRWISE_MAX_N`](crate::PAIRWISE_MAX_N) unscoped can come under it
10480    /// scoped and succeed, because the work the cap refuses is work this call
10481    /// no longer does. A filtered count still over the cap is still refused.
10482    #[allow(clippy::type_complexity)]
10483    pub fn pairwise_similar_scoped(
10484        &self,
10485        keys: &[&str],
10486        field: &str,
10487        k: usize,
10488        min: f64,
10489        mask: &crate::mask::NodeMask,
10490    ) -> Result<Vec<(String, Vec<(String, f64)>)>> {
10491        let visible: Vec<&str> = keys
10492            .iter()
10493            .copied()
10494            .filter(|key| mask.contains_node(self, key))
10495            .collect();
10496        self.pairwise_similar(&visible, field, k, min)
10497    }
10498
10499    /// Neighbours of packed row `i`: drop self, keep `score >= min`, sort
10500    /// `(sim desc, key asc)`, truncate to `k`. Packed srcs with no survivors
10501    /// still appear as `(src, [])`.
10502    fn topk_from_row(
10503        src_keys: &[String],
10504        i: usize,
10505        scores: &[f64],
10506        k: usize,
10507        min: f64,
10508    ) -> (String, Vec<(String, f64)>) {
10509        let mut neigh: Vec<(String, f64)> = scores
10510            .iter()
10511            .enumerate()
10512            .filter_map(|(j, &sim)| {
10513                if i == j || sim < min {
10514                    return None;
10515                }
10516                Some((src_keys[j].clone(), sim))
10517            })
10518            .collect();
10519        neigh.sort_by(|a, b| {
10520            b.1.partial_cmp(&a.1)
10521                .unwrap_or(std::cmp::Ordering::Equal)
10522                .then_with(|| a.0.cmp(&b.0))
10523        });
10524        neigh.truncate(k);
10525        (src_keys[i].clone(), neigh)
10526    }
10527
10528    /// Re-score HNSW candidates from the `f64` vectors, drop hidden / below-`min`
10529    /// hits, order by score then key. The index's own `f32` similarity is discarded.
10530    fn score_masked_hnsw_hits(
10531        &self,
10532        hits: &[(u32, f64)],
10533        field: &str,
10534        q_unit: &[f64],
10535        min: f64,
10536        mask: &crate::mask::NodeMask,
10537    ) -> Vec<(String, f64)> {
10538        let view = self.view_masked(mask);
10539        let mut out: Vec<(String, f64)> = hits
10540            .iter()
10541            .copied()
10542            .filter(|&(id, _)| mask.contains_id(id))
10543            .filter_map(|(id, _)| {
10544                let sim = exact_vector_similarity(&view, id, field, q_unit)?;
10545                if sim < min {
10546                    return None;
10547                }
10548                Some((self.ids.key_of(id)?.to_string(), sim))
10549            })
10550            .collect();
10551        out.sort_by(|a, b| {
10552            b.1.partial_cmp(&a.1)
10553                .unwrap_or(std::cmp::Ordering::Equal)
10554                .then_with(|| a.0.cmp(&b.0))
10555        });
10556        out
10557    }
10558
10559    /// Read a single property from an edge.
10560    ///
10561    /// Returns `None` when the edge does not exist, the field is absent, or any
10562    /// of the string keys cannot be resolved to interned ids.  Only edge props
10563    /// written by rules (weight fields) are accessible without a `set_edge_prop`
10564    /// binding; topology-only edges (no props set) return `None` for every field.
10565    pub fn get_edge_prop(
10566        &self,
10567        edge_type: &str,
10568        src_key: &str,
10569        dst_key: &str,
10570        field: &str,
10571    ) -> Option<Value> {
10572        let etype = self.syms.get(edge_type)?;
10573        let src = self.ids.get(src_key)?;
10574        let dst = self.ids.get(dst_key)?;
10575        self.edge_props_view().get(etype, src, dst, field)
10576    }
10577
10578    /// Lex → parse → plan → execute `cypher` over a read-only view.
10579    /// Every pipeline `Err(String)` becomes `GraphError::QueryError` with a
10580    /// stage prefix (`lex:` / `parse:` / `plan:` / `execute:`).
10581    pub fn query(&self, cypher: &str, params: &BTreeMap<String, Value>) -> Result<ResultSet> {
10582        let tokens = lex(cypher).map_err(|e| GraphError::QueryError {
10583            detail: format!("lex: {e}"),
10584        })?;
10585        let union = parse_read(&tokens).map_err(|e| GraphError::QueryError {
10586            detail: format!("parse: {e}"),
10587        })?;
10588        let t0 = std::time::Instant::now();
10589        let result = execute_union(&self.view(), &union, &Params(params)).map_err(|e| {
10590            GraphError::QueryError {
10591                detail: format!("execute: {e}"),
10592            }
10593        });
10594        let elapsed_ms = t0.elapsed().as_millis() as u64;
10595        let threshold = self.slow_query_threshold_ms;
10596        if threshold > 0 && elapsed_ms >= threshold {
10597            eprintln!("[mushroomdb] slow query ({elapsed_ms}ms): {cypher}");
10598            let entry = SlowQueryEntry {
10599                ms: elapsed_ms,
10600                query: cypher.to_string(),
10601                at_commit: self.commit_seq,
10602            };
10603            if let Ok(mut log) = self.slow_queries.lock() {
10604                if log.entries.len() == SLOW_QUERY_RING_CAP {
10605                    log.entries.pop_front();
10606                }
10607                log.entries.push_back(entry);
10608                log.total += 1;
10609            }
10610        }
10611        result
10612    }
10613
10614    /// Convenience entry-point that accepts a slice of `(name, value)` pairs
10615    /// instead of a pre-built `BTreeMap`.  Equivalent to building the map and
10616    /// calling [`GraphDb::query`].
10617    pub fn query_with_params(&self, cypher: &str, params: &[(&str, Value)]) -> Result<ResultSet> {
10618        let map: BTreeMap<String, Value> = params
10619            .iter()
10620            .map(|(k, v)| (k.to_string(), v.clone()))
10621            .collect();
10622        self.query(cypher, &map)
10623    }
10624
10625    /// Execute a Cypher write statement (CREATE / MATCH…SET / MATCH…DELETE / MERGE).
10626    ///
10627    /// All mutations flow through the same `insert_node` / `set_prop` /
10628    /// `delete_edge` / `insert_edge` path as the Rust API so the rule engine
10629    /// fires and the WAL captures everything with one fsync per statement.
10630    ///
10631    /// Returns a one-row [`ResultSet`] with columns `created`, `properties_set`,
10632    /// and `deleted` matching the write-result contract.
10633    ///
10634    /// **Mutation routing**: mutations are collected into a single
10635    /// [`BatchBuilder`] and committed atomically (one WAL `Batch` frame, one
10636    /// fsync). The MATCH phase for SET/DELETE uses a read-only `execute` call
10637    /// over `self.view()` — the borrow is dropped before the batch is opened.
10638    ///
10639    /// **Limitations (v1)**:
10640    /// - SET RHS must be a literal, `$param`, or arithmetic; bare property copy → named error.
10641    /// - `DETACH DELETE n` → calls `delete_node` for each matched node (removes all edges).
10642    /// - Bare `DELETE n` → error if n has any incident edges; succeeds for isolated nodes.
10643    /// - MERGE supports `ON CREATE SET` / `ON MATCH SET` in the same write batch.
10644    /// - Deleting a derived edge → named error "cannot delete derived edge".
10645    pub fn query_write(
10646        &mut self,
10647        cypher: &str,
10648        params: &BTreeMap<String, Value>,
10649    ) -> Result<ResultSet> {
10650        let tokens = lex(cypher).map_err(|e| GraphError::QueryError {
10651            detail: format!("lex: {e}"),
10652        })?;
10653        let stmt = parse_write(&tokens).map_err(|e| GraphError::QueryError {
10654            detail: format!("parse: {e}"),
10655        })?;
10656        self.exec_write_stmt(stmt, params)
10657    }
10658
10659    fn exec_write_stmt(
10660        &mut self,
10661        stmt: WriteStatement,
10662        params: &BTreeMap<String, Value>,
10663    ) -> Result<ResultSet> {
10664        match stmt {
10665            WriteStatement::Create(s) => self.exec_create(s, params),
10666            WriteStatement::MatchSet(s) => self.exec_match_set(s, params),
10667            WriteStatement::MatchDelete(s) => self.exec_match_delete(s, params),
10668            WriteStatement::MatchDeleteNode(s) => self.exec_match_delete_node(s, params),
10669            WriteStatement::Merge(s) => self.exec_merge(s, params),
10670        }
10671    }
10672
10673    fn exec_create(
10674        &mut self,
10675        stmt: core_query::cypher::CreateStmt,
10676        params: &BTreeMap<String, Value>,
10677    ) -> Result<ResultSet> {
10678        // Extract the node key from props: require a string-valued `id` field.
10679        let mut var_to_key: BTreeMap<String, String> = BTreeMap::new();
10680        for node in &stmt.nodes {
10681            let var = node.var.as_deref().unwrap_or("_cn0");
10682            let key = node
10683                .props
10684                .iter()
10685                .find(|(f, _)| f == "id")
10686                .and_then(|(_, v)| {
10687                    if let Value::Str(s) = v {
10688                        Some(s.clone())
10689                    } else {
10690                        None
10691                    }
10692                })
10693                .ok_or_else(|| GraphError::QueryError {
10694                    detail: format!(
10695                        "CREATE node ({}:{}) requires a string 'id' property",
10696                        var, node.label
10697                    ),
10698                })?;
10699            var_to_key.insert(var.to_string(), key);
10700        }
10701
10702        let mut batch = self.batch();
10703        let mut created: usize = 0;
10704        for node in &stmt.nodes {
10705            let var = node.var.as_deref().unwrap_or("_cn0");
10706            let key = &var_to_key[var];
10707            batch.insert_node(&node.label, key, node.props.clone());
10708            created += 1;
10709        }
10710        for edge in &stmt.edges {
10711            let src_key = var_to_key
10712                .get(&edge.src_var)
10713                .ok_or_else(|| GraphError::QueryError {
10714                    detail: format!("CREATE edge src variable '{}' is not bound", edge.src_var),
10715                })?;
10716            let dst_key = var_to_key
10717                .get(&edge.dst_var)
10718                .ok_or_else(|| GraphError::QueryError {
10719                    detail: format!("CREATE edge dst variable '{}' is not bound", edge.dst_var),
10720                })?;
10721            batch.insert_edge(&edge.etype, src_key, dst_key);
10722        }
10723        batch.commit()?;
10724
10725        // Optional RETURN clause: project created bindings as a read result.
10726        if let Some(returns) = stmt.returns {
10727            // Each created node is looked up by its key via a separate MATCH pattern.
10728            // Multiple single-node patterns cross-join to produce 1 output row with
10729            // all variables bound (each pattern returns exactly 1 row).
10730            let patterns: Vec<Pattern> = stmt
10731                .nodes
10732                .iter()
10733                .map(|node| {
10734                    let var = node.var.as_deref().unwrap_or("_cn0");
10735                    let key = var_to_key[var].clone();
10736                    Pattern {
10737                        start: NodePat {
10738                            var: Some(var.to_string()),
10739                            label: Some(node.label.clone()),
10740                            props: vec![("id".to_string(), Operand::Lit(Value::Str(key)))],
10741                        },
10742                        chain: vec![],
10743                        shortest: false,
10744                    }
10745                })
10746                .collect();
10747            let q = Query {
10748                matches: patterns,
10749                optional_clauses: vec![],
10750                where_expr: None,
10751                unwinds: vec![],
10752                post_unwind_where: None,
10753                stages: vec![],
10754                returns,
10755                distinct: false,
10756                order_by: vec![],
10757                skip: None,
10758                limit: None,
10759            };
10760            let ops = plan(&q).map_err(|e| GraphError::QueryError {
10761                detail: format!("plan: {e}"),
10762            })?;
10763            return execute(&self.view(), &ops, &Params(params)).map_err(|e| {
10764                GraphError::QueryError {
10765                    detail: format!("execute: {e}"),
10766                }
10767            });
10768        }
10769
10770        let mut rs = write_result_set();
10771        rs.push_row(vec![
10772            Some(Value::Int(created as i64)),
10773            Some(Value::Int(0)),
10774            Some(Value::Int(0)),
10775        ]);
10776        Ok(rs)
10777    }
10778
10779    fn exec_match_set(
10780        &mut self,
10781        stmt: core_query::cypher::MatchSetStmt,
10782        params: &BTreeMap<String, Value>,
10783    ) -> Result<ResultSet> {
10784        let project_returns = stmt.returns.clone();
10785        // Collect unique node vars targeted by SET clauses, plus RETURN bindings
10786        // so the post-write projection can look them up by key.
10787        let mut set_vars: Vec<String> = Vec::new();
10788        for s in &stmt.sets {
10789            if !set_vars.contains(&s.var) {
10790                set_vars.push(s.var.clone());
10791            }
10792        }
10793        let rel_vars = pattern_rel_vars(&stmt.matches);
10794        // `count` is the engine's, on an edge: it is the insert-count §5.13
10795        // maintains, and a `SET` that overwrote it would make the number mean
10796        // whatever the last writer said rather than how many times the pair was
10797        // inserted. Refused by name here, before the match runs, so the caller
10798        // is told what is actually wrong instead of meeting the executor's
10799        // generic "did not resolve to a node key" — and so the answer does not
10800        // depend on whether the pattern happened to match a row. The same name
10801        // on a *node* is an ordinary property and is untouched.
10802        for s in &stmt.sets {
10803            if s.field == EDGE_COUNT_PROP && rel_vars.iter().any(|r| r == &s.var) {
10804                return Err(GraphError::QueryError {
10805                    detail: format!(
10806                        "cannot SET {}.{EDGE_COUNT_PROP}: `{EDGE_COUNT_PROP}` is a reserved edge \
10807                         property holding the pair's insert count",
10808                        s.var
10809                    ),
10810                });
10811            }
10812        }
10813        let mut lookup_vars = set_vars.clone();
10814        for v in pattern_node_vars(&stmt.matches) {
10815            add_var(&mut lookup_vars, &v);
10816        }
10817        if let Some(ref returns) = project_returns {
10818            for v in ret_node_vars(returns) {
10819                if !rel_vars.iter().any(|r| r == &v) {
10820                    add_var(&mut lookup_vars, &v);
10821                }
10822            }
10823        }
10824
10825        // Synthesize a read query: MATCH … WHERE … RETURN <lookup_vars>, <set_values…>
10826        // SET values are projected as ScalarExpr items so that arithmetic expressions
10827        // (e.g. `SET n.score = n.score * 1.5`) are evaluated in the matched-row context.
10828        let mut set_returns: Vec<RetItem> = lookup_vars
10829            .iter()
10830            .map(|v| RetItem {
10831                value: RetVal::Var(v.clone()),
10832                alias: None,
10833            })
10834            .collect();
10835        // One computed column per SET clause; alias is `__sv_<i>`.
10836        let set_val_cols: Vec<String> = stmt
10837            .sets
10838            .iter()
10839            .enumerate()
10840            .map(|(i, _)| format!("__sv_{i}"))
10841            .collect();
10842        for (sc, col) in stmt.sets.iter().zip(&set_val_cols) {
10843            set_returns.push(RetItem {
10844                value: RetVal::ScalarExpr(sc.value.clone()),
10845                alias: Some(col.clone()),
10846            });
10847        }
10848        // Capture relationship types while r is bound; SET does not change them.
10849        for r in &rel_vars {
10850            set_returns.push(RetItem {
10851                value: RetVal::FuncCall {
10852                    name: "type".into(),
10853                    args: vec![Operand::Var(r.clone())],
10854                },
10855                alias: Some(rel_type_alias(r)),
10856            });
10857        }
10858
10859        let read_q = Query {
10860            matches: stmt.matches.clone(),
10861            optional_clauses: vec![],
10862            where_expr: stmt.where_expr.clone(),
10863            unwinds: vec![],
10864            post_unwind_where: None,
10865            stages: vec![],
10866            returns: set_returns,
10867            distinct: false,
10868            order_by: vec![],
10869            skip: None,
10870            limit: None,
10871        };
10872        let ops = plan(&read_q).map_err(|e| GraphError::QueryError {
10873            detail: format!("plan: {e}"),
10874        })?;
10875        // MATCH phase is read-only; borrow ends before batch opens.
10876        //
10877        // When a role-scoped write is in flight, run the MATCH read through
10878        // view_masked so hidden nodes are invisible → hidden ≡ absent ≡
10879        // zero-rows (no SetProp ops generated, no existence-oracle 403).
10880        // Full-authority writes (pending_write_authz=None) keep view().
10881        let match_rs = {
10882            let mask_opt = self.pending_write_authz.as_ref().map(|a| a.mask.clone());
10883            if let Some(ref mask) = mask_opt {
10884                execute(&self.view_masked(mask), &ops, &Params(params))
10885            } else {
10886                execute(&self.view(), &ops, &Params(params))
10887            }
10888        }
10889        .map_err(|e| GraphError::QueryError {
10890            detail: format!("execute: {e}"),
10891        })?;
10892
10893        // Collect (key, field, value) for each matched row × each SET clause.
10894        let mut set_ops: Vec<(String, String, Value)> = Vec::new();
10895        for row_i in 0..match_rs.len() {
10896            for (sc, col) in stmt.sets.iter().zip(&set_val_cols) {
10897                let key = match match_rs.get(row_i, &sc.var) {
10898                    Some(Value::Str(k)) => k.clone(),
10899                    _ => {
10900                        return Err(GraphError::QueryError {
10901                            detail: format!(
10902                                "SET variable '{}' did not resolve to a node key",
10903                                sc.var
10904                            ),
10905                        })
10906                    }
10907                };
10908                // The SET value was already evaluated by the executor.
10909                let value = match match_rs.get(row_i, col) {
10910                    Some(v) => v.clone(),
10911                    None => {
10912                        return Err(GraphError::QueryError {
10913                            detail: format!(
10914                                "SET value for {}.{} evaluated to null",
10915                                sc.var, sc.field
10916                            ),
10917                        })
10918                    }
10919                };
10920                set_ops.push((key, sc.field.clone(), value));
10921            }
10922        }
10923
10924        // Apply as one atomic batch.
10925        let props_set = set_ops.len();
10926        let mut batch = self.batch();
10927        for (key, field, value) in set_ops {
10928            batch.set_prop(&key, &field, value);
10929        }
10930        batch.commit()?;
10931
10932        if let Some(returns) = project_returns {
10933            return project_set_return_rows(self, &rel_vars, &match_rs, &returns, params);
10934        }
10935
10936        let mut rs = write_result_set();
10937        rs.push_row(vec![
10938            Some(Value::Int(0)),
10939            Some(Value::Int(props_set as i64)),
10940            Some(Value::Int(0)),
10941        ]);
10942        Ok(rs)
10943    }
10944
10945    fn exec_match_delete(
10946        &mut self,
10947        stmt: core_query::cypher::MatchDeleteStmt,
10948        params: &BTreeMap<String, Value>,
10949    ) -> Result<ResultSet> {
10950        // Collect unique node vars needed to identify edge endpoints.
10951        let mut node_vars: Vec<String> = Vec::new();
10952        for ed in &stmt.deletes {
10953            if !node_vars.contains(&ed.src_var) {
10954                node_vars.push(ed.src_var.clone());
10955            }
10956            if !node_vars.contains(&ed.dst_var) {
10957                node_vars.push(ed.dst_var.clone());
10958            }
10959        }
10960
10961        // Synthesize read query.
10962        let returns: Vec<RetItem> = node_vars
10963            .iter()
10964            .map(|v| RetItem {
10965                value: RetVal::Var(v.clone()),
10966                alias: None,
10967            })
10968            .collect();
10969        let read_q = Query {
10970            matches: stmt.matches,
10971            optional_clauses: vec![],
10972            where_expr: stmt.where_expr,
10973            unwinds: vec![],
10974            post_unwind_where: None,
10975            stages: vec![],
10976            returns,
10977            distinct: false,
10978            order_by: vec![],
10979            skip: None,
10980            limit: None,
10981        };
10982        let ops = plan(&read_q).map_err(|e| GraphError::QueryError {
10983            detail: format!("plan: {e}"),
10984        })?;
10985        // Role-scoped writes: mask the MATCH read phase so hidden nodes are
10986        // invisible → hidden ≡ absent ≡ zero-rows (spec §3.1, hidden ≡ absent).
10987        let match_rs = {
10988            let mask_opt = self.pending_write_authz.as_ref().map(|a| a.mask.clone());
10989            if let Some(ref mask) = mask_opt {
10990                execute(&self.view_masked(mask), &ops, &Params(params))
10991            } else {
10992                execute(&self.view(), &ops, &Params(params))
10993            }
10994        }
10995        .map_err(|e| GraphError::QueryError {
10996            detail: format!("execute: {e}"),
10997        })?;
10998
10999        // Collect (etype, src_key, dst_key) for each row × each delete target.
11000        let mut del_ops: Vec<(String, String, String)> = Vec::new();
11001        for row_i in 0..match_rs.len() {
11002            for ed in &stmt.deletes {
11003                let src_key = match match_rs.get(row_i, &ed.src_var) {
11004                    Some(Value::Str(k)) => k.clone(),
11005                    _ => {
11006                        return Err(GraphError::QueryError {
11007                            detail: format!(
11008                                "DELETE src variable '{}' did not resolve to a node key",
11009                                ed.src_var
11010                            ),
11011                        })
11012                    }
11013                };
11014                let dst_key = match match_rs.get(row_i, &ed.dst_var) {
11015                    Some(Value::Str(k)) => k.clone(),
11016                    _ => {
11017                        return Err(GraphError::QueryError {
11018                            detail: format!(
11019                                "DELETE dst variable '{}' did not resolve to a node key",
11020                                ed.dst_var
11021                            ),
11022                        })
11023                    }
11024                };
11025                del_ops.push((ed.etype.clone(), src_key, dst_key));
11026            }
11027        }
11028
11029        // Apply as one atomic batch.
11030        let deleted = del_ops.len();
11031        let mut batch = self.batch();
11032        for (etype, src_key, dst_key) in del_ops {
11033            batch.delete_edge(&etype, &src_key, &dst_key);
11034        }
11035        batch.commit().map_err(|e| match e {
11036            GraphError::RuleOwned { .. } => GraphError::QueryError {
11037                detail: "cannot delete derived edge; retract via the rule or change the property"
11038                    .to_string(),
11039            },
11040            other => other,
11041        })?;
11042
11043        let mut rs = write_result_set();
11044        rs.push_row(vec![
11045            Some(Value::Int(0)),
11046            Some(Value::Int(0)),
11047            Some(Value::Int(deleted as i64)),
11048        ]);
11049        Ok(rs)
11050    }
11051
11052    /// Execute `MATCH … [DETACH] DELETE <node_var> [, …]`.
11053    ///
11054    /// Collects the matching node keys via an ephemeral read query, then calls
11055    /// `delete_node` on each one.  When `stmt.detach` is `false` (bare DELETE)
11056    /// the executor first checks that the node has no incident edges; if any
11057    /// remain it returns a named error matching openCypher semantics.
11058    fn exec_match_delete_node(
11059        &mut self,
11060        stmt: MatchDeleteNodeStmt,
11061        params: &BTreeMap<String, Value>,
11062    ) -> Result<ResultSet> {
11063        // Build a read query returning only the node keys we need.
11064        let returns: Vec<RetItem> = stmt
11065            .node_vars
11066            .iter()
11067            .map(|v| RetItem {
11068                value: RetVal::Var(v.clone()),
11069                alias: None,
11070            })
11071            .collect();
11072        let read_q = Query {
11073            matches: stmt.matches,
11074            optional_clauses: vec![],
11075            where_expr: stmt.where_expr,
11076            unwinds: vec![],
11077            post_unwind_where: None,
11078            stages: vec![],
11079            returns,
11080            distinct: false,
11081            order_by: vec![],
11082            skip: None,
11083            limit: None,
11084        };
11085        let ops = plan(&read_q).map_err(|e| GraphError::QueryError {
11086            detail: format!("plan: {e}"),
11087        })?;
11088        // Role-scoped writes: mask the MATCH read phase so hidden nodes are
11089        // invisible → hidden ≡ absent ≡ zero-rows (spec §3.1, hidden ≡ absent).
11090        let match_rs = {
11091            let mask_opt = self.pending_write_authz.as_ref().map(|a| a.mask.clone());
11092            if let Some(ref mask) = mask_opt {
11093                execute(&self.view_masked(mask), &ops, &Params(params))
11094            } else {
11095                execute(&self.view(), &ops, &Params(params))
11096            }
11097        }
11098        .map_err(|e| GraphError::QueryError {
11099            detail: format!("execute: {e}"),
11100        })?;
11101
11102        // Collect unique node keys to delete (deduplicate across rows × vars).
11103        let mut keys: Vec<String> = Vec::new();
11104        for row_i in 0..match_rs.len() {
11105            for var in &stmt.node_vars {
11106                if let Some(Value::Str(k)) = match_rs.get(row_i, var) {
11107                    if !keys.contains(k) {
11108                        keys.push(k.clone());
11109                    }
11110                }
11111            }
11112        }
11113
11114        if !stmt.detach {
11115            // openCypher bare DELETE: error if any matched node has incident edges.
11116            for key in &keys {
11117                if let Some(id) = self.ids.get(key) {
11118                    let tv = self.topo_view();
11119                    let has_edges = tv.etypes().any(|et| {
11120                        !tv.neighbors(et, Direction::Out, id).is_empty()
11121                            || !tv.neighbors(et, Direction::In, id).is_empty()
11122                    });
11123                    if has_edges {
11124                        return Err(GraphError::QueryError {
11125                            detail: format!(
11126                                "Cannot delete node `{key}` because it still has incident edges. \
11127                                 Use DETACH DELETE to remove the node and all its edges."
11128                            ),
11129                        });
11130                    }
11131                }
11132            }
11133        }
11134
11135        let mut nodes_deleted = 0i64;
11136        let mut edges_deleted = 0i64;
11137        for key in keys {
11138            match self.delete_node(&key) {
11139                Ok(report) => {
11140                    nodes_deleted += 1;
11141                    edges_deleted += (report.manual_edges + report.derived_edges) as i64;
11142                }
11143                Err(GraphError::KeyNotFound { .. }) => {
11144                    // Node may have been deleted by an earlier iteration (e.g., via
11145                    // multiple MATCH rows for the same node).  Safe to skip.
11146                }
11147                Err(e) => return Err(e),
11148            }
11149        }
11150
11151        let mut rs = write_result_set();
11152        rs.push_row(vec![
11153            Some(Value::Int(0)),
11154            Some(Value::Int(0)),
11155            Some(Value::Int(nodes_deleted + edges_deleted)),
11156        ]);
11157        Ok(rs)
11158    }
11159
11160    /// Props the MERGE create arm inserts: the identifying key, plus `ns` when
11161    /// the pattern named one, or the executing role's sole namespace when it
11162    /// did not. A role bound to two or more namespaces cannot choose, and is
11163    /// refused with [`MERGE_CREATE_NEEDS_ONE_NAMESPACE`]. The authorizer still
11164    /// refuses a named `ns` the role cannot write.
11165    fn merge_create_props(
11166        &self,
11167        key_field: &str,
11168        key_value: &Value,
11169        named_ns: Option<&Value>,
11170    ) -> Result<Vec<(String, Value)>> {
11171        let mut props = vec![(key_field.to_string(), key_value.clone())];
11172        if let Some(ns) = named_ns {
11173            props.push((NS_PROP.to_string(), ns.clone()));
11174            return Ok(props);
11175        }
11176        if let Some(ns) = self.merge_create_stamp_ns()? {
11177            props.push((NS_PROP.to_string(), Value::Str(ns)));
11178        }
11179        Ok(props)
11180    }
11181
11182    /// The namespace a role-scoped MERGE create stamps when the pattern does
11183    /// not name `ns`. `None` = unscoped / full authority, so the node lands in
11184    /// `default`.
11185    fn merge_create_stamp_ns(&self) -> Result<Option<String>> {
11186        let Some(authz) = self.pending_write_authz.as_ref() else {
11187            return Ok(None);
11188        };
11189        let Some(def) = self.role_def_for(&authz.role) else {
11190            return Ok(None);
11191        };
11192        match def.namespaces.as_deref() {
11193            Some([only]) => Ok(Some(only.clone())),
11194            Some(_) => Err(GraphError::RoleWriteDenied {
11195                reason: MERGE_CREATE_NEEDS_ONE_NAMESPACE.to_string(),
11196            }),
11197            None => Ok(None),
11198        }
11199    }
11200
11201    fn exec_merge(
11202        &mut self,
11203        stmt: core_query::cypher::MergeStmt,
11204        params: &BTreeMap<String, Value>,
11205    ) -> Result<ResultSet> {
11206        // MERGE: check if a node with the given key already exists.
11207        let key = match &stmt.key_value {
11208            Value::Str(s) => s.clone(),
11209            _ => {
11210                return Err(GraphError::QueryError {
11211                    detail: format!(
11212                        "MERGE key value must be a string (got {:?})",
11213                        stmt.key_value
11214                    ),
11215                })
11216            }
11217        };
11218
11219        if let Some(var) = stmt.var.as_deref() {
11220            for sc in stmt.on_create.iter().chain(&stmt.on_match) {
11221                if sc.var != var {
11222                    return Err(GraphError::QueryError {
11223                        detail: format!(
11224                            "SET variable '{}' does not match MERGE variable '{var}'",
11225                            sc.var
11226                        ),
11227                    });
11228                }
11229            }
11230        }
11231
11232        // ── MERGE authz pre-check (when role-scoped) ─────────────────────────
11233        //
11234        // MERGE scope precondition: check create OR update scope for the
11235        // declared label BEFORE calling `has_node` (timing-oracle closure,
11236        // spec §6.2 "MERGE visibility oracle" item: hidden ≡ absent for
11237        // unscoped roles — the scope denial fires without touching the key store).
11238        //
11239        // Clone to avoid holding a borrow on `self.pending_write_authz` while
11240        // also calling `self.ids.get(key)`.
11241        let merge_existed: bool = if let Some(authz) = self.pending_write_authz.clone() {
11242            let has_create = authz.scope.create_labels.contains(&stmt.label);
11243            let has_update = authz.scope.update_labels.contains(&stmt.label);
11244            if !has_create && !has_update {
11245                // Scope-before-lookup: 403 without has_node call (timing oracle
11246                // closure — see test_merge_unscoped_no_key_lookup).
11247                return Err(GraphError::RoleWriteDenied {
11248                    reason: format!(
11249                        "role-bound token: label '{}' not in write scope (create_labels)",
11250                        stmt.label
11251                    ),
11252                });
11253            }
11254            // Key lookup under mask.
11255            match self.ids.get(key.as_str()) {
11256                Some(id) if authz.mask.contains_id(id) => {
11257                    // Visible: must have update scope to proceed to match arm.
11258                    if !has_update {
11259                        return Err(GraphError::RoleWriteDenied {
11260                            reason: format!(
11261                                "role-bound token: label '{}' not in write scope (update_labels)",
11262                                stmt.label
11263                            ),
11264                        });
11265                    }
11266                    true // existed = true → match arm
11267                }
11268                Some(_) => {
11269                    // Hidden: same error as absent to the role (spec §3.1/§3.3).
11270                    return Err(GraphError::RoleWriteDenied {
11271                        reason: "role-bound token: target node not visible".into(),
11272                    });
11273                }
11274                None => {
11275                    // Absent: must have create scope to proceed to the create arm.
11276                    //
11277                    // Update-only roles (create_labels empty, update_labels set):
11278                    // return the SAME "not visible" error as the hidden-key branch
11279                    // so hidden ≡ absent — no distinguishing oracle (spec §6.1
11280                    // "confirm existence of hidden nodes: No").
11281                    //
11282                    // Create-scoped roles (has_create=true): absent → create arm
11283                    // as before.  The accepted structural key-existence disclosure
11284                    // (§THREAT-MODEL) applies only when the role holds create scope.
11285                    if !has_create {
11286                        return Err(GraphError::RoleWriteDenied {
11287                            reason: "role-bound token: target node not visible".into(),
11288                        });
11289                    }
11290                    false // existed = false → create arm
11291                }
11292            }
11293        } else {
11294            // Full authority: use the existing non-masked has_node check.
11295            self.has_node(&key)
11296        };
11297
11298        let existed = merge_existed;
11299        let create_props = if existed {
11300            None
11301        } else {
11302            Some(self.merge_create_props(&stmt.key_field, &stmt.key_value, stmt.ns.as_ref())?)
11303        };
11304        let mut created = 0i64;
11305        if create_props.is_some() || !stmt.on_match.is_empty() {
11306            let mut batch = self.batch();
11307            if let Some(props) = create_props {
11308                batch.insert_node(&stmt.label, &key, props);
11309                for sc in &stmt.on_create {
11310                    let value = resolve_merge_set_value(&sc.value, params)?;
11311                    batch.set_prop(&key, &sc.field, value);
11312                }
11313                created = 1;
11314            } else {
11315                for sc in &stmt.on_match {
11316                    let value = resolve_merge_set_value(&sc.value, params)?;
11317                    batch.set_prop(&key, &sc.field, value);
11318                }
11319            }
11320            batch.commit()?;
11321        }
11322
11323        // Refresh the role mask so the just-created node is visible to this
11324        // statement's RETURN (read-after-write). Safe: create_labels ⊆ read labels
11325        // (apply_schema subset rule), so the new node's label is already in the
11326        // role's read scope — this never widens beyond the role's declared labels.
11327        if !existed {
11328            if let Some(role) = self.pending_write_authz.as_ref().map(|a| a.role.clone()) {
11329                let new_mask = self.mask_for_role(&role)?;
11330                if let Some(a) = self.pending_write_authz.as_mut() {
11331                    a.mask = new_mask;
11332                }
11333            }
11334        }
11335
11336        // Optional RETURN clause: project the node (created or matched) as a read result.
11337        if let Some(returns) = stmt.returns {
11338            let var = stmt.var.as_deref().unwrap_or("_mn0");
11339            let q = Query {
11340                matches: vec![Pattern {
11341                    start: NodePat {
11342                        var: Some(var.to_string()),
11343                        label: Some(stmt.label.clone()),
11344                        props: vec![("id".to_string(), Operand::Lit(stmt.key_value.clone()))],
11345                    },
11346                    chain: vec![],
11347                    shortest: false,
11348                }],
11349                optional_clauses: vec![],
11350                where_expr: None,
11351                unwinds: vec![],
11352                post_unwind_where: None,
11353                stages: vec![],
11354                returns,
11355                distinct: false,
11356                order_by: vec![],
11357                skip: None,
11358                limit: None,
11359            };
11360            let ops = plan(&q).map_err(|e| GraphError::QueryError {
11361                detail: format!("plan: {e}"),
11362            })?;
11363            // Use view_masked when a role-scoped write is in flight so the
11364            // post-merge projection is consistent with the masked read phase.
11365            let mask_opt = self.pending_write_authz.as_ref().map(|a| a.mask.clone());
11366            return (if let Some(ref mask) = mask_opt {
11367                execute(&self.view_masked(mask), &ops, &Params(params))
11368            } else {
11369                execute(&self.view(), &ops, &Params(params))
11370            })
11371            .map_err(|e| GraphError::QueryError {
11372                detail: format!("execute: {e}"),
11373            });
11374        }
11375
11376        let mut rs = write_result_set();
11377        rs.push_row(vec![
11378            Some(Value::Int(created)),
11379            Some(Value::Int(0)),
11380            Some(Value::Int(0)),
11381        ]);
11382        Ok(rs)
11383    }
11384
11385    /// Return all rule-owned edges between `key_a` and `key_b` (either direction),
11386    /// annotated with rule name, edge type, direction, and weight.
11387    /// Results are sorted by (rule, edge_type).
11388    /// Returns `Err(KeyNotFound)` if either key is unknown.
11389    pub fn explain(&self, key_a: &str, key_b: &str) -> Result<Vec<Explanation>> {
11390        self.ensure_v8_base_sections_loaded();
11391        let id_a = self
11392            .ids
11393            .get(key_a)
11394            .ok_or_else(|| GraphError::KeyNotFound { key: key_a.into() })?;
11395        let id_b = self
11396            .ids
11397            .get(key_b)
11398            .ok_or_else(|| GraphError::KeyNotFound { key: key_b.into() })?;
11399
11400        let mut results = Vec::new();
11401
11402        // Walk the smaller incident set so explain is O(min(deg(a), deg(b)))
11403        // rather than O(total provenance).
11404        let scan = if self.engine.provenance_touching_len(id_a)
11405            <= self.engine.provenance_touching_len(id_b)
11406        {
11407            id_a
11408        } else {
11409            id_b
11410        };
11411        for (rule_name, etype, src, dst) in self.engine.provenance_touching(scan) {
11412            if !((src == id_a && dst == id_b) || (src == id_b && dst == id_a)) {
11413                continue;
11414            }
11415            let Some(rule_def) = self.engine.rules().find(|r| r.name == rule_name) else {
11416                continue;
11417            };
11418            let edge_type = match self.syms.resolve(etype) {
11419                Some(s) => s.to_string(),
11420                None => continue,
11421            };
11422            // Provenance (src, dst) ids come from the archived PROVENANCE section
11423            // (large, no eager CRC).  A corrupt section can produce ids that are
11424            // out of range; return Corrupt rather than panic.
11425            let src_key = self
11426                .ids
11427                .key_of(src)
11428                .ok_or_else(|| GraphError::Corrupt {
11429                    detail: format!("v8: provenance src id {src} not in id table"),
11430                })?
11431                .to_string();
11432            let dst_key = self
11433                .ids
11434                .key_of(dst)
11435                .ok_or_else(|| GraphError::Corrupt {
11436                    detail: format!("v8: provenance dst id {dst} not in id table"),
11437                })?
11438                .to_string();
11439            let stored = rule_def.weight_prop.as_deref().and_then(|prop| {
11440                self.edge_props_view()
11441                    .get(etype, src, dst, prop)
11442                    .and_then(|v| {
11443                        if let Value::Float(f) = v {
11444                            Some(f)
11445                        } else {
11446                            None
11447                        }
11448                    })
11449            });
11450            // Rules that store no weight (KeyMatch/FieldEqual defaults, auto-FK)
11451            // still have a score: recompute it from the predicate so explain
11452            // never reports "no score" for an edge the engine scored.  Via-hop
11453            // rules score over their via set, not over (src, dst), so leave
11454            // those None rather than report a number the rule did not produce.
11455            let weight = stored.or_else(|| {
11456                if rule_def.via_edge.is_some() {
11457                    return None;
11458                }
11459                let props_view = build_props_view(&self.props, &self.base);
11460                let src_get = |field: &str| props_view.get(src, field).map(|vr| vr.into_value());
11461                let dst_get = |field: &str| props_view.get(dst, field).map(|vr| vr.into_value());
11462                let src_view = NodeView {
11463                    key: &src_key,
11464                    props: &src_get,
11465                };
11466                let dst_view = NodeView {
11467                    key: &dst_key,
11468                    props: &dst_get,
11469                };
11470                evaluate(&rule_def.predicate, &src_view, &dst_view)
11471            });
11472            results.push(Explanation {
11473                rule: rule_name.to_string(),
11474                edge_type,
11475                src_key,
11476                dst_key,
11477                weight,
11478                predicate: PredicateSummary {
11479                    approximate: rule_def.approximate,
11480                    ..PredicateSummary::from(&rule_def.predicate)
11481                },
11482                via_edge: rule_def.via_edge.clone(),
11483            });
11484        }
11485
11486        results.sort_by(|a, b| a.rule.cmp(&b.rule).then(a.edge_type.cmp(&b.edge_type)));
11487        Ok(results)
11488    }
11489
11490    /// [`explain`](Self::explain) with the scoped read contract (§5.3): both
11491    /// endpoints are subject-checked, and any explanation whose evidence runs
11492    /// through a hidden node is **dropped entirely, not redacted**.
11493    ///
11494    /// A plain two-node rule's evidence is the pair itself, so once both
11495    /// subjects are visible there is nothing left to hide. A **via-hop** rule is
11496    /// different: it fires `src → dst` because some node carrying `via_label`
11497    /// sits between them, and [`Explanation`] carries the hop's edge *type*
11498    /// (`via_edge`) and never the hop's key. There is no field to blank, so a
11499    /// redacted explanation would still say "these two are linked through
11500    /// something you cannot see" — which discloses that the something exists.
11501    /// The explanation is therefore kept only when at least one **visible** via
11502    /// node satisfies the rule on its own.
11503    ///
11504    /// The weight is the **visible corpus's** number, not the store's: a via-hop
11505    /// rule stores the max over every via it hopped through, so the stored value
11506    /// can be a score only a hidden via produced. It is recomputed over the
11507    /// visible vias alone.
11508    ///
11509    /// Hidden or unknown `key_a` or `key_b` → [`GraphError::KeyNotFound`].
11510    pub fn explain_scoped(
11511        &self,
11512        key_a: &str,
11513        key_b: &str,
11514        mask: &crate::mask::NodeMask,
11515    ) -> Result<Vec<Explanation>> {
11516        for key in [key_a, key_b] {
11517            if !mask.contains_node(self, key) {
11518                return Err(GraphError::KeyNotFound { key: key.into() });
11519            }
11520        }
11521        Ok(self
11522            .explain(key_a, key_b)?
11523            .into_iter()
11524            .filter_map(|e| self.scoped_explanation(e, mask))
11525            .collect())
11526    }
11527
11528    /// `e` as a caller limited to `mask` may have it, or `None` when it must be
11529    /// dropped entirely.
11530    ///
11531    /// Every non-via-hop explanation passes through untouched: its only nodes
11532    /// are the two subjects, which [`explain_scoped`](Self::explain_scoped) has
11533    /// already checked, and its weight is scored over that pair alone.
11534    ///
11535    /// A via-hop explanation is kept only when some via node the caller may see
11536    /// satisfies the rule on its own — and then its weight is recomputed as the
11537    /// max over exactly those vias. The engine writes the max over **all** of
11538    /// them (`core-rules::engine`, `best = prev.max(score)`), so passing the
11539    /// stored number through would let a hidden node set a figure the caller
11540    /// reads: the same disclosure dropping the explanation exists to prevent.
11541    ///
11542    /// A rule that stores no weight still reports none. The recomputed score is
11543    /// a sanitised version of a number `explain` already returned, never a new
11544    /// one — a scoped read must not say more than the unscoped read it narrows.
11545    fn scoped_explanation(
11546        &self,
11547        e: Explanation,
11548        mask: &crate::mask::NodeMask,
11549    ) -> Option<Explanation> {
11550        let Some(via_edge) = e.via_edge.clone() else {
11551            return Some(e);
11552        };
11553        let Some(rule_def) = self.engine.rules().find(|r| r.name == e.rule) else {
11554            // The rule is gone but its provenance is not; nothing can vouch for
11555            // the hop, so nothing is shown.
11556            return None;
11557        };
11558        let Some(via_label) = rule_def.via_label.as_deref() else {
11559            return Some(e);
11560        };
11561        let (Some(src), Some(dst)) = (self.ids.get(&e.src_key), self.ids.get(&e.dst_key)) else {
11562            return None;
11563        };
11564        let (Some(via_etype), Some(via_sym)) = (self.syms.get(&via_edge), self.syms.get(via_label))
11565        else {
11566            return None;
11567        };
11568        let via_dir = rule_def.via_dir.unwrap_or(Direction::Out);
11569        let props_view = build_props_view(&self.props, &self.base);
11570        let dst_get = |field: &str| props_view.get(dst, field).map(|vr| vr.into_value());
11571        let dst_view = NodeView {
11572            key: &e.dst_key,
11573            props: &dst_get,
11574        };
11575        // The rule's own namespace test, the one the engine applies to each via
11576        // candidate (`core-rules::engine::rule_sees_node`). Without it a
11577        // visible, out-of-namespace via — right label, satisfying predicate —
11578        // vouches for a hop the engine never made, and an explanation whose real
11579        // evidence is a hidden in-namespace node is kept.
11580        let rule_sees = |id: u32| match rule_def.namespace.as_deref() {
11581            None => true,
11582            Some(ns) => {
11583                let value = props_view.get(id, NS_PROP).map(|vr| vr.into_value());
11584                namespace_of_value(value.as_ref()) == ns
11585            }
11586        };
11587        let best = self
11588            .topo_view()
11589            .neighbors(via_etype, via_dir, src)
11590            .iter()
11591            .copied()
11592            .filter_map(|via| {
11593                if !mask.contains_id(via) {
11594                    return None;
11595                }
11596                if self.labels.get(via as usize).copied() != Some(via_sym) {
11597                    return None;
11598                }
11599                if !rule_sees(via) {
11600                    return None;
11601                }
11602                let via_key = self.ids.key_of(via)?;
11603                let via_get = |field: &str| props_view.get(via, field).map(|vr| vr.into_value());
11604                let via_view = NodeView {
11605                    key: via_key,
11606                    props: &via_get,
11607                };
11608                evaluate(&rule_def.predicate, &via_view, &dst_view)
11609            })
11610            .fold(None::<f64>, |best, score| {
11611                Some(match best {
11612                    None => score,
11613                    Some(prev) => prev.max(score),
11614                })
11615            })?;
11616        let weight = e.weight.map(|_| best);
11617        Some(Explanation { weight, ..e })
11618    }
11619
11620    pub fn neighbors(&self, key: &str, edge_type: &str, dir: Direction) -> Result<Vec<String>> {
11621        let id = self
11622            .ids
11623            .get(key)
11624            .ok_or_else(|| GraphError::KeyNotFound { key: key.into() })?;
11625        let Some(sym) = self.syms.get(edge_type) else {
11626            return Ok(Vec::new());
11627        };
11628        self.topo_view()
11629            .neighbors(sym, dir, id)
11630            .iter()
11631            .map(|&n| {
11632                self.ids
11633                    .key_of(n)
11634                    .map(|k| k.to_string())
11635                    .ok_or_else(|| GraphError::Corrupt {
11636                        detail: format!("topology id {n} has no key"),
11637                    })
11638            })
11639            .collect::<Result<Vec<_>>>()
11640    }
11641
11642    /// Unique directed degree of `key`. Unknown key → [`GraphError::KeyNotFound`].
11643    /// Unknown `edge_type` → 0. [`crate::algo::AlgoDir::Both`] is out + in (sum).
11644    pub fn degree(
11645        &self,
11646        key: &str,
11647        edge_type: Option<&str>,
11648        direction: crate::algo::AlgoDir,
11649    ) -> Result<u64> {
11650        let id = self
11651            .ids
11652            .get(key)
11653            .ok_or_else(|| GraphError::KeyNotFound { key: key.into() })?;
11654        let topo = self.topo_view();
11655        Ok(Self::unique_directed_degree(
11656            &topo, &self.syms, id, edge_type, direction,
11657        ))
11658    }
11659
11660    /// [`degree`](Self::degree) summing each pair's **insert count** instead of
11661    /// counting each pair once (§5.13).
11662    ///
11663    /// The unique degree asks how many neighbours there are; this asks how many
11664    /// times they were inserted. A pair with no recorded count contributes 1,
11665    /// so on a store that never called
11666    /// [`enable_multiplicity`](Self::enable_multiplicity) this returns exactly
11667    /// what [`degree`](Self::degree) returns rather than erroring — the
11668    /// distinction is a readout preference, not a demand the store cannot meet.
11669    ///
11670    /// `AlgoDir::Both` still sums out + in, so a pair visible on both sides
11671    /// still contributes twice: multiplicity changes what a pair is worth, never
11672    /// how a direction is counted.
11673    pub fn degree_multiplicity(
11674        &self,
11675        key: &str,
11676        edge_type: Option<&str>,
11677        direction: crate::algo::AlgoDir,
11678    ) -> Result<u64> {
11679        let id = self
11680            .ids
11681            .get(key)
11682            .ok_or_else(|| GraphError::KeyNotFound { key: key.into() })?;
11683        Ok(Self::multiplicity_directed_degree(
11684            &self.topo_view(),
11685            &self.edge_props_view(),
11686            &self.syms,
11687            id,
11688            edge_type,
11689            direction,
11690            None,
11691        ))
11692    }
11693
11694    /// [`degree_multiplicity`](Self::degree_multiplicity) under a scope.
11695    ///
11696    /// The sum covers **visible pairs only**. A hidden neighbour's inserts stay
11697    /// out of it for the reason
11698    /// [`degree_scoped`](Self::degree_scoped) documents, and more sharply: an
11699    /// unscoped multiplicity count discloses not only that a hidden neighbour
11700    /// exists but how often it was written.
11701    ///
11702    /// Hidden or unknown `key` → [`GraphError::KeyNotFound`].
11703    pub fn degree_scoped_multiplicity(
11704        &self,
11705        key: &str,
11706        edge_type: Option<&str>,
11707        direction: crate::algo::AlgoDir,
11708        mask: &crate::mask::NodeMask,
11709    ) -> Result<u64> {
11710        let id = self
11711            .ids
11712            .get(key)
11713            .filter(|&id| mask.contains_id(id))
11714            .ok_or_else(|| GraphError::KeyNotFound { key: key.into() })?;
11715        Ok(Self::multiplicity_directed_degree(
11716            &self.topo_view(),
11717            &self.edge_props_view(),
11718            &self.syms,
11719            id,
11720            edge_type,
11721            direction,
11722            Some(mask),
11723        ))
11724    }
11725
11726    /// [`degree`](Self::degree) counting **only neighbours the mask admits**.
11727    ///
11728    /// The filter is a correctness requirement, not an optimisation: an
11729    /// unfiltered count discloses the existence of a hidden neighbour to a
11730    /// caller who cannot see it, which is the same leak
11731    /// [`node_edges_scoped`](Self::node_edges_scoped) exists to prevent —
11732    /// reached by arithmetic instead of by name.
11733    ///
11734    /// Hidden or unknown `key` → [`GraphError::KeyNotFound`]. Unknown
11735    /// `edge_type` is still 0, as it is unscoped.
11736    pub fn degree_scoped(
11737        &self,
11738        key: &str,
11739        edge_type: Option<&str>,
11740        direction: crate::algo::AlgoDir,
11741        mask: &crate::mask::NodeMask,
11742    ) -> Result<u64> {
11743        let id = self
11744            .ids
11745            .get(key)
11746            .filter(|&id| mask.contains_id(id))
11747            .ok_or_else(|| GraphError::KeyNotFound { key: key.into() })?;
11748        let topo = self.topo_view();
11749        Ok(Self::visible_directed_degree(
11750            &topo, &self.syms, id, edge_type, direction, mask,
11751        ))
11752    }
11753
11754    /// Unique directed degree for a subset or a label scan.
11755    ///
11756    /// Unknown keys in `keys` are omitted (mask-like). `keys = Some(&[])` →
11757    /// empty `Ok(vec![])`. `limit` is applied after sorting degree desc, key
11758    /// asc, and only when `Some`. Invalid `where_` → `QueryError`.
11759    #[allow(clippy::too_many_arguments)]
11760    pub fn degrees(
11761        &self,
11762        keys: Option<&[String]>,
11763        label: Option<&str>,
11764        where_: Option<&PropPredicate>,
11765        edge_type: Option<&str>,
11766        direction: crate::algo::AlgoDir,
11767        limit: Option<usize>,
11768    ) -> Result<Vec<(String, u64)>> {
11769        self.degrees_inner(
11770            keys, label, where_, edge_type, direction, limit, None, false,
11771        )
11772    }
11773
11774    /// [`degrees`](Self::degrees) reporting each row's **insert-count** sum
11775    /// instead of its unique neighbour count, as
11776    /// [`degree_multiplicity`](Self::degree_multiplicity) does for one key.
11777    ///
11778    /// The sort is still degree descending, key ascending — over the counts this
11779    /// reading produces — and `limit` still applies after it.
11780    #[allow(clippy::too_many_arguments)]
11781    pub fn degrees_multiplicity(
11782        &self,
11783        keys: Option<&[String]>,
11784        label: Option<&str>,
11785        where_: Option<&PropPredicate>,
11786        edge_type: Option<&str>,
11787        direction: crate::algo::AlgoDir,
11788        limit: Option<usize>,
11789    ) -> Result<Vec<(String, u64)>> {
11790        self.degrees_inner(keys, label, where_, edge_type, direction, limit, None, true)
11791    }
11792
11793    /// [`degrees_scoped`](Self::degrees_scoped) reporting insert counts.
11794    ///
11795    /// Both filters apply: a hidden key stays out of the result, and every
11796    /// row's sum covers its **visible** pairs only.
11797    #[allow(clippy::too_many_arguments)]
11798    pub fn degrees_scoped_multiplicity(
11799        &self,
11800        keys: Option<&[String]>,
11801        label: Option<&str>,
11802        where_: Option<&PropPredicate>,
11803        edge_type: Option<&str>,
11804        direction: crate::algo::AlgoDir,
11805        limit: Option<usize>,
11806        mask: &crate::mask::NodeMask,
11807    ) -> Result<Vec<(String, u64)>> {
11808        self.degrees_inner(
11809            keys,
11810            label,
11811            where_,
11812            edge_type,
11813            direction,
11814            limit,
11815            Some(mask),
11816            true,
11817        )
11818    }
11819
11820    /// [`degrees`](Self::degrees) with the scope applied on both sides: a hidden
11821    /// key is omitted from the input — whether it arrived in `keys` or came out
11822    /// of the `label`/`where_` scan — and every row's count is the count of its
11823    /// **visible** neighbours, for the reason
11824    /// [`degree_scoped`](Self::degree_scoped) documents.
11825    ///
11826    /// Unlike `degree_scoped`, a hidden key here is not
11827    /// [`GraphError::KeyNotFound`]: `degrees` already drops unknown keys
11828    /// silently, so hidden and absent stay one answer by staying out of the
11829    /// result. `limit` still applies after the sort, and so counts visible rows.
11830    #[allow(clippy::too_many_arguments)]
11831    pub fn degrees_scoped(
11832        &self,
11833        keys: Option<&[String]>,
11834        label: Option<&str>,
11835        where_: Option<&PropPredicate>,
11836        edge_type: Option<&str>,
11837        direction: crate::algo::AlgoDir,
11838        limit: Option<usize>,
11839        mask: &crate::mask::NodeMask,
11840    ) -> Result<Vec<(String, u64)>> {
11841        self.degrees_inner(
11842            keys,
11843            label,
11844            where_,
11845            edge_type,
11846            direction,
11847            limit,
11848            Some(mask),
11849            false,
11850        )
11851    }
11852
11853    /// The body shared by [`degrees`](Self::degrees) and
11854    /// [`degrees_scoped`](Self::degrees_scoped). `mask = None` is the unscoped
11855    /// contract unchanged.
11856    #[allow(clippy::too_many_arguments)]
11857    fn degrees_inner(
11858        &self,
11859        keys: Option<&[String]>,
11860        label: Option<&str>,
11861        where_: Option<&PropPredicate>,
11862        edge_type: Option<&str>,
11863        direction: crate::algo::AlgoDir,
11864        limit: Option<usize>,
11865        mask: Option<&crate::mask::NodeMask>,
11866        multiplicity: bool,
11867    ) -> Result<Vec<(String, u64)>> {
11868        if let Some(pred) = where_ {
11869            pred.validate_named("where")
11870                .map_err(|detail| GraphError::QueryError { detail })?;
11871        }
11872        if matches!(keys, Some(ks) if ks.is_empty()) {
11873            return Ok(Vec::new());
11874        }
11875        let view = self.view();
11876        let ids: Vec<u32> = match keys {
11877            Some(ks) => {
11878                let mut seen = HashSet::new();
11879                let mut out = Vec::new();
11880                for k in ks {
11881                    let Some(id) = view.ids.get(k) else {
11882                        continue;
11883                    };
11884                    if !seen.insert(id) {
11885                        continue;
11886                    }
11887                    if let Some(pred) = where_ {
11888                        let holds = match view.prop(id, &pred.field) {
11889                            None => pred.holds(None),
11890                            Some(vr) => pred.holds(Some(vr.as_value())),
11891                        };
11892                        if !holds {
11893                            continue;
11894                        }
11895                    }
11896                    out.push(id);
11897                }
11898                out
11899            }
11900            None => Self::vector_candidates(&view, label, where_),
11901        };
11902        let mut out: Vec<(String, u64)> = ids
11903            .into_iter()
11904            // A hidden candidate leaves as quietly as an unknown key does.
11905            .filter(|&id| mask.is_none_or(|m| m.contains_id(id)))
11906            .filter_map(|id| {
11907                let key = self.ids.key_of(id)?.to_string();
11908                let deg = match (multiplicity, mask) {
11909                    // The same `view` the unique arms read, so the per-row
11910                    // rebuild F9 measured is gone and all three arms agree on
11911                    // the state they are reading.
11912                    (true, m) => Self::multiplicity_directed_degree(
11913                        &view.topo,
11914                        &view.edge_props,
11915                        view.syms,
11916                        id,
11917                        edge_type,
11918                        direction,
11919                        m,
11920                    ),
11921                    (false, Some(m)) => Self::visible_directed_degree(
11922                        &view.topo, view.syms, id, edge_type, direction, m,
11923                    ),
11924                    (false, None) => Self::unique_directed_degree(
11925                        &view.topo, view.syms, id, edge_type, direction,
11926                    ),
11927                };
11928                Some((key, deg))
11929            })
11930            .collect();
11931        out.sort_by(|a, b| b.1.cmp(&a.1).then_with(|| a.0.cmp(&b.0)));
11932        if let Some(lim) = limit {
11933            out.truncate(lim);
11934        }
11935        Ok(out)
11936    }
11937
11938    /// Unique neighbour count for `id` across `edge_type` (or all types) and
11939    /// `direction`. Unknown `edge_type` → 0. `Both` sums out + in.
11940    fn unique_directed_degree(
11941        topo: &TopologyView<'_>,
11942        syms: &Interner,
11943        id: u32,
11944        edge_type: Option<&str>,
11945        direction: crate::algo::AlgoDir,
11946    ) -> u64 {
11947        let dirs: &[Direction] = match direction {
11948            crate::algo::AlgoDir::Out => &[Direction::Out],
11949            crate::algo::AlgoDir::In => &[Direction::In],
11950            crate::algo::AlgoDir::Both => &[Direction::Out, Direction::In],
11951        };
11952        match edge_type {
11953            Some(name) => {
11954                let Some(et) = syms.get(name) else {
11955                    return 0;
11956                };
11957                dirs.iter().map(|&d| topo.degree(et, d, id) as u64).sum()
11958            }
11959            None => topo
11960                .etypes()
11961                .map(|et| {
11962                    dirs.iter()
11963                        .map(|&d| topo.degree(et, d, id) as u64)
11964                        .sum::<u64>()
11965                })
11966                .sum(),
11967        }
11968    }
11969
11970    /// [`unique_directed_degree`](Self::unique_directed_degree) counting only
11971    /// neighbours `mask` admits.
11972    ///
11973    /// Same shape, one substitution: `topo.degree` is a length, so it cannot be
11974    /// filtered; the neighbour list it measures can. `Both` still sums out + in,
11975    /// so a node visible on both sides still counts twice — the filter changes
11976    /// which neighbours are counted, never how a degree is defined.
11977    fn visible_directed_degree(
11978        topo: &TopologyView<'_>,
11979        syms: &Interner,
11980        id: u32,
11981        edge_type: Option<&str>,
11982        direction: crate::algo::AlgoDir,
11983        mask: &crate::mask::NodeMask,
11984    ) -> u64 {
11985        let dirs: &[Direction] = match direction {
11986            crate::algo::AlgoDir::Out => &[Direction::Out],
11987            crate::algo::AlgoDir::In => &[Direction::In],
11988            crate::algo::AlgoDir::Both => &[Direction::Out, Direction::In],
11989        };
11990        let visible = |et: u32| -> u64 {
11991            dirs.iter()
11992                .map(|&d| {
11993                    topo.neighbors(et, d, id)
11994                        .iter()
11995                        .filter(|&&n| mask.contains_id(n))
11996                        .count() as u64
11997                })
11998                .sum()
11999        };
12000        match edge_type {
12001            Some(name) => syms.get(name).map_or(0, visible),
12002            None => topo.etypes().map(visible).sum(),
12003        }
12004    }
12005
12006    /// Sum of the insert counts of `id`'s pairs (§5.13), over `edge_type` (or
12007    /// all types) and `direction`, restricted to what `mask` admits when one is
12008    /// given.
12009    ///
12010    /// The same neighbour lists the unique reading measures, with each entry
12011    /// worth its pair's count rather than worth 1 — so the filter decides which
12012    /// pairs are in the sum and the count decides what each contributes. A
12013    /// direction decides which way round the pair is addressed: an `In`
12014    /// neighbour `n` of `id` is the pair `(et, n, id)`.
12015    ///
12016    /// Takes its views as parameters, exactly as the unique helpers do, because
12017    /// it is called once per row from a label scan. `edge_props_view()` reaches
12018    /// into the mmap'd base's rkyv section on every call, so building the two
12019    /// views inside made an N-row `degrees(multiplicity=True)` do N section
12020    /// accesses where the unique reading does one: worth 2.57 ms of 16.68 ms
12021    /// over 20 000 rows, about 0.13 us per row (defect #29,
12022    /// `tests/f9_bench.rs`). Most of that call's cost is the per-neighbour
12023    /// count lookup and is inherent, so this is a hoist, not a rescue.
12024    ///
12025    /// The views are exactly `self.view()`'s own `topo` and `edge_props`, so a
12026    /// caller that already has a view passes its halves and reads the same
12027    /// state it reads everything else from.
12028    fn multiplicity_directed_degree(
12029        topo: &TopologyView<'_>,
12030        edge_props: &EdgePropsView<'_>,
12031        syms: &Interner,
12032        id: u32,
12033        edge_type: Option<&str>,
12034        direction: crate::algo::AlgoDir,
12035        mask: Option<&crate::mask::NodeMask>,
12036    ) -> u64 {
12037        let dirs: &[Direction] = match direction {
12038            crate::algo::AlgoDir::Out => &[Direction::Out],
12039            crate::algo::AlgoDir::In => &[Direction::In],
12040            crate::algo::AlgoDir::Both => &[Direction::Out, Direction::In],
12041        };
12042        let count_of = |et: u32, src: u32, dst: u32| -> u64 {
12043            match edge_props.get(et, src, dst, EDGE_COUNT_PROP) {
12044                Some(Value::Int(n)) if n > 0 => n as u64,
12045                _ => 1,
12046            }
12047        };
12048        let per_etype = |et: u32| -> u64 {
12049            dirs.iter()
12050                .map(|&d| {
12051                    topo.neighbors(et, d, id)
12052                        .iter()
12053                        .filter(|&&n| mask.is_none_or(|m| m.contains_id(n)))
12054                        .map(|&n| match d {
12055                            Direction::Out => count_of(et, id, n),
12056                            Direction::In => count_of(et, n, id),
12057                        })
12058                        .sum::<u64>()
12059                })
12060                .sum()
12061        };
12062        match edge_type {
12063            Some(name) => syms.get(name).map_or(0, per_etype),
12064            None => topo.etypes().map(per_etype).sum(),
12065        }
12066    }
12067
12068    /// Return the last-change commit sequence for `key`, or `None` if the node
12069    /// does not exist or has never been mutated since the last V5-V7 snapshot
12070    /// (horizon-bounded for legacy stores).
12071    ///
12072    /// The returned sequence is a monotonically increasing counter that starts
12073    /// at 1 for the first commit after `open` and increments with every
12074    /// successful write.  WAL replay at open also assigns sequences (1..N for N
12075    /// replayed frames), so sequences are consistent across snapshot+WAL cycles.
12076    ///
12077    /// For V5-V7 stores opened without a V8 snapshot, nodes that were present
12078    /// in the snapshot but not touched by any WAL frame will return `None`
12079    /// (horizon-bounded: CAS against such nodes is only safe after the first
12080    /// V8 snapshot or after the node is next mutated).
12081    pub fn last_changed(&self, key: &str) -> Option<u64> {
12082        let id = self.ids.get(key)?;
12083        self.last_change.get(&id).copied()
12084    }
12085
12086    /// Which loaded store this handle is.
12087    ///
12088    /// Paired with [`commit_seq`](GraphDb::commit_seq) it identifies a graph
12089    /// state outright, which `commit_seq` alone does not: two stores of the same
12090    /// age share a sequence, and a reload can return to one. Memos of dense
12091    /// node ids that the handle does not own are stamped with both; see
12092    /// [`StoreStamp`](crate::mask::StoreStamp).
12093    pub(crate) fn store_id(&self) -> crate::mask::StoreId {
12094        self.store_id
12095    }
12096
12097    /// The current commit sequence (number of successful commits since open,
12098    /// including WAL replay frames).  Useful for recording a baseline before
12099    /// a read-modify-write cycle.
12100    pub fn commit_seq(&self) -> u64 {
12101        self.commit_seq
12102    }
12103
12104    /// Check that all `preconds` are satisfied against the current db state.
12105    /// Returns `Err(GraphError::CasConflict)` on the first failing precondition.
12106    pub(crate) fn check_preconditions(&self, preconds: &[Precondition]) -> Result<()> {
12107        for precond in preconds {
12108            match precond {
12109                Precondition::NodeUnchangedSince { key, expected } => {
12110                    // Missing entry means the node predates the WAL window or
12111                    // does not exist; treat as 0 (before any commit).
12112                    let actual = self.last_changed(key).unwrap_or_default();
12113                    if actual != *expected {
12114                        return Err(GraphError::CasConflict {
12115                            key: key.clone(),
12116                            expected: *expected,
12117                            actual,
12118                        });
12119                    }
12120                }
12121                Precondition::NodeAbsent { key } => {
12122                    // Node must not exist (not live).
12123                    if self.ids.get(key).is_some() {
12124                        let actual = self.last_changed(key).unwrap_or(0);
12125                        return Err(GraphError::CasConflict {
12126                            key: key.clone(),
12127                            expected: u64::MAX,
12128                            actual,
12129                        });
12130                    }
12131                }
12132            }
12133        }
12134        Ok(())
12135    }
12136
12137    /// Apply a batch of mutations with compare-and-set preconditions.
12138    ///
12139    /// All preconditions are checked atomically before any operation is applied.
12140    /// If any precondition fails, the entire batch is rejected with
12141    /// [`GraphError::CasConflict`] and no WAL frame is written.
12142    ///
12143    /// # Returns
12144    /// `(nodes_inserted, edges_inserted)` on success, same as [`write_batch`].
12145    ///
12146    /// # Errors
12147    /// - [`GraphError::CasConflict`] if any precondition is not satisfied.
12148    /// - Any error that [`write_batch`] would return for the ops themselves.
12149    pub fn write_batch_cas(
12150        &mut self,
12151        preconds: Vec<Precondition>,
12152        ops: Vec<BatchOp>,
12153    ) -> Result<(usize, usize)> {
12154        self.check_preconditions(&preconds)?;
12155        self.commit_logged_batch(ops, None, None).map(inserted_pair)
12156    }
12157
12158    /// Update the per-node last-change map for a WAL record at commit `seq`.
12159    ///
12160    /// Called after a successful apply to record which nodes were touched.
12161    /// For replay, called with the WAL-frame's replayed seq.
12162    ///
12163    /// Touch definition (see [`Precondition`] doc):
12164    /// - InsertNode / InsertNodeId / SetProp / SetPropId / RemoveProp → the node.
12165    /// - InsertEdge / InsertEdgeId / DeleteEdge → both src and dst.
12166    /// - DeleteNode → node tombstoned; last_changed() returns None so no update needed.
12167    /// - DerivedEdge markers, Intern, rule/view records → no-ops.
12168    /// - Batch → recurse into inner records.
12169    fn update_last_change_from_rec(&mut self, rec: &WalRecord, seq: u64) {
12170        match rec {
12171            WalRecord::InsertNode { key, .. }
12172            | WalRecord::SetProp { key, .. }
12173            | WalRecord::RemoveProp { key, .. } => {
12174                if let Some(id) = self.ids.get(key) {
12175                    self.last_change.insert(id, seq);
12176                }
12177            }
12178            WalRecord::InsertNodeId { key, .. } => {
12179                if let Some(id) = self.ids.get(key) {
12180                    self.last_change.insert(id, seq);
12181                }
12182            }
12183            WalRecord::SetPropId { id, .. } => {
12184                self.last_change.insert(*id, seq);
12185            }
12186            WalRecord::InsertEdge {
12187                src_key, dst_key, ..
12188            }
12189            | WalRecord::DeleteEdge {
12190                src_key, dst_key, ..
12191            } => {
12192                if let Some(src_id) = self.ids.get(src_key) {
12193                    self.last_change.insert(src_id, seq);
12194                }
12195                if let Some(dst_id) = self.ids.get(dst_key) {
12196                    self.last_change.insert(dst_id, seq);
12197                }
12198            }
12199            WalRecord::InsertEdgeId { src, dst, .. } => {
12200                self.last_change.insert(*src, seq);
12201                self.last_change.insert(*dst, seq);
12202            }
12203            // A count record touches the pair, so it touches both endpoints —
12204            // the same reading `InsertEdgeId` gets, because a duplicate insert
12205            // that raises the count *is* a mutation of that pair. The opt-in
12206            // declaration touches nothing.
12207            WalRecord::SetEdgeCount { src, dst, .. } if !rec.is_multiplicity_decl() => {
12208                self.last_change.insert(*src, seq);
12209                self.last_change.insert(*dst, seq);
12210            }
12211            WalRecord::SetEdgeCount { .. } => {}
12212            // DeleteNode: node is tombstoned; last_changed(key) returns None for
12213            // deleted keys (ids.get() returns None post-tombstone), so no update needed.
12214            // History markers: state no-ops; the underlying mutation already
12215            // touched the relevant nodes' last_change entries.
12216            WalRecord::DeleteNode { .. }
12217            | WalRecord::DerivedEdgeAdded { .. }
12218            | WalRecord::DerivedEdgeRetracted { .. }
12219            | WalRecord::Intern { .. }
12220            | WalRecord::CreateRule { .. }
12221            | WalRecord::DeleteRule { .. }
12222            | WalRecord::RebuildRule { .. }
12223            | WalRecord::CreateView { .. }
12224            | WalRecord::DeleteView { .. }
12225            | WalRecord::EnableFulltext { .. }
12226            | WalRecord::DisableFulltext { .. }
12227            | WalRecord::EnableIndex { .. }
12228            | WalRecord::DisableIndex { .. } => {}
12229            // RenameNode: node id is stable; update last_change via the new key.
12230            // Called after apply(), so ids already reflects new_key.
12231            WalRecord::RenameNode { new_key, .. } => {
12232                if let Some(id) = self.ids.get(new_key) {
12233                    self.last_change.insert(id, seq);
12234                }
12235            }
12236            WalRecord::Batch(inner) => {
12237                for inner_rec in inner {
12238                    self.update_last_change_from_rec(inner_rec, seq);
12239                }
12240            }
12241        }
12242    }
12243
12244    pub fn node_count(&self) -> usize {
12245        self.ids.len()
12246    }
12247
12248    /// Configure archive retention: keep the `N` newest WAL archives at each
12249    /// [`snapshot_with`] call when `archive_wal: true`.
12250    ///
12251    /// `Some(N)` where N > 0 → prune oldest archives keeping the newest N.
12252    /// `Some(0)` or `None` → unlimited (no pruning).
12253    ///
12254    /// Pruning only ever happens inside [`snapshot_with`]; this method only
12255    /// stores the policy.  Archives below the retention limit are deleted
12256    /// oldest-first.  The horizon floor is updated so that
12257    /// [`was_linked`] / history APIs return `CommitOutOfRange` for commits
12258    /// in pruned archives rather than silently returning wrong data.
12259    pub fn set_wal_archive_retention(&mut self, keep: Option<u32>) {
12260        self.wal_archive_retention = keep;
12261    }
12262
12263    /// Delete any WAL archives that are fully below the current horizon floor.
12264    ///
12265    /// Orphaned archives arise when the floor is written first during retention
12266    /// pruning and then a crash interrupts the archive-delete sequence.  The
12267    /// opening cleanup ensures no subsequent read path sees stale data.
12268    ///
12269    /// Under the monotonic naming scheme, the archive name N equals the
12270    /// cumulative end-frame index of the archive in global commit space (i.e.
12271    /// the archive covers global frames `[prev_n, N)`).  An archive is
12272    /// fully orphaned when `N <= wal_horizon_floor`: all of its frames fall
12273    /// below the floor and have already been counted in it.
12274    fn cleanup_orphaned_archives(&mut self) -> Result<()> {
12275        if self.wal_horizon_floor == 0 {
12276            // Floor at 0 means no pruning has ever occurred; nothing to clean.
12277            return Ok(());
12278        }
12279        let archive_ns = self.fs.list_archives()?;
12280        for n in archive_ns {
12281            if n <= self.wal_horizon_floor {
12282                // Archive N ends at global frame N; all its frames are below
12283                // the floor (floor already accounts for them) → orphaned.
12284                self.fs.delete_archive(n).map_err(GraphError::Io)?;
12285            } else {
12286                // Archives are sorted ascending; first one above floor stops scan.
12287                break;
12288            }
12289        }
12290        Ok(())
12291    }
12292
12293    /// Collect all WAL frames from surviving archives (oldest-first) then the
12294    /// live WAL into one flat list, and return the total along with the number
12295    /// of archive frames at the front of the list.
12296    ///
12297    /// Commit indices into the returned list are LOCAL (0 = first frame of
12298    /// oldest surviving archive).  To obtain the GLOBAL index add
12299    /// `self.wal_horizon_floor`.
12300    /// How many frames the surviving archives hold, without materialising them.
12301    ///
12302    /// The same count `all_frames` puts at the front of its list. Used to seed
12303    /// [`wal_frames_written`](GraphDb::wal_frames_written) at open without
12304    /// decoding the live WAL a second time; free on a store with no archives,
12305    /// which is most of them.
12306    fn archive_frame_count(&self) -> Result<u64> {
12307        let mut n = 0u64;
12308        for a in self.fs.list_archives()? {
12309            let bytes = self.fs.read_archive(a)?;
12310            let (frames, _) = decode_all(&bytes);
12311            n += frames.len() as u64;
12312        }
12313        Ok(n)
12314    }
12315
12316    fn all_frames(&self) -> Result<(Vec<WalRecord>, u64)> {
12317        let archive_ns = self.fs.list_archives()?;
12318        let mut all: Vec<WalRecord> = Vec::new();
12319        for n in archive_ns {
12320            let bytes = self.fs.read_archive(n)?;
12321            let (frames, _) = decode_all(&bytes);
12322            all.extend(frames);
12323        }
12324        let archive_count = all.len() as u64;
12325        let live_bytes = self.fs.read(FileId::Wal)?;
12326        let (live_frames, _) = decode_all(&live_bytes);
12327        all.extend(live_frames);
12328        Ok((all, archive_count))
12329    }
12330
12331    /// Return the total number of committed WAL frames visible in the current
12332    /// horizon window, including frames in surviving WAL archives.
12333    ///
12334    /// This is the exclusive upper bound for valid `at_commit` indices in
12335    /// `was_linked`. Valid indices are `wal_horizon_floor()..wal_total_commits()`.
12336    ///
12337    /// Returns the horizon floor when all surviving history is empty.
12338    pub fn wal_total_commits(&self) -> Result<u64> {
12339        let (frames, _) = self.all_frames()?;
12340        Ok(self.wal_horizon_floor + frames.len() as u64)
12341    }
12342
12343    /// The global frame index of the first commit reachable through surviving
12344    /// archives (0 when no archives have been pruned).
12345    pub fn wal_horizon_floor(&self) -> u64 {
12346        self.wal_horizon_floor
12347    }
12348
12349    /// Return the per-node change history for `key` by scanning the on-disk WAL.
12350    ///
12351    /// ## Horizon
12352    ///
12353    /// History reaches back only to the last WAL-truncating snapshot, exactly like `open_at`.
12354    /// Snapshots written with `keep_wal: true` preserve deeper history. This is the honest,
12355    /// zero-cost contract; a durable history log is out of scope.
12356    ///
12357    /// ## Derived edges
12358    ///
12359    /// Rule-created (derived) edges are **not** in the WAL and therefore do not appear in
12360    /// history. Only edges written directly by the application are recorded.
12361    ///
12362    /// ## Deleted nodes
12363    ///
12364    /// For nodes that have been deleted, dense-id records (SetPropId, InsertEdgeId) that
12365    /// predate the deletion may not resolve (the id is tombstoned in the live map). The
12366    /// string-keyed `DeleteNode` record still matches and produces a `NodeDeleted` entry.
12367    /// Prop/edge history of a deleted node may therefore be partially unresolvable.
12368    ///
12369    /// ## Dense-id edge entries and tombstoned partners
12370    ///
12371    /// Edge entries from dense-id WAL records (`InsertEdgeId`) are omitted when the partner
12372    /// endpoint's dense id is tombstoned. As a result, a live node's history can contain an
12373    /// `EdgeRemoved` (string-keyed, always resolves) without a corresponding `EdgeAdded`.
12374    /// Build commit-bounded alias intervals for `queried_key`.
12375    ///
12376    /// Returns a list of `(key, valid_from_inclusive, valid_until_exclusive)` tuples.
12377    /// A record written under `key` at commit `c` matches the queried identity iff
12378    /// `c >= valid_from && (valid_until.is_none() || c < valid_until)`.
12379    ///
12380    /// Each alias entry carries both a lower and an upper bound so that key-reuse
12381    /// after a rename is handled correctly: if "a" is renamed to "b" at commit 5,
12382    /// then a NEW node is created as "a" at commit 7 and renamed to "c" at commit 10,
12383    /// querying "c" must NOT surface identity-1's events (commits 0–4 under "a");
12384    /// only identity-2's events (commits 7–9 under "a") are in scope.
12385    ///
12386    /// Only **forward aliasing**: querying the *new* key surfaces events written
12387    /// under the *old* key.  The reverse direction is not supported.
12388    fn build_key_alias_intervals(
12389        &self,
12390        frames: &[core_storage::wal::WalRecord],
12391        queried_key: &str,
12392    ) -> Vec<(String, u64, Option<u64>)> {
12393        use core_storage::wal::WalRecord;
12394
12395        // Pre-pass: build reverse_rename and key_starts maps.
12396        let mut reverse_rename: HashMap<String, (String, u64)> = HashMap::new();
12397        let mut key_starts: HashMap<String, Vec<u64>> = HashMap::new();
12398
12399        for (local_i, frame) in frames.iter().enumerate() {
12400            let commit = self.wal_horizon_floor + local_i as u64;
12401            let records: &[WalRecord] = match frame {
12402                WalRecord::Batch(inner) => inner.as_slice(),
12403                single => std::slice::from_ref(single),
12404            };
12405            for rec in records {
12406                match rec {
12407                    WalRecord::InsertNode { key, .. } | WalRecord::InsertNodeId { key, .. } => {
12408                        key_starts.entry(key.clone()).or_default().push(commit);
12409                    }
12410                    WalRecord::RenameNode { old_key, new_key } => {
12411                        // new_key came into existence at this commit.
12412                        key_starts.entry(new_key.clone()).or_default().push(commit);
12413                        // Record the reverse rename: new_key was introduced by renaming old_key.
12414                        reverse_rename.insert(new_key.clone(), (old_key.clone(), commit));
12415                    }
12416                    _ => {}
12417                }
12418            }
12419        }
12420
12421        // Build alias intervals by following the reverse rename chain.
12422        let mut result: Vec<(String, u64, Option<u64>)> = Vec::new();
12423        let mut current_key = queried_key.to_string();
12424        let mut current_valid_until: Option<u64> = None;
12425
12426        loop {
12427            // valid_from: the most recent commit where current_key was assigned to this
12428            // identity.  For aliases (valid_until = Some(vu)), find the last start event
12429            // for the key strictly before vu — this is where the alias's occupancy by
12430            // this identity began, correctly excluding prior identities that reused the key.
12431            let valid_from = if let Some(vu) = current_valid_until {
12432                key_starts
12433                    .get(&current_key)
12434                    .and_then(|starts| starts.iter().rev().find(|&&s| s < vu).copied())
12435                    .unwrap_or(self.wal_horizon_floor)
12436            } else {
12437                // Queried key — no upper bound; may have been introduced at any commit.
12438                self.wal_horizon_floor
12439            };
12440
12441            result.push((current_key.clone(), valid_from, current_valid_until));
12442
12443            match reverse_rename.get(&current_key) {
12444                Some((old_key, rename_commit)) => {
12445                    current_valid_until = Some(*rename_commit);
12446                    current_key = old_key.clone();
12447                }
12448                None => break,
12449            }
12450        }
12451
12452        result
12453    }
12454
12455    /// Returns true if `record_key` matches any alias interval that covers `commit`.
12456    fn aliases_match(
12457        intervals: &[(String, u64, Option<u64>)],
12458        record_key: &str,
12459        commit: u64,
12460    ) -> bool {
12461        intervals
12462            .iter()
12463            .any(|(k, vf, vu)| k == record_key && commit >= *vf && vu.is_none_or(|u| commit < u))
12464    }
12465
12466    /// Return the change history of node `key` by scanning the on-disk WAL.
12467    ///
12468    /// ## Horizon
12469    ///
12470    /// History reaches back only as far as the retained WAL. The returned
12471    /// [`HistoryResult`](crate::history::HistoryResult) carries `total_commits`
12472    /// (the exclusive upper bound for valid commit indices) and `horizon` (the
12473    /// oldest commit still reachable). When `horizon > 0`, older events were
12474    /// pruned and are not in `items`.
12475    pub fn node_history(
12476        &self,
12477        key: &str,
12478    ) -> Result<crate::history::HistoryResult<crate::history::HistoryEntry>> {
12479        use crate::history::{HistoryChange, HistoryEntry, HistoryResult};
12480        use core_storage::wal::WalRecord;
12481
12482        let (frames, _) = self.all_frames()?;
12483        let total_commits = self.wal_horizon_floor + frames.len() as u64;
12484
12485        // Resolve commit-bounded alias intervals for `key` (handles renames in the WAL).
12486        let alias_intervals = self.build_key_alias_intervals(&frames, key);
12487
12488        let mut out: Vec<HistoryEntry> = Vec::new();
12489
12490        for (local_i, frame) in frames.iter().enumerate() {
12491            let commit = self.wal_horizon_floor + local_i as u64;
12492            // Collect the inner records to process — Batch is one commit, single records are one commit.
12493            let records: &[WalRecord] = match frame {
12494                WalRecord::Batch(inner) => inner.as_slice(),
12495                single => std::slice::from_ref(single),
12496            };
12497
12498            for rec in records {
12499                let change = match rec {
12500                    WalRecord::InsertNode { label, key: k, .. }
12501                        if Self::aliases_match(&alias_intervals, k, commit) =>
12502                    {
12503                        Some(HistoryChange::NodeInserted {
12504                            label: label.clone(),
12505                        })
12506                    }
12507                    WalRecord::InsertNodeId { label, key: k, .. }
12508                        if Self::aliases_match(&alias_intervals, k, commit) =>
12509                    {
12510                        let label_str = match self.syms.resolve(*label) {
12511                            Some(s) => s.to_string(),
12512                            None => continue,
12513                        };
12514                        Some(HistoryChange::NodeInserted { label: label_str })
12515                    }
12516                    WalRecord::SetProp {
12517                        key: k,
12518                        field,
12519                        value,
12520                    } if Self::aliases_match(&alias_intervals, k, commit) => {
12521                        Some(HistoryChange::PropSet {
12522                            field: field.clone(),
12523                            value: value.clone(),
12524                        })
12525                    }
12526                    WalRecord::SetPropId { id, field, value } => {
12527                        // Use key_of_historical (not key_of) so a node's prop_set
12528                        // events remain visible after the node is later deleted:
12529                        // key_of returns None for a tombstoned id, which would
12530                        // silently drop every PropSet between insert and delete.
12531                        // Mirrors the InsertEdgeId arm below and edge_history's
12532                        // own id-keyed arms.
12533                        match self.ids.key_of_historical(*id) {
12534                            // key_of_historical returns the last-known (possibly
12535                            // post-rename, possibly post-delete) key; compare to queried key.
12536                            Some(resolved) if resolved == key => {
12537                                let field_str = match self.syms.resolve(*field) {
12538                                    Some(s) => s.to_string(),
12539                                    None => continue,
12540                                };
12541                                Some(HistoryChange::PropSet {
12542                                    field: field_str,
12543                                    value: value.clone(),
12544                                })
12545                            }
12546                            _ => None,
12547                        }
12548                    }
12549                    WalRecord::RemoveProp { key: k, field }
12550                        if Self::aliases_match(&alias_intervals, k, commit) =>
12551                    {
12552                        Some(HistoryChange::PropRemoved {
12553                            field: field.clone(),
12554                        })
12555                    }
12556                    WalRecord::InsertEdge {
12557                        edge_type,
12558                        src_key,
12559                        dst_key,
12560                    } => {
12561                        if Self::aliases_match(&alias_intervals, src_key, commit) {
12562                            Some(HistoryChange::EdgeAdded {
12563                                edge_type: edge_type.clone(),
12564                                other: dst_key.clone(),
12565                                outgoing: true,
12566                            })
12567                        } else if Self::aliases_match(&alias_intervals, dst_key, commit) {
12568                            Some(HistoryChange::EdgeAdded {
12569                                edge_type: edge_type.clone(),
12570                                other: src_key.clone(),
12571                                outgoing: false,
12572                            })
12573                        } else {
12574                            None
12575                        }
12576                    }
12577                    WalRecord::InsertEdgeId { etype, src, dst } => {
12578                        let etype_str = match self.syms.resolve(*etype) {
12579                            Some(s) => s.to_string(),
12580                            None => continue,
12581                        };
12582                        // key_of_historical (not key_of): an edge added before
12583                        // either endpoint was later deleted must still resolve —
12584                        // see the SetPropId arm above and edge_history's
12585                        // InsertEdgeId arm, which use the same lookup for the
12586                        // same reason.
12587                        let src_key = self.ids.key_of_historical(*src);
12588                        let dst_key = self.ids.key_of_historical(*dst);
12589                        if src_key == Some(key) {
12590                            let other = match dst_key {
12591                                Some(s) => s.to_string(),
12592                                None => continue,
12593                            };
12594                            Some(HistoryChange::EdgeAdded {
12595                                edge_type: etype_str,
12596                                other,
12597                                outgoing: true,
12598                            })
12599                        } else if dst_key == Some(key) {
12600                            let other = match src_key {
12601                                Some(s) => s.to_string(),
12602                                None => continue,
12603                            };
12604                            Some(HistoryChange::EdgeAdded {
12605                                edge_type: etype_str,
12606                                other,
12607                                outgoing: false,
12608                            })
12609                        } else {
12610                            None
12611                        }
12612                    }
12613                    WalRecord::DeleteEdge {
12614                        edge_type,
12615                        src_key,
12616                        dst_key,
12617                    } => {
12618                        if Self::aliases_match(&alias_intervals, src_key, commit) {
12619                            Some(HistoryChange::EdgeRemoved {
12620                                edge_type: edge_type.clone(),
12621                                other: dst_key.clone(),
12622                                outgoing: true,
12623                            })
12624                        } else if Self::aliases_match(&alias_intervals, dst_key, commit) {
12625                            Some(HistoryChange::EdgeRemoved {
12626                                edge_type: edge_type.clone(),
12627                                other: src_key.clone(),
12628                                outgoing: false,
12629                            })
12630                        } else {
12631                            None
12632                        }
12633                    }
12634                    WalRecord::DeleteNode { key: k }
12635                        if Self::aliases_match(&alias_intervals, k, commit) =>
12636                    {
12637                        Some(HistoryChange::NodeDeleted)
12638                    }
12639                    // Skip: rule/view/fulltext/intern metadata; Batch wrapper handled above.
12640                    _ => None,
12641                };
12642
12643                if let Some(change) = change {
12644                    out.push(HistoryEntry { commit, change });
12645                }
12646            }
12647        }
12648
12649        Ok(HistoryResult {
12650            items: out,
12651            total_commits,
12652            horizon: self.wal_horizon_floor,
12653        })
12654    }
12655
12656    /// Return the per-edge change history between nodes `a` and `b` by scanning
12657    /// the on-disk WAL.
12658    ///
12659    /// ## Horizon
12660    ///
12661    /// History reaches back only to the last WAL-truncating snapshot, exactly
12662    /// like `node_history` and `open_at`. The returned [`HistoryResult`] carries
12663    /// `total_commits` (= number of WAL frames), which is the exclusive upper
12664    /// bound for valid commit indices.
12665    ///
12666    /// ## Derived edges
12667    ///
12668    /// Rule-derived edges appear via `DerivedEdgeAdded` / `DerivedEdgeRetracted`
12669    /// WAL markers written by `log_then_apply_with` after each rule-firing
12670    /// mutation. The `rule` field of those events carries the rule name.
12671    ///
12672    /// ## DeleteNode
12673    ///
12674    /// When a node is deleted, its manual incident edges are swept inline without
12675    /// individual `DeleteEdge` WAL records. `edge_history` detects `DeleteNode`
12676    /// events for either endpoint and synthesises `Retracted(rule:None)` events
12677    /// for each manual edge that was active at that point. Derived edges active at
12678    /// the time of deletion are handled by the `DerivedEdgeRetracted` marker that
12679    /// the engine appends immediately after the `DeleteNode` record; those events
12680    /// carry correct rule attribution and are emitted by the marker arm, not the
12681    /// synthetic sweep.
12682    ///
12683    /// ## Masks
12684    ///
12685    /// Like `node_history`, this method has no mask parameter and returns WAL
12686    /// history regardless of any role mask. For masked history semantics, apply
12687    /// the mask at the caller level.
12688    pub fn edge_history(
12689        &self,
12690        a: &str,
12691        b: &str,
12692    ) -> Result<crate::history::HistoryResult<crate::history::EdgeHistoryEvent>> {
12693        use crate::history::{EdgeEvent, EdgeHistoryEvent, HistoryResult};
12694        use core_storage::wal::WalRecord;
12695
12696        let (frames, _) = self.all_frames()?;
12697        let total_commits = self.wal_horizon_floor + frames.len() as u64;
12698
12699        // Resolve all historical names for a and b (handles RenameNode in the WAL).
12700        // Intervals are commit-bounded so recycled keys don't contaminate histories.
12701        let alias_a = self.build_key_alias_intervals(&frames, a);
12702        let alias_b = self.build_key_alias_intervals(&frames, b);
12703
12704        // Active edges between a and b tracked as (edge_type, src_key, dst_key, is_derived).
12705        // The is_derived flag is used by the DeleteNode sweep: manual edges are
12706        // swept with a synthetic Retracted(rule:None); derived edges are skipped
12707        // because the engine writes a DerivedEdgeRetracted marker immediately after
12708        // the DeleteNode record, which carries the correct rule attribution.
12709        let mut active: Vec<(String, String, String, bool)> = Vec::new();
12710        let mut out: Vec<EdgeHistoryEvent> = Vec::new();
12711
12712        for (local_i, frame) in frames.iter().enumerate() {
12713            let commit = self.wal_horizon_floor + local_i as u64;
12714            let records: &[WalRecord] = match frame {
12715                WalRecord::Batch(inner) => inner.as_slice(),
12716                single => std::slice::from_ref(single),
12717            };
12718
12719            for rec in records {
12720                match rec {
12721                    WalRecord::InsertEdge {
12722                        edge_type,
12723                        src_key,
12724                        dst_key,
12725                    } => {
12726                        let is_ab = Self::aliases_match(&alias_a, src_key, commit)
12727                            && Self::aliases_match(&alias_b, dst_key, commit);
12728                        let is_ba = Self::aliases_match(&alias_b, src_key, commit)
12729                            && Self::aliases_match(&alias_a, dst_key, commit);
12730                        if is_ab || is_ba {
12731                            active.push((
12732                                edge_type.clone(),
12733                                src_key.clone(),
12734                                dst_key.clone(),
12735                                false,
12736                            ));
12737                            out.push(EdgeHistoryEvent {
12738                                edge_type: edge_type.clone(),
12739                                commit,
12740                                event: EdgeEvent::Added,
12741                                rule: None,
12742                            });
12743                        }
12744                    }
12745                    WalRecord::InsertEdgeId { etype, src, dst } => {
12746                        let etype_str = match self.syms.resolve(*etype) {
12747                            Some(s) => s.to_string(),
12748                            None => continue,
12749                        };
12750                        // Use key_of_historical so tombstoned nodes (deleted
12751                        // later in the WAL) still resolve during the scan.
12752                        let src_key = self.ids.key_of_historical(*src);
12753                        let dst_key = self.ids.key_of_historical(*dst);
12754                        let is_ab = src_key == Some(a) && dst_key == Some(b);
12755                        let is_ba = src_key == Some(b) && dst_key == Some(a);
12756                        if is_ab || is_ba {
12757                            let src_str = src_key.unwrap().to_string();
12758                            let dst_str = dst_key.unwrap().to_string();
12759                            active.push((etype_str.clone(), src_str, dst_str, false));
12760                            out.push(EdgeHistoryEvent {
12761                                edge_type: etype_str,
12762                                commit,
12763                                event: EdgeEvent::Added,
12764                                rule: None,
12765                            });
12766                        }
12767                    }
12768                    WalRecord::DeleteEdge {
12769                        edge_type,
12770                        src_key,
12771                        dst_key,
12772                    } => {
12773                        let is_ab = Self::aliases_match(&alias_a, src_key, commit)
12774                            && Self::aliases_match(&alias_b, dst_key, commit);
12775                        let is_ba = Self::aliases_match(&alias_b, src_key, commit)
12776                            && Self::aliases_match(&alias_a, dst_key, commit);
12777                        if is_ab || is_ba {
12778                            // Remove the first matching active entry (flag ignored).
12779                            if let Some(pos) = active.iter().position(|(et, s, d, _)| {
12780                                et == edge_type && s == src_key && d == dst_key
12781                            }) {
12782                                active.remove(pos);
12783                            }
12784                            out.push(EdgeHistoryEvent {
12785                                edge_type: edge_type.clone(),
12786                                commit,
12787                                event: EdgeEvent::Retracted,
12788                                rule: None,
12789                            });
12790                        }
12791                    }
12792                    WalRecord::DeleteNode { key: k }
12793                        if Self::aliases_match(&alias_a, k, commit)
12794                            || Self::aliases_match(&alias_b, k, commit) =>
12795                    {
12796                        // Sweep: implicitly retract only MANUAL active edges.
12797                        // Derived active edges are skipped here because the rule
12798                        // engine appends a DerivedEdgeRetracted marker immediately
12799                        // after this DeleteNode record; that marker produces the
12800                        // single correctly-attributed Retracted event.  Derived
12801                        // entries are dropped from `active` (the marker arm's
12802                        // idempotent retain finds nothing to remove).
12803                        for (et, _, _, is_derived) in active.drain(..) {
12804                            if !is_derived {
12805                                out.push(EdgeHistoryEvent {
12806                                    edge_type: et,
12807                                    commit,
12808                                    event: EdgeEvent::Retracted,
12809                                    rule: None,
12810                                });
12811                            }
12812                            // Derived: drop silently; marker carries the Retracted event.
12813                        }
12814                    }
12815                    WalRecord::DerivedEdgeAdded {
12816                        rule,
12817                        edge_type: et,
12818                        src_key,
12819                        dst_key,
12820                    } => {
12821                        let is_ab = Self::aliases_match(&alias_a, src_key, commit)
12822                            && Self::aliases_match(&alias_b, dst_key, commit);
12823                        let is_ba = Self::aliases_match(&alias_b, src_key, commit)
12824                            && Self::aliases_match(&alias_a, dst_key, commit);
12825                        if is_ab || is_ba {
12826                            active.push((et.clone(), src_key.clone(), dst_key.clone(), true));
12827                            out.push(EdgeHistoryEvent {
12828                                edge_type: et.clone(),
12829                                commit,
12830                                event: EdgeEvent::Added,
12831                                rule: Some(rule.clone()),
12832                            });
12833                        }
12834                    }
12835                    WalRecord::DerivedEdgeRetracted {
12836                        rule,
12837                        edge_type: et,
12838                        src_key,
12839                        dst_key,
12840                    } => {
12841                        let is_ab = Self::aliases_match(&alias_a, src_key, commit)
12842                            && Self::aliases_match(&alias_b, dst_key, commit);
12843                        let is_ba = Self::aliases_match(&alias_b, src_key, commit)
12844                            && Self::aliases_match(&alias_a, dst_key, commit);
12845                        if is_ab || is_ba {
12846                            // Push unconditionally: a derived edge whose Added marker
12847                            // predates the history horizon has no `active` entry, but
12848                            // the retraction is still a real in-window event.
12849                            // Remove from active idempotently if present.
12850                            active.retain(|(aet, s, d, _)| {
12851                                !(aet == et && s == src_key && d == dst_key)
12852                            });
12853                            out.push(EdgeHistoryEvent {
12854                                edge_type: et.clone(),
12855                                commit,
12856                                event: EdgeEvent::Retracted,
12857                                rule: Some(rule.clone()),
12858                            });
12859                        }
12860                    }
12861                    // All other records (InsertNode, SetProp, CreateRule, etc.)
12862                    // do not affect edges between a and b.
12863                    _ => {}
12864                }
12865            }
12866        }
12867
12868        Ok(HistoryResult {
12869            items: out,
12870            total_commits,
12871            horizon: self.wal_horizon_floor,
12872        })
12873    }
12874
12875    /// Return `true` iff an edge of `edge_type` existed between `a` and `b`
12876    /// (in either direction) at the WAL commit `at_commit`.
12877    ///
12878    /// ## Horizon
12879    ///
12880    /// Valid commit indices are `0..total_commits` where `total_commits` is the
12881    /// number of WAL frames. An `at_commit >= total_commits` is outside the
12882    /// visible horizon and returns [`GraphError::CommitOutOfRange`].
12883    ///
12884    /// ## Derived edges
12885    ///
12886    /// Rule-derived edges are tracked via `DerivedEdgeAdded` / `DerivedEdgeRetracted`
12887    /// WAL markers appended at firing time (Task 1). `was_linked` reads these markers
12888    /// and therefore includes derived edges in its point-in-time evaluation,
12889    /// matching `edge_history`'s fidelity.
12890    pub fn was_linked(&self, a: &str, b: &str, edge_type: &str, at_commit: u64) -> Result<bool> {
12891        use core_storage::wal::WalRecord;
12892
12893        let (frames, _) = self.all_frames()?;
12894        let total_commits = self.wal_horizon_floor + frames.len() as u64;
12895
12896        // Horizon floor: commits in pruned archives are unreachable.
12897        if at_commit < self.wal_horizon_floor {
12898            return Err(GraphError::CommitOutOfRange {
12899                commit: at_commit,
12900                total: total_commits,
12901                floor: self.wal_horizon_floor,
12902            });
12903        }
12904        if at_commit >= total_commits {
12905            return Err(GraphError::CommitOutOfRange {
12906                commit: at_commit,
12907                total: total_commits,
12908                floor: self.wal_horizon_floor,
12909            });
12910        }
12911
12912        // Resolve all historical names for a and b (handles RenameNode in the WAL).
12913        // Intervals are commit-bounded so recycled keys don't contaminate point-in-time reads.
12914        let alias_a = self.build_key_alias_intervals(&frames, a);
12915        let alias_b = self.build_key_alias_intervals(&frames, b);
12916
12917        // Local index into surviving frames (0 = first frame of oldest archive).
12918        let local_commit = at_commit - self.wal_horizon_floor;
12919
12920        // Replay local frames 0..=local_commit, tracking active edges.
12921        let mut active: BTreeSet<(String, String, String)> = BTreeSet::new();
12922
12923        for (local_i, frame) in frames.iter().enumerate().take((local_commit + 1) as usize) {
12924            let commit = self.wal_horizon_floor + local_i as u64;
12925            let records: &[WalRecord] = match frame {
12926                WalRecord::Batch(inner) => inner.as_slice(),
12927                single => std::slice::from_ref(single),
12928            };
12929
12930            for rec in records {
12931                match rec {
12932                    WalRecord::InsertEdge {
12933                        edge_type: et,
12934                        src_key,
12935                        dst_key,
12936                    } => {
12937                        let is_ab = Self::aliases_match(&alias_a, src_key, commit)
12938                            && Self::aliases_match(&alias_b, dst_key, commit);
12939                        let is_ba = Self::aliases_match(&alias_b, src_key, commit)
12940                            && Self::aliases_match(&alias_a, dst_key, commit);
12941                        if is_ab || is_ba {
12942                            active.insert((et.clone(), src_key.clone(), dst_key.clone()));
12943                        }
12944                    }
12945                    WalRecord::InsertEdgeId { etype, src, dst } => {
12946                        let etype_str = match self.syms.resolve(*etype) {
12947                            Some(s) => s.to_string(),
12948                            None => continue,
12949                        };
12950                        // Use key_of_historical so tombstoned nodes resolve.
12951                        let src_key = self.ids.key_of_historical(*src);
12952                        let dst_key = self.ids.key_of_historical(*dst);
12953                        let is_ab = src_key == Some(a) && dst_key == Some(b);
12954                        let is_ba = src_key == Some(b) && dst_key == Some(a);
12955                        if is_ab || is_ba {
12956                            active.insert((
12957                                etype_str,
12958                                src_key.unwrap().to_string(),
12959                                dst_key.unwrap().to_string(),
12960                            ));
12961                        }
12962                    }
12963                    WalRecord::DeleteEdge {
12964                        edge_type: et,
12965                        src_key,
12966                        dst_key,
12967                    } => {
12968                        let is_ab = Self::aliases_match(&alias_a, src_key, commit)
12969                            && Self::aliases_match(&alias_b, dst_key, commit);
12970                        let is_ba = Self::aliases_match(&alias_b, src_key, commit)
12971                            && Self::aliases_match(&alias_a, dst_key, commit);
12972                        if is_ab || is_ba {
12973                            active.remove(&(et.clone(), src_key.clone(), dst_key.clone()));
12974                        }
12975                    }
12976                    WalRecord::DeleteNode { key: k }
12977                        if Self::aliases_match(&alias_a, k, commit)
12978                            || Self::aliases_match(&alias_b, k, commit) =>
12979                    {
12980                        // All edges touching the deleted node are gone.
12981                        active.retain(|(_, s, d)| s != k && d != k);
12982                    }
12983                    WalRecord::DerivedEdgeAdded {
12984                        edge_type: et,
12985                        src_key,
12986                        dst_key,
12987                        ..
12988                    } => {
12989                        let is_ab = Self::aliases_match(&alias_a, src_key, commit)
12990                            && Self::aliases_match(&alias_b, dst_key, commit);
12991                        let is_ba = Self::aliases_match(&alias_b, src_key, commit)
12992                            && Self::aliases_match(&alias_a, dst_key, commit);
12993                        if is_ab || is_ba {
12994                            active.insert((et.clone(), src_key.clone(), dst_key.clone()));
12995                        }
12996                    }
12997                    WalRecord::DerivedEdgeRetracted {
12998                        edge_type: et,
12999                        src_key,
13000                        dst_key,
13001                        ..
13002                    } => {
13003                        let is_ab = Self::aliases_match(&alias_a, src_key, commit)
13004                            && Self::aliases_match(&alias_b, dst_key, commit);
13005                        let is_ba = Self::aliases_match(&alias_b, src_key, commit)
13006                            && Self::aliases_match(&alias_a, dst_key, commit);
13007                        if is_ab || is_ba {
13008                            active.remove(&(et.clone(), src_key.clone(), dst_key.clone()));
13009                        }
13010                    }
13011                    _ => {}
13012                }
13013            }
13014        }
13015
13016        Ok(active.iter().any(|(et, _, _)| et == edge_type))
13017    }
13018
13019    /// Every edge incident to `key` — either endpoint — that existed at WAL
13020    /// commit `commit`, from ONE scan of the WAL.
13021    ///
13022    /// This is the bulk form of [`was_linked`](GraphDb::was_linked): answering
13023    /// "what did K's relationships look like at commit C" with one call instead
13024    /// of one [`edge_history`](GraphDb::edge_history) per candidate partner.
13025    /// The two agree edge for edge.
13026    ///
13027    /// Results are sorted by `(edge_type, src_key, dst_key)`.
13028    ///
13029    /// ## Horizon
13030    ///
13031    /// Valid commit indices are `wal_horizon_floor()..wal_total_commits()`;
13032    /// anything outside is [`GraphError::CommitOutOfRange`], exactly like
13033    /// `was_linked`. An unknown key is not an error — it simply had no edges.
13034    ///
13035    /// ## Derived edges
13036    ///
13037    /// `DerivedEdgeAdded` / `DerivedEdgeRetracted` markers carry rule
13038    /// attribution, so a rule-owned edge comes back with `derived: true` and
13039    /// `rule: Some(name)`.
13040    ///
13041    /// ## Renames
13042    ///
13043    /// `key` is matched through the same commit-bounded alias intervals
13044    /// `edge_history` uses, so querying a node's *current* key surfaces edges
13045    /// written under an earlier name. Endpoint keys in the result are reported
13046    /// under the name the node carries today, so they can be fed straight back
13047    /// into `node_info`, `explain` or another `edges_at`.
13048    ///
13049    /// ## Masks
13050    ///
13051    /// Like `edge_history` and `node_history`, this reads the WAL regardless of
13052    /// any role mask. Apply masking at the caller level.
13053    pub fn edges_at(&self, key: &str, commit: u64) -> Result<Vec<EdgeAt>> {
13054        use core_storage::wal::WalRecord;
13055
13056        let (frames, _) = self.all_frames()?;
13057        let total_commits = self.wal_horizon_floor + frames.len() as u64;
13058
13059        // Horizon floor: commits in pruned archives are unreachable.
13060        if commit < self.wal_horizon_floor || commit >= total_commits {
13061            return Err(GraphError::CommitOutOfRange {
13062                commit,
13063                total: total_commits,
13064                floor: self.wal_horizon_floor,
13065            });
13066        }
13067
13068        // Commit-bounded historical names of `key` (handles RenameNode).
13069        let alias = self.build_key_alias_intervals(&frames, key);
13070
13071        // Forward rename chain, for reporting endpoints under their current
13072        // names: old key → [(commit, new key)] in ascending commit order.
13073        // Built over the whole WAL, not just the prefix up to `commit`, because
13074        // a rename after `commit` still changes what the node is called today.
13075        let mut renames: HashMap<String, Vec<(u64, String)>> = HashMap::new();
13076        for (local_i, frame) in frames.iter().enumerate() {
13077            let c = self.wal_horizon_floor + local_i as u64;
13078            let records: &[WalRecord] = match frame {
13079                WalRecord::Batch(inner) => inner.as_slice(),
13080                single => std::slice::from_ref(single),
13081            };
13082            for rec in records {
13083                if let WalRecord::RenameNode { old_key, new_key } = rec {
13084                    renames
13085                        .entry(old_key.clone())
13086                        .or_default()
13087                        .push((c, new_key.clone()));
13088                }
13089            }
13090        }
13091
13092        // The name a node written as `k` at commit `from` carries today.
13093        // Follows the first rename at or after `from`, then keeps going. The
13094        // iteration cap bounds a rename cycle inside a single batch.
13095        let canon = |k: &str, from: u64| -> String {
13096            if renames.is_empty() {
13097                return k.to_string();
13098            }
13099            let mut cur = k.to_string();
13100            let mut at = from;
13101            for _ in 0..64 {
13102                match renames
13103                    .get(&cur)
13104                    .and_then(|v| v.iter().find(|(c, _)| *c >= at))
13105                {
13106                    Some((c, new)) => {
13107                        at = *c;
13108                        cur = new.clone();
13109                    }
13110                    None => break,
13111                }
13112            }
13113            cur
13114        };
13115
13116        let local_commit = commit - self.wal_horizon_floor;
13117        // (edge_type, src_key, dst_key) → (derived, rule)
13118        let mut active: BTreeMap<(String, String, String), (bool, Option<String>)> =
13119            BTreeMap::new();
13120
13121        for (local_i, frame) in frames.iter().enumerate().take((local_commit + 1) as usize) {
13122            let c = self.wal_horizon_floor + local_i as u64;
13123            let records: &[WalRecord] = match frame {
13124                WalRecord::Batch(inner) => inner.as_slice(),
13125                single => std::slice::from_ref(single),
13126            };
13127
13128            for rec in records {
13129                match rec {
13130                    WalRecord::InsertEdge {
13131                        edge_type,
13132                        src_key,
13133                        dst_key,
13134                    } => {
13135                        if Self::aliases_match(&alias, src_key, c)
13136                            || Self::aliases_match(&alias, dst_key, c)
13137                        {
13138                            active.insert(
13139                                (edge_type.clone(), canon(src_key, c), canon(dst_key, c)),
13140                                (false, None),
13141                            );
13142                        }
13143                    }
13144                    WalRecord::InsertEdgeId { etype, src, dst } => {
13145                        let Some(etype_str) = self.syms.resolve(*etype) else {
13146                            continue;
13147                        };
13148                        // `key_of_historical` resolves tombstoned ids too, and
13149                        // already returns the node's current key — no rename
13150                        // canonicalisation needed on this arm.
13151                        let (Some(src_key), Some(dst_key)) = (
13152                            self.ids.key_of_historical(*src),
13153                            self.ids.key_of_historical(*dst),
13154                        ) else {
13155                            continue;
13156                        };
13157                        if src_key == key || dst_key == key {
13158                            active.insert(
13159                                (
13160                                    etype_str.to_string(),
13161                                    src_key.to_string(),
13162                                    dst_key.to_string(),
13163                                ),
13164                                (false, None),
13165                            );
13166                        }
13167                    }
13168                    WalRecord::DeleteEdge {
13169                        edge_type,
13170                        src_key,
13171                        dst_key,
13172                    } => {
13173                        if Self::aliases_match(&alias, src_key, c)
13174                            || Self::aliases_match(&alias, dst_key, c)
13175                        {
13176                            active.remove(&(
13177                                edge_type.clone(),
13178                                canon(src_key, c),
13179                                canon(dst_key, c),
13180                            ));
13181                        }
13182                    }
13183                    WalRecord::DeleteNode { key: k } => {
13184                        if active.is_empty() {
13185                            continue;
13186                        }
13187                        if Self::aliases_match(&alias, k, c) {
13188                            // Our node is gone; every incident edge goes with it.
13189                            active.clear();
13190                        } else {
13191                            // A partner is gone; its edges to us go with it.
13192                            let ck = canon(k, c);
13193                            active.retain(|(_, s, d), _| *s != ck && *d != ck);
13194                        }
13195                    }
13196                    WalRecord::DerivedEdgeAdded {
13197                        rule,
13198                        edge_type,
13199                        src_key,
13200                        dst_key,
13201                    } => {
13202                        if Self::aliases_match(&alias, src_key, c)
13203                            || Self::aliases_match(&alias, dst_key, c)
13204                        {
13205                            active.insert(
13206                                (edge_type.clone(), canon(src_key, c), canon(dst_key, c)),
13207                                (true, Some(rule.clone())),
13208                            );
13209                        }
13210                    }
13211                    WalRecord::DerivedEdgeRetracted {
13212                        edge_type,
13213                        src_key,
13214                        dst_key,
13215                        ..
13216                    } => {
13217                        if Self::aliases_match(&alias, src_key, c)
13218                            || Self::aliases_match(&alias, dst_key, c)
13219                        {
13220                            active.remove(&(
13221                                edge_type.clone(),
13222                                canon(src_key, c),
13223                                canon(dst_key, c),
13224                            ));
13225                        }
13226                    }
13227                    // InsertNode, SetProp, CreateRule, … do not move edges.
13228                    _ => {}
13229                }
13230            }
13231        }
13232
13233        // BTreeMap iteration is already (edge_type, src, dst) order.
13234        Ok(active
13235            .into_iter()
13236            .map(|((edge_type, src_key, dst_key), (derived, rule))| EdgeAt {
13237                edge_type,
13238                src_key,
13239                dst_key,
13240                derived,
13241                rule,
13242            })
13243            .collect())
13244    }
13245
13246    /// The derived edges that would be retracted and derived if `key.field`
13247    /// were set to `value` — computed WITHOUT writing anything.
13248    ///
13249    /// Nothing is committed and nothing on `self` is mutated: the rule engine's
13250    /// provenance, its candidate indexes, the topology and the property columns
13251    /// are all cloned first, the change is applied to the clone, and the real
13252    /// per-node re-derivation (`RuleEngine::on_node_changed` — the same call
13253    /// `set_prop` makes during apply) runs against it. The derived-edge deltas
13254    /// it emits are the answer, so rule semantics — predicates, top-k,
13255    /// via-hops, chaining, weights — are the engine's, not a re-implementation.
13256    ///
13257    /// Works on a read-only handle.
13258    ///
13259    /// **While a rule's vector index is still building** (`RuleStats::building`)
13260    /// the clone carries no pending-build state, so this reports the edges that
13261    /// rule would derive — which the live store will not derive until its
13262    /// backfill runs. Right about the end state, early about the timing.
13263    ///
13264    /// Returns `Err(KeyNotFound)` for an unknown or tombstoned key and
13265    /// `Err(ViewPropReadOnly)` for a field a view owns — matching
13266    /// [`set_prop`](GraphDb::set_prop)'s validation. A change with no effect
13267    /// (the node already holds `value`, or no rule watches `field`) returns
13268    /// empty lists.
13269    ///
13270    /// ## Cost
13271    ///
13272    /// One clone of the property columns, the topology overlay, the symbol
13273    /// interner, the edge properties and the provenance map, plus one candidate
13274    /// re-index (O(nodes × rules)). That is much cheaper than copying the store
13275    /// directory, but it is not free — this is an interactive "what if", not a
13276    /// hot path.
13277    pub fn what_if_set_prop(&self, key: &str, field: &str, value: Value) -> Result<WhatIf> {
13278        // The engine's provenance, HNSW and IVF state live in the mmap'd base
13279        // until something asks for them. On a store opened cold from a snapshot
13280        // this is the first ask, and without it the clone below starts from an
13281        // empty provenance map: nothing to retract, so `lost` comes back empty.
13282        self.ensure_v8_base_sections_loaded();
13283
13284        let empty = WhatIf {
13285            lost: Vec::new(),
13286            gained: Vec::new(),
13287        };
13288
13289        if let Some(view_name) = self.view_store.view_for_prop(field) {
13290            return Err(GraphError::ViewPropReadOnly {
13291                view_name: view_name.to_string(),
13292            });
13293        }
13294        MutPreview::new(self).check_live_key(key)?;
13295        let id = self
13296            .ids
13297            .get(key)
13298            .ok_or_else(|| GraphError::KeyNotFound { key: key.into() })?;
13299
13300        let rules: Vec<RuleDef> = self.engine.rules().cloned().collect();
13301        if rules.is_empty() {
13302            return Ok(empty);
13303        }
13304
13305        // No rule watches this field → no derivation can change.
13306        if !rules.iter().any(|r| r.watched_fields().contains(field)) {
13307            return Ok(empty);
13308        }
13309
13310        let old_value = build_props_view(&self.props, &self.base)
13311            .get(id, field)
13312            .map(|vr| vr.into_value());
13313        if old_value.as_ref() == Some(&value) {
13314            return Ok(empty);
13315        }
13316
13317        // --- Clone every piece of state the re-derivation writes to. ---
13318        // `what_if` mutates its own throwaway copies and writes nothing, so it
13319        // takes owned clones rather than sharing the `Arc`s. Deref-clone, not
13320        // `Arc::clone`: sharing here would make `make_mut` copy on first touch
13321        // anyway, and an owned local keeps the rest of this function unchanged.
13322        let mut props = (*self.props).clone();
13323        let mut topo = (*self.topo).clone();
13324        let mut syms = (*self.syms).clone();
13325        let mut edge_props = (*self.edge_props).clone();
13326
13327        let mut tripped: BTreeMap<String, bool> = BTreeMap::new();
13328        let mut fires: BTreeMap<String, u64> = BTreeMap::new();
13329        for r in &rules {
13330            tripped.insert(r.name.clone(), self.engine.is_tripped(&r.name));
13331            fires.insert(r.name.clone(), self.engine.fire_count(&r.name));
13332        }
13333        // `provenance()` decodes retained snapshot bytes on first use; the
13334        // engine clone needs the real map, not an empty one.
13335        let provenance = self.engine.provenance().clone();
13336        let mut engine = core_rules::RuleEngine::from_persist(rules, provenance, tripped, fires);
13337
13338        // Build the candidate indexes from the state BEFORE the change, exactly
13339        // as apply() sees them: `on_node_changed` withdraws the node under its
13340        // old value and refiles it under the new one, so the index must not
13341        // already reflect the change.
13342        engine.reindex_all_load_state(
13343            &self.ids,
13344            &syms,
13345            &self.labels,
13346            build_props_view(&self.props, &self.base),
13347            self.engine.export_ivf_state(),
13348            self.engine.export_hnsw_state_passthrough(),
13349        );
13350        engine.set_emit_deltas(true);
13351
13352        // --- Apply the hypothetical change and re-derive. ---
13353        props.set(id, field, value);
13354        {
13355            let mut gm = make_graph_mut(
13356                &self.ids,
13357                &mut syms,
13358                &self.labels,
13359                build_props_view(&props, &self.base),
13360                &mut topo,
13361                &self.base,
13362                &mut edge_props,
13363            );
13364            engine.on_node_changed(id, Some((field, old_value)), &mut gm);
13365        }
13366
13367        let mut lost: BTreeSet<EdgeAt> = BTreeSet::new();
13368        let mut gained: BTreeSet<EdgeAt> = BTreeSet::new();
13369        for d in engine.drain_deltas() {
13370            let edge = EdgeAt {
13371                edge_type: d.edge_type,
13372                src_key: d.src_key,
13373                dst_key: d.dst_key,
13374                derived: true,
13375                rule: Some(d.rule),
13376            };
13377            if d.fired {
13378                gained.insert(edge);
13379            } else {
13380                lost.insert(edge);
13381            }
13382        }
13383        // An edge retracted and re-derived within the same re-derivation (top-k
13384        // churn) is not a change the caller would see.
13385        let churn: Vec<EdgeAt> = lost.intersection(&gained).cloned().collect();
13386        for e in churn {
13387            lost.remove(&e);
13388            gained.remove(&e);
13389        }
13390
13391        Ok(WhatIf {
13392            lost: lost.into_iter().collect(),
13393            gained: gained.into_iter().collect(),
13394        })
13395    }
13396
13397    pub fn edge_count(&self) -> u64 {
13398        self.topo_view().edge_count()
13399    }
13400
13401    /// Live/tombstone/edge counts plus per-rule provenance size, trip latch,
13402    /// and fire counter (includes rebuild evaluations). Rules are sorted by name.
13403    pub fn stats(&self) -> Stats {
13404        self.ensure_v8_base_sections_loaded();
13405        let building = self.engine.builds_in_progress();
13406        let rules: Vec<RuleStats> = self
13407            .engine
13408            .rules()
13409            .map(|r| RuleStats {
13410                name: r.name.clone(),
13411                edges: self
13412                    .engine
13413                    .provenance()
13414                    .get(&r.name)
13415                    .map(|s| s.len() as u64)
13416                    .unwrap_or(0),
13417                tripped: self.engine.is_tripped(&r.name),
13418                fires: self.engine.fire_count(&r.name),
13419                approximate: r.approximate,
13420                building: building.iter().find(|b| b.rule == r.name).cloned(),
13421            })
13422            .collect();
13423        Stats {
13424            nodes_live: self.ids.live_len(),
13425            nodes_tombstoned: self.ids.len() - self.ids.live_len(),
13426            edges: self.topo_view().edge_count(),
13427            rules,
13428            chain_truncations: self.engine.chain_truncations(),
13429            history_floor: self.wal_horizon_floor,
13430            namespaces: self.namespace_stats(),
13431        }
13432    }
13433
13434    /// On-disk size of the WAL file in bytes.
13435    ///
13436    /// Reads file metadata without loading WAL contents.  Returns `Err` for
13437    /// in-memory (`SimFs`) databases where no WAL file exists on disk.
13438    pub fn wal_size_bytes(&self) -> std::io::Result<u64> {
13439        let path = self.fs.wal_path().ok_or_else(|| {
13440            std::io::Error::new(
13441                std::io::ErrorKind::Unsupported,
13442                "wal_path not available for this Fs implementation",
13443            )
13444        })?;
13445        Ok(std::fs::metadata(path)?.len())
13446    }
13447
13448    /// Set the slow-query threshold.  Queries whose execution time equals or
13449    /// exceeds `ms` milliseconds are logged.  Pass `0` to disable.
13450    ///
13451    /// Use this setter in tests — the environment variable
13452    /// `MUSHROOMDB_SLOW_QUERY_MS` is process-global and races parallel test
13453    /// threads.
13454    pub fn set_slow_query_threshold_ms(&mut self, ms: u64) {
13455        self.slow_query_threshold_ms = ms;
13456    }
13457
13458    /// Snapshot of the slow-query ring buffer and lifetime counter.
13459    pub fn slow_query_snapshot(&self) -> SlowQuerySnapshot {
13460        let log = self.slow_queries.lock().unwrap_or_else(|e| e.into_inner());
13461        SlowQuerySnapshot {
13462            threshold_ms: self.slow_query_threshold_ms,
13463            count: log.total,
13464            last: log.entries.iter().cloned().collect(),
13465        }
13466    }
13467
13468    /// Instant the database was opened.  Used by consumers (e.g. `/metrics`)
13469    /// to compute uptime.
13470    pub fn started_at(&self) -> std::time::Instant {
13471        self.started_at
13472    }
13473
13474    /// The on-disk snapshot version a store that has opted in to nothing
13475    /// writes — the **floor**, not the whole answer.
13476    ///
13477    /// It is not "the version this binary writes", and it is not "the version
13478    /// this binary reads". Since v0.6.10 this binary writes 9 **or** 10
13479    /// depending on the store — [`snapshot::version_for`] decides, and a store
13480    /// that has called [`enable_multiplicity`](Self::enable_multiplicity)
13481    /// writes 10 — and it reads 5 through 10. A caller comparing a store's
13482    /// stamp against this value must use `>=`, not `==`, or it will report an
13483    /// opted-in store as needing a migration *down*; `cli::run_migrate` is the
13484    /// worked example.
13485    ///
13486    /// The name is kept for compatibility: it is public API reachable from the
13487    /// CLI and from any embedder, and respelling it would break them for a
13488    /// doc-level clarification.
13489    ///
13490    /// [`snapshot::version_for`]: core_storage::snapshot::version_for
13491    pub fn format_version() -> u16 {
13492        core_storage::snapshot::VERSION
13493    }
13494
13495    /// Test-support: total bytes appended (SimFs only usage).
13496    pub fn fs_total_appended(&self) -> usize
13497    where
13498        F: FsIntrospect,
13499    {
13500        self.fs.total_appended()
13501    }
13502
13503    /// Test-support: successful `Fs::sync` calls (SimFs / counting fs).
13504    pub fn fs_sync_count(&self) -> usize
13505    where
13506        F: FsIntrospect,
13507    {
13508        self.fs.sync_count()
13509    }
13510
13511    /// Consume the db, returning its fs (for crash simulation).
13512    pub fn into_fs(self) -> F {
13513        self.fs
13514    }
13515
13516    pub fn snapshot(&mut self) -> Result<()> {
13517        self.snapshot_with(SnapshotOptions::default())
13518    }
13519
13520    /// Snapshot with explicit options.
13521    ///
13522    /// # `keep_wal`
13523    ///
13524    /// When `keep_wal` is `false` (the default, same as [`snapshot`]):
13525    ///   - The WAL is replaced with a minimal baseline containing one
13526    ///     `EnableFulltext` record per active declaration.  All pre-snapshot
13527    ///     history is discarded; `open_at` can only reach post-snapshot commits.
13528    ///
13529    /// When `keep_wal` is `true`:
13530    ///   - The WAL is left intact.  All pre-snapshot commits remain reachable
13531    ///     via `open_at`.  The existing WAL already contains the original
13532    ///     `EnableFulltext` records, so no baseline re-write is needed; the
13533    ///     recovery guards in `apply()` silently skip any duplicate records on
13534    ///     replay.
13535    ///   - Crash window: a crash after the snapshot write but before the next
13536    ///     WAL write leaves the full pre-snapshot WAL intact.  On reopen the
13537    ///     snapshot is loaded and the WAL replayed idempotently over it — safe
13538    ///     because every `apply()` arm is idempotent when replayed over an
13539    ///     already-current snapshot.
13540    pub fn snapshot_with(&mut self, opts: SnapshotOptions) -> Result<()> {
13541        if self.read_only {
13542            return Err(GraphError::ReadOnly);
13543        }
13544        // A snapshot rewrites `wal.bin` through a tmp+rename, so a peer that is
13545        // appending ends up holding a descriptor on an unlinked inode and loses
13546        // commits it believes durable. Snapshotting therefore requires the
13547        // cross-process write lock, exactly as appending does. Unlike the WAL
13548        // append path this does not go through `log_then_apply_with`, so both
13549        // guards are repeated here.
13550        if self.degraded {
13551            return Err(GraphError::Io(std::io::Error::other(
13552                "database degraded after group-commit fsync failure; reopen required",
13553            )));
13554        }
13555        if self.lock_denied {
13556            return Err(GraphError::Busy { holder: None });
13557        }
13558        // Capture whether snapshot.bin already existed BEFORE this snapshot write.
13559        // Used by the archive path's conservative genesis-chain check: if a prior
13560        // snapshot exists but wal.truncated does not, we cannot distinguish a
13561        // legacy store (may have been truncated in an older code version) from a
13562        // new store that only used keep_wal=true.  Conservative: refuse genesis in
13563        // both cases.  Must be sampled here, before the snapshot write below.
13564        //
13565        // `snapshot_preserved_history` is the one case where the answer is not a
13566        // guess: a snapshot *this handle* took, on a store that had none when it
13567        // opened, and that kept the WAL. The proxy defers to it, because
13568        // otherwise `enable_multiplicity` — whose forced snapshot is exactly
13569        // that — would permanently disqualify the store from a genesis chain it
13570        // is fully entitled to (defect #23).
13571        let had_prior_snapshot = self.fs.snapshot_path().map(|p| p.exists()).unwrap_or(false)
13572            && !self.snapshot_preserved_history;
13573        // Which version this store writes. V9 unless it has opted in to
13574        // multiplicity, in which case V10 — the stamp that makes a reader which
13575        // does not know WAL discriminant 23 refuse the open instead of
13576        // truncating the WAL at the first such frame. The container is
13577        // identical either way; only these two header bytes move.
13578        let snapshot_version = core_storage::snapshot::version_for(self.multiplicity);
13579        self.ensure_v8_base_sections_loaded();
13580        // Ensure provenance is decoded before to_persist() clones it.
13581        self.engine.ensure_provenance_loaded_mut();
13582        let (rule_defs_typed, provenance, rule_tripped, rule_fires) = self.engine.to_persist();
13583        let rule_defs = rule_defs_typed
13584            .iter()
13585            .map(|r| bincode::serialize(r).expect("RuleDef serialize cannot fail"))
13586            .collect();
13587        // Collect HNSW state and IVF state.  When indexes are not yet
13588        // populated (clean open, no mutation since open), pass the retained
13589        // raw bytes through directly so that migrate/snapshot does not
13590        // silently discard fitted approximate-rule indexes.
13591        let hnsw_state = self.engine.export_hnsw_state_passthrough();
13592        let ivf_bytes = if !self.engine.indexes_populated() {
13593            // Pass retained IVF bytes through unchanged (no re-encode).
13594            self.engine.retained_ivf_bytes_clone().unwrap_or_default()
13595        } else {
13596            // Indexes live: encode from current state.
13597            let raw_ivf = self.engine.export_ivf_state();
13598            let ivf_state_map: BTreeMap<String, core_storage::snapshot::PerRuleIvfState> = raw_ivf
13599                .into_iter()
13600                .map(|(name, ((sc, sa, sd), (dc, da, dd)))| {
13601                    (
13602                        name,
13603                        core_storage::snapshot::PerRuleIvfState {
13604                            src: core_storage::snapshot::SideIvfState {
13605                                centroids: sc,
13606                                clusters: sa,
13607                                drift: sd,
13608                            },
13609                            dst: core_storage::snapshot::SideIvfState {
13610                                centroids: dc,
13611                                clusters: da,
13612                                drift: dd,
13613                            },
13614                        },
13615                    )
13616                })
13617                .collect();
13618            if ivf_state_map.is_empty() {
13619                Vec::new()
13620            } else {
13621                bincode::serialize(&ivf_state_map).expect("IVF state serialize cannot fail")
13622            }
13623        };
13624        let view_defs: Vec<Vec<u8>> = self
13625            .view_store
13626            .views()
13627            .map(|v| bincode::serialize(v).expect("ViewDef serialize cannot fail"))
13628            .collect();
13629        if self.base.is_some() {
13630            // V8 merge-snapshot path: encode base+overlay into a new V8 snapshot,
13631            // write it atomically, remap it as the new base, then clear the overlay.
13632            let meta = V8Meta {
13633                labels: (*self.labels).clone(),
13634                edge_props: (*self.edge_props).clone(),
13635                rule_defs,
13636                provenance,
13637                rule_tripped,
13638                rule_fires,
13639                ivf_bytes,
13640                view_defs,
13641                wal_truncated: !opts.keep_wal,
13642                hnsw: hnsw_state,
13643                last_change: self.last_change.clone(),
13644            };
13645            let mut buf: Vec<u8> = Vec::new();
13646            {
13647                // Clone the Arc so the old base stays alive while we encode.
13648                // The borrow of archived_csr (into old_base's mmap) is released
13649                // at the end of this block, before we replace self.base.
13650                let old_base = self.base.clone().expect("is_some checked above");
13651                let archived_csr = old_base.topology().map_err(|e| GraphError::Corrupt {
13652                    detail: format!("v8 snapshot: topology section: {e:?}"),
13653                })?;
13654                let archived_cols = old_base.columns().map_err(|e| GraphError::Corrupt {
13655                    detail: format!("v8 snapshot: columns section: {e:?}"),
13656                })?;
13657                // `None` when the base predates V9 — the migration path: its
13658                // string columns still carry their own tables and this snapshot
13659                // is the rewrite that collapses them into section 12.
13660                let archived_strings =
13661                    old_base
13662                        .string_table()
13663                        .transpose()
13664                        .map_err(|e| GraphError::Corrupt {
13665                            detail: format!("v8 snapshot: strings section: {e:?}"),
13666                        })?;
13667                let archived_edge_props =
13668                    old_base
13669                        .edge_props_section()
13670                        .map_err(|e| GraphError::Corrupt {
13671                            detail: format!("v8 snapshot: edge_props section: {e:?}"),
13672                        })?;
13673                let edge_props_raw =
13674                    old_base
13675                        .edge_props_raw_bytes()
13676                        .map_err(|e| GraphError::Corrupt {
13677                            detail: format!("v8 snapshot: edge_props raw bytes: {e:?}"),
13678                        })?;
13679                let prov_raw =
13680                    old_base
13681                        .provenance_raw_bytes()
13682                        .map_err(|e| GraphError::Corrupt {
13683                            detail: format!("v8 snapshot: provenance raw bytes: {e:?}"),
13684                        })?;
13685                encode_v8(
13686                    Some(archived_csr),
13687                    Some(archived_cols),
13688                    archived_strings,
13689                    Some((archived_edge_props, edge_props_raw)),
13690                    Some(prov_raw),
13691                    &self.topo,
13692                    &self.props,
13693                    &self.ids,
13694                    &self.syms,
13695                    &meta,
13696                    &mut buf,
13697                )?;
13698            }
13699            core_storage::snapshot::stamp_container_version(&mut buf, snapshot_version)?;
13700            self.fs.write_atomic(FileId::Snapshot, &buf)?;
13701            // Remap the freshly-written snapshot as the new base.
13702            // C2: use file mmap on RealFs; fall back to from_bytes on SimFs.
13703            let new_base = if let Some(snap_path) = self.fs.snapshot_path() {
13704                core_storage::v8::MappedBase::map(&snap_path)
13705            } else {
13706                core_storage::v8::MappedBase::from_bytes(buf)
13707            }
13708            .map_err(|e| GraphError::Corrupt {
13709                detail: format!("v8 snapshot: remap new base: {e:?}"),
13710            })?;
13711            self.base = Some(Arc::new(new_base));
13712            // Clear the overlay and prop tombstones — all data is now in the new base.
13713            self.topo = Arc::new(Topology::new());
13714            self.props = Arc::new(core_storage::columns::ColumnStore::new());
13715        } else {
13716            // Legacy path (V5–V7 stores without a V8 base).
13717            //
13718            // Memory-diet path: build V8Meta directly from &self — no SnapshotState
13719            // clone and no encode_v8_from_state intermediate clones.  The big
13720            // structures (self.topo, self.props) are borrowed, not cloned.
13721            // self.edge_props is moved (not cloned) because we immediately clear it
13722            // when we remap the new V8 snapshot as self.base (see below).
13723            //
13724            // Eliminates from peak RSS vs. the old SnapshotState path:
13725            //   • self.topo.clone()      (~topology HashMap footprint)
13726            //   • self.props.clone()     (~column-store footprint)
13727            //   • encode_v8_from_state V8Meta secondary clones (labels, edge_props, …)
13728            let meta = V8Meta {
13729                labels: (*self.labels).clone(),
13730                wal_truncated: !opts.keep_wal,
13731                // Move edge_props out so the large overlay is freed when meta
13732                // drops at end of this block (self.edge_props is now empty; reads
13733                // after base assignment go through the mmap'd base section).
13734                edge_props: std::mem::take(Arc::make_mut(&mut self.edge_props)),
13735                rule_defs,
13736                provenance,
13737                rule_tripped,
13738                rule_fires,
13739                ivf_bytes,
13740                view_defs,
13741                hnsw: hnsw_state,
13742                last_change: self.last_change.clone(),
13743            };
13744            let mut buf = Vec::new();
13745            encode_v8(
13746                None,
13747                None,
13748                None,
13749                None,
13750                None,
13751                &self.topo,
13752                &self.props,
13753                &self.ids,
13754                &self.syms,
13755                &meta,
13756                &mut buf,
13757            )?;
13758            // meta (and the moved edge_props inside it) is no longer needed;
13759            // drop it before the write to keep the peak window narrow.
13760            drop(meta);
13761            core_storage::snapshot::stamp_container_version(&mut buf, snapshot_version)?;
13762            self.fs.write_atomic(FileId::Snapshot, &buf)?;
13763            // Remap the freshly-written V8 snapshot as self.base.
13764            // On RealFs: drop the encode buffer before mmap to recover ~1.9 GiB.
13765            // On SimFs (tests): pass buf to from_bytes.
13766            let new_base = if let Some(snap_path) = self.fs.snapshot_path() {
13767                drop(buf);
13768                core_storage::v8::MappedBase::map(&snap_path)
13769            } else {
13770                core_storage::v8::MappedBase::from_bytes(buf)
13771            }
13772            .map_err(|e| GraphError::Corrupt {
13773                detail: format!("v8 snapshot: remap new base (legacy path): {e:?}"),
13774            })?;
13775            self.base = Some(Arc::new(new_base));
13776            // Free the large heap-allocated decoded state — all data is now in the
13777            // mmap'd base.  Mirrors the V8 merge-snapshot path (see above).
13778            // self.edge_props was already moved into meta and is effectively empty.
13779            self.topo = Arc::new(Topology::new());
13780            self.props = Arc::new(core_storage::columns::ColumnStore::new());
13781        }
13782
13783        if opts.archive_wal {
13784            // History-preserving snapshot (Task 4):
13785            //   1. Snapshot already written above (write_atomic → fsynced).
13786            //   2. Rename WAL → wal.<commit_seq>.archive  (atomic, same fs).
13787            //      Crash window B: crash here leaves archive present, WAL
13788            //      absent.  Reopen: snapshot loaded (full state), no WAL
13789            //      replay.  Archive is NOT replayed into live state — it is
13790            //      pre-snapshot by construction.  Safe.
13791            //   3. Optionally write genesis marker (first archive only, no
13792            //      prior WAL truncation).
13793            //   4. Prune old archives (retention), update horizon floor.
13794            //      Pruning invalidates the genesis chain; delete marker.
13795            //   5. Write new minimal baseline WAL (write_atomic).
13796            //      Crash window C: crash here leaves new archive plus no live
13797            //      WAL.  Same as window B — handled above.
13798            //
13799            // Sample existing archives BEFORE the rename so we can detect
13800            // whether this is the first archive.
13801            let existing_archives = self.fs.list_archives()?;
13802            let is_first_archive = existing_archives.is_empty();
13803
13804            // Compute a globally-monotonic archive name: the name equals the
13805            // cumulative end-frame index of the archive in global commit space.
13806            //
13807            // Using `commit_seq` directly is UNSOUND across sessions: on reopen
13808            // commit_seq is seeded from max(last_change), which underestimates
13809            // the WAL depth when trailing commits (e.g. insert_edge) do not
13810            // update last_change.  A session-2 archive could then receive a name
13811            // ≤ the session-1 archive, causing incorrect sort order or collision.
13812            //
13813            // Instead: read and decode the live WAL here (before the rename) to
13814            // get its exact frame count, then add it to the last known global
13815            // end-frame index (the name of the most recent existing archive, or
13816            // wal_horizon_floor if no archives exist).  This is O(WAL size) but
13817            // snapshot is already serialising the full graph state, so the cost
13818            // is dominated.
13819            let live_wal_bytes_for_name = self.fs.read(FileId::Wal)?;
13820            let (live_frames_for_name, _) = decode_all(&live_wal_bytes_for_name);
13821            let archive_n = existing_archives
13822                .last()
13823                .copied()
13824                .unwrap_or(self.wal_horizon_floor)
13825                + live_frames_for_name.len() as u64;
13826            self.fs.archive_wal(archive_n)?;
13827
13828            // The replacement WAL goes in **immediately**, with no fallible call
13829            // between it and the rename above.
13830            //
13831            // The rename is what removes the store's live declarations — the
13832            // multiplicity opt-in, and every `EnableFulltext` / `EnableIndex` —
13833            // and this write is what puts them back. Every call that used to sit
13834            // in between (the genesis marker, the retention sweep's reads, the
13835            // floor write, the archive deletes) was a `?` that could leave the
13836            // store with neither, so a single transient `Err` was enough to lose
13837            // a declaration that no rebuild can recover (defect #22).
13838            //
13839            // Ordering alone cannot close the crash window between two
13840            // filesystem calls; for the multiplicity declaration the V10 stamp
13841            // does that on the open path. What ordering does close is the much
13842            // wider window in which an ordinary I/O error did it — and that half
13843            // covers all three declarations, not just the one with a stamp.
13844            let mut baseline_wal: Vec<u8> = Vec::new();
13845            // The multiplicity opt-in is a declaration like the two below it,
13846            // and it is re-emitted for the same reason: truncation must not
13847            // silently opt the store back out and stop counting.
13848            if self.multiplicity {
13849                baseline_wal
13850                    .extend_from_slice(&encode_record(&core_storage::wal::MULTIPLICITY_ENABLED));
13851            }
13852            for (label, field) in self.fulltext.enabled_pairs() {
13853                let rec = WalRecord::EnableFulltext {
13854                    label: label.clone(),
13855                    field: field.clone(),
13856                };
13857                baseline_wal.extend_from_slice(&encode_record(&rec));
13858            }
13859            for (label, field) in self.prop_index.enabled_pairs() {
13860                let rec = WalRecord::EnableIndex {
13861                    label: label.clone(),
13862                    field: field.clone(),
13863                };
13864                baseline_wal.extend_from_slice(&encode_record(&rec));
13865            }
13866            self.fs.write_atomic(FileId::Wal, &baseline_wal)?;
13867
13868            // Genesis marker: written once when the first archive is taken
13869            // from a store that has never undergone a WAL-truncating snapshot.
13870            // When present, `open_at` may replay archive-resident commits from
13871            // empty state (the archive chain covers from global index 0).
13872            //
13873            // Two conditions must ALL hold:
13874            //   1. This is the first archive (existing_archives was empty).
13875            //   2. No snapshot.bin existed before this operation (had_prior_snapshot=false).
13876            //      A WAL-truncating snapshot (keep_wal=false) always writes snapshot.bin
13877            //      before truncating the WAL, so if any prior truncating snapshot was taken
13878            //      — even in a previous session — snapshot.bin is present and this condition
13879            //      is false.  This subsumes the cross-session truncation case without
13880            //      requiring a separate wal.truncated sidecar file.
13881            //      For legacy stores (snapshot.bin written by an older code version that
13882            //      may have truncated the WAL), the same conservative refusal applies:
13883            //      we cannot prove the chain is complete, so we refuse genesis (cost =
13884            //      no as-of-through-archives; never silent wrong data).
13885            //      The one exception is a snapshot this handle took itself, on a store
13886            //      that had none when it opened, with the WAL kept: there the answer is
13887            //      known rather than guessed, and `snapshot_preserved_history` says so.
13888            //      Without that exception `enable_multiplicity`'s forced keep_wal
13889            //      snapshot would disqualify the store forever (defect #23).
13890            //      On SimFs (snapshot_path() == None) had_prior_snapshot is always false,
13891            //      so SimFs always passes this check.
13892            if is_first_archive && !had_prior_snapshot {
13893                self.fs.write_genesis_marker()?;
13894                self.archive_genesis_chain = true;
13895            }
13896
13897            // Retention pruning: keep newest `keep` archives; delete oldest.
13898            // Pruning is the ONLY deletion site for archives.
13899            //
13900            // Crash-safety ordering (C1 fix):
13901            //   1. Count frames in surplus archives (reads only — no mutation).
13902            //   2. Advance and PERSIST the horizon floor FIRST via write-then-
13903            //      rename (atomic).  A crash after this point leaves orphaned
13904            //      archives on disk, but the floor is correct.  The opening
13905            //      cleanup sweep (`cleanup_orphaned_archives`) removes them on
13906            //      the next open, so the store is always safe to reopen.
13907            //   3. Delete the genesis marker (floor > 0 already blocks open_at
13908            //      via the conjunctive gate; marker cleanup is belt-and-suspenders).
13909            //   4. Delete surplus archives.  A crash between any two deletes
13910            //      leaves the floor committed and orphaned archives cleaned at
13911            //      next open — never a stale floor with a missing archive prefix.
13912            if let Some(keep) = self.wal_archive_retention {
13913                if keep > 0 {
13914                    let archives = self.fs.list_archives()?;
13915                    // archives is sorted ascending (oldest first)
13916                    if archives.len() as u32 > keep {
13917                        let surplus = archives.len() - keep as usize;
13918                        // Step 1: count pruned frames (reads, no mutation).
13919                        let mut pruned_frames = 0u64;
13920                        for &n in &archives[..surplus] {
13921                            let bytes = self.fs.read_archive(n)?;
13922                            let (frames, _) = decode_all(&bytes);
13923                            pruned_frames += frames.len() as u64;
13924                        }
13925                        // Step 2: advance and persist floor FIRST.
13926                        self.wal_horizon_floor += pruned_frames;
13927                        self.fs.write_horizon_floor(self.wal_horizon_floor)?;
13928                        // The time map must not outlive the commits it
13929                        // describes: an entry below the new floor would resolve
13930                        // a date to a commit the engine can no longer replay,
13931                        // which is worse than having no entry at all.
13932                        self.commit_times.truncate_below(self.wal_horizon_floor);
13933                        self.rewrite_commit_times();
13934                        // Step 3: delete genesis marker (floor > 0 already
13935                        // blocks open_at; this is belt-and-suspenders cleanup).
13936                        if pruned_frames > 0 && self.archive_genesis_chain {
13937                            self.fs.delete_genesis_marker()?;
13938                            self.archive_genesis_chain = false;
13939                        }
13940                        // Step 4: delete surplus archives.  Crash here →
13941                        // orphaned archives; cleaned at next open.
13942                        for &n in &archives[..surplus] {
13943                            self.fs.delete_archive(n)?;
13944                        }
13945                    }
13946                }
13947            }
13948        } else if opts.keep_wal {
13949            // keep_wal=true: WAL is left untouched.  The existing WAL already
13950            // contains the EnableFulltext records from the original enable calls;
13951            // replay is idempotent (guards in apply() skip already-live entries).
13952            // No baseline re-write is needed or safe here — the full WAL history
13953            // must remain intact for open_at to reach pre-snapshot commits.
13954        } else {
13955            // keep_wal=false (default): truncate by replacing the WAL with a
13956            // minimal baseline of one EnableFulltext record per active pair.
13957            //
13958            // Crash-ordering: write_atomic is atomic.
13959            //   • Crash before snapshot write  → WAL unchanged.  Safe.
13960            //   • Crash after snapshot write but before this WAL write → full
13961            //     pre-snapshot WAL still present; open_with replays idempotently.
13962            //   • Crash after both writes → normal post-snapshot state.
13963            //
13964            // Genesis chain: a WAL-truncating snapshot breaks the archive chain
13965            // for any archives taken AFTER this point (their WAL slices would
13966            // not start at genesis).  Delete any existing genesis marker so that
13967            // open_at refuses archive-resident commits.  Future sessions are
13968            // covered by had_prior_snapshot: snapshot.bin written here persists
13969            // across sessions and prevents a later archiving session from
13970            // incorrectly claiming a complete genesis chain.
13971            if self.archive_genesis_chain {
13972                self.fs.delete_genesis_marker()?;
13973                self.archive_genesis_chain = false;
13974            }
13975            // And this handle can no longer prove the WAL is whole: it is about
13976            // to truncate it itself. Same-session archives after this point get
13977            // the conservative answer, exactly as cross-session ones do.
13978            self.snapshot_preserved_history = false;
13979            let mut baseline_wal: Vec<u8> = Vec::new();
13980            // The multiplicity opt-in is a declaration like the two below it,
13981            // and it is re-emitted for the same reason: truncation must not
13982            // silently opt the store back out and stop counting.
13983            if self.multiplicity {
13984                baseline_wal
13985                    .extend_from_slice(&encode_record(&core_storage::wal::MULTIPLICITY_ENABLED));
13986            }
13987            for (label, field) in self.fulltext.enabled_pairs() {
13988                let rec = WalRecord::EnableFulltext {
13989                    label: label.clone(),
13990                    field: field.clone(),
13991                };
13992                baseline_wal.extend_from_slice(&encode_record(&rec));
13993            }
13994            for (label, field) in self.prop_index.enabled_pairs() {
13995                let rec = WalRecord::EnableIndex {
13996                    label: label.clone(),
13997                    field: field.clone(),
13998                };
13999                baseline_wal.extend_from_slice(&encode_record(&rec));
14000            }
14001            self.fs.write_atomic(FileId::Wal, &baseline_wal)?;
14002            // This branch discards history rather than archiving it: the frames
14003            // the map describes are gone and the replacement WAL renumbers from
14004            // the floor, so every surviving entry now names a different frame.
14005            // Keeping them would resolve a date onto an unrelated commit — and
14006            // the stamps written after this point would sit far above the
14007            // store's own frame count, which is how the date surface came to
14008            // refuse commits the index path served perfectly well.
14009            //
14010            // A store that cannot answer a date says so by name
14011            // (`NoRecordedTime`). That is the honest state after discarding the
14012            // history the dates addressed.
14013            self.commit_times = core_storage::commit_times::CommitTimes::default();
14014            self.rewrite_commit_times();
14015        }
14016        // After snapshot the overlay may have changed (V8 merge path clears
14017        // self.topo and self.props). Refresh the MVCC fold so future readers
14018        // see the post-snapshot state rather than stale overlay data.
14019        self.fold_now();
14020        // We wrote the snapshot and (unless keep_wal) replaced the WAL, so both
14021        // markers this handle uses to detect other processes' work must be
14022        // re-taken from disk. Skipping this would make our own snapshot look
14023        // like a peer's on the next staleness check and force a needless
14024        // reload.
14025        self.wal_consumed = self.fs.wal_len().map_err(GraphError::Io)?;
14026        // A snapshot can replace the live WAL with a baseline, which renumbers
14027        // every frame after it. Re-derive the frame cursor from what the store
14028        // now actually holds rather than carrying the pre-snapshot count
14029        // forward — `wal_total_commits` is the same sequence the history
14030        // surfaces index, and the snapshot has already paid a far larger cost
14031        // than one decode.
14032        self.wal_frames_written = self.wal_total_commits()?;
14033        self.snapshot_ident = self.fs.snapshot_ident().map_err(GraphError::Io)?;
14034        Ok(())
14035    }
14036}
14037
14038/// What a batch node insert does when its key is already taken.
14039///
14040/// A mirror rebuild writes a frame onto a store that already has content, so
14041/// "the key exists" is a routine answer rather than a failure. The decision is
14042/// made during the batch's existing validate pass, from one id-map lookup per
14043/// row, so the frame stays atomic and re-ingest stays O(n).
14044#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)]
14045pub enum OnConflict {
14046    /// Refuse the whole frame with [`GraphError::DuplicateKey`]. The default,
14047    /// and the only behaviour before v0.6.10.
14048    #[default]
14049    Error,
14050    /// Leave the stored node exactly as it is — properties, label and edges —
14051    /// and count it in [`BatchOutcome::skipped`].
14052    Skip,
14053    /// Keep the key and make the node's properties **exactly** the supplied
14054    /// props: supplied fields are set, fields absent from the supplied props
14055    /// are removed. A supplied label that differs from the stored one, and a
14056    /// supplied `ns` that would move the node, are row errors — relabelling is
14057    /// [`GraphDb::rename_node`], not a side effect of a rebuild.
14058    ///
14059    /// Two properties are outside "exactly", both because they are not the
14060    /// caller's to supply:
14061    ///
14062    /// - `ns` is immutable, so an omitted `ns` leaves the node where it is
14063    ///   rather than moving it to `default`;
14064    /// - a property a **view** owns is kept, not removed. Supplying one is a
14065    ///   row error, so omitting it cannot be a request to delete it, and
14066    ///   refusing the row instead would make `Replace` impossible for the whole
14067    ///   population a view has written to. Each such field kept is counted in
14068    ///   [`BatchOutcome::kept_view_owned`] — the row is still `replaced` and
14069    ///   still raises no row error, so that count is the only signal a caller
14070    ///   gets that the stored node carries a field its frame did not describe.
14071    Replace,
14072}
14073
14074/// What one [`OnConflict::Replace`] row resolves to.
14075///
14076/// The property writes that make the node exactly the supplied props —
14077/// `Some(value)` is a set, `None` a removal — paired with how many view-owned
14078/// fields the row kept instead of removing, which is the one way the result is
14079/// not exactly the supplied props. See [`MutPreview::plan_replace`].
14080type ReplacePlan = (Vec<(String, Option<Value>)>, usize);
14081
14082/// What one committed batch did.
14083///
14084/// [`BatchBuilder::commit`] returns the first two fields as a tuple; the rest
14085/// exist for [`OnConflict`] and are always zero / empty without it.
14086#[derive(Clone, Debug, Default, PartialEq, Eq)]
14087pub struct BatchOutcome {
14088    /// Node records actually written.
14089    pub nodes_inserted: usize,
14090    /// Edge records actually written. A duplicate edge is a silent no-op under
14091    /// every policy — adjacency is a set — and is not counted.
14092    pub edges_inserted: usize,
14093    /// Rows whose key was taken and whose policy was [`OnConflict::Skip`].
14094    pub skipped: usize,
14095    /// Rows whose key was taken and whose policy was [`OnConflict::Replace`].
14096    pub replaced: usize,
14097    /// View-owned properties an [`OnConflict::Replace`] row **kept** although
14098    /// the caller did not supply them — counted per field, so one row that
14099    /// keeps two contributes two.
14100    ///
14101    /// This is the one respect in which `Replace` does not make a node's props
14102    /// exactly the supplied ones (see [`OnConflict::Replace`]). Those rows
14103    /// still count in `replaced` and still raise no `row_errors`, because
14104    /// nothing went wrong: a view's property is not the caller's to supply or
14105    /// to remove. A mirror rebuild that needs its copy to be byte-exact reads
14106    /// this to learn that the store kept fields its frame did not describe.
14107    pub kept_view_owned: usize,
14108    /// `(row, why)` for rows an [`OnConflict::Replace`] refused. `row` counts
14109    /// node-insert ops in this batch from zero, which for a caller that queues
14110    /// its nodes in order is the index of the offending node. The rest of the
14111    /// frame still commits; the refused row changes nothing.
14112    pub row_errors: Vec<(usize, String)>,
14113}
14114
14115/// The `(nodes_inserted, edges_inserted)` pair every pre-0.6.10 commit entry
14116/// point returns. Keeps those signatures unchanged now that the validate pass
14117/// produces a [`BatchOutcome`].
14118fn inserted_pair(outcome: BatchOutcome) -> (usize, usize) {
14119    (outcome.nodes_inserted, outcome.edges_inserted)
14120}
14121
14122/// One entry of a frame the validate pass has decided on, before
14123/// [`GraphDb::rewrite_wal_dense_planned`] turns it into dense-id records.
14124///
14125/// Almost every entry is already a finished [`WalRecord`]. The exception is a
14126/// duplicate edge insert: its count names a dense triple, and on the batch path
14127/// the endpoints and the edge type may all be created by earlier records in the
14128/// *same* frame, so no id for them exists until the dense rewrite allocates it.
14129/// Carrying the keys this far and resolving them there is what lets the count
14130/// survive the shape a mirror rebuild writes (defect #24).
14131enum PlannedRec {
14132    Rec(WalRecord),
14133    DuplicateCount {
14134        edge_type: String,
14135        src_key: String,
14136        dst_key: String,
14137    },
14138}
14139
14140/// Queued mutation for a [`BatchBuilder`] or [`GraphDb::commit_group`].
14141///
14142/// The `submit_batch` / `commit_group` APIs accept `Vec<BatchOp>` so that
14143/// callers can build a set of mutations without holding `&mut GraphDb` and
14144/// hand them off to the group-committing writer for durable, batched I/O.
14145pub enum BatchOp {
14146    InsertNode {
14147        label: String,
14148        key: String,
14149        props: Vec<(String, Value)>,
14150    },
14151    InsertEdge {
14152        edge_type: String,
14153        src_key: String,
14154        dst_key: String,
14155    },
14156    SetProp {
14157        key: String,
14158        field: String,
14159        value: Value,
14160    },
14161    RemoveProp {
14162        key: String,
14163        field: String,
14164    },
14165    DeleteEdge {
14166        edge_type: String,
14167        src_key: String,
14168        dst_key: String,
14169    },
14170    DeleteNode {
14171        key: String,
14172    },
14173    CreateRule(RuleDef),
14174    DeleteRule {
14175        name: String,
14176    },
14177    /// Rename a node's key. Validated: old must exist, new must not.
14178    RenameNode {
14179        old_key: String,
14180        new_key: String,
14181    },
14182    /// Insert an edge, auto-creating any missing endpoint as a plain node with
14183    /// `placeholder_label` and no props. Rules fire and last-change is updated
14184    /// for each created endpoint (normal InsertNode semantics in the batch frame).
14185    InsertEdgeUpsert {
14186        edge_type: String,
14187        src_key: String,
14188        dst_key: String,
14189        placeholder_label: String,
14190    },
14191    /// Insert `key`, or — when the key is already taken — do what `on_conflict`
14192    /// says. Queued by [`BatchBuilder::insert_node_on_conflict`]; `Error`
14193    /// queues a plain [`BatchOp::InsertNode`] instead, so this variant only
14194    /// ever carries `Skip` or `Replace`.
14195    InsertNodeOnConflict {
14196        label: String,
14197        key: String,
14198        props: Vec<(String, Value)>,
14199        on_conflict: OnConflict,
14200    },
14201}
14202
14203/// Three-way node visibility status used by `check_single_op_authz`.
14204enum NodeAuthzStatus {
14205    /// Node exists in the store and is in the role's read mask.
14206    Visible(String), // carries the node's label
14207    /// Node exists in the store but is NOT in the role's read mask.
14208    Hidden,
14209    /// Node does not exist in the store.
14210    Absent,
14211}
14212
14213/// Overlay of ops already accepted earlier in the same batch. Never written
14214/// back to the database — validation only.
14215#[derive(Default)]
14216struct Overlay {
14217    extra_keys: BTreeSet<String>,
14218    /// Label of each node inserted earlier in this batch. The store does not
14219    /// have these keys yet, so `label_of` cannot answer for them, and
14220    /// `OnConflict::Replace` has to compare labels.
14221    extra_labels: BTreeMap<String, String>,
14222    deleted_keys: BTreeSet<String>,
14223    extra_props: BTreeMap<(String, String), Value>,
14224    removed_props: BTreeSet<(String, String)>,
14225    extra_edges: BTreeSet<(String, String, String)>,
14226    deleted_edges: BTreeSet<(String, String, String)>,
14227    extra_rules: BTreeSet<String>,
14228    deleted_rules: BTreeSet<String>,
14229    /// `rule name → (via_edge, edge_type)` for every via-hop rule accepted
14230    /// earlier in this batch. Feeds the rule-chain cycle check, which otherwise
14231    /// sees only the rules already committed to the engine. Keyed by name so a
14232    /// later `DeleteRule` in the same batch drops the arc with the rule.
14233    extra_rule_arcs: BTreeMap<String, (String, String)>,
14234}
14235
14236/// Read-only view of live db state plus a batch overlay. Shared by single-op
14237/// public methods (empty overlay) and `commit_batch`.
14238struct MutPreview<'a, F: Fs> {
14239    db: &'a GraphDb<F>,
14240    overlay: Overlay,
14241}
14242
14243/// Shortest path from `start` to `target` following `arcs` (`from → to`), or
14244/// `None` if `target` is unreachable.
14245///
14246/// Used for rule-chain cycle detection, where an arc is "a rule hops over
14247/// `from` and writes `to`". Breadth-first over BTree-ordered adjacency, so the
14248/// reported path is stable for a given rule set, and iterative so a pathological
14249/// rule graph cannot overflow the stack.
14250fn find_cycle_through(arcs: &[(String, String)], start: &str, target: &str) -> Option<Vec<String>> {
14251    let mut adj: BTreeMap<&str, BTreeSet<&str>> = BTreeMap::new();
14252    for (from, to) in arcs {
14253        adj.entry(from.as_str()).or_default().insert(to.as_str());
14254    }
14255    let mut parent: BTreeMap<&str, &str> = BTreeMap::new();
14256    let mut visited: BTreeSet<&str> = BTreeSet::new();
14257    let mut queue: std::collections::VecDeque<&str> = std::collections::VecDeque::new();
14258    visited.insert(start);
14259    queue.push_back(start);
14260    while let Some(node) = queue.pop_front() {
14261        if node == target {
14262            let mut path = vec![node.to_string()];
14263            let mut cur = node;
14264            while let Some(&p) = parent.get(cur) {
14265                path.push(p.to_string());
14266                cur = p;
14267            }
14268            path.reverse();
14269            return Some(path);
14270        }
14271        for &next in adj.get(node).into_iter().flatten() {
14272            if visited.insert(next) {
14273                parent.insert(next, node);
14274                queue.push_back(next);
14275            }
14276        }
14277    }
14278    None
14279}
14280
14281impl<'a, F: Fs> MutPreview<'a, F> {
14282    fn new(db: &'a GraphDb<F>) -> Self {
14283        Self {
14284            db,
14285            overlay: Overlay::default(),
14286        }
14287    }
14288
14289    fn has_key(&self, key: &str) -> bool {
14290        if self.overlay.extra_keys.contains(key) {
14291            return true;
14292        }
14293        if self.overlay.deleted_keys.contains(key) {
14294            return false;
14295        }
14296        self.db.ids.get(key).is_some()
14297    }
14298
14299    fn has_prop(&self, key: &str, field: &str) -> bool {
14300        if !self.has_key(key) {
14301            return false;
14302        }
14303        let k = (key.to_string(), field.to_string());
14304        if self.overlay.removed_props.contains(&k) {
14305            return false;
14306        }
14307        if self.overlay.extra_props.contains_key(&k) {
14308            return true;
14309        }
14310        // Fresh identity (first insert in this batch, or delete+reinsert):
14311        // ignore props still sitting on the soon-to-be-tombstoned slot.
14312        if self.overlay.extra_keys.contains(key) {
14313            return false;
14314        }
14315        self.db.get_prop(key, field).is_some()
14316    }
14317
14318    fn has_edge(&self, edge_type: &str, src_key: &str, dst_key: &str) -> bool {
14319        let k = (
14320            edge_type.to_string(),
14321            src_key.to_string(),
14322            dst_key.to_string(),
14323        );
14324        if self.overlay.deleted_edges.contains(&k) {
14325            return false;
14326        }
14327        if self.overlay.extra_edges.contains(&k) {
14328            return true;
14329        }
14330        // A key created in this batch (including reinsert) has no db edges.
14331        if self.overlay.extra_keys.contains(src_key) || self.overlay.extra_keys.contains(dst_key) {
14332            return false;
14333        }
14334        if self.overlay.deleted_keys.contains(src_key)
14335            || self.overlay.deleted_keys.contains(dst_key)
14336        {
14337            return false;
14338        }
14339        let Some(src) = self.db.ids.get(src_key) else {
14340            return false;
14341        };
14342        let Some(dst) = self.db.ids.get(dst_key) else {
14343            return false;
14344        };
14345        let Some(sym) = self.db.syms.get(edge_type) else {
14346            return false;
14347        };
14348        self.db
14349            .topo_view()
14350            .neighbors(sym, Direction::Out, src)
14351            .binary_search(&dst)
14352            .is_ok()
14353    }
14354
14355    fn has_rule(&self, name: &str) -> bool {
14356        if self.overlay.extra_rules.contains(name) {
14357            return true;
14358        }
14359        if self.overlay.deleted_rules.contains(name) {
14360            return false;
14361        }
14362        self.db.engine.rules().any(|r| r.name == name)
14363    }
14364
14365    fn is_rule_owned(&self, edge_type: &str, src_key: &str, dst_key: &str) -> bool {
14366        if self.overlay.extra_keys.contains(src_key) || self.overlay.extra_keys.contains(dst_key) {
14367            return false;
14368        }
14369        if self.overlay.deleted_keys.contains(src_key)
14370            || self.overlay.deleted_keys.contains(dst_key)
14371        {
14372            return false;
14373        }
14374        let Some(src) = self.db.ids.get(src_key) else {
14375            return false;
14376        };
14377        let Some(dst) = self.db.ids.get(dst_key) else {
14378            return false;
14379        };
14380        let Some(et) = self.db.syms.get(edge_type) else {
14381            return false;
14382        };
14383        // extra_rules is deliberately not consulted: a CreateRule earlier in
14384        // this batch has not fired, so it contributes no provenance. That is
14385        // the documented rule-window gap (see GraphDb::batch).
14386        if self.overlay.deleted_rules.is_empty() {
14387            return self.db.engine.is_owned(et, src, dst);
14388        }
14389        for (rule, triples) in self.db.engine.provenance() {
14390            if self.overlay.deleted_rules.contains(rule) {
14391                continue;
14392            }
14393            if triples.contains(&(et, src, dst)) {
14394                return true;
14395            }
14396        }
14397        false
14398    }
14399
14400    /// The refusals a node creation makes, in the order it makes them.
14401    ///
14402    /// A view owns its property, and creating a node that carries one is a
14403    /// write to it exactly as `set_prop` is — so it is refused here, at the one
14404    /// choke-point `GraphDb::insert_node`, `BatchOp::InsertNode` and the
14405    /// no-conflict arm of `BatchOp::InsertNodeOnConflict` all pass through.
14406    ///
14407    /// Leaving creation exempt was not harmless. The value was stored and
14408    /// served: a created node the view has no reason to revisit keeps the
14409    /// caller's number for the life of the handle, and the backfill at the next
14410    /// open overwrites it — so the store answered `deg = 777` before a restart
14411    /// and `deg = 0` after, for a property every other surface calls read-only.
14412    /// It also split one op two ways: supplying a view-owned field under
14413    /// `OnConflict::Replace` was already a row error on a taken key while the
14414    /// same field on a fresh key was accepted.
14415    ///
14416    /// Checked before the key, like [`MutPreview::prepare_remove_prop`], so the
14417    /// answer does not depend on whether the key exists.
14418    fn check_insert_node(&self, key: &str, props: &[(String, Value)]) -> Result<()> {
14419        for (field, _) in props {
14420            if let Some(view_name) = self.db.view_store.view_for_prop(field) {
14421                return Err(GraphError::ViewPropReadOnly {
14422                    view_name: view_name.to_string(),
14423                });
14424            }
14425        }
14426        if self.has_key(key) {
14427            Err(GraphError::DuplicateKey { key: key.into() })
14428        } else {
14429            Ok(())
14430        }
14431    }
14432
14433    fn check_live_key(&self, key: &str) -> Result<()> {
14434        if self.has_key(key) {
14435            Ok(())
14436        } else {
14437            Err(GraphError::KeyNotFound { key: key.into() })
14438        }
14439    }
14440
14441    fn prepare_insert_edge(&self, edge_type: &str, src_key: &str, dst_key: &str) -> Result<bool> {
14442        for k in [src_key, dst_key] {
14443            if !self.has_key(k) {
14444                return Err(GraphError::KeyNotFound { key: k.into() });
14445            }
14446        }
14447        if self.is_rule_owned(edge_type, src_key, dst_key) {
14448            return Err(GraphError::RuleOwned {
14449                detail: format!("edge {edge_type} {src_key}→{dst_key} is rule-owned"),
14450            });
14451        }
14452        // A user-written edge stays inside one namespace. Derived edges do not
14453        // come through here — the engine adds them directly — and the rule
14454        // scoping check is what keeps those pure.
14455        let src_ns = self.namespace_in_batch(src_key);
14456        let dst_ns = self.namespace_in_batch(dst_key);
14457        if src_ns != dst_ns {
14458            return Err(GraphError::CrossNamespace {
14459                src: src_key.to_string(),
14460                src_ns,
14461                dst: dst_key.to_string(),
14462                dst_ns,
14463            });
14464        }
14465        Ok(!self.has_edge(edge_type, src_key, dst_key))
14466    }
14467
14468    fn prepare_remove_prop(&self, key: &str, field: &str) -> Result<bool> {
14469        // A view owns its property, and the refusal has to live here rather
14470        // than on `GraphDb::remove_prop`: `BatchOp::RemoveProp` never meets
14471        // that one, and it is what the HTTP `DELETE /node/{key}/prop/{field}`
14472        // route, `Batch::remove_prop` and the CLI all submit. This is the one
14473        // choke-point every removal passes, exactly as it is for `ns` below.
14474        // Checked before the key, so the answer does not depend on whether the
14475        // key exists — which is also what `GraphDb::remove_prop` answered when
14476        // it carried the only copy of this guard.
14477        if let Some(view_name) = self.db.view_store.view_for_prop(field) {
14478            return Err(GraphError::ViewPropReadOnly {
14479                view_name: view_name.to_string(),
14480            });
14481        }
14482        self.check_live_key(key)?;
14483        // Removing `ns` is changing the namespace — to `default`, the namespace
14484        // an absent property names. It goes through this one choke-point and NOT
14485        // through `rewrite_wal_dense` (a `RemoveProp` needs no dense rewrite), so
14486        // the immutability rule has to be stated here as well. Without it the
14487        // node silently lands in `default` on the next open: the cross-namespace
14488        // edge guard is defeated and a default-bound role reads a tenant's node.
14489        if field == NS_PROP {
14490            let from = self.namespace_in_batch(key);
14491            if from != NS_DEFAULT {
14492                return Err(GraphError::NamespaceImmutable {
14493                    key: key.to_string(),
14494                    from,
14495                    to: NS_DEFAULT.to_string(),
14496                });
14497            }
14498            // Already in `default`: the removal changes no namespace. It is the
14499            // no-op `set_prop` to the current namespace is, not an error.
14500            return Ok(false);
14501        }
14502        Ok(self.has_prop(key, field))
14503    }
14504
14505    fn prepare_delete_edge(&self, edge_type: &str, src_key: &str, dst_key: &str) -> Result<bool> {
14506        for k in [src_key, dst_key] {
14507            if !self.has_key(k) {
14508                return Err(GraphError::KeyNotFound { key: k.into() });
14509            }
14510        }
14511        // Provenance-owned OR a live rule would derive this pair. User-first
14512        // edges that a later rule matches are not in `owned`, but deleting
14513        // them would leave a hole `rebuild_rule` immediately fills.
14514        if self.is_rule_owned(edge_type, src_key, dst_key) {
14515            return Err(GraphError::RuleOwned {
14516                detail: format!(
14517                    "edge {edge_type} {src_key}→{dst_key} is rule-owned; \
14518                     delete or change the owning rule"
14519                ),
14520            });
14521        }
14522        if self.would_derive(edge_type, src_key, dst_key) {
14523            return Err(GraphError::RuleOwned {
14524                detail: format!(
14525                    "edge {edge_type} {src_key}→{dst_key} is rule-owned; \
14526                     delete or change the owning rule, or a live rule would re-derive it"
14527                ),
14528            });
14529        }
14530        Ok(self.has_edge(edge_type, src_key, dst_key))
14531    }
14532
14533    /// True if any live rule (minus overlay-deleted names) would derive
14534    /// `(edge_type, src, dst)` from current overlay-visible props/labels.
14535    /// CreateRule names in `extra_rules` are ignored — same documented
14536    /// same-batch rule-window as [`Self::is_rule_owned`].
14537    fn would_derive(&self, edge_type: &str, src_key: &str, dst_key: &str) -> bool {
14538        if src_key == dst_key {
14539            return false;
14540        }
14541        let Some(src_label) = self.label_of(src_key) else {
14542            return false;
14543        };
14544        let Some(dst_label) = self.label_of(dst_key) else {
14545            return false;
14546        };
14547        for rule in self.db.engine.rules() {
14548            if self.overlay.deleted_rules.contains(&rule.name) {
14549                continue;
14550            }
14551            if rule.edge_type != edge_type {
14552                continue;
14553            }
14554            if rule.src_label != src_label || rule.dst_label != dst_label {
14555                continue;
14556            }
14557            let src_props = |f: &str| self.prop_value(src_key, f);
14558            let dst_props = |f: &str| self.prop_value(dst_key, f);
14559            let src_view = NodeView {
14560                key: src_key,
14561                props: &src_props,
14562            };
14563            let dst_view = NodeView {
14564                key: dst_key,
14565                props: &dst_props,
14566            };
14567            if evaluate(&rule.predicate, &src_view, &dst_view).is_some() {
14568                return true;
14569            }
14570        }
14571        false
14572    }
14573
14574    fn label_of(&self, key: &str) -> Option<String> {
14575        if self.overlay.deleted_keys.contains(key) {
14576            return None;
14577        }
14578        // Fresh identities created in this batch have no stored label in the
14579        // overlay; they cannot be provenance-owned yet either.
14580        let id = self.db.ids.get(key)?;
14581        let sym = self.db.labels.get(id as usize).copied()?;
14582        if sym == u32::MAX {
14583            return None;
14584        }
14585        self.db.syms.resolve(sym).map(str::to_string)
14586    }
14587
14588    /// The label `key` carries as this batch sees it — including a node
14589    /// inserted earlier in the same batch, which the store does not have yet.
14590    fn label_in_batch(&self, key: &str) -> Option<String> {
14591        if self.overlay.deleted_keys.contains(key) {
14592            return None;
14593        }
14594        if let Some(label) = self.overlay.extra_labels.get(key) {
14595            return Some(label.clone());
14596        }
14597        self.label_of(key)
14598    }
14599
14600    /// The property writes that make `key`'s props exactly `props`, or why the
14601    /// row is refused.
14602    ///
14603    /// `Some(value)` is a set and `None` is a removal. `store_fields` is every
14604    /// field name the store knows, hoisted by the caller so a frame of N
14605    /// replaces reads the field list once rather than N times.
14606    ///
14607    /// The second half of the pair is how many view-owned fields this row kept
14608    /// rather than removed — the one part of "exactly the supplied props" that
14609    /// does not hold, and the caller's only signal that it did not.
14610    ///
14611    /// The refusals are row errors, not frame errors: a mirror rebuild should
14612    /// learn which of its rows disagree with the store without losing the rows
14613    /// that agree.
14614    fn plan_replace(
14615        &self,
14616        label: &str,
14617        key: &str,
14618        props: &[(String, Value)],
14619        store_fields: &[String],
14620    ) -> std::result::Result<ReplacePlan, String> {
14621        // A different label is a relabel, and a rebuild does not relabel: that
14622        // is `rename_node` or an explicit write, never a side effect here.
14623        let stored = self.label_in_batch(key).unwrap_or_default();
14624        if stored != label {
14625            return Err(format!(
14626                "node {key}: on_conflict=\"replace\" will not relabel {stored:?} to {label:?}; \
14627                 relabelling is rename_node or an explicit write"
14628            ));
14629        }
14630        // `ns` is immutable. Replace removes what the supplied props omit, so
14631        // an omitted `ns` is a move to `default` exactly as a different `ns` is
14632        // a move to that one; both are the same refusal.
14633        let from = self.namespace_in_batch(key);
14634        let to = match props.iter().find(|(field, _)| field == NS_PROP) {
14635            Some((_, Value::Str(ns))) => ns.clone(),
14636            Some((_, value)) => {
14637                return Err(format!(
14638                    "node {key}: {NS_PROP} must be a string naming a namespace, got {value:?}"
14639                ));
14640            }
14641            None => NS_DEFAULT.to_string(),
14642        };
14643        if to != from {
14644            return Err(format!(
14645                "node {key}: {NS_PROP} is immutable; on_conflict=\"replace\" cannot move it \
14646                 from {from:?} to {to:?}"
14647            ));
14648        }
14649
14650        if let Some(why) = self.supplied_view_owned_prop(key, props) {
14651            return Err(why);
14652        }
14653
14654        let supplied: BTreeSet<&str> = props.iter().map(|(field, _)| field.as_str()).collect();
14655        let mut writes = Vec::new();
14656        for (field, value) in props {
14657            // `ns` names the namespace the node is already in, so the write is
14658            // the no-op the dense-rewrite seam would drop anyway.
14659            if field == NS_PROP {
14660                continue;
14661            }
14662            // Already exactly this value: a rebuild of an unchanged row should
14663            // cost no WAL record.
14664            if self.prop_value(key, field).as_ref() == Some(value) {
14665                continue;
14666            }
14667            writes.push((field.clone(), Some(value.clone())));
14668        }
14669        // Everything the node still carries that the supplied props do not.
14670        // `ns` is never removed: it is immutable, and the check above has
14671        // already established the node stays where it is.
14672        let overlay_fields = self
14673            .overlay
14674            .extra_props
14675            .keys()
14676            .filter(|(k, _)| k == key)
14677            .map(|(_, field)| field.as_str());
14678        //
14679        // A view-owned field is filtered out rather than refused. It is not the
14680        // caller's to supply (supplying one is still the row error above) and
14681        // so it is not part of what "exactly the supplied ones" ranges over:
14682        // omitting it is not a request to delete it. Refusing here instead
14683        // would make `replace` impossible for every node a view has written to
14684        // — which on a store carrying a view is the whole population a mirror
14685        // rebuild has to cover.
14686        let omitted: BTreeSet<&str> = store_fields
14687            .iter()
14688            .map(String::as_str)
14689            .chain(overlay_fields)
14690            .filter(|field| {
14691                *field != NS_PROP && !supplied.contains(field) && self.has_prop(key, field)
14692            })
14693            .collect();
14694        // The view-owned half is kept, and counted: the row still commits and
14695        // still reports no error, so without this number a mirror rebuild is
14696        // told it got exactly what it asked for when it did not (defect #18).
14697        let (stale, kept): (Vec<&str>, Vec<&str>) = omitted
14698            .into_iter()
14699            .partition(|field| self.db.view_store.view_for_prop(field).is_none());
14700        writes.extend(stale.into_iter().map(|field| (field.to_string(), None)));
14701        Ok((writes, kept.len()))
14702    }
14703
14704    /// The row error a supplied view-owned field earns, or `None`.
14705    ///
14706    /// Shared by [`MutPreview::plan_replace`] and the no-conflict arm of
14707    /// `BatchOp::InsertNodeOnConflict` so that one op answers a supplied
14708    /// view-owned field the same way whether or not the key was already taken.
14709    fn supplied_view_owned_prop(&self, key: &str, props: &[(String, Value)]) -> Option<String> {
14710        props.iter().find_map(|(field, _)| {
14711            self.db.view_store.view_for_prop(field).map(|view_name| {
14712                format!(
14713                    "node {key}: property {field:?} is owned by view {view_name:?} and is \
14714                     read-only"
14715                )
14716            })
14717        })
14718    }
14719
14720    /// The namespace `key` is in as this batch sees it — including a node
14721    /// inserted earlier in the same batch, which the store does not have yet.
14722    fn namespace_in_batch(&self, key: &str) -> String {
14723        namespace_of_value(self.prop_value(key, NS_PROP).as_ref()).to_string()
14724    }
14725
14726    fn prop_value(&self, key: &str, field: &str) -> Option<Value> {
14727        if !self.has_key(key) {
14728            return None;
14729        }
14730        let k = (key.to_string(), field.to_string());
14731        if self.overlay.removed_props.contains(&k) {
14732            return None;
14733        }
14734        if let Some(v) = self.overlay.extra_props.get(&k) {
14735            return Some(v.clone());
14736        }
14737        if self.overlay.extra_keys.contains(key) {
14738            return None;
14739        }
14740        self.db.get_prop(key, field)
14741    }
14742
14743    fn check_create_rule(&self, def: &RuleDef) -> Result<()> {
14744        def.validate()
14745            .map_err(|e| GraphError::RuleInvalid { detail: e })?;
14746        if self.has_rule(&def.name) {
14747            return Err(GraphError::RuleInvalid {
14748                detail: format!("rule {:?} already exists", def.name),
14749            });
14750        }
14751        // Rule-chain cycle rejection. Derived edges feed via-hop rules, so a
14752        // rule set forms a graph whose arcs are "hops over `via_edge`, writes
14753        // `edge_type`". A cycle in that graph is a rule set that would re-fire
14754        // itself forever; the engine's depth cap would silently truncate it
14755        // instead, leaving an arbitrary partial result. Reject it here, the one
14756        // place that sees the whole rule set.
14757        //
14758        // Rules accepted earlier in the same batch count too: the overlay
14759        // carries their arcs, so a cycle cannot be assembled one op at a time.
14760        if let Some(via) = def.via_edge.as_deref() {
14761            if via == def.edge_type {
14762                return Err(GraphError::RuleInvalid {
14763                    detail: format!("rule chain cycle: {} -> {}", via, def.edge_type),
14764                });
14765            }
14766            let mut arcs: Vec<(String, String)> = self
14767                .db
14768                .engine
14769                .rules()
14770                .filter(|r| !self.overlay.deleted_rules.contains(&r.name))
14771                .filter_map(|r| r.via_edge.clone().map(|v| (v, r.edge_type.clone())))
14772                .collect();
14773            arcs.extend(self.overlay.extra_rule_arcs.values().cloned());
14774            arcs.push((via.to_string(), def.edge_type.clone()));
14775            if let Some(path) = find_cycle_through(&arcs, &def.edge_type, via) {
14776                return Err(GraphError::RuleInvalid {
14777                    detail: format!("rule chain cycle: {} -> {}", via, path.join(" -> ")),
14778                });
14779            }
14780        }
14781        Ok(())
14782    }
14783
14784    fn check_delete_rule(&self, name: &str) -> Result<()> {
14785        if self.has_rule(name) {
14786            Ok(())
14787        } else {
14788            Err(GraphError::RuleNotFound { name: name.into() })
14789        }
14790    }
14791
14792    fn note_insert_node(&mut self, label: &str, key: &str, props: &[(String, Value)]) {
14793        self.overlay.deleted_keys.remove(key);
14794        self.overlay.extra_keys.insert(key.to_string());
14795        self.overlay
14796            .extra_labels
14797            .insert(key.to_string(), label.to_string());
14798        self.overlay.extra_props.retain(|(k, _), _| k != key);
14799        self.overlay.removed_props.retain(|(k, _)| k != key);
14800        for (field, value) in props {
14801            self.overlay
14802                .extra_props
14803                .insert((key.to_string(), field.clone()), value.clone());
14804        }
14805    }
14806
14807    fn note_insert_edge(&mut self, edge_type: &str, src_key: &str, dst_key: &str) {
14808        let k = (
14809            edge_type.to_string(),
14810            src_key.to_string(),
14811            dst_key.to_string(),
14812        );
14813        self.overlay.deleted_edges.remove(&k);
14814        self.overlay.extra_edges.insert(k);
14815    }
14816
14817    fn note_set_prop(&mut self, key: &str, field: &str, value: &Value) {
14818        let k = (key.to_string(), field.to_string());
14819        self.overlay.removed_props.remove(&k);
14820        self.overlay.extra_props.insert(k, value.clone());
14821    }
14822
14823    fn note_remove_prop(&mut self, key: &str, field: &str) {
14824        let k = (key.to_string(), field.to_string());
14825        self.overlay.extra_props.remove(&k);
14826        self.overlay.removed_props.insert(k);
14827    }
14828
14829    fn note_delete_edge(&mut self, edge_type: &str, src_key: &str, dst_key: &str) {
14830        let k = (
14831            edge_type.to_string(),
14832            src_key.to_string(),
14833            dst_key.to_string(),
14834        );
14835        self.overlay.extra_edges.remove(&k);
14836        self.overlay.deleted_edges.insert(k);
14837    }
14838
14839    fn note_delete_node(&mut self, key: &str) {
14840        self.overlay.extra_keys.remove(key);
14841        self.overlay.deleted_keys.insert(key.to_string());
14842        self.overlay.extra_props.retain(|(k, _), _| k != key);
14843        self.overlay.removed_props.retain(|(k, _)| k != key);
14844        self.overlay
14845            .extra_edges
14846            .retain(|(_, s, d)| s != key && d != key);
14847        self.overlay
14848            .deleted_edges
14849            .retain(|(_, s, d)| s != key && d != key);
14850    }
14851
14852    fn note_create_rule(&mut self, def: &RuleDef) {
14853        self.overlay.deleted_rules.remove(&def.name);
14854        self.overlay.extra_rules.insert(def.name.clone());
14855        // Rules accepted earlier in this batch are not in the engine yet, so
14856        // the cycle check would not see their arcs. Keep the arc, not just the
14857        // name, so a batch cannot smuggle in a cycle one op at a time.
14858        if let Some(via) = def.via_edge.clone() {
14859            self.overlay
14860                .extra_rule_arcs
14861                .insert(def.name.clone(), (via, def.edge_type.clone()));
14862        }
14863    }
14864
14865    fn check_rename_node(&self, old: &str, new: &str) -> Result<()> {
14866        if !self.has_key(old) {
14867            return Err(GraphError::KeyNotFound { key: old.into() });
14868        }
14869        if self.has_key(new) {
14870            return Err(GraphError::DuplicateKey { key: new.into() });
14871        }
14872        Ok(())
14873    }
14874
14875    fn note_rename_node(&mut self, old: &str, new: &str) {
14876        // Mark old as deleted so subsequent batch ops cannot reference it.
14877        self.overlay.extra_keys.remove(old);
14878        self.overlay.deleted_keys.insert(old.to_string());
14879        // Mark new as extra so subsequent batch ops can reference it.
14880        self.overlay.deleted_keys.remove(new);
14881        self.overlay.extra_keys.insert(new.to_string());
14882        // Migrate any overlay props from old key to new key.
14883        let new_str = new.to_string();
14884        let transferred: Vec<((String, String), Value)> = self
14885            .overlay
14886            .extra_props
14887            .iter()
14888            .filter(|((k, _), _)| k.as_str() == old)
14889            .map(|((_, f), v)| ((new_str.clone(), f.clone()), v.clone()))
14890            .collect();
14891        self.overlay
14892            .extra_props
14893            .retain(|(k, _), _| k.as_str() != old);
14894        for (k, v) in transferred {
14895            self.overlay.extra_props.insert(k, v);
14896        }
14897        // Migrate removed_props.
14898        let transferred_removed: Vec<(String, String)> = self
14899            .overlay
14900            .removed_props
14901            .iter()
14902            .filter(|(k, _)| k.as_str() == old)
14903            .map(|(_, f)| (new_str.clone(), f.clone()))
14904            .collect();
14905        self.overlay
14906            .removed_props
14907            .retain(|(k, _)| k.as_str() != old);
14908        for k in transferred_removed {
14909            self.overlay.removed_props.insert(k);
14910        }
14911    }
14912
14913    fn note_delete_rule(&mut self, name: &str) {
14914        self.overlay.extra_rules.remove(name);
14915        // Drop its chain arc too: a rule created and then deleted in the same
14916        // batch must not make a later, legal rule look like a cycle.
14917        self.overlay.extra_rule_arcs.remove(name);
14918        self.overlay.deleted_rules.insert(name.to_string());
14919        // Treat the deleted rule's current provenance as gone so a later
14920        // delete_edge of those triples is a no-op (matches sequential).
14921        if let Some(triples) = self.db.engine.provenance().get(name) {
14922            for &(et, s, d) in triples {
14923                let Some(etype) = self.db.syms.resolve(et) else {
14924                    continue;
14925                };
14926                let Some(src) = self.db.ids.key_of(s) else {
14927                    continue;
14928                };
14929                let Some(dst) = self.db.ids.key_of(d) else {
14930                    continue;
14931                };
14932                let k = (etype.to_string(), src.to_string(), dst.to_string());
14933                self.overlay.extra_edges.remove(&k);
14934                self.overlay.deleted_edges.insert(k);
14935            }
14936        }
14937    }
14938}
14939
14940/// Collects mutations and commits them as one WAL `Batch` frame.
14941///
14942/// Holds `&mut GraphDb` for its lifetime. Queue with the same method names
14943/// as [`GraphDb`]; call [`commit`](Self::commit) to validate, log, and apply.
14944/// See [`GraphDb::batch`] for validation and atomicity rules.
14945pub struct BatchBuilder<'a, F: Fs> {
14946    db: &'a mut GraphDb<F>,
14947    ops: Vec<BatchOp>,
14948}
14949
14950impl<'a, F: Fs> BatchBuilder<'a, F> {
14951    pub fn insert_node(
14952        &mut self,
14953        label: &str,
14954        key: &str,
14955        props: Vec<(String, Value)>,
14956    ) -> &mut Self {
14957        self.ops.push(BatchOp::InsertNode {
14958            label: label.into(),
14959            key: key.into(),
14960            props,
14961        });
14962        self
14963    }
14964
14965    /// Queue a node insert whose answer to a taken key is `on_conflict`.
14966    ///
14967    /// [`OnConflict::Error`] queues exactly the op [`insert_node`](Self::insert_node)
14968    /// does, so the default path is unchanged.
14969    pub fn insert_node_on_conflict(
14970        &mut self,
14971        label: &str,
14972        key: &str,
14973        props: Vec<(String, Value)>,
14974        on_conflict: OnConflict,
14975    ) -> &mut Self {
14976        self.ops.push(match on_conflict {
14977            OnConflict::Error => BatchOp::InsertNode {
14978                label: label.into(),
14979                key: key.into(),
14980                props,
14981            },
14982            on_conflict => BatchOp::InsertNodeOnConflict {
14983                label: label.into(),
14984                key: key.into(),
14985                props,
14986                on_conflict,
14987            },
14988        });
14989        self
14990    }
14991
14992    pub fn insert_edge(&mut self, edge_type: &str, src_key: &str, dst_key: &str) -> &mut Self {
14993        self.ops.push(BatchOp::InsertEdge {
14994            edge_type: edge_type.into(),
14995            src_key: src_key.into(),
14996            dst_key: dst_key.into(),
14997        });
14998        self
14999    }
15000
15001    pub fn set_prop(&mut self, key: &str, field: &str, value: Value) -> &mut Self {
15002        self.ops.push(BatchOp::SetProp {
15003            key: key.into(),
15004            field: field.into(),
15005            value,
15006        });
15007        self
15008    }
15009
15010    pub fn remove_prop(&mut self, key: &str, field: &str) -> &mut Self {
15011        self.ops.push(BatchOp::RemoveProp {
15012            key: key.into(),
15013            field: field.into(),
15014        });
15015        self
15016    }
15017
15018    pub fn delete_edge(&mut self, edge_type: &str, src_key: &str, dst_key: &str) -> &mut Self {
15019        self.ops.push(BatchOp::DeleteEdge {
15020            edge_type: edge_type.into(),
15021            src_key: src_key.into(),
15022            dst_key: dst_key.into(),
15023        });
15024        self
15025    }
15026
15027    pub fn delete_node(&mut self, key: &str) -> &mut Self {
15028        self.ops.push(BatchOp::DeleteNode { key: key.into() });
15029        self
15030    }
15031
15032    pub fn create_rule(&mut self, def: RuleDef) -> &mut Self {
15033        self.ops.push(BatchOp::CreateRule(def));
15034        self
15035    }
15036
15037    pub fn delete_rule(&mut self, name: &str) -> &mut Self {
15038        self.ops.push(BatchOp::DeleteRule { name: name.into() });
15039        self
15040    }
15041
15042    /// Queue a node-rename in this batch.
15043    ///
15044    /// Validation (old exists, new not taken) runs at commit time.
15045    pub fn rename_node(&mut self, old_key: &str, new_key: &str) -> &mut Self {
15046        self.ops.push(BatchOp::RenameNode {
15047            old_key: old_key.into(),
15048            new_key: new_key.into(),
15049        });
15050        self
15051    }
15052
15053    /// Queue an edge insert with endpoint auto-creation.
15054    ///
15055    /// Any missing endpoint is created as a plain node `{key, label:
15056    /// placeholder_label, no props}` inside this batch frame. Rules fire and
15057    /// last-change is updated for each auto-created node.
15058    pub fn insert_edge_upsert(
15059        &mut self,
15060        edge_type: &str,
15061        src_key: &str,
15062        dst_key: &str,
15063        placeholder_label: &str,
15064    ) -> &mut Self {
15065        self.ops.push(BatchOp::InsertEdgeUpsert {
15066            edge_type: edge_type.into(),
15067            src_key: src_key.into(),
15068            dst_key: dst_key.into(),
15069            placeholder_label: placeholder_label.into(),
15070        });
15071        self
15072    }
15073
15074    /// Validate every queued op, then log one `Batch` frame and apply.
15075    /// Empty / all-noop batches return `Ok(())` without writing the WAL.
15076    /// A second `commit()` after a successful one is an empty-batch no-op
15077    /// (queued ops were taken).
15078    /// Takes `&mut self` so it chains after the queue methods (`b.insert_node(..).commit()`)
15079    /// and also works as `let mut b = db.batch(); b.insert_node(..); b.commit()`.
15080    ///
15081    /// **Rule-window limitation:** batch validation cannot see edges that a
15082    /// rule created earlier in the *same* batch will derive at apply time, so
15083    /// a `delete_edge` / `insert_edge` in that window is silently no-oped
15084    /// where sequential calls would return `Err(RuleOwned)`. State integrity
15085    /// is unaffected (idempotent apply, provenance intact). Create rules in
15086    /// their own batch, or sequentially, when later ops may touch derived
15087    /// edges.
15088    /// Validate every queued op and commit atomically.
15089    ///
15090    /// Returns `(nodes_inserted, edges_inserted)` — the counts of node and edge
15091    /// WAL records actually written (duplicate edges are silent no-ops and are
15092    /// NOT counted). Both are 0 when the batch is empty or all-noop.
15093    pub fn commit(&mut self) -> Result<(usize, usize)> {
15094        let ops = std::mem::take(&mut self.ops);
15095        self.db.commit_batch(ops)
15096    }
15097
15098    /// [`commit`](Self::commit) with the full [`BatchOutcome`] — the counts a
15099    /// caller needs when its rows carry an [`OnConflict`] policy.
15100    pub fn commit_outcome(&mut self) -> Result<BatchOutcome> {
15101        let ops = std::mem::take(&mut self.ops);
15102        self.db.commit_logged_batch(ops, None, None)
15103    }
15104
15105    /// Same as [`commit`](Self::commit) but tail the inner events with
15106    /// [`MutationEvent::Ingested`] instead of [`MutationEvent::BatchApplied`].
15107    pub(crate) fn commit_ingest(&mut self, label: &str, inserted: usize) -> Result<(usize, usize)> {
15108        let ops = std::mem::take(&mut self.ops);
15109        self.db
15110            .commit_logged_batch(ops, Some((label.to_string(), inserted)), None)
15111            .map(inserted_pair)
15112    }
15113}
15114
15115pub struct NodeRef<'a, F: Fs> {
15116    db: &'a GraphDb<F>,
15117    id: u32,
15118}
15119
15120impl<'a, F: Fs> NodeRef<'a, F> {
15121    pub fn key(&self) -> &str {
15122        self.db.ids.key_of(self.id).expect("dense ids")
15123    }
15124
15125    pub fn label(&self) -> &str {
15126        let sym = self
15127            .db
15128            .labels
15129            .get(self.id as usize)
15130            .copied()
15131            .filter(|&s| s != u32::MAX)
15132            .expect("real nodes always have a label; u32::MAX sentinel cannot occur");
15133        self.db.syms.resolve(sym).expect("interned label symbol")
15134    }
15135
15136    pub fn prop(&self, field: &str) -> Option<Value> {
15137        self.db
15138            .props_view()
15139            .get(self.id, field)
15140            .map(|vr| vr.into_value())
15141    }
15142
15143    /// All stored fields for this node, sorted by field name.
15144    ///
15145    /// Reads from the full base+overlay view so that props stored only in the
15146    /// V8 snapshot base (i.e. before any post-snapshot WAL writes) are visible.
15147    pub fn props(&self) -> BTreeMap<String, Value> {
15148        let mut out = BTreeMap::new();
15149        let pv = self.db.props_view();
15150        for field in pv.field_names() {
15151            if let Some(vr) = pv.get(self.id, &field) {
15152                out.insert(field, vr.into_value());
15153            }
15154        }
15155        out
15156    }
15157
15158    /// depth-N BFS as a ResultSet: columns ["key","label","depth"], BFS order.
15159    pub fn neighborhood(&self, depth: u32, edge_types: Option<&[&str]>, dir: Dir) -> ResultSet {
15160        let view = self.db.view();
15161        let resolved: Option<Vec<u32>> = edge_types.map(|names| {
15162            names
15163                .iter()
15164                .filter_map(|name| view.syms.get(name))
15165                .collect()
15166        });
15167        let nb = neighborhood(&view, self.id, depth, resolved.as_deref(), dir);
15168        let mut rs = ResultSet::new(vec!["key".into(), "label".into(), "depth".into()]);
15169        for (nid, d) in nb.nodes {
15170            let key = view.key_of(nid);
15171            let label = view
15172                .label_of(nid)
15173                .expect("real nodes always have a label; u32::MAX sentinel cannot occur");
15174            rs.push_row(vec![
15175                Some(Value::Str(key.to_string())),
15176                Some(Value::Str(label.to_string())),
15177                Some(Value::Int(d as i64)),
15178            ]);
15179        }
15180        rs
15181    }
15182
15183    /// 1-hop, Both directions: edge-type name → sorted unique neighbor keys.
15184    pub fn grouped_by_edge_type(&self) -> BTreeMap<String, Vec<String>> {
15185        let view = self.db.view();
15186        let mut groups: BTreeMap<String, BTreeSet<String>> = BTreeMap::new();
15187        for e in expand(&view, self.id, None, Dir::Both) {
15188            // Skip edges with unknown etypes (only possible from corrupt large
15189            // TOPOLOGY section; function returns BTreeMap not Result).
15190            let Some(etype) = view.syms.resolve(e.etype) else {
15191                continue;
15192            };
15193            let etype = etype.to_string();
15194            let nbr = if e.src == self.id { e.dst } else { e.src };
15195            groups
15196                .entry(etype)
15197                .or_default()
15198                .insert(view.key_of(nbr).to_string());
15199        }
15200        groups
15201            .into_iter()
15202            .map(|(k, v)| (k, v.into_iter().collect()))
15203            .collect()
15204    }
15205}
15206
15207#[cfg(test)]
15208mod tests {
15209    use super::*;
15210    use core_rules::Predicate;
15211
15212    fn tmp_dir(name: &str) -> std::path::PathBuf {
15213        let d =
15214            std::env::temp_dir().join(format!("graphdb-db-unit-{}-{}", name, std::process::id()));
15215        let _ = std::fs::remove_dir_all(&d);
15216        d
15217    }
15218
15219    fn fk_rule() -> RuleDef {
15220        RuleDef {
15221            name: "works_at".into(),
15222            src_label: "Person".into(),
15223            dst_label: "Org".into(),
15224            predicate: Predicate::KeyMatch {
15225                field: "org_id".into(),
15226            },
15227            edge_type: "WORKS_AT".into(),
15228            weight_prop: None,
15229            max_edges: None,
15230            approximate: false,
15231            via_label: None,
15232            via_edge: None,
15233            via_dir: None,
15234            namespace: None,
15235        }
15236    }
15237
15238    /// Regression guard for the no-views delta-copy fast path.
15239    ///
15240    /// When no views are defined, `pending_deltas_since().to_vec()` must never
15241    /// be called — even during a large CreateRule backfill. The DELTA_COPY_COUNT
15242    /// thread-local is incremented inside every `if !view_store.is_empty()` block;
15243    /// a count of 0 after the entire sequence proves the guard fires correctly.
15244    #[test]
15245    fn no_delta_copy_when_no_views() {
15246        DELTA_COPY_COUNT.with(|c| c.set(0));
15247        let dir = tmp_dir("no-delta-copy");
15248        {
15249            let mut db = GraphDb::open(&dir).unwrap();
15250            // Insert 50 Org + 50 Person nodes with FK links.
15251            for i in 0..50u32 {
15252                db.insert_node("Org", &format!("o{i}"), vec![]).unwrap();
15253            }
15254            for i in 0..50u32 {
15255                db.insert_node(
15256                    "Person",
15257                    &format!("p{i}"),
15258                    vec![("org_id".into(), Value::Str(format!("o{i}")))],
15259                )
15260                .unwrap();
15261            }
15262            // CreateRule backfill should NOT invoke to_vec() when no views are defined.
15263            db.create_rule(fk_rule()).unwrap();
15264
15265            // Counter must stay 0 — no views, no copies.
15266            let copies = DELTA_COPY_COUNT.with(|c| c.get());
15267            assert_eq!(
15268                copies, 0,
15269                "pending_deltas_since().to_vec() called despite no views"
15270            );
15271
15272            // Derived edges must still be correct (the guard skips only the
15273            // empty delta propagation loop, not the rule application itself).
15274            let nbrs = db.neighbors("p0", "WORKS_AT", Direction::Out).unwrap();
15275            assert_eq!(
15276                nbrs,
15277                vec!["o0"],
15278                "rule must derive edges even with no views"
15279            );
15280        }
15281        let _ = std::fs::remove_dir_all(&dir);
15282    }
15283
15284    /// Gating regression: subscribe AFTER a backfill must see no stale events.
15285    /// subscribe BEFORE a backfill must see every edge-fire event.
15286    #[test]
15287    fn subscribe_after_backfill_no_stale_events() {
15288        let dir = tmp_dir("sub-after-backfill");
15289        {
15290            let mut db = GraphDb::open(&dir).unwrap();
15291            for i in 0..10u32 {
15292                db.insert_node("Org", &format!("o{i}"), vec![]).unwrap();
15293                db.insert_node(
15294                    "Person",
15295                    &format!("p{i}"),
15296                    vec![("org_id".into(), Value::Str(format!("o{i}")))],
15297                )
15298                .unwrap();
15299            }
15300            // Create rule BEFORE subscribing — emit_deltas is false during backfill.
15301            db.create_rule(fk_rule()).unwrap();
15302
15303            // Subscribe AFTER the backfill — queue must be empty (no stale events).
15304            let sub = db.subscribe_all_rules().unwrap();
15305            // No events should have queued for the prior backfill.
15306            assert!(
15307                sub.try_recv().is_none(),
15308                "subscribe after backfill must see no stale events"
15309            );
15310
15311            // Inserting a new node now should fire an event (emit_deltas is now true).
15312            db.insert_node("Org", "o_new", vec![]).unwrap();
15313            db.insert_node(
15314                "Person",
15315                "p_new",
15316                vec![("org_id".into(), Value::Str("o_new".into()))],
15317            )
15318            .unwrap();
15319            let ev = sub.recv_timeout(std::time::Duration::from_millis(200));
15320            assert!(
15321                ev.is_some(),
15322                "edge-fire event must arrive after subscribe (emit_deltas=true)"
15323            );
15324        }
15325        let _ = std::fs::remove_dir_all(&dir);
15326    }
15327
15328    /// Gating regression: subscribe BEFORE a backfill → events flow.
15329    #[test]
15330    fn subscribe_before_backfill_events_flow() {
15331        let dir = tmp_dir("sub-before-backfill");
15332        {
15333            let mut db = GraphDb::open(&dir).unwrap();
15334            // Subscribe FIRST — emit_deltas becomes true.
15335            let sub = db.subscribe_all_rules().unwrap();
15336
15337            for i in 0..5u32 {
15338                db.insert_node("Org", &format!("o{i}"), vec![]).unwrap();
15339                db.insert_node(
15340                    "Person",
15341                    &format!("p{i}"),
15342                    vec![("org_id".into(), Value::Str(format!("o{i}")))],
15343                )
15344                .unwrap();
15345            }
15346            // Backfill fires with emit_deltas=true → events queued.
15347            db.create_rule(fk_rule()).unwrap();
15348
15349            // Should receive at least one edge-fired event from the backfill.
15350            let mut received = 0usize;
15351            while sub.try_recv().is_some() {
15352                received += 1;
15353            }
15354            assert!(
15355                received > 0,
15356                "subscribe before backfill must receive edge-fire events (got 0)"
15357            );
15358        }
15359        let _ = std::fs::remove_dir_all(&dir);
15360    }
15361
15362    /// Companion: when a view IS defined, the delta path fires and view values update.
15363    #[test]
15364    fn delta_copy_fires_when_view_exists() {
15365        use core_rules::ViewSource;
15366        DELTA_COPY_COUNT.with(|c| c.set(0));
15367        let dir = tmp_dir("delta-copy-with-view");
15368        {
15369            let mut db = GraphDb::open(&dir).unwrap();
15370            db.insert_node("Org", "o1", vec![]).unwrap();
15371            db.insert_node(
15372                "Person",
15373                "p1",
15374                vec![("org_id".into(), Value::Str("o1".into()))],
15375            )
15376            .unwrap();
15377            // Declare a Degree view so is_empty() returns false.
15378            db.create_view(ViewDef {
15379                name: "degree_out".into(),
15380                label: "Person".into(),
15381                view_prop: "degree_out".into(),
15382                source: ViewSource::Degree {
15383                    edge_type: "WORKS_AT".into(),
15384                    direction: Direction::Out,
15385                },
15386            })
15387            .unwrap();
15388            db.create_rule(fk_rule()).unwrap();
15389
15390            // At least one delta copy should have happened (CreateRule backfill).
15391            let copies = DELTA_COPY_COUNT.with(|c| c.get());
15392            assert!(
15393                copies > 0,
15394                "expected delta copy to fire when a view is defined"
15395            );
15396
15397            // View value should be computed: p1 has one WORKS_AT out-edge.
15398            let info = db.node_info("p1").unwrap();
15399            let degree = info.props.get("degree_out");
15400            assert!(
15401                degree.is_some(),
15402                "view prop should be written to node props"
15403            );
15404        }
15405        let _ = std::fs::remove_dir_all(&dir);
15406    }
15407
15408    /// Regression: `open_at_with` must call `rebuild_all` after WAL replay so
15409    /// derived-edge-driven view values reflect the as-of state rather than just
15410    /// the initial backfill written at `CreateView` time.
15411    ///
15412    /// Base WAL frames (indices 0..=5 before history markers):
15413    ///   0: insert Org "o1"
15414    ///   1: create_view "employee_count" (Degree / WORKS_AT / In) on Org
15415    ///   2: create_rule fk_rule (WORKS_AT, Person→Org via org_id)
15416    ///   3: insert Person "p1" → rule fires WORKS_AT p1→o1 (degree = 1)  ← mid
15417    ///   4: insert Person "p2" → rule fires WORKS_AT p2→o1 (degree = 2)
15418    ///   5: insert Person "p3" → rule fires WORKS_AT p3→o1 (degree = 3)  ← latest
15419    ///
15420    /// Each rule-fire also appends a DerivedEdgeAdded history-marker frame (state
15421    /// no-op), so the total commit count is higher than the base frame count.
15422    /// The "latest" open_at commit is computed dynamically via `wal_commit_count_at`.
15423    ///
15424    /// Without `rebuild_all`, the as-of instance's "emp" view stays at the
15425    /// initial backfill value (0) instead of reflecting the replayed derived edges.
15426    #[test]
15427    fn open_at_derived_edge_view_values_correct() {
15428        use core_rules::ViewSource;
15429        let dir = tmp_dir("open-at-view-rebuild");
15430        {
15431            let mut db = GraphDb::open(&dir).unwrap();
15432            // frame 0
15433            db.insert_node("Org", "o1", vec![]).unwrap();
15434            // frame 1: create view — initial backfill sees 0 derived edges (none fired yet)
15435            db.create_view(ViewDef {
15436                name: "employee_count".into(),
15437                label: "Org".into(),
15438                view_prop: "emp".into(),
15439                source: ViewSource::Degree {
15440                    edge_type: "WORKS_AT".into(),
15441                    direction: Direction::In,
15442                },
15443            })
15444            .unwrap();
15445            // frame 2: create rule — no Persons yet; backfill is a no-op
15446            db.create_rule(fk_rule()).unwrap();
15447            // frame 3: p1 — rule fires WORKS_AT p1→o1; degree = 1
15448            db.insert_node(
15449                "Person",
15450                "p1",
15451                vec![("org_id".into(), Value::Str("o1".into()))],
15452            )
15453            .unwrap();
15454            // frame 4: p2 — degree = 2
15455            db.insert_node(
15456                "Person",
15457                "p2",
15458                vec![("org_id".into(), Value::Str("o1".into()))],
15459            )
15460            .unwrap();
15461            // frame 5: p3 — degree = 3
15462            db.insert_node(
15463                "Person",
15464                "p3",
15465                vec![("org_id".into(), Value::Str("o1".into()))],
15466            )
15467            .unwrap();
15468            // Sanity: normal open sees degree = 3.
15469            assert_eq!(
15470                db.get_view_prop("o1", "emp"),
15471                Some(Value::Int(3)),
15472                "normal db must show degree 3 after 3 derived edges"
15473            );
15474        } // WAL flushed
15475
15476        // Re-open normally to get the authoritative reference value.
15477        let normal_db = GraphDb::open(&dir).unwrap();
15478        let normal_emp = normal_db.get_view_prop("o1", "emp");
15479        assert_eq!(
15480            normal_emp,
15481            Some(Value::Int(3)),
15482            "re-opened normal db must show degree 3"
15483        );
15484
15485        // Latest as-of (last WAL commit): must match the normal open.
15486        // History-marker frames are appended after each rule-fire, so the total
15487        // commit count is computed dynamically rather than hardcoded.
15488        let total = crate::wal_commit_count_at(&dir).unwrap();
15489        let aof_latest = GraphDb::open_at(&dir, total - 1).unwrap();
15490        assert_eq!(
15491            aof_latest.get_view_prop("o1", "emp"),
15492            normal_emp,
15493            "open_at latest: derived-edge view must equal normal open (rebuild_all required)"
15494        );
15495
15496        // Mid-history as-of (commit 3 = p1 insert Batch frame): only p1; degree = 1.
15497        // The DerivedEdgeAdded marker for p1 is at frame 4 (state no-op on replay),
15498        // so replaying 0..=3 correctly re-derives only the p1→o1 edge.
15499        let aof_mid = GraphDb::open_at(&dir, 3).unwrap();
15500        assert_eq!(
15501            aof_mid.get_view_prop("o1", "emp"),
15502            Some(Value::Int(1)),
15503            "open_at mid-history: only p1 exists at frame 3, degree must be 1"
15504        );
15505
15506        let _ = std::fs::remove_dir_all(&dir);
15507    }
15508
15509    /// Pin: subscribe_* on an as-of instance must return Err(ReadOnly) —
15510    /// as-of instances never commit, so distribute_events never runs and any
15511    /// subscription would wait forever.
15512    #[test]
15513    fn subscribe_on_as_of_returns_read_only_error() {
15514        let dir = tmp_dir("sub-as-of-read-only");
15515        {
15516            let mut db = GraphDb::open(&dir).unwrap();
15517            db.insert_node("Org", "o1", vec![]).unwrap();
15518            db.create_rule(fk_rule()).unwrap();
15519        }
15520        let mut aof = GraphDb::open_at(&dir, 0).unwrap();
15521
15522        assert!(
15523            matches!(
15524                aof.subscribe_all_rules(),
15525                Err(core_storage::GraphError::ReadOnly)
15526            ),
15527            "subscribe_all_rules on as-of must return ReadOnly"
15528        );
15529        assert!(
15530            matches!(
15531                aof.subscribe_writes(),
15532                Err(core_storage::GraphError::ReadOnly)
15533            ),
15534            "subscribe_writes on as-of must return ReadOnly"
15535        );
15536        assert!(
15537            matches!(
15538                aof.subscribe_rule("works_at"),
15539                Err(core_storage::GraphError::ReadOnly)
15540            ),
15541            "subscribe_rule on as-of must return ReadOnly"
15542        );
15543        let _ = std::fs::remove_dir_all(&dir);
15544    }
15545
15546    /// Regression: a failed dense WAL rewrite must not leave speculative
15547    /// interns in `syms`. If it does, the next successful mutation logs an
15548    /// `Intern` record with an inflated id; replay (which never saw the
15549    /// orphans) assigns a smaller id and the WAL becomes unreplayable.
15550    #[test]
15551    fn dense_rewrite_error_rolls_back_speculative_interns() {
15552        let dir = tmp_dir("dense-rewrite-rollback");
15553        {
15554            let mut db = GraphDb::open(&dir).unwrap();
15555            db.insert_node("Person", "a", vec![]).unwrap();
15556
15557            // Bypass MutPreview validation to hit the rewrite's own error path
15558            // (same shape as an id-exhaustion failure mid-rewrite). The
15559            // InsertEdge arm interns the edge type before it resolves keys.
15560            let err = db.rewrite_wal_dense(vec![WalRecord::InsertEdge {
15561                edge_type: "ORPHAN_TYPE".into(),
15562                src_key: "missing".into(),
15563                dst_key: "a".into(),
15564            }]);
15565            assert!(err.is_err(), "rewrite of a missing src key must fail");
15566            assert_eq!(
15567                db.syms.get("ORPHAN_TYPE"),
15568                None,
15569                "failed rewrite must roll back speculative interns"
15570            );
15571
15572            // A later successful mutation must produce a replayable WAL.
15573            db.set_prop("a", "later_field", Value::Int(2)).unwrap();
15574        }
15575        let db = GraphDb::open(&dir).expect("WAL must replay after failed rewrite");
15576        assert_eq!(db.get_prop("a", "later_field"), Some(Value::Int(2)));
15577        let _ = std::fs::remove_dir_all(&dir);
15578    }
15579}