core_api/db.rs
1use crate::ingest::{IngestOptions, IngestReport};
2use crate::roles::{PropPredicate, RoleDef, RolesFile, WriteScope};
3use crate::subscription::{
4 event_matches, DbEvent, SubEntry, SubFilter, SubInner, Subscription, DEFAULT_SUB_CAPACITY,
5};
6use core_query::cypher::ast::{ret_val_label, ArithOp};
7use core_query::cypher::{
8 execute, execute_union, is_subscribable, is_write_tokens, lex, parse, parse_read, parse_write,
9 plan, Expr, MatchDeleteNodeStmt, NodePat, Operand, Params, Pattern, PlanOp, Query, RetItem,
10 RetVal, WriteStatement,
11};
12use core_query::{eval_cmp, eval_filter, expand, neighborhood, Dir, Filter, GraphView, ResultSet};
13use core_rules::{
14 decode_rule_def, ef_max, evaluate, BuildProgress, EngineEdgeDelta, GraphMut, NodeView,
15 Predicate, RuleDef, RuleEngine, ViewDef, ViewStore,
16};
17use core_storage::fs::{FileId, Fs, FsIntrospect, RealFs};
18use core_storage::fulltext::FulltextIndex;
19use core_storage::property_index::PropertyIndex;
20use core_storage::v8::encode::{
21 archived_hnsw_to_owned, archived_rules_meta_to_owned, archived_to_idmap, archived_to_interner,
22 archived_views_to_owned, decode_last_change_bytes, decode_meta, encode_v8, V8Meta,
23};
24use core_storage::v8::seam::TopologyView;
25use core_storage::wal::{decode_all, encode_record, WalRecord};
26use core_storage::EdgePropsView;
27use core_storage::{
28 namespace_of_value, ColumnStore, Direction, EdgeProps, GraphError, IdMap, Interner, Result,
29 Topology, Value,
30};
31pub use core_storage::{valid_namespace, NS_DEFAULT, NS_MAX_LEN, NS_PROP};
32
33/// Index of [`NS_DEFAULT`] in `GraphDb::ns_names` — always zero, so the
34/// open-time pass over a store with no `ns` column fills `node_ns` with one
35/// constant and allocates no names.
36const NS_DEFAULT_IDX: u32 = 0;
37
38/// The reserved edge property holding a pair's insert count (§5.13).
39///
40/// Absent means 1 — the count is written only from the second insert of a
41/// triple onward, and only on a store that called
42/// [`GraphDb::enable_multiplicity`]. The engine owns the name: Cypher `SET` on
43/// it is refused, as the other reserved names are.
44pub const EDGE_COUNT_PROP: &str = "count";
45use serde::{Deserialize, Serialize};
46use std::collections::{BTreeMap, BTreeSet, HashMap, HashSet};
47use std::sync::Arc;
48
49/// Print a timing checkpoint when MUSHROOMDB_TRACE_OPEN is set.
50/// Zero-cost when the env var is absent (the var check is O(1) after first call).
51macro_rules! trace_open {
52 ($phase:literal, $t:expr) => {
53 if std::env::var("MUSHROOMDB_TRACE_OPEN").is_ok() {
54 eprintln!(
55 "[MUSHROOMDB_TRACE_OPEN] {:40} {:>9.3?}",
56 $phase,
57 $t.elapsed()
58 );
59 }
60 };
61}
62
63/// Print a migration phase checkpoint when MUSHROOMDB_TRACE_MIGRATE is set.
64/// Zero-cost when the env var is absent (the var check is O(1) after first call).
65macro_rules! trace_migrate {
66 ($phase:literal, $t:expr) => {
67 if std::env::var("MUSHROOMDB_TRACE_MIGRATE").is_ok() {
68 eprintln!(
69 "[MUSHROOMDB_TRACE_MIGRATE] {:40} {:>9.3?}",
70 $phase,
71 $t.elapsed()
72 );
73 }
74 };
75}
76
77// Test-only: counts how many times `pending_deltas_since().to_vec()` actually
78// executes (i.e., at least one view is defined). Used to verify the fast-path
79// guard skips the allocation when `view_store.is_empty()`.
80#[cfg(test)]
81thread_local! {
82 static DELTA_COPY_COUNT: std::cell::Cell<usize> = const { std::cell::Cell::new(0) };
83}
84
85// Per-thread count of query-subscription `execute` calls in `distribute_events`.
86//
87// Incremented each time a query subscription actually runs its plan (i.e.,
88// the label-skip fast-path did not fire). Because `distribute_events` is
89// called synchronously on the writer thread, this thread-local correctly
90// isolates each test thread's count even when integration tests run in
91// parallel. Read via [`query_sub_exec_count`].
92thread_local! {
93 static QUERY_SUB_EXECS_TL: std::cell::Cell<usize> = const { std::cell::Cell::new(0) };
94}
95
96/// Return the number of query-subscription re-executions logged on this
97/// thread since the process started (or since last reset via
98/// [`reset_query_sub_exec_count`]).
99///
100/// Primarily for integration tests that verify the label-skip fast-path.
101#[doc(hidden)]
102pub fn query_sub_exec_count() -> usize {
103 QUERY_SUB_EXECS_TL.with(|c| c.get())
104}
105
106/// Reset the per-thread query-subscription execution counter to zero.
107#[doc(hidden)]
108pub fn reset_query_sub_exec_count() {
109 QUERY_SUB_EXECS_TL.with(|c| c.set(0));
110}
111
112// Exact-versus-approximate warnings emitted on this thread. Thread-local for
113// the same reason [`QUERY_SUB_EXECS_TL`] is: integration tests run in parallel
114// and each gets its own thread, so a neighbour's masked search cannot be
115// mistaken for this test's.
116thread_local! {
117 static AMBIGUOUS_EXACTNESS_WARNS: std::cell::Cell<u64> = const { std::cell::Cell::new(0) };
118 static AMBIGUOUS_EXACTNESS_LAST: std::cell::RefCell<Option<String>> =
119 const { std::cell::RefCell::new(None) };
120}
121
122/// How many times a masked, non-exact vector search has explained itself on
123/// this thread since the last [`ambiguous_exactness_warns_reset`].
124///
125/// The line itself is the product; this counter exists so a test can assert it
126/// is printed **once per index** rather than once per call.
127///
128/// **Single-threaded assertions only.** The suppression set this counts is a
129/// `Mutex<HashSet<_>>` on the `GraphDb` — shared by every thread — while the
130/// counter is thread-local. Under a concurrent caller (`serve`, which is the
131/// deployment the warning exists for) the thread that prints the line is not
132/// necessarily the thread that asked, so a zero here does not mean the line was
133/// not printed and a one does not mean it was printed once. It answers
134/// "once per index" only in a test that owns the store.
135#[doc(hidden)]
136pub fn ambiguous_exactness_warns() -> u64 {
137 AMBIGUOUS_EXACTNESS_WARNS.with(|c| c.get())
138}
139
140/// The most recent exactness warning printed on this thread, verbatim.
141///
142/// The advice a caller reads has to be advice that caller can act on, which the
143/// counter alone cannot witness — see
144/// `the_hybrid_path_advises_a_call_the_hybrid_caller_can_make`. Carries the
145/// same single-threaded caveat as [`ambiguous_exactness_warns`].
146#[doc(hidden)]
147pub fn ambiguous_exactness_last_warning() -> Option<String> {
148 AMBIGUOUS_EXACTNESS_LAST.with(|c| c.borrow().clone())
149}
150
151/// Reset this thread's exactness-warning counter and recorded line.
152#[doc(hidden)]
153pub fn ambiguous_exactness_warns_reset() {
154 AMBIGUOUS_EXACTNESS_WARNS.with(|c| c.set(0));
155 AMBIGUOUS_EXACTNESS_LAST.with(|c| *c.borrow_mut() = None);
156}
157
158/// Which signature reached the approximate masked vector leg.
159///
160/// One code path, two entry points, and the advice cannot be the same: a
161/// warning that names an argument the caller's function does not take sends
162/// them looking for a parameter that is not there. `search_hybrid` and
163/// `search_hybrid_scoped` take `(text_field, query_text, vector_field,
164/// query_vec, label, k[, mask])` — no `exact`, no `where`.
165///
166/// The warning is not suppressed on the hybrid path. The approximation is the
167/// same one, and a caller who read `mask=` as a promise of exhaustiveness is
168/// the reader it was written for whichever door they came in by; only the
169/// remedy differs, so only the remedy changes.
170#[derive(Clone, Copy, PartialEq, Eq, Hash, Debug)]
171enum ExactnessCaller {
172 /// `find_similar` / `find_similar_vector_*` — `exact` and `where` are its
173 /// own parameters.
174 Vector,
175 /// `search_hybrid` / `search_hybrid_scoped` — neither argument exists, and
176 /// the leg is one half of a fusion.
177 Hybrid,
178}
179
180impl ExactnessCaller {
181 fn subject(self) -> &'static str {
182 match self {
183 Self::Vector => "a masked vector search",
184 Self::Hybrid => "the vector leg of a masked hybrid search",
185 }
186 }
187
188 fn advice(self) -> &'static str {
189 match self {
190 Self::Vector => "pass exact=True or a where= predicate.",
191 // Names the call that does take the argument, because this one
192 // does not: the caller's own next step, not a parameter hunt.
193 Self::Hybrid => {
194 "run the vector leg on its own with find_similar(field, vector, mask=…, \
195 exact=True) and fuse it with search() yourself — search_hybrid itself \
196 takes no exactness argument."
197 }
198 }
199 }
200}
201
202/// Internal state for a single `subscribe_query` subscription.
203///
204/// On every commit, `distribute_events` re-executes `ops` against the current
205/// graph state, diffs the result against `prev_rows`, and pushes
206/// `DbEvent::QueryRowAdded` / `QueryRowRemoved` events to `inner`.
207///
208/// **Full re-run per commit; use LIMIT to bound execution cost.**
209/// (Differential evaluation is roadmap / Phase 5.)
210pub(crate) struct QuerySubEntry {
211 /// Compiled plan for the subscribed Cypher query.
212 ops: Vec<PlanOp>,
213 /// Column names from the first execution (fixed for the subscription lifetime).
214 columns: Vec<String>,
215 /// Serialized (JSON) row key → row data, representing the result set at
216 /// the end of the last commit. Used to diff against the new result.
217 prev_row_map: std::collections::HashMap<String, Vec<Option<Value>>>,
218 /// Weak pointer to the subscriber queue; dead Weak → subscription dropped.
219 inner: std::sync::Weak<SubInner>,
220 /// Interned label sym captured at subscribe time from the plan's leading scan
221 /// (`ScanLabel`, `IndexScan`, or `IndexIntersect` with a concrete label).
222 ///
223 /// `None` means the plan has an `Expand` op (or no recognizable leading scan
224 /// with a concrete label), and this subscription must re-execute on every
225 /// commit without skipping. This is the conservative v0.4.3 boundary: Expand
226 /// queries are never skipped because edges can alter join results regardless
227 /// of which node labels were written.
228 scan_label: Option<u32>,
229}
230
231/// A post-commit mutation notification.
232///
233/// Emitted from `log_then_apply` after the WAL append, fsync, and
234/// in-memory `apply` all succeed. Never emitted for rejected operations
235/// (validation errors, [`GraphError::RuleOwned`], duplicate keys, no-op
236/// deletes/removes). Event payloads carry user keys and rule names, never
237/// internal ids.
238///
239/// **Replay:** [`GraphDb::open`] / [`GraphDb::open_with`] replay the WAL via
240/// `apply` only. Emission lives exclusively in `log_then_apply`, so
241/// recovery is silent even if a sink were installed (it cannot be: the
242/// sink is in-memory and set after open).
243///
244/// **Ordering:** a `Batch` WAL frame emits one event per inner record, then
245/// [`MutationEvent::BatchApplied`]. An ingest commit emits those same inner
246/// events, then [`MutationEvent::Ingested`] (not `BatchApplied`). An empty
247/// or all-noop batch writes no WAL and emits nothing (including no summary).
248///
249/// **Derived edges:** rule-created or retracted edges are not individually
250/// evented — they are recoverable from the triggering mutation plus the live
251/// rule set. Only the triggering record is emitted.
252///
253/// **Wire form:** externally tagged snake_case JSON
254/// (`{"node_inserted":{"label":"A","key":"k"}}`).
255#[derive(Debug, Clone, PartialEq, Eq, Serialize)]
256#[serde(rename_all = "snake_case")]
257pub enum MutationEvent {
258 NodeInserted {
259 label: String,
260 key: String,
261 },
262 PropSet {
263 key: String,
264 field: String,
265 },
266 PropRemoved {
267 key: String,
268 field: String,
269 },
270 EdgeInserted {
271 edge_type: String,
272 src: String,
273 dst: String,
274 },
275 EdgeDeleted {
276 edge_type: String,
277 src: String,
278 dst: String,
279 },
280 NodeDeleted {
281 key: String,
282 },
283 RuleCreated {
284 name: String,
285 },
286 RuleDeleted {
287 name: String,
288 },
289 RuleRebuilt {
290 name: String,
291 },
292 BatchApplied {
293 ops: usize,
294 },
295 Ingested {
296 label: String,
297 inserted: usize,
298 },
299}
300
301/// Whether `rec` is the frame [`GraphDb::create_rule`] logs: a `CreateRule`,
302/// alone or behind the `Intern` record for its edge type.
303///
304/// `create_rule` goes through the dense rewrite like every other write, so its
305/// frame is `Batch([Intern { edge_type }, CreateRule])` rather than a bare
306/// `CreateRule`. Anything that asks "was this commit a rule creation" must
307/// accept both shapes, or it silently stops recognising the one the
308/// standalone call writes. A batch carrying anything else is a user batch and
309/// is not this frame.
310fn is_create_rule_frame(rec: &WalRecord) -> bool {
311 match rec {
312 WalRecord::CreateRule { .. } => true,
313 WalRecord::Batch(inner) => {
314 inner
315 .iter()
316 .any(|r| matches!(r, WalRecord::CreateRule { .. }))
317 && inner
318 .iter()
319 .all(|r| matches!(r, WalRecord::CreateRule { .. } | WalRecord::Intern { .. }))
320 }
321 _ => false,
322 }
323}
324
325fn event_from_record(rec: &WalRecord, intern: &Interner, ids: &IdMap) -> Option<MutationEvent> {
326 match rec {
327 WalRecord::InsertNode { label, key, .. } => Some(MutationEvent::NodeInserted {
328 label: label.clone(),
329 key: key.clone(),
330 }),
331 WalRecord::InsertNodeId { label, key, .. } => Some(MutationEvent::NodeInserted {
332 label: intern.resolve(*label)?.to_string(),
333 key: key.clone(),
334 }),
335 WalRecord::SetProp { key, field, .. } => Some(MutationEvent::PropSet {
336 key: key.clone(),
337 field: field.clone(),
338 }),
339 WalRecord::SetPropId { id, field, .. } => Some(MutationEvent::PropSet {
340 key: ids.key_of(*id)?.to_string(),
341 field: intern.resolve(*field)?.to_string(),
342 }),
343 WalRecord::RemoveProp { key, field } => Some(MutationEvent::PropRemoved {
344 key: key.clone(),
345 field: field.clone(),
346 }),
347 WalRecord::InsertEdge {
348 edge_type,
349 src_key,
350 dst_key,
351 } => Some(MutationEvent::EdgeInserted {
352 edge_type: edge_type.clone(),
353 src: src_key.clone(),
354 dst: dst_key.clone(),
355 }),
356 WalRecord::InsertEdgeId { etype, src, dst } => Some(MutationEvent::EdgeInserted {
357 edge_type: intern.resolve(*etype)?.to_string(),
358 src: ids.key_of(*src)?.to_string(),
359 dst: ids.key_of(*dst)?.to_string(),
360 }),
361 WalRecord::DeleteEdge {
362 edge_type,
363 src_key,
364 dst_key,
365 } => Some(MutationEvent::EdgeDeleted {
366 edge_type: edge_type.clone(),
367 src: src_key.clone(),
368 dst: dst_key.clone(),
369 }),
370 WalRecord::DeleteNode { key } => Some(MutationEvent::NodeDeleted { key: key.clone() }),
371 WalRecord::CreateRule { def_bytes } => {
372 let def: RuleDef = decode_rule_def(def_bytes).ok()?;
373 Some(MutationEvent::RuleCreated { name: def.name })
374 }
375 WalRecord::DeleteRule { name } => Some(MutationEvent::RuleDeleted { name: name.clone() }),
376 WalRecord::RebuildRule { name } => Some(MutationEvent::RuleRebuilt { name: name.clone() }),
377 WalRecord::Batch(_)
378 | WalRecord::CreateView { .. }
379 | WalRecord::DeleteView { .. }
380 | WalRecord::EnableFulltext { .. }
381 | WalRecord::DisableFulltext { .. }
382 | WalRecord::EnableIndex { .. }
383 | WalRecord::DisableIndex { .. }
384 | WalRecord::Intern { .. }
385 // History markers are no-ops for mutation events — they carry no new
386 // state and rules re-derive deterministically on replay.
387 | WalRecord::DerivedEdgeAdded { .. }
388 | WalRecord::DerivedEdgeRetracted { .. }
389 // A count changes neither the node nor the edge population: the pair it
390 // counts was already there, which is why it is written at all.
391 | WalRecord::SetEdgeCount { .. }
392 // RenameNode carries no node/edge count change; no special event.
393 | WalRecord::RenameNode { .. } => None,
394 }
395}
396
397/// Database-wide counters plus per-rule budget/fire stats.
398#[derive(Debug, Clone, PartialEq, Eq, Serialize)]
399pub struct Stats {
400 pub nodes_live: usize,
401 pub nodes_tombstoned: usize,
402 pub edges: u64,
403 pub rules: Vec<RuleStats>,
404 /// How many writes hit the rule-chaining depth cap with work still pending,
405 /// since this handle was opened. Non-zero means some derived edges beyond
406 /// the cap are stale and no single later write will repair them: split the
407 /// rule chain or shorten it. Never persisted, so it resets on reopen.
408 #[serde(default)]
409 pub chain_truncations: u64,
410 /// The oldest commit index history still reaches (the WAL horizon floor).
411 /// `0` means nothing has been pruned and history is complete; a non-zero
412 /// value means events before that commit were pruned and are gone.
413 #[serde(default)]
414 pub history_floor: u64,
415 /// Live node counts per namespace, in name order. Always carries
416 /// `default` — a store is at least its default namespace — so a
417 /// single-tenant store reads `[{"name":"default", …}]` and a reader can
418 /// tell "no namespaces in use" from one entry.
419 #[serde(default)]
420 pub namespaces: Vec<NamespaceStats>,
421}
422
423/// Live node count for one namespace; one entry of [`Stats::namespaces`].
424#[derive(Debug, Clone, PartialEq, Eq, Serialize)]
425pub struct NamespaceStats {
426 pub name: String,
427 pub nodes_live: usize,
428}
429
430/// One rule's provenance size, trip latch, and fire counter.
431///
432/// `tripped` is a one-way latch: once set, the engine adds no new edges for
433/// that rule until [`GraphDb::rebuild_rule`] (and only if the full desired
434/// set then fits). `fires` counts `on_node_changed` evaluations plus
435/// backfill/rebuild participant ticks (rebuild counts even when it is a
436/// provenance no-op).
437#[derive(Debug, Clone, PartialEq, Eq, Serialize)]
438pub struct RuleStats {
439 pub name: String,
440 pub edges: u64,
441 pub tripped: bool,
442 pub fires: u64,
443 /// Whether this rule uses the approximate IVF-Flat candidate path.
444 pub approximate: bool,
445 /// `Some` while this rule's vector index is still being built.
446 ///
447 /// The rule derives **no** edges until it is `None`: the backfill is one
448 /// commit that runs after the index is whole, so a caller never sees a
449 /// partial edge set. Absent from the JSON when the rule is not building,
450 /// which is every rule created over a corpus at or below
451 /// [`core_rules::HNSW_BUILD_BATCH`] vectors.
452 #[serde(default, skip_serializing_if = "Option::is_none")]
453 pub building: Option<BuildProgress>,
454}
455
456/// One entry in the slow-query ring buffer.
457#[derive(Debug, Clone, Serialize)]
458pub struct SlowQueryEntry {
459 /// Execution time in whole milliseconds.
460 pub ms: u64,
461 /// The Cypher query string that was slow.
462 pub query: String,
463 /// The commit sequence number at the time the query ran.
464 pub at_commit: u64,
465}
466
467/// Snapshot of the slow-query log returned by [`GraphDb::slow_query_snapshot`].
468#[derive(Debug, Clone, Serialize)]
469pub struct SlowQuerySnapshot {
470 /// Current threshold in milliseconds (0 = disabled).
471 pub threshold_ms: u64,
472 /// Total number of slow queries ever recorded (not capped by ring size).
473 pub count: u64,
474 /// Most-recent slow queries (up to 16), oldest first.
475 pub last: Vec<SlowQueryEntry>,
476}
477
478/// Internal ring-buffer state protected by a `Mutex` so `query(&self)` can
479/// write to it without a mutable borrow.
480struct SlowQueryLog {
481 entries: std::collections::VecDeque<SlowQueryEntry>,
482 total: u64,
483}
484
485/// Maximum number of entries kept in the slow-query ring buffer.
486const SLOW_QUERY_RING_CAP: usize = 16;
487
488/// Wire summary of a [`Predicate`]. JSON only — `Explanation` is never
489/// bincode-persisted (WAL/snapshots store `RuleDef` bytes, not this type).
490#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
491pub struct PredicateSummary {
492 pub kind: String,
493 pub fields: Vec<String>,
494 pub min: Option<f64>,
495 pub tolerance: Option<f64>,
496 pub km: Option<f64>,
497 pub parts: Option<Vec<PredicateSummary>>,
498 /// True when the owning rule has `approximate=true` (IVF-Flat candidate path).
499 /// Always false for predicates reported without rule context (sub-predicates in `parts`).
500 #[serde(default)]
501 pub approximate: bool,
502}
503
504impl From<&Predicate> for PredicateSummary {
505 fn from(p: &Predicate) -> Self {
506 match p {
507 Predicate::KeyMatch { field } => PredicateSummary {
508 kind: "key_match".into(),
509 fields: vec![field.clone()],
510 min: None,
511 tolerance: None,
512 km: None,
513 parts: None,
514 approximate: false,
515 },
516 Predicate::FieldEqual { field } => PredicateSummary {
517 kind: "field_equal".into(),
518 fields: vec![field.clone()],
519 min: None,
520 tolerance: None,
521 km: None,
522 parts: None,
523 approximate: false,
524 },
525 Predicate::Overlap { field, min } => PredicateSummary {
526 kind: "overlap".into(),
527 fields: vec![field.clone()],
528 min: Some(*min),
529 tolerance: None,
530 km: None,
531 parts: None,
532 approximate: false,
533 },
534 Predicate::NumericWithin { field, tolerance } => PredicateSummary {
535 kind: "numeric_within".into(),
536 fields: vec![field.clone()],
537 min: None,
538 tolerance: Some(*tolerance),
539 km: None,
540 parts: None,
541 approximate: false,
542 },
543 Predicate::GeoRadius { field, km } => PredicateSummary {
544 kind: "geo_radius".into(),
545 fields: vec![field.clone()],
546 min: None,
547 tolerance: None,
548 km: Some(*km),
549 parts: None,
550 approximate: false,
551 },
552 Predicate::VectorSimilar { field, min } => PredicateSummary {
553 kind: "vector_similar".into(),
554 fields: vec![field.clone()],
555 min: Some(*min),
556 tolerance: None,
557 km: None,
558 parts: None,
559 approximate: false,
560 },
561 Predicate::All(inner) => {
562 let parts: Vec<PredicateSummary> = inner.iter().map(Self::from).collect();
563 let mut fields = Vec::new();
564 for part in &parts {
565 for f in &part.fields {
566 if !fields.contains(f) {
567 fields.push(f.clone());
568 }
569 }
570 }
571 PredicateSummary {
572 kind: "all".into(),
573 fields,
574 min: None,
575 tolerance: None,
576 km: None,
577 parts: Some(parts),
578 approximate: false,
579 }
580 }
581 Predicate::Any(inner) => {
582 let parts: Vec<PredicateSummary> = inner.iter().map(Self::from).collect();
583 let mut fields = Vec::new();
584 for part in &parts {
585 for f in &part.fields {
586 if !fields.contains(f) {
587 fields.push(f.clone());
588 }
589 }
590 }
591 PredicateSummary {
592 kind: "any".into(),
593 fields,
594 min: None,
595 tolerance: None,
596 km: None,
597 parts: Some(parts),
598 approximate: false,
599 }
600 }
601 }
602 }
603}
604
605/// Snapshot of a live node's key, label, and columnar properties.
606///
607/// `props` is a [`BTreeMap`] so field order is deterministic (sorted by name)
608/// regardless of insert order or the columnar store's `HashMap` iteration.
609///
610/// Deliberately does not derive `Serialize`: `Value`'s serde form is
611/// internally tagged. Wire JSON is built by `value_to_json` in the server.
612#[derive(Debug, Clone, PartialEq)]
613pub struct NodeInfo {
614 pub key: String,
615 pub label: String,
616 pub props: BTreeMap<String, Value>,
617}
618
619/// Counts returned by [`GraphDb::delete_node`].
620#[derive(Debug, Clone, PartialEq, Eq, Default)]
621pub struct DeleteReport {
622 /// Number of manual (user-inserted) edges removed.
623 pub manual_edges: u64,
624 /// Number of derived (rule-owned) edges retracted.
625 pub derived_edges: u64,
626}
627
628/// One directed edge incident on a node, with provenance membership.
629///
630/// `derived` is true iff `(edge_type, src, dst)` is in the rule engine's
631/// Plan-8 `by_node` provenance index.
632#[derive(Debug, Clone, PartialEq, Eq, Serialize)]
633pub struct EdgeInfo {
634 pub edge_type: String,
635 pub src_key: String,
636 pub dst_key: String,
637 pub derived: bool,
638}
639
640/// One directed edge incident on a node at a point in WAL history, with the
641/// rule that derived it when it is rule-owned.
642///
643/// Returned by [`GraphDb::edges_at`] (sorted by `(edge_type, src_key, dst_key)`)
644/// and by [`GraphDb::what_if_set_prop`].
645#[derive(Debug, Clone, PartialEq, Eq, PartialOrd, Ord, Serialize)]
646pub struct EdgeAt {
647 pub edge_type: String,
648 pub src_key: String,
649 pub dst_key: String,
650 /// `true` when a rule wrote the edge (`DerivedEdgeAdded` in the WAL, or a
651 /// live provenance entry).
652 pub derived: bool,
653 /// The rule that derived the edge. `None` for a manual edge.
654 pub rule: Option<String>,
655}
656
657/// The derived edges a hypothetical property change would retract and derive.
658///
659/// Returned by [`GraphDb::what_if_set_prop`]. Both lists are sorted by
660/// `(edge_type, src_key, dst_key)` and every entry is rule-derived.
661#[derive(Debug, Clone, PartialEq, Eq, Serialize)]
662pub struct WhatIf {
663 /// Derived edges that exist now and would be retracted.
664 pub lost: Vec<EdgeAt>,
665 /// Derived edges that do not exist now and would be derived.
666 pub gained: Vec<EdgeAt>,
667}
668
669/// An edge with mask-aware endpoint visibility.
670///
671/// Returned by [`GraphDb::node_edges_masked`] in [`crate::mask::MaskMode::Stub`]
672/// mode — hidden endpoints carry `*_restricted: true`.
673#[derive(Debug, Clone, PartialEq, Eq)]
674pub struct MaskedEdge {
675 pub edge_type: String,
676 pub src_key: String,
677 /// `true` when `src_key` is in the DB but hidden from the mask.
678 pub src_restricted: bool,
679 pub dst_key: String,
680 /// `true` when `dst_key` is in the DB but hidden from the mask.
681 pub dst_restricted: bool,
682 pub derived: bool,
683}
684
685/// Result of a mask-aware node lookup via [`GraphDb::node_info_masked`].
686///
687/// `None` from that method means the key does not exist (→ 404).
688/// `Some(Restricted)` is only produced when `mask.mode() == MaskMode::Stub`.
689#[derive(Debug, PartialEq)]
690pub enum MaskedNodeResult {
691 Visible(NodeInfo),
692 /// Node exists in the DB but is hidden from this mask.
693 Restricted,
694}
695
696/// One rule-owned edge between two nodes, with the rule name, edge type,
697/// direction (src_key → dst_key), and weight if the rule stores one.
698#[derive(Debug, Clone, PartialEq, Serialize)]
699pub struct Explanation {
700 pub rule: String,
701 pub edge_type: String,
702 pub src_key: String,
703 pub dst_key: String,
704 pub weight: Option<f64>,
705 pub predicate: PredicateSummary,
706 /// For a via-hop rule, the edge type the rule hops over to reach its
707 /// candidates. `None` for a plain two-node rule. A via-hop rule whose
708 /// `via_edge` is itself rule-derived is the chaining case: the hop edge
709 /// was written by another rule in the same commit.
710 #[serde(default)]
711 pub via_edge: Option<String>,
712}
713
714/// Report returned by [`GraphDb::backup_to`].
715#[derive(Debug, Clone)]
716pub struct BackupReport {
717 /// Filenames copied into the destination directory (sorted ascending).
718 pub files: Vec<String>,
719 /// Total bytes written across all copied files.
720 pub bytes: u64,
721 /// `true` when the destination opened cleanly and passed post-copy checks.
722 ///
723 /// For stores that have a `snapshot.bin` this means: all V8 section CRCs
724 /// matched **and** the destination opened without error.
725 ///
726 /// For WAL-only stores (no `snapshot.bin`) there is no snapshot to
727 /// CRC-check; `verified` is `true` when the destination opened and
728 /// replayed the WAL without error (record-level checksums in the WAL
729 /// provide the integrity signal, not section CRCs).
730 pub verified: bool,
731}
732
733/// One directed edge in export form, with optional rule attribution for derived edges.
734///
735/// Returned by [`GraphDb::all_edges_for_export`].
736///
737/// Does not derive `Eq`/`Ord`: `weight` is an `f64` and NaN breaks a total
738/// order. Callers that need a stable edge ordering already sort by
739/// `(edge_type, src, dst)` explicitly (see `all_edges_for_export`).
740#[derive(Debug, Clone, PartialEq, PartialOrd)]
741pub struct ExportEdge {
742 pub edge_type: String,
743 pub src: String,
744 pub dst: String,
745 pub derived: bool,
746 /// Rule name that created this edge, if derived. `None` for manual edges.
747 pub rule: Option<String>,
748 /// The creating rule's declared `weight_prop`, read off this edge, when
749 /// derived and numeric (`Int`/`Float`). `None` for manual edges, derived
750 /// edges whose rule declares no `weight_prop`, or a non-numeric value.
751 pub weight: Option<f64>,
752}
753
754/// One edge type's shape, as [`GraphDb::edge_type_census`] counts it.
755///
756/// Deliberately per *type* and not per edge: everything here is a summary a
757/// caller can print in one line, and none of it costs a record per edge.
758#[derive(Debug, Clone, PartialEq, Eq)]
759pub struct EdgeTypeCensus {
760 pub edge_type: String,
761 /// Directed edges of this type. Counted the way
762 /// [`GraphDb::edge_count`] counts: each edge once, from its source.
763 pub edges: u64,
764 /// Every label seen on a source of this type, sorted.
765 pub src_labels: Vec<String>,
766 /// Every label seen on a destination of this type, sorted.
767 pub dst_labels: Vec<String>,
768 /// The rules that declare this `edge_type`, sorted. Empty for a type
769 /// written by hand.
770 pub rules: Vec<String>,
771 /// `(src key, dst key)` of the first edge of this type in the store's own
772 /// id order — a real pair to quote in an example.
773 pub sample: Option<(String, String)>,
774}
775
776/// Construct the standard write-query result set (columns: created, properties_set, deleted).
777fn write_result_set() -> ResultSet {
778 ResultSet::new(vec![
779 "created".into(),
780 "properties_set".into(),
781 "deleted".into(),
782 ])
783}
784
785fn resolve_merge_set_value(op: &Operand, params: &BTreeMap<String, Value>) -> Result<Value> {
786 match op {
787 Operand::Lit(v) => Ok(v.clone()),
788 Operand::Param(name) => params
789 .get(name)
790 .cloned()
791 .ok_or_else(|| GraphError::QueryError {
792 detail: format!("missing parameter `{name}`"),
793 }),
794 _ => Err(GraphError::QueryError {
795 detail: "ON CREATE/ON MATCH SET value must be a literal or $parameter".into(),
796 }),
797 }
798}
799
800fn operand_node_vars(op: &Operand, out: &mut Vec<String>) {
801 match op {
802 Operand::Prop { var, .. } | Operand::Var(var) => {
803 if !out.contains(var) {
804 out.push(var.clone());
805 }
806 }
807 Operand::FuncCall { args, .. } => {
808 for arg in args {
809 operand_node_vars(arg, out);
810 }
811 }
812 Operand::BinArith { left, right, .. } => {
813 operand_node_vars(left, out);
814 operand_node_vars(right, out);
815 }
816 Operand::Case { branches, default } => {
817 // Branch conditions reference vars already bound (and mask-filtered)
818 // by the MATCH phase, so collecting from the value operands + ELSE
819 // is sufficient for RETURN-projection var discovery.
820 for (_, value) in branches {
821 operand_node_vars(value, out);
822 }
823 if let Some(d) = default {
824 operand_node_vars(d, out);
825 }
826 }
827 Operand::Index { base, index } => {
828 operand_node_vars(base, out);
829 operand_node_vars(index, out);
830 }
831 Operand::Lit(_) | Operand::Param(_) => {}
832 }
833}
834
835fn ret_node_vars(items: &[RetItem]) -> Vec<String> {
836 let mut out = Vec::new();
837 for item in items {
838 match &item.value {
839 RetVal::Var(v) | RetVal::Prop { var: v, .. } => {
840 if !out.contains(v) {
841 out.push(v.clone());
842 }
843 }
844 RetVal::FuncCall { args, .. } => {
845 for arg in args {
846 operand_node_vars(arg, &mut out);
847 }
848 }
849 RetVal::ScalarExpr(op) => operand_node_vars(op, &mut out),
850 RetVal::Agg { .. } => {}
851 }
852 }
853 out
854}
855
856fn add_var(out: &mut Vec<String>, v: &str) {
857 if !out.iter().any(|x| x == v) {
858 out.push(v.to_string());
859 }
860}
861
862fn pattern_node_vars(pats: &[Pattern]) -> Vec<String> {
863 let mut out = Vec::new();
864 for p in pats {
865 if let Some(v) = &p.start.var {
866 add_var(&mut out, v);
867 }
868 for (_, dest) in &p.chain {
869 if let Some(v) = &dest.var {
870 add_var(&mut out, v);
871 }
872 }
873 }
874 out
875}
876
877fn pattern_rel_vars(pats: &[Pattern]) -> Vec<String> {
878 let mut out = Vec::new();
879 for p in pats {
880 for (rel, _) in &p.chain {
881 if rel.hops.is_none() {
882 if let Some(v) = &rel.var {
883 add_var(&mut out, v);
884 }
885 }
886 }
887 }
888 out
889}
890
891fn rel_type_alias(var: &str) -> String {
892 format!("__rt_{var}")
893}
894
895fn ret_column_name(item: &RetItem) -> String {
896 if let Some(alias) = &item.alias {
897 return alias.clone();
898 }
899 // The same naming rule the planner and the executor use, so a
900 // write-statement RETURN names its columns exactly as a read query does.
901 // An aggregate is not legal in a write-statement RETURN; it keeps the
902 // placeholder it always had.
903 ret_val_label(&item.value).unwrap_or_else(|| "<agg>".to_string())
904}
905
906fn eval_set_return_operand<F: Fs>(
907 db: &GraphDb<F>,
908 match_rs: &ResultSet,
909 row: usize,
910 rel_vars: &[String],
911 op: &Operand,
912 params: &BTreeMap<String, Value>,
913) -> Result<Option<Value>> {
914 match op {
915 Operand::Lit(v) => Ok(Some(v.clone())),
916 Operand::Param(name) => params.get(name).cloned().ok_or_else(|| GraphError::QueryError {
917 detail: format!("missing parameter `{name}`"),
918 }).map(Some),
919 Operand::Var(name) if rel_vars.iter().any(|r| r == name) => Err(GraphError::QueryError {
920 detail: format!(
921 "cannot return relationship variable '{name}' bare; return its properties ({name}.field) instead"
922 ),
923 }),
924 Operand::Var(name) => Ok(match_rs.get(row, name).cloned()),
925 Operand::Prop { var, field } => {
926 if rel_vars.iter().any(|r| r == var) {
927 return Ok(None);
928 }
929 let Some(Value::Str(key)) = match_rs.get(row, var) else {
930 return Ok(None);
931 };
932 if let Some(v) = db.get_prop(key, field) {
933 return Ok(Some(v));
934 }
935 // Same stored-wins identity fallback as the read path:
936 // n.key / n.id / n.label, not only get_prop.
937 Ok(match field.as_str() {
938 "key" | "id" => Some(Value::Str(key.clone())),
939 "label" => db
940 .node_ref(key)
941 .map(|n| Value::Str(n.label().to_owned())),
942 _ => None,
943 })
944 }
945 Operand::FuncCall { name, args } => {
946 eval_set_return_func(db, match_rs, row, rel_vars, name, args, params)
947 }
948 Operand::BinArith { op, left, right } => {
949 let lv = eval_set_return_operand(db, match_rs, row, rel_vars, left, params)?;
950 let rv = eval_set_return_operand(db, match_rs, row, rel_vars, right, params)?;
951 eval_set_return_arith(op, lv, rv)
952 }
953 Operand::Case { branches, default } => {
954 for (cond, value) in branches {
955 if eval_set_return_expr(db, match_rs, row, rel_vars, cond, params, 0)? {
956 return eval_set_return_operand(db, match_rs, row, rel_vars, value, params);
957 }
958 }
959 match default {
960 Some(d) => eval_set_return_operand(db, match_rs, row, rel_vars, d, params),
961 None => Ok(None),
962 }
963 }
964 Operand::Index { base, index } => {
965 let base_val = eval_set_return_operand(db, match_rs, row, rel_vars, base, params)?;
966 let idx_val = eval_set_return_operand(db, match_rs, row, rel_vars, index, params)?;
967 Ok(core_query::value_ops::index_list(base_val, idx_val))
968 }
969 }
970}
971
972fn eval_set_return_expr<F: Fs>(
973 db: &GraphDb<F>,
974 match_rs: &ResultSet,
975 row: usize,
976 rel_vars: &[String],
977 expr: &Expr,
978 params: &BTreeMap<String, Value>,
979 depth: u32,
980) -> Result<bool> {
981 if depth > 256 {
982 return Err(GraphError::QueryError {
983 detail: "expression nesting too deep".into(),
984 });
985 }
986 match expr {
987 Expr::And(lhs, rhs) => {
988 let l = eval_set_return_expr(db, match_rs, row, rel_vars, lhs, params, depth + 1)?;
989 let r = eval_set_return_expr(db, match_rs, row, rel_vars, rhs, params, depth + 1)?;
990 Ok(l && r)
991 }
992 Expr::Or(lhs, rhs) => {
993 let l = eval_set_return_expr(db, match_rs, row, rel_vars, lhs, params, depth + 1)?;
994 let r = eval_set_return_expr(db, match_rs, row, rel_vars, rhs, params, depth + 1)?;
995 Ok(l || r)
996 }
997 Expr::Not(inner) => Ok(!eval_set_return_expr(
998 db,
999 match_rs,
1000 row,
1001 rel_vars,
1002 inner,
1003 params,
1004 depth + 1,
1005 )?),
1006 Expr::Cmp { lhs, op, rhs } => {
1007 let l = eval_set_return_operand(db, match_rs, row, rel_vars, lhs, params)?;
1008 let r = eval_set_return_operand(db, match_rs, row, rel_vars, rhs, params)?;
1009 match (l, r) {
1010 (Some(a), Some(b)) => Ok(eval_cmp(op, &a, &b)),
1011 _ => Ok(false),
1012 }
1013 }
1014 Expr::Truthy(op) => {
1015 let val = eval_set_return_operand(db, match_rs, row, rel_vars, op, params)?;
1016 Ok(match val {
1017 None => false,
1018 Some(Value::Bool(b)) => b,
1019 Some(Value::Int(n)) => n != 0,
1020 Some(Value::Float(f)) => f != 0.0,
1021 Some(Value::Str(s)) => !s.is_empty(),
1022 Some(Value::List(v)) => !v.is_empty(),
1023 Some(Value::Map(m)) => !m.is_empty(),
1024 })
1025 }
1026 Expr::IsNull(op) => {
1027 let val = eval_set_return_operand(db, match_rs, row, rel_vars, op, params)?;
1028 Ok(val.is_none())
1029 }
1030 Expr::IsNotNull(op) => {
1031 let val = eval_set_return_operand(db, match_rs, row, rel_vars, op, params)?;
1032 Ok(val.is_some())
1033 }
1034 Expr::In { expr, list } => {
1035 let Some(needle) = eval_set_return_operand(db, match_rs, row, rel_vars, expr, params)?
1036 else {
1037 return Ok(false);
1038 };
1039 for item_op in list {
1040 match eval_set_return_operand(db, match_rs, row, rel_vars, item_op, params)? {
1041 None => {}
1042 Some(Value::List(items)) => {
1043 for item in items {
1044 if eval_cmp(&core_query::CmpOp::Eq, &needle, &item) {
1045 return Ok(true);
1046 }
1047 }
1048 }
1049 Some(item) if eval_cmp(&core_query::CmpOp::Eq, &needle, &item) => {
1050 return Ok(true);
1051 }
1052 Some(_) => {}
1053 }
1054 }
1055 Ok(false)
1056 }
1057 }
1058}
1059
1060fn eval_set_return_arith(
1061 op: &ArithOp,
1062 lv: Option<Value>,
1063 rv: Option<Value>,
1064) -> Result<Option<Value>> {
1065 match (lv, rv) {
1066 (None, _) | (_, None) => Ok(None),
1067 (Some(Value::Int(a)), Some(Value::Int(b))) => {
1068 let result = match op {
1069 ArithOp::Sub => a.saturating_sub(b),
1070 ArithOp::Mul => a.saturating_mul(b),
1071 ArithOp::Add => a.saturating_add(b),
1072 ArithOp::Div => {
1073 if b == 0 {
1074 return Err(GraphError::QueryError {
1075 detail: "division by zero".into(),
1076 });
1077 }
1078 a.checked_div(b).unwrap_or(i64::MAX)
1079 }
1080 };
1081 Ok(Some(Value::Int(result)))
1082 }
1083 (Some(lv), Some(rv)) => {
1084 let a = match &lv {
1085 Value::Float(f) => *f,
1086 Value::Int(i) => *i as f64,
1087 _ => {
1088 return Err(GraphError::QueryError {
1089 detail: format!("arithmetic operand must be numeric, got {lv:?}"),
1090 })
1091 }
1092 };
1093 let b = match &rv {
1094 Value::Float(f) => *f,
1095 Value::Int(i) => *i as f64,
1096 _ => {
1097 return Err(GraphError::QueryError {
1098 detail: format!("arithmetic operand must be numeric, got {rv:?}"),
1099 })
1100 }
1101 };
1102 let result = match op {
1103 ArithOp::Sub => a - b,
1104 ArithOp::Mul => a * b,
1105 ArithOp::Add => a + b,
1106 ArithOp::Div => {
1107 if b == 0.0 {
1108 return Err(GraphError::QueryError {
1109 detail: "division by zero".into(),
1110 });
1111 }
1112 a / b
1113 }
1114 };
1115 Ok(Some(Value::Float(result)))
1116 }
1117 }
1118}
1119
1120fn eval_set_return_func<F: Fs>(
1121 db: &GraphDb<F>,
1122 match_rs: &ResultSet,
1123 row: usize,
1124 rel_vars: &[String],
1125 name: &str,
1126 args: &[Operand],
1127 params: &BTreeMap<String, Value>,
1128) -> Result<Option<Value>> {
1129 let norm = name.to_ascii_lowercase();
1130 if norm == "type" {
1131 if args.len() != 1 {
1132 return Err(GraphError::QueryError {
1133 detail: format!("type() requires exactly 1 argument, got {}", args.len()),
1134 });
1135 }
1136 let Operand::Var(rel) = &args[0] else {
1137 return Err(GraphError::QueryError {
1138 detail: "type() argument must be a relationship variable (e.g. type(r))".into(),
1139 });
1140 };
1141 return Ok(match_rs.get(row, &rel_type_alias(rel)).cloned());
1142 }
1143 if norm == "key" || norm == "id" {
1144 let fname = if norm == "id" { "id" } else { "key" };
1145 if args.len() != 1 {
1146 return Err(GraphError::QueryError {
1147 detail: format!("{fname}() requires exactly 1 argument, got {}", args.len()),
1148 });
1149 }
1150 let Operand::Var(var) = &args[0] else {
1151 return Err(GraphError::QueryError {
1152 detail: format!("{fname}() argument must be a node variable (e.g. {fname}(n))"),
1153 });
1154 };
1155 if rel_vars.iter().any(|r| r == var) {
1156 return Err(GraphError::QueryError {
1157 detail: format!("{fname}() argument `{var}` is a relationship, not a node"),
1158 });
1159 }
1160 // MATCH rows bind node variables to their key string, so the column
1161 // value *is* the key. `id()` aliases `key()`.
1162 return Ok(match_rs.get(row, var).cloned());
1163 }
1164 let mut vals = Vec::with_capacity(args.len());
1165 for arg in args {
1166 vals.push(eval_set_return_operand(
1167 db, match_rs, row, rel_vars, arg, params,
1168 )?);
1169 }
1170 match norm.as_str() {
1171 "tolower" => {
1172 if vals.len() != 1 {
1173 return Err(GraphError::QueryError {
1174 detail: format!("toLower() requires exactly 1 argument, got {}", vals.len()),
1175 });
1176 }
1177 Ok(vals[0].clone().map(|val| match val {
1178 Value::Str(s) => Value::Str(s.to_ascii_lowercase()),
1179 other => other,
1180 }))
1181 }
1182 "toupper" => {
1183 if vals.len() != 1 {
1184 return Err(GraphError::QueryError {
1185 detail: format!("toUpper() requires exactly 1 argument, got {}", vals.len()),
1186 });
1187 }
1188 Ok(vals[0].clone().map(|val| match val {
1189 Value::Str(s) => Value::Str(s.to_ascii_uppercase()),
1190 other => other,
1191 }))
1192 }
1193 "size" => match vals.first().cloned().flatten() {
1194 None => Ok(None),
1195 Some(Value::Str(s)) => Ok(Some(Value::Int(s.len() as i64))),
1196 Some(Value::List(items)) => Ok(Some(Value::Int(items.len() as i64))),
1197 Some(_) => Ok(None),
1198 },
1199 "coalesce" => Ok(vals.into_iter().flatten().next()),
1200 "abs" => match vals.first().cloned().flatten() {
1201 None => Ok(None),
1202 Some(Value::Int(n)) => Ok(Some(Value::Int(n.saturating_abs()))),
1203 Some(Value::Float(f)) => Ok(Some(Value::Float(f.abs()))),
1204 Some(_) => Ok(None),
1205 },
1206 "round" => match vals.first().cloned().flatten() {
1207 None => Ok(None),
1208 Some(Value::Float(f)) => Ok(Some(Value::Float(f.round()))),
1209 Some(Value::Int(n)) => Ok(Some(Value::Int(n))),
1210 Some(_) => Ok(None),
1211 },
1212 "decay" => {
1213 if vals.len() != 3 {
1214 return Err(GraphError::QueryError {
1215 detail: format!("decay() requires exactly 3 arguments, got {}", vals.len()),
1216 });
1217 }
1218 match (vals[0].clone(), vals[1].clone(), vals[2].clone()) {
1219 (None, _, _) | (_, None, _) | (_, _, None) => Ok(None),
1220 (Some(b), Some(a), Some(h)) => {
1221 let numeric = |v: Value| -> Result<f64> {
1222 match v {
1223 Value::Int(n) => Ok(n as f64),
1224 Value::Float(f) => Ok(f),
1225 other => Err(GraphError::QueryError {
1226 detail: format!(
1227 "decay() requires numeric arguments, got {other:?}"
1228 ),
1229 }),
1230 }
1231 };
1232 let b = numeric(b)?;
1233 let a = numeric(a)?;
1234 let h = numeric(h)?;
1235 if h <= 0.0 {
1236 return Err(GraphError::QueryError {
1237 detail: "decay() requires halflife > 0".into(),
1238 });
1239 }
1240 Ok(Some(Value::Float(b * 0.5f64.powf(a / h))))
1241 }
1242 }
1243 }
1244 _ => Err(GraphError::QueryError {
1245 detail: format!(
1246 "unknown function `{name}`; supported: toLower, toUpper, size, coalesce, type, abs, round, decay, key, id"
1247 ),
1248 }),
1249 }
1250}
1251
1252fn eval_set_return_item<F: Fs>(
1253 db: &GraphDb<F>,
1254 match_rs: &ResultSet,
1255 row: usize,
1256 rel_vars: &[String],
1257 item: &RetItem,
1258 params: &BTreeMap<String, Value>,
1259) -> Result<Option<Value>> {
1260 match &item.value {
1261 RetVal::Var(v) => eval_set_return_operand(
1262 db,
1263 match_rs,
1264 row,
1265 rel_vars,
1266 &Operand::Var(v.clone()),
1267 params,
1268 ),
1269 RetVal::Prop { var, field } => eval_set_return_operand(
1270 db,
1271 match_rs,
1272 row,
1273 rel_vars,
1274 &Operand::Prop {
1275 var: var.clone(),
1276 field: field.clone(),
1277 },
1278 params,
1279 ),
1280 RetVal::FuncCall { name, args } => {
1281 eval_set_return_func(db, match_rs, row, rel_vars, name, args, params)
1282 }
1283 RetVal::ScalarExpr(op) => eval_set_return_operand(db, match_rs, row, rel_vars, op, params),
1284 RetVal::Agg { .. } => Err(GraphError::QueryError {
1285 detail: "aggregates are not supported in MATCH … SET … RETURN".into(),
1286 }),
1287 }
1288}
1289
1290/// Project user RETURN from original MATCH rows after SET. No rematch.
1291fn project_set_return_rows<F: Fs>(
1292 db: &GraphDb<F>,
1293 rel_vars: &[String],
1294 match_rs: &ResultSet,
1295 returns: &[RetItem],
1296 params: &BTreeMap<String, Value>,
1297) -> Result<ResultSet> {
1298 let columns: Vec<String> = returns.iter().map(ret_column_name).collect();
1299 let mut out = ResultSet::new(columns);
1300 for row in 0..match_rs.len() {
1301 let mut cells = Vec::with_capacity(returns.len());
1302 for item in returns {
1303 cells.push(eval_set_return_item(
1304 db, match_rs, row, rel_vars, item, params,
1305 )?);
1306 }
1307 out.push_row(cells);
1308 }
1309 Ok(out)
1310}
1311
1312/// Single construction point for a `GraphMut` view over the split-borrowed graph fields.
1313/// Callers use `std::mem::take` on the engine before calling this, then restore it after.
1314/// Extract a `Vec<f64>` from a `Value::List` whose items are all numeric.
1315/// Returns `None` for non-list values or lists with non-numeric elements.
1316/// Extra candidates pulled from an approximate index before re-scoring, over and
1317/// above the `k` asked for.
1318///
1319/// The index orders candidates by `f32` distances, which agree with the exact
1320/// `f64` cosine to about 1e-6. Re-scoring can therefore only reshuffle
1321/// candidates inside a band that narrow — it cannot move a hit past one that is
1322/// further away by more than 1e-6 — so the only way a true top-`k` member can be
1323/// lost is if the index ranked it just outside `k` on the `f32` order. Fetching
1324/// `k + 16` covers any such band up to 16 members wide, which at 1e-6 means 16
1325/// vectors within a millionth of each other in cosine: a duplicate cluster, and
1326/// then the members are interchangeable anyway. `min` is applied to the exact
1327/// score, never to the index's, so a hit sitting on the threshold is decided
1328/// exactly.
1329const VECTOR_RESCORE_MARGIN: usize = 16;
1330
1331/// Cosine similarity between an already-unit query and node `id`'s `field`
1332/// vector, read from the **`f64`** properties. `None` when the node has no
1333/// numeric-list vector there, or its norm is zero.
1334///
1335/// The single definition of the score this API reports. Both the brute-force
1336/// scan and the re-scoring step that follows an index lookup go through it, so
1337/// the two paths cannot disagree — which is the property
1338/// `index_and_brute_force_agree_on_scores` pins.
1339fn exact_vector_similarity(
1340 view: &GraphView<'_>,
1341 id: u32,
1342 field: &str,
1343 q_unit: &[f64],
1344) -> Option<f64> {
1345 let v = view.prop(id, field)?;
1346 let xs = value_as_float_list(&v.into_value())?;
1347 let v_norm: f64 = xs.iter().map(|x| x * x).sum::<f64>().sqrt();
1348 if v_norm == 0.0 {
1349 return None;
1350 }
1351 Some(
1352 q_unit
1353 .iter()
1354 .zip(xs.iter())
1355 .map(|(a, b)| a * (b / v_norm))
1356 .sum(),
1357 )
1358}
1359
1360fn value_as_float_list(v: &Value) -> Option<Vec<f64>> {
1361 match v {
1362 Value::List(items) => items
1363 .iter()
1364 .map(|item| match item {
1365 Value::Float(f) => Some(*f),
1366 Value::Int(i) => Some(*i as f64),
1367 _ => None,
1368 })
1369 .collect(),
1370 _ => None,
1371 }
1372}
1373
1374fn make_graph_mut<'a>(
1375 ids: &'a IdMap,
1376 syms: &'a mut Interner,
1377 labels: &'a [u32],
1378 props: core_storage::v8::seam::ColumnsView<'a>,
1379 topo: &'a mut Topology,
1380 base: &'a Option<std::sync::Arc<core_storage::v8::MappedBase>>,
1381 edge_props: &'a mut EdgeProps,
1382) -> GraphMut<'a> {
1383 GraphMut {
1384 ids,
1385 syms,
1386 labels,
1387 props,
1388 topo,
1389 base_topo: base_csr(base),
1390 edge_props,
1391 }
1392}
1393
1394/// The archived CSR of an open V8 snapshot, for the rule engine's graph reads.
1395///
1396/// A store opened from a snapshot keeps its edges in the mapping and its
1397/// overlay empty, so a rule that reads the graph's shape has to see both.
1398fn base_csr(
1399 base: &Option<std::sync::Arc<core_storage::v8::MappedBase>>,
1400) -> Option<&core_storage::v8::layout::ArchivedCsr> {
1401 base.as_ref().map(|b| {
1402 b.topology()
1403 .expect("base topology section bounds validated at open")
1404 })
1405}
1406
1407/// Build a `ColumnsView` from the disjoint `props` overlay and optional V8 base.
1408///
1409/// Takes explicit field references rather than `&self` so the caller can hold
1410/// simultaneous mutable borrows of other fields (e.g. `syms`, `topo`).
1411fn build_props_view<'a>(
1412 props: &'a ColumnStore,
1413 base: &'a Option<std::sync::Arc<core_storage::v8::MappedBase>>,
1414) -> core_storage::v8::seam::ColumnsView<'a> {
1415 match base {
1416 None => core_storage::v8::seam::ColumnsView::owned(props),
1417 Some(b) => {
1418 let archived = b
1419 .columns()
1420 .expect("base columns section bounds validated at open");
1421 core_storage::v8::seam::ColumnsView::with_base_cached(props, archived, b.mixed_cache())
1422 .with_shared_strings(base_string_table(b))
1423 }
1424 }
1425}
1426
1427/// The base columns section paired with the string table that resolves its
1428/// string ids — what `ViewStore` needs to read a neighbour's string property
1429/// out of a V9 snapshot.
1430fn base_columns(
1431 base: &Option<std::sync::Arc<core_storage::v8::MappedBase>>,
1432) -> Option<core_storage::v8::seam::BaseColumns<'_>> {
1433 base.as_ref().map(|b| core_storage::v8::seam::BaseColumns {
1434 cols: b
1435 .columns()
1436 .expect("base columns section bounds validated at open"),
1437 strings: base_string_table(b),
1438 })
1439}
1440
1441/// The shared string table of a V9 base, or `None` for a pre-V9 one.
1442///
1443/// Every `ColumnsView` built over a base must carry it: without it a V9
1444/// snapshot's string columns, whose own tables are empty, read back as absent.
1445fn base_string_table(
1446 base: &core_storage::v8::MappedBase,
1447) -> Option<&core_storage::v8::layout::ArchivedStringTable> {
1448 base.string_table()
1449 .transpose()
1450 .expect("base strings section bounds validated at open")
1451}
1452
1453fn build_topo_view<'a>(
1454 overlay: &'a Topology,
1455 base: &'a Option<std::sync::Arc<core_storage::v8::MappedBase>>,
1456) -> core_storage::v8::seam::TopologyView<'a> {
1457 match base {
1458 None => core_storage::v8::seam::TopologyView::owned(overlay),
1459 Some(b) => {
1460 let archived_csr = b
1461 .topology()
1462 .expect("base topology section bounds validated at open");
1463 core_storage::v8::seam::TopologyView::with_base(overlay, archived_csr)
1464 }
1465 }
1466}
1467
1468/// When [`GraphDb`] calls `Fs::sync` after a WAL append.
1469///
1470/// Default is [`Strict`](FsyncPolicy::Strict): every `log_then_apply_with`
1471/// fsyncs (single `insert_node` / `set_prop`). Ingest and `write_batch`
1472/// emit one `WalRecord::Batch` and fsync once at that frame (Batched).
1473/// [`Relaxed`](FsyncPolicy::Relaxed) skips WAL sync; [`GraphDb::snapshot`]
1474/// is still durable via `write_atomic`. Crash-recovery DST stays Strict.
1475#[derive(Clone, Copy, Debug, PartialEq, Eq, Default)]
1476pub enum FsyncPolicy {
1477 /// Every WAL commit calls `fs.sync` (today's behavior).
1478 #[default]
1479 Strict,
1480 /// Sync only at a `Batch` frame end. Single-op path stays Strict unless
1481 /// this policy is set on the database.
1482 Batched,
1483 /// Never call `fs.sync`. [`GraphDb::snapshot`] still syncs via `write_atomic`.
1484 Relaxed,
1485}
1486
1487/// A precondition for a compare-and-set batch write.
1488///
1489/// All preconditions in a [`GraphDb::write_batch_cas`] or
1490/// [`crate::SharedDb::submit_batch_cas`] call are checked atomically before
1491/// any operation in the batch is applied. If any precondition fails, the
1492/// entire batch is rejected with [`GraphError::CasConflict`] and no WAL frame
1493/// is written.
1494///
1495/// # Touch definition
1496///
1497/// A node's last-change commit (`last_changed`) is updated when any of the
1498/// following state-changing WAL records touch it:
1499///
1500/// - `InsertNode` / `InsertNodeId` — the newly-inserted node.
1501/// - `SetProp` / `SetPropId` / `RemoveProp` — the property-bearing node.
1502/// - `InsertEdge` / `InsertEdgeId` / `DeleteEdge` — **both** src and dst
1503/// endpoints (an edge change touches both sides).
1504/// - `DeleteNode` — the node is tombstoned; `last_changed` returns `None`
1505/// for deleted keys so the pre-deletion entry is never observed.
1506///
1507/// History markers (`DerivedEdgeAdded` / `DerivedEdgeRetracted`) are
1508/// state no-ops. The underlying mutation that triggered rule firing already
1509/// updated the relevant nodes' last-change entries. Rule-management records
1510/// (`CreateRule`, `DeleteRule`, `RebuildRule`) and view/full-text declarations
1511/// do not touch any node's last-change.
1512#[derive(Debug, Clone, PartialEq, Eq)]
1513pub enum Precondition {
1514 /// The node's last-change commit must equal `expected`.
1515 ///
1516 /// Fails with [`GraphError::CasConflict`] when:
1517 /// - The node does not exist (`last_changed` returns `None`), or
1518 /// - The recorded commit seq does not match `expected`.
1519 NodeUnchangedSince { key: String, expected: u64 },
1520 /// The node must not exist (not inserted, or already deleted).
1521 ///
1522 /// Fails with [`GraphError::CasConflict`] (expected=`u64::MAX`,
1523 /// actual=`last_changed(key).unwrap_or(0)`) when the node is live.
1524 NodeAbsent { key: String },
1525}
1526
1527pub struct GraphDb<F: Fs> {
1528 fs: F,
1529 ids: Arc<IdMap>,
1530 syms: Arc<Interner>,
1531 topo: Arc<Topology>,
1532 props: Arc<ColumnStore>,
1533 labels: Arc<Vec<u32>>, // node id -> label symbol
1534 /// Namespace names by index; index [`NS_DEFAULT_IDX`] is always
1535 /// [`NS_DEFAULT`]. Derived beside [`Self::node_ns`], never persisted.
1536 ///
1537 /// A private table rather than the shared [`Interner`]: interning
1538 /// `"default"` at open would add a symbol to the store's symbol table and
1539 /// change the bytes of the next snapshot of a store that has no namespaces
1540 /// at all.
1541 ns_names: Vec<String>,
1542 /// Namespace index per dense node id, into [`Self::ns_names`];
1543 /// [`NS_DEFAULT_IDX`] for a node with no `ns` property.
1544 ///
1545 /// Derived: built by one pass over the `ns` column at open (which reads
1546 /// nothing when the column does not exist) and maintained at every node
1547 /// insert. Never written to a snapshot or the WAL, because the property it
1548 /// mirrors already is. A namespace cannot change, so no other record shape
1549 /// can move a node between namespaces.
1550 node_ns: Vec<u32>,
1551 edge_props: Arc<EdgeProps>,
1552 engine: RuleEngine,
1553 view_store: ViewStore,
1554 /// Incremental inverted index for full-text-lite search.
1555 /// Rebuild-on-open: populated from WAL replay + rebuild_all at open end.
1556 fulltext: Arc<FulltextIndex>,
1557 /// Opt-in equality index over scalar node properties.
1558 /// Rebuild-on-open: declarations replay from the WAL, postings rebuild at
1559 /// open end (mirrors `fulltext`).
1560 prop_index: PropertyIndex,
1561 /// Whether this store records insert-count multiplicity (§5.13).
1562 ///
1563 /// Declared like `prop_index`'s enabled pairs — a WAL record replayed at
1564 /// open, re-emitted into the baseline by a truncating snapshot — but it
1565 /// gates a *format* step rather than an index: `WalRecord::SetEdgeCount`
1566 /// (discriminant 23) is written only when this is `true`, so a store that
1567 /// never opts in stays readable by a binary that predates the record.
1568 multiplicity: bool,
1569 event_sink: Option<Box<dyn Fn(MutationEvent) + Send + Sync>>,
1570 /// WAL fsync cadence. Default [`FsyncPolicy::Strict`].
1571 fsync: FsyncPolicy,
1572 /// Monotonically increasing per-commit counter. A single `log_then_apply_with`
1573 /// call increments this once; all events emitted from that call share the same
1574 /// `commit_seq` value.
1575 commit_seq: u64,
1576 /// Commit → wall-clock map, loaded from the `commit_times.bin` sidecar at
1577 /// open and appended to by `log_then_apply_with` — the one place a commit
1578 /// is born. Replay does **not** stamp: `apply_frames` re-applies commits
1579 /// that already happened, and `SystemTime::now()` there would record replay
1580 /// time as commit time. Empty on a store written before v0.6.11, which
1581 /// makes every date query answer `NoRecordedTime` rather than guess.
1582 commit_times: core_storage::commit_times::CommitTimes,
1583 /// When set, subsequent commits are recorded at this instant instead of the
1584 /// system clock.
1585 ///
1586 /// Sticky on purpose. A backfill replays history that happened over months,
1587 /// and a day's worth of rows genuinely share one instant — a one-shot flag
1588 /// would mean setting it before every row of a bulk load, and forgetting one
1589 /// would stamp that row "now" in the middle of 2026-06. Sticky makes the
1590 /// failure visible instead: forget to move it and every commit carries the
1591 /// same timestamp, which a date query answers oddly and an inspection shows
1592 /// at once.
1593 commit_time_override: Option<i64>,
1594 /// `true` only while the open path is replaying, where `load_from_disk`
1595 /// calls `fulltext.rebuild_all` unconditionally afterwards.
1596 ///
1597 /// Replaying an `EnableFulltext` record backfills its pair with a full
1598 /// `0..ids.len()` scan, and every snapshot re-emits one such record per
1599 /// enabled pair — so on a snapshotted store the open does that scan once per
1600 /// pair and then `rebuild_all` clears every posting and does it all again.
1601 /// The backfill is pure waste *when a rebuild follows*, which is true of the
1602 /// open path and **false** of `refresh()`: refresh applies peer frames and
1603 /// then only folds, so its backfill is the only thing that indexes them.
1604 fulltext_rebuild_follows: bool,
1605 /// `true` when `commit_times.bin` was present but would not decode.
1606 ///
1607 /// Mirrors `roles: None`: a damaged map must not read as "this store
1608 /// records no times", because that is also what an honest pre-v0.6.11 store
1609 /// says. Date queries answer `Corrupt` instead, and nothing is appended to
1610 /// a file already known to be damaged.
1611 commit_times_poisoned: bool,
1612 /// RBAC role definitions loaded from `roles.json` at open.
1613 ///
1614 /// `Some(roles)` — loaded successfully (may be empty when no roles are defined).
1615 /// `None` — `roles.json` was present but corrupt; `mask_for_role` returns
1616 /// `Err` for any request (fail-loud, never silently grant empty visibility).
1617 roles: Option<Vec<RoleDef>>,
1618 /// Memo for [`mask_for_role`](GraphDb::mask_for_role), keyed by
1619 /// `(role, commit_seq)` — a scoped reader between two writes resolves once.
1620 ///
1621 /// Shared by `Arc` with every [`ReaderSnapshot`](crate::reader::ReaderSnapshot)
1622 /// taken from this handle. Replaced (not cleared) whenever the role
1623 /// definitions change or the store is reloaded, which `commit_seq` does not
1624 /// record; see [`RoleMaskCache`](crate::mask::RoleMaskCache).
1625 role_masks: Arc<crate::mask::RoleMaskCache>,
1626 /// Which loaded store this handle is, for memos that outlive it.
1627 ///
1628 /// `role_masks` needs no such thing — the handle owns it and replaces it —
1629 /// but a [`Scope`](crate::mask::Scope) is the caller's, so its resolved key
1630 /// leg is stamped with this alongside `commit_seq`. Minted fresh here and
1631 /// again in [`reset_for_reload`](GraphDb::reset_for_reload), at exactly the
1632 /// two points a fresh `RoleMaskCache` is installed; see
1633 /// [`StoreStamp`](crate::mask::StoreStamp) for the invariant.
1634 store_id: crate::mask::StoreId,
1635 /// Live subscriptions. Entries with a dead `Weak` are pruned on the next
1636 /// distribute_events call.
1637 subscriptions: Vec<SubEntry>,
1638 /// Live query subscriptions. Re-executed on every commit when non-empty.
1639 /// Dead `Weak` entries are pruned inside `distribute_events`.
1640 query_subscriptions: Vec<QuerySubEntry>,
1641 /// Queue capacity for new subscriptions created by this db. Default is
1642 /// [`DEFAULT_SUB_CAPACITY`]; can be overridden via [`set_sub_capacity`]
1643 /// to test Lagged behaviour with small queues.
1644 sub_capacity: usize,
1645 /// True for as-of instances opened via [`GraphDb::open_at`].
1646 /// Every mutation method and `snapshot()` returns [`GraphError::ReadOnly`]
1647 /// when this flag is set.
1648 read_only: bool,
1649 /// Total WAL commit count at the time [`open_at`] was called.
1650 /// 0 for normal (non-as-of) instances.
1651 total_wal_commits: u64,
1652 /// Immutable mmap-backed base snapshot (V8). When `Some`, `self.topo` is
1653 /// the WAL-replay overlay (empty at open time, populated by apply()) and
1654 /// reads go through a merged `TopologyView`. `self.props` is always
1655 /// fully materialized (base + WAL replay) for HNSW/IVF and view compat.
1656 base: Option<Arc<core_storage::v8::MappedBase>>,
1657 // ── MVCC epoch reader state ───────────────────────────────────────────────
1658 /// Most-recent full overlay clone. Initialized at end of `open_with` /
1659 /// `open_at_with`; refreshed every `FOLD_EVERY_K` commits.
1660 /// `None` only between struct creation and the first fold.
1661 fold_overlay: Option<Arc<crate::reader::FrozenOverlay>>,
1662 /// Per-commit deltas accumulated since the last fold.
1663 delta_tail: Vec<Arc<crate::reader::CommitDelta>>,
1664 /// How many commits have occurred since the last fold.
1665 commits_since_fold: usize,
1666 /// When true, `log_then_apply_with` buffers event notifications instead of
1667 /// firing them immediately. Used by the group-commit drain thread to defer
1668 /// events until after the group fsync (R2: durability before notification).
1669 /// Cleared to false once the drain thread flushes or discards the buffer.
1670 defer_events: bool,
1671 /// Buffered events accumulated while `defer_events` is true.
1672 deferred_events: Vec<DeferredEvent>,
1673 /// Set to true by the group-commit drain thread when a group fsync fails
1674 /// after WAL truncation. All subsequent mutation attempts return an IO
1675 /// error until the database is reopened.
1676 degraded: bool,
1677 /// Set to `true` after `ensure_v8_base_sections_loaded` has read provenance,
1678 /// HNSW, and IVF sections from the mmap base into the engine's retained
1679 /// fields. `false` on all opens until first use; always `true` for non-V8
1680 /// opens (base is None, fast-path sets flag immediately).
1681 v8_sections_loaded: std::sync::atomic::AtomicBool,
1682 /// Serializes the one-time section population in `ensure_v8_base_sections_loaded`.
1683 v8_sections_mutex: std::sync::Mutex<()>,
1684 /// Per-node last-change commit sequence. `last_change[node_id] = seq` means
1685 /// the node was last modified by commit `seq`.
1686 ///
1687 /// Loaded from V8 section 11 at open; updated on every state-changing commit
1688 /// and WAL replay frame. V5-V7 stores start with an empty map; pre-WAL-horizon
1689 /// nodes return `None` from `last_changed` until they are next mutated.
1690 ///
1691 /// See [`Precondition`] for the full touch definition.
1692 last_change: HashMap<u32, u64>,
1693 /// WAL archive retention policy set by [`set_wal_archive_retention`].
1694 /// `None` = unlimited (keep all archives); `Some(N)` = keep N newest archives,
1695 /// pruning older ones at snapshot time. 0 is treated as unlimited.
1696 wal_archive_retention: Option<u32>,
1697 /// Global frame index of the first commit that is still reachable through
1698 /// surviving archives. Persisted to `wal.floor` sidecar when pruning occurs.
1699 /// Default 0 = all history reachable.
1700 wal_horizon_floor: u64,
1701 /// True when the surviving archive chain forms a continuous WAL history
1702 /// starting from the store's first commit (the genesis chain).
1703 ///
1704 /// `open_at` may replay archive-resident commits from empty state only when
1705 /// this flag is true AND `wal_horizon_floor == 0`. Cleared whenever:
1706 /// - a WAL-truncating snapshot (`keep_wal=false`) is taken after archives
1707 /// already exist (breaks the chain for subsequent archives), or
1708 /// - any archive is pruned (floor advances past zero).
1709 ///
1710 /// Persisted via the `wal.genesis` marker file; loaded from it at open.
1711 archive_genesis_chain: bool,
1712 /// True when this handle can *prove* the live WAL has never been truncated:
1713 /// there was no `snapshot.bin` when it opened the store, and it has taken no
1714 /// truncating snapshot since.
1715 ///
1716 /// The archive path's genesis check asks "did a snapshot exist before this
1717 /// one?" as a proxy for "was the WAL ever truncated". The proxy is sound
1718 /// across sessions — this binary cannot tell a history-preserving snapshot
1719 /// from a truncating one once the handle that took it is gone — but inside
1720 /// one session it is not, and `enable_multiplicity` made that visible: its
1721 /// forced `keep_wal` snapshot left the WAL entirely intact and yet
1722 /// permanently disqualified the store from ever receiving a genesis marker
1723 /// (defect #23). This flag is what the proxy defers to when the answer is
1724 /// actually known.
1725 snapshot_preserved_history: bool,
1726 /// Transient write-authz context set by `write_batch_authz` /
1727 /// `query_write_authz` for the duration of ONE mutation call.
1728 /// Always `None` at rest. Never serialized, never WAL-replayed.
1729 pending_write_authz: Option<WriteAuthz>,
1730 /// Slow-query threshold in milliseconds. 0 = disabled.
1731 /// Seeded from `MUSHROOMDB_SLOW_QUERY_MS` at open; override via
1732 /// [`GraphDb::set_slow_query_threshold_ms`] (tests must use the setter
1733 /// — env vars are process-global and race parallel test threads).
1734 slow_query_threshold_ms: u64,
1735 /// Ring buffer of recent slow queries (interior-mutable so `query(&self)`
1736 /// can record entries without requiring `&mut self`).
1737 slow_queries: std::sync::Mutex<SlowQueryLog>,
1738 /// `(field, label, caller)` triples whose exact-versus-approximate
1739 /// ambiguity this handle has already explained once. See
1740 /// [`note_ambiguous_exactness`](GraphDb::note_ambiguous_exactness).
1741 /// The caller shape is part of the key because the two shapes give
1742 /// different advice — silencing one with the other would leave a caller
1743 /// reading advice meant for a signature it does not have.
1744 /// Advice bookkeeping, not graph state: a reload keeps it, as the
1745 /// slow-query log does.
1746 warned_ambiguous_exactness: std::sync::Mutex<HashSet<(String, String, ExactnessCaller)>>,
1747 /// Instant at which the database was opened (used by `/metrics` uptime).
1748 started_at: std::time::Instant,
1749 // ── Multi-process state (cross-process lock + WAL tailing) ────────────────
1750 /// Byte offset of the WAL prefix already applied to in-memory state.
1751 ///
1752 /// Advanced by exactly the encoded length of every frame this handle
1753 /// appends, and by the decoded byte count of every tail
1754 /// [`refresh`](GraphDb::refresh) absorbs. Rewound by
1755 /// [`set_wal_consumed`](GraphDb::set_wal_consumed) when the group-commit
1756 /// drain thread truncates a failed group. Compared against the WAL's
1757 /// on-disk length to decide staleness.
1758 wal_consumed: u64,
1759 /// The **global 0-based frame index the next appended WAL frame will
1760 /// occupy** — `wal_horizon_floor` plus every frame currently reachable
1761 /// through archives and the live WAL.
1762 ///
1763 /// This is the space every history surface addresses: `edges_at`,
1764 /// `was_linked`, both history readouts and `open_at` all index the sequence
1765 /// [`all_frames`](GraphDb::all_frames) returns, and
1766 /// [`wal_total_commits`](GraphDb::wal_total_commits) counts it.
1767 ///
1768 /// It exists because **a commit is not a frame**. `commit_seq` counts
1769 /// commits; a commit whose rules fire appends a *second* frame — the
1770 /// derived-edge history marker — that no counter of commits ever sees. The
1771 /// two diverge by one frame per rule-firing commit, cumulatively, so
1772 /// deriving a frame index from `commit_seq` under-reports by more and more
1773 /// as history grows and resolves every date to an earlier graph. Silently:
1774 /// an older graph is a plausible answer, not an error.
1775 ///
1776 /// Maintained in lockstep with [`wal_consumed`](GraphDb::wal_consumed) —
1777 /// the same appends advance both, one in frames and one in bytes — so the
1778 /// two are seeded and rewound at exactly the same places. Keep it that way.
1779 wal_frames_written: u64,
1780 /// Identity of the snapshot this handle's base state came from, as
1781 /// `(len, mtime_nanos)`. A different value means another process replaced
1782 /// the snapshot and the WAL no longer continues our state: refresh reloads.
1783 snapshot_ident: Option<(u64, u64)>,
1784 /// The options this handle was opened with. Replayed verbatim when
1785 /// `refresh` has to rebuild from disk.
1786 open_opts: OpenOptions,
1787 /// True when this handle holds the cross-process write lock for its whole
1788 /// lifetime (a plain read-write open). Per-write lock acquisition is a
1789 /// no-op on such a handle, and never releases the lock.
1790 holds_lifetime_lock: bool,
1791 /// True between a failed lock acquisition and the end of the write scope
1792 /// that failed. Makes every WAL-appending mutation in that scope return
1793 /// [`GraphError::Busy`] instead of writing.
1794 lock_denied: bool,
1795 /// True for an as-of view opened via [`GraphDb::open_at`]. Such a view is
1796 /// pinned to one commit, so it is never stale and never refreshes — later
1797 /// commits by any process are deliberately invisible to it.
1798 pinned: bool,
1799}
1800
1801/// One group of deferred event notifications, held until the group fsync
1802/// completes. Replayed by [`GraphDb::flush_deferred_events`].
1803struct DeferredEvent {
1804 rec: core_storage::WalRecord,
1805 engine_deltas: Vec<EngineEdgeDelta>,
1806 seq: u64,
1807 ingest: Option<(String, usize)>,
1808}
1809
1810/// Options for [`GraphDb::open_with_options`].
1811#[derive(Clone, Copy, Debug)]
1812pub struct OpenOptions {
1813 /// Rewrite an old-format snapshot to the current VERSION after a
1814 /// successful load (default `true`). The old snapshot is kept as
1815 /// `snapshot.bin.bak` until the next clean open at the current version,
1816 /// at which point the `.bak` is deleted.
1817 ///
1818 /// Set to `false` to open a store without touching any on-disk files
1819 /// (useful for read-only inspection of a store at an older format).
1820 pub auto_migrate: bool,
1821
1822 /// Write the valid WAL prefix back over a torn tail on open (default
1823 /// `true`). Truncating a genuinely torn tail is correct crash recovery.
1824 ///
1825 /// Set to `false` for an unattended reader. The valid prefix is still
1826 /// decoded and replayed in memory, but nothing is written: a reader that
1827 /// opens while another process is mid-append would otherwise discard a
1828 /// frame that writer believes durable. `mushroomdb recall`, which runs on
1829 /// every prompt, passes `false` for exactly this reason.
1830 pub repair_wal: bool,
1831
1832 /// Open without ever writing to the store (default `false`).
1833 ///
1834 /// A read-only handle:
1835 /// - returns [`GraphError::ReadOnly`] from every mutation and from
1836 /// `snapshot()`;
1837 /// - performs no disk write at open — no WAL repair write-back and no
1838 /// auto-migration rewrite, whatever the other two flags say;
1839 /// - never takes the cross-process write lock, so it opens immediately even
1840 /// while another process is writing, and never makes a writer wait.
1841 ///
1842 /// [`refresh`](GraphDb::refresh) and [`is_stale`](GraphDb::is_stale) work
1843 /// normally, so a read-only handle can follow another process's commits.
1844 pub read_only: bool,
1845}
1846
1847impl Default for OpenOptions {
1848 fn default() -> Self {
1849 Self {
1850 auto_migrate: true,
1851 repair_wal: true,
1852 read_only: false,
1853 }
1854 }
1855}
1856
1857/// How long a writer polls for the cross-process write lock before giving up
1858/// with [`GraphError::Busy`].
1859///
1860/// Long enough to ride out another process's commit (a batch apply plus one
1861/// fsync), short enough that a stuck peer surfaces as an error rather than a
1862/// hang.
1863pub const WRITE_LOCK_WAIT: std::time::Duration = std::time::Duration::from_secs(2);
1864
1865/// Refusal when a `MERGE` create cannot choose a namespace.
1866///
1867/// A role bound to two or more namespaces cannot have its create arm land in
1868/// `default`, and the statement did not name `ns`. The role must name one.
1869pub const MERGE_CREATE_NEEDS_ONE_NAMESPACE: &str =
1870 "role-bound token: MERGE create requires the role to name one namespace";
1871
1872/// Interval between poll attempts while waiting for the cross-process lock.
1873pub(crate) const LOCK_POLL_INTERVAL: std::time::Duration = std::time::Duration::from_millis(10);
1874
1875/// Why `load_from_disk` is running, which decides whether it may repair.
1876#[derive(Clone, Copy, Debug, PartialEq, Eq)]
1877enum LoadOrigin {
1878 /// A fresh open. Crash recovery is this handle's job: a torn WAL tail is
1879 /// the signature of a crash and truncating it is correct, and archives
1880 /// orphaned by an interrupted prune can be swept.
1881 Open,
1882 /// A reload driven by [`GraphDb::refresh`], because another process
1883 /// replaced the snapshot. Nothing here is crash recovery — the store is
1884 /// live and someone else is writing it — so this origin writes nothing.
1885 Reload,
1886}
1887
1888/// Authorization context carried by `write_batch_authz` / `query_write_authz`.
1889///
1890/// `None` at the call site = full authority (today's zero-cost behavior).
1891/// `Some(WriteAuthz)` = role-scoped: the decision table (plan §"authz decision
1892/// table") is evaluated per-op inside `commit_logged_batch` BEFORE any WAL
1893/// record is built. A denial returns an error with no WAL frame written.
1894///
1895/// The mask is ALWAYS `Omit`-mode: role-token paths must never acknowledge
1896/// hidden-node existence to callers.
1897#[derive(Clone, Debug)]
1898pub struct WriteAuthz {
1899 pub role: String,
1900 pub scope: WriteScope,
1901 /// Resolved by `mask_for_role` under the same write guard as the mutation.
1902 /// Always `Omit`-mode — never `Stub`.
1903 pub mask: crate::mask::NodeMask,
1904}
1905
1906/// The error every role surface gives when `roles.json` did not parse at open.
1907///
1908/// One text, so `mask_for_role` and [`GraphDb::roles_checked`] cannot drift
1909/// apart on the same cause.
1910fn roles_poisoned() -> GraphError {
1911 GraphError::Corrupt {
1912 detail: "roles.json was corrupt at open; fix the file and re-open to restore role access"
1913 .into(),
1914 }
1915}
1916
1917/// Write `bytes` to `snapshot.bin.bak` atomically with full fsync.
1918///
1919/// Uses [`RealFs::write_atomic`] which applies `F_FULLFSYNC` on macOS and
1920/// `sync_all` on other platforms, then renames the `.tmp` file into place and
1921/// syncs the directory entry. This is the only correct path for writing the
1922/// `.bak` — plain `std::fs::write + sync_all` misses both `F_FULLFSYNC` and
1923/// the directory sync.
1924pub fn write_snapshot_bak(dir: &std::path::Path, bytes: &[u8]) -> crate::Result<()> {
1925 use core_storage::fs::{FileId, Fs as _};
1926 RealFs::new(dir)
1927 .map_err(core_storage::GraphError::Io)?
1928 .write_atomic(FileId::SnapshotBak, bytes)
1929 .map_err(core_storage::GraphError::Io)
1930}
1931
1932/// Return the on-disk snapshot format version without decoding the full snapshot.
1933///
1934/// Reads only the 6-byte header (magic + version LE). Returns `None` when no
1935/// snapshot file exists (WAL-only store). Returns an error if the header is
1936/// malformed.
1937pub fn snapshot_version_at(dir: &std::path::Path) -> crate::Result<Option<u16>> {
1938 use std::io::Read as _;
1939 let path = dir.join("snapshot.bin");
1940 let mut header = [0u8; 6];
1941 let n = match std::fs::File::open(&path) {
1942 Ok(mut f) => f.read(&mut header).map_err(core_storage::GraphError::Io)?,
1943 Err(e) if e.kind() == std::io::ErrorKind::NotFound => return Ok(None),
1944 Err(e) => return Err(core_storage::GraphError::Io(e)),
1945 };
1946 core_storage::snapshot::peek_version(&header[..n])
1947}
1948
1949/// Options for [`GraphDb::snapshot_with`].
1950#[derive(Debug, Clone, Default)]
1951pub struct SnapshotOptions {
1952 /// When `true`, the WAL is preserved after the snapshot write.
1953 /// Pre-snapshot commits remain reachable via [`GraphDb::open_at`].
1954 /// When `false` (the default), the WAL is truncated to a minimal
1955 /// baseline so cold-start replay stays fast.
1956 pub keep_wal: bool,
1957 /// When `true`, the current WAL is renamed to `wal.<commit_seq>.archive`
1958 /// before a fresh WAL baseline is written (history-preserving snapshot).
1959 ///
1960 /// This is the feature opt-in: `false` (the default) leaves the existing
1961 /// truncation / keep-wal behaviour byte-identical. `archive_wal` takes
1962 /// precedence over `keep_wal` when both are set.
1963 ///
1964 /// Archives can be scanned by [`GraphDb::node_history`],
1965 /// [`GraphDb::edge_history`], [`GraphDb::was_linked`], and
1966 /// [`GraphDb::open_at`], extending the reachable history horizon across
1967 /// snapshot boundaries.
1968 pub archive_wal: bool,
1969}
1970
1971/// Derive the scan-label sym for the commit-skip fast-path.
1972///
1973/// Walks `ops` to find the plan's leading scan op (`ScanLabel`, `IndexScan`,
1974/// or `IndexIntersect`) with a concrete label string, then interns it.
1975///
1976/// Returns `None` in all cases where skipping is unsafe:
1977/// - Any `Expand` op is present (edge traversal; edges change results regardless
1978/// of node labels).
1979/// - The leading scan has no label (`ScanLabel { label: None }` — full scan).
1980/// - No recognizable leading scan op is found.
1981///
1982/// This is the conservative v0.4.3 boundary. The caller stores the result in
1983/// [`QuerySubEntry::scan_label`] at subscribe time; `None` means always execute.
1984fn extract_scan_label(ops: &[PlanOp], syms: &mut Interner) -> Option<u32> {
1985 // Any Expand → must always re-execute (edges can change join results).
1986 if ops.iter().any(|op| matches!(op, PlanOp::Expand { .. })) {
1987 return None;
1988 }
1989 for op in ops {
1990 match op {
1991 PlanOp::ScanLabel {
1992 label: Some(label), ..
1993 } => return Some(syms.intern(label)),
1994 PlanOp::IndexScan {
1995 label: Some(label), ..
1996 } => return Some(syms.intern(label)),
1997 PlanOp::IndexIntersect {
1998 label: Some(label), ..
1999 } => return Some(syms.intern(label)),
2000 _ => {}
2001 }
2002 }
2003 None
2004}
2005
2006/// How an as-of read is restricted — the argument to
2007/// [`GraphDb::query_at_scoped`].
2008///
2009/// Every variant is resolved against the graph **as it was at the requested
2010/// commit**, not against the current graph.
2011#[derive(Debug, Clone, Copy)]
2012pub enum AsOfScope<'a> {
2013 /// Everything the named role may see. The role *definition* is the current
2014 /// one — `roles.json` is a sidecar and has no past version — but its
2015 /// `keys` and `labels` are resolved against the as-of graph.
2016 Role(&'a str),
2017 /// An explicit node-key allow-list. Keys that did not exist at that commit
2018 /// resolve to nothing.
2019 Keys(&'a [String]),
2020 /// A role intersected with a client-supplied allow-list. The intersection
2021 /// is the never-widen rule: a client mask can only narrow a role.
2022 RoleAndKeys(&'a str, &'a [String]),
2023 /// Every live node in one namespace, as the graph was at that commit.
2024 ///
2025 /// A namespace cannot change — it is set at insert and immutable — so the
2026 /// answer is simply "the nodes that existed then and are in this
2027 /// namespace". A name no node uses resolves to nothing, never to
2028 /// everything.
2029 Namespace(&'a str),
2030}
2031
2032impl GraphDb<RealFs> {
2033 /// Open the database at `dir` with default options.
2034 ///
2035 /// Equivalent to `open_with_options(dir, OpenOptions::default())`.
2036 /// Old-format snapshots (V5, V6) are automatically migrated to the
2037 /// current version on a successful load (see [`OpenOptions::auto_migrate`]).
2038 pub fn open(dir: &std::path::Path) -> Result<Self> {
2039 Self::open_with_options(dir, OpenOptions::default())
2040 }
2041
2042 /// Open the database at `dir` with explicit options.
2043 ///
2044 /// When `opts.auto_migrate` is `true` (the default) and the on-disk
2045 /// snapshot is an older format version, this function:
2046 /// 1. Copies the current `snapshot.bin` to `snapshot.bin.bak` (atomic
2047 /// + fsynced) before any modification.
2048 /// 2. Rewrites `snapshot.bin` at the current format version via
2049 /// [`GraphDb::snapshot_with`] with `keep_wal: true` (WAL preserved).
2050 ///
2051 /// If migration fails the error is returned and the original files are
2052 /// intact (the `.bak` was written before the new snapshot was attempted).
2053 ///
2054 /// A clean open that finds the snapshot already at the current version
2055 /// deletes any leftover `.bak` file.
2056 ///
2057 /// WAL-only stores (no snapshot) are never auto-migrated on open.
2058 ///
2059 /// `opts.repair_wal` controls the other write this function can make; see
2060 /// [`OpenOptions::repair_wal`]. With both flags `false` the open touches
2061 /// no file on disk.
2062 pub fn open_with_options(dir: &std::path::Path, opts: OpenOptions) -> Result<Self> {
2063 Self::open_dir(dir, opts, true)
2064 }
2065
2066 /// Open without taking the cross-process write lock for the handle's
2067 /// lifetime.
2068 ///
2069 /// Only [`SharedDb`](crate::SharedDb) uses this: a long-lived server holds
2070 /// its handle open indefinitely, so it takes the lock per write instead of
2071 /// keeping every other process out of the store for as long as it runs.
2072 pub(crate) fn open_unlocked(dir: &std::path::Path) -> Result<Self> {
2073 Self::open_dir(dir, OpenOptions::default(), false)
2074 }
2075
2076 fn open_dir(dir: &std::path::Path, opts: OpenOptions, hold_lock: bool) -> Result<Self> {
2077 // Header-only peek — 6 bytes, no full decode.
2078 let snap_version = snapshot_version_at(dir)?;
2079
2080 // Full load: decode snapshot + replay WAL + rebuild indexes.
2081 let mut db = Self::open_generic(RealFs::new(dir)?, opts, hold_lock)?;
2082
2083 // A read-only handle writes nothing at open, so it never migrates —
2084 // the old-format snapshot is loaded and left exactly as it is.
2085 if opts.auto_migrate && !opts.read_only {
2086 match snap_version {
2087 Some(ver) if ver < core_storage::snapshot::VERSION => {
2088 let _tm = std::time::Instant::now();
2089 // Copy the original snapshot to .bak at OS level — no in-memory
2090 // buffer required for a 2+ GiB file.
2091 //
2092 // Crash-safety: snapshot.bin remains intact (write_atomic inside
2093 // snapshot_with uses a .tmp+rename) until the V8 write succeeds.
2094 // A torn .bak on crash is acceptable because the original
2095 // snapshot.bin is the authoritative source until after the rename.
2096 std::fs::copy(dir.join("snapshot.bin"), dir.join("snapshot.bin.bak"))
2097 .map_err(core_storage::GraphError::Io)?;
2098 trace_migrate!("bak copy done", _tm);
2099 // Rewrite snapshot at current version; keep WAL intact.
2100 db.snapshot_with(SnapshotOptions {
2101 keep_wal: true,
2102 ..SnapshotOptions::default()
2103 })?;
2104 trace_migrate!("snapshot_with done", _tm);
2105 }
2106 Some(_) => {
2107 // Already current version: remove any leftover .bak.
2108 let bak = dir.join("snapshot.bin.bak");
2109 if bak.exists() {
2110 std::fs::remove_file(&bak).map_err(core_storage::GraphError::Io)?;
2111 }
2112 }
2113 None => {
2114 // WAL-only store — nothing to migrate on open.
2115 }
2116 }
2117 }
2118
2119 Ok(db)
2120 }
2121
2122 /// Open a read-only view of the database as it existed after `commit`.
2123 ///
2124 /// Commit indices are 0-based over the current WAL: commit 0 is the state
2125 /// after the first WAL frame, commit N-1 is the state after the N-th (most
2126 /// recent) frame. Call [`GraphDb::open`] to read the full current state.
2127 ///
2128 /// **Replay base.** [`GraphDb::snapshot`] truncates the WAL when it runs,
2129 /// so as-of can only reach commits recorded in the current WAL (those
2130 /// written after the most recent snapshot, or all commits if no snapshot
2131 /// was ever taken). Commit 0 in `open_at` always refers to the first
2132 /// frame in the WAL that exists on disk, not the first ever write to the
2133 /// database. When the on-disk snapshot recorded that it truncated the
2134 /// WAL (V7, default `keep_wal: false`), it is loaded as the base state
2135 /// before frame replay, so the as-of view includes all pre-snapshot data.
2136 /// Snapshots written with `keep_wal: true` (and legacy V5/V6 snapshots)
2137 /// are ignored and replay is WAL-only, as before.
2138 ///
2139 /// **Read-only.** Every mutation method and `snapshot()` on the returned
2140 /// instance returns [`GraphError::ReadOnly`]. Queries, `explain()`, and
2141 /// `stats()` work normally.
2142 ///
2143 /// # Errors
2144 /// - [`GraphError::CommitOutOfRange`] if `commit >= wal_commit_count` (including
2145 /// when the WAL is empty after a snapshot).
2146 pub fn open_at(dir: &std::path::Path, commit: u64) -> Result<Self> {
2147 Self::open_at_with(RealFs::new(dir)?, commit)
2148 }
2149
2150 /// Run a **read-only** Cypher query against the graph as it existed at
2151 /// `commit` — the "time-travel" / agent-replay query. Opens a temporal view
2152 /// of this store's directory at that commit and executes the read there.
2153 ///
2154 /// The current instance is unaffected. Write statements are rejected (the
2155 /// temporal view is read-only). `commit` is a 0-based WAL commit index;
2156 /// `commit == wal_commit_count` (or `open_at`'s range) yields the newest
2157 /// state. Prefer this over holding many historical instances open.
2158 ///
2159 /// # Errors
2160 /// - [`GraphError::CommitOutOfRange`] if `commit` is past the WAL horizon.
2161 /// - A query error for a malformed or write query.
2162 pub fn query_at(
2163 &self,
2164 commit: u64,
2165 cypher: &str,
2166 params: &std::collections::BTreeMap<String, Value>,
2167 ) -> Result<ResultSet> {
2168 let temporal = self.open_at_for_read(commit, cypher)?;
2169 temporal.query(cypher, params)
2170 }
2171
2172 /// Run a **read-only** Cypher query at `commit`, restricted by `scope`.
2173 ///
2174 /// The **graph** is as of `commit`; the **role definition** is as it is
2175 /// now, because `roles.json` is a sidecar and is never a WAL record — it
2176 /// has no past version to read. A role's `keys` and `labels` are resolved
2177 /// against the commit-`commit` graph, so a role that may see a label sees
2178 /// exactly the nodes that carried it then, and an explicit key that did
2179 /// not exist yet resolves to nothing.
2180 ///
2181 /// [`AsOfScope::RoleAndKeys`] intersects the two: a client allow-list can
2182 /// only narrow what a role may see, never widen it.
2183 ///
2184 /// Write statements are rejected, exactly as [`GraphDb::query_at`] rejects
2185 /// them.
2186 ///
2187 /// # Errors
2188 /// - [`GraphError::CommitOutOfRange`] if `commit` is outside the retained
2189 /// range; the error carries that range.
2190 /// - [`GraphError::KeyNotFound`] with a `role:` prefix for an unknown role,
2191 /// or [`GraphError::Corrupt`] when `roles.json` was corrupt at open.
2192 /// - A query error for a malformed or write query.
2193 pub fn query_at_scoped(
2194 &self,
2195 commit: u64,
2196 cypher: &str,
2197 params: &std::collections::BTreeMap<String, Value>,
2198 scope: AsOfScope<'_>,
2199 ) -> Result<ResultSet> {
2200 let temporal = self.open_at_for_read(commit, cypher)?;
2201 let mask = temporal.mask_at_scope(scope)?;
2202 temporal.query_masked(cypher, params, &mask)
2203 }
2204
2205 /// As [`GraphDb::query_at_scoped`], with `namespace` intersected into
2206 /// whatever `scope` resolves to.
2207 ///
2208 /// This is what a surface needs when a caller passes `namespace` beside a
2209 /// `role` or a client mask on a time-travel read: [`AsOfScope`] names one
2210 /// restriction, and the namespace is a second one that composes with it
2211 /// rather than replacing it. The intersection is the never-widen rule — a
2212 /// namespace can only narrow what the scope already allows — and both legs
2213 /// are resolved against the graph as it was at `commit`.
2214 ///
2215 /// `AsOfScope::Namespace(ns)` is still the way to ask for a namespace alone.
2216 pub fn query_at_scoped_in_namespace(
2217 &self,
2218 commit: u64,
2219 cypher: &str,
2220 params: &std::collections::BTreeMap<String, Value>,
2221 scope: AsOfScope<'_>,
2222 namespace: &str,
2223 ) -> Result<ResultSet> {
2224 let temporal = self.open_at_for_read(commit, cypher)?;
2225 let mask = temporal
2226 .mask_at_scope(scope)?
2227 .intersect(&temporal.mask_for_namespace(namespace));
2228 temporal.query_masked(cypher, params, &mask)
2229 }
2230
2231 /// Run a **read-only** Cypher query at `commit`, restricted by a
2232 /// [`Scope`](crate::mask::Scope).
2233 ///
2234 /// [`AsOfScope`] names *one* restriction — a role, a key list, a namespace,
2235 /// or a role-and-keys pair. A `Scope` is the general shape a handle carries,
2236 /// and nesting can give it several role or namespace legs at once, so it
2237 /// cannot be spelled as an `AsOfScope`. This is the entry point a scoped
2238 /// handle uses for time travel; `query_at_scoped` stays the way to ask for
2239 /// one named restriction.
2240 ///
2241 /// Both the graph and the scope's key and namespace legs are resolved
2242 /// against `commit`; a role's *definition* is the current one, because
2243 /// `roles.json` is a sidecar with no past version — the same split
2244 /// [`GraphDb::query_at_scoped`] documents.
2245 ///
2246 /// The scope resolves **cold** here: a temporal handle is its own store, so
2247 /// its ids could never be served to a live read, but filling the scope's
2248 /// one-entry key memo from a handle thrown away at the end of this call
2249 /// would evict the live entry for nothing. See
2250 /// [`Scope::resolve_uncached`](crate::mask::Scope::resolve_uncached).
2251 ///
2252 /// # Errors
2253 /// - [`GraphError::CommitOutOfRange`] if `commit` is outside the retained
2254 /// range; the error carries that range.
2255 /// - [`GraphError::KeyNotFound`] with a `role:` prefix for an unknown role,
2256 /// or [`GraphError::Corrupt`] when `roles.json` was corrupt at open.
2257 /// - A query error for a malformed or write query.
2258 pub fn query_at_with_scope(
2259 &self,
2260 commit: u64,
2261 cypher: &str,
2262 params: &std::collections::BTreeMap<String, Value>,
2263 scope: &crate::mask::Scope,
2264 ) -> Result<ResultSet> {
2265 let temporal = self.open_at_for_read(commit, cypher)?;
2266 let mask = scope.resolve_uncached(&temporal)?;
2267 temporal.query_masked(cypher, params, &mask)
2268 }
2269
2270 /// Open the temporal view for a time-travel read and refuse write Cypher.
2271 ///
2272 /// Shared by [`GraphDb::query_at`] and [`GraphDb::query_at_scoped`] so both
2273 /// resolve the commit and reject writes identically.
2274 fn open_at_for_read(&self, commit: u64, cypher: &str) -> Result<Self> {
2275 let dir = self.fs.dir().to_path_buf();
2276 let temporal = Self::open_at(&dir, commit)?;
2277 if is_write_tokens(&lex(cypher).map_err(|e| GraphError::QueryError {
2278 detail: format!("lex: {e}"),
2279 })?) {
2280 return Err(GraphError::QueryError {
2281 detail: "query_at is read-only: write statements are not permitted in a \
2282 time-travel query"
2283 .into(),
2284 });
2285 }
2286 Ok(temporal)
2287 }
2288}
2289
2290impl<F: Fs> GraphDb<F> {
2291 /// Open over an arbitrary [`Fs`], repairing a torn WAL tail as usual.
2292 pub fn open_with(fs: F) -> Result<Self> {
2293 Self::open_with_repair(fs, true)
2294 }
2295
2296 /// As [`GraphDb::open_with`], but `repair_wal: false` decodes the valid WAL
2297 /// prefix without writing the truncation back. See
2298 /// [`OpenOptions::repair_wal`].
2299 pub fn open_with_repair(fs: F, repair_wal: bool) -> Result<Self> {
2300 Self::open_generic(
2301 fs,
2302 OpenOptions {
2303 repair_wal,
2304 ..OpenOptions::default()
2305 },
2306 true,
2307 )
2308 }
2309
2310 /// Shared open path.
2311 ///
2312 /// `hold_lock` requests the cross-process write lock for the whole handle
2313 /// lifetime — the right behaviour for a plain read-write `GraphDb`, whose
2314 /// owner writes through it directly. [`SharedDb`](crate::SharedDb) passes
2315 /// `false` and takes the lock per write instead, so that a long-lived
2316 /// server does not keep every other process out of the store.
2317 ///
2318 /// A read-only open never takes the lock regardless of `hold_lock`.
2319 fn open_generic(fs: F, opts: OpenOptions, hold_lock: bool) -> Result<Self> {
2320 let mut db = Self::new_empty(fs, opts);
2321 db.read_only = opts.read_only;
2322 if hold_lock && !opts.read_only {
2323 if !db.poll_lock(WRITE_LOCK_WAIT)? {
2324 return Err(GraphError::Busy { holder: None });
2325 }
2326 db.holds_lifetime_lock = true;
2327 }
2328 db.load_from_disk(LoadOrigin::Open)?;
2329 Ok(db)
2330 }
2331
2332 /// A handle with no state loaded: every field at its empty value, the
2333 /// filesystem and options in place. Only [`load_from_disk`] makes it
2334 /// usable.
2335 fn new_empty(fs: F, opts: OpenOptions) -> Self {
2336 Self {
2337 fs,
2338 ids: Arc::new(IdMap::new()),
2339 syms: Arc::new(Interner::new()),
2340 topo: Arc::new(Topology::new()),
2341 props: Arc::new(ColumnStore::new()),
2342 labels: Arc::new(Vec::new()),
2343 ns_names: vec![NS_DEFAULT.to_string()],
2344 node_ns: Vec::new(),
2345 edge_props: Arc::new(EdgeProps::new()),
2346 engine: RuleEngine::new(),
2347 view_store: ViewStore::new(),
2348 fulltext: Arc::new(FulltextIndex::new()),
2349 prop_index: PropertyIndex::new(),
2350 multiplicity: false,
2351 event_sink: None,
2352 fsync: FsyncPolicy::Strict,
2353 commit_seq: 0,
2354 commit_times: core_storage::commit_times::CommitTimes::default(),
2355 commit_times_poisoned: false,
2356 fulltext_rebuild_follows: false,
2357 commit_time_override: None,
2358 roles: Some(vec![]),
2359 role_masks: Arc::new(crate::mask::RoleMaskCache::new()),
2360 store_id: crate::mask::StoreId::next(),
2361 subscriptions: Vec::new(),
2362 query_subscriptions: Vec::new(),
2363 sub_capacity: DEFAULT_SUB_CAPACITY,
2364 read_only: false,
2365 total_wal_commits: 0,
2366 base: None,
2367 fold_overlay: None,
2368 delta_tail: Vec::new(),
2369 commits_since_fold: 0,
2370 defer_events: false,
2371 deferred_events: Vec::new(),
2372 degraded: false,
2373 v8_sections_loaded: std::sync::atomic::AtomicBool::new(false),
2374 v8_sections_mutex: std::sync::Mutex::new(()),
2375 last_change: HashMap::new(),
2376 wal_archive_retention: None,
2377 wal_horizon_floor: 0,
2378 archive_genesis_chain: false,
2379 // Nothing is proven until `load_from_disk` has looked at the store.
2380 snapshot_preserved_history: false,
2381 pending_write_authz: None,
2382 slow_query_threshold_ms: std::env::var("MUSHROOMDB_SLOW_QUERY_MS")
2383 .ok()
2384 .and_then(|v| v.parse().ok())
2385 .unwrap_or(100),
2386 slow_queries: std::sync::Mutex::new(SlowQueryLog {
2387 entries: std::collections::VecDeque::new(),
2388 total: 0,
2389 }),
2390 warned_ambiguous_exactness: std::sync::Mutex::new(HashSet::new()),
2391 started_at: std::time::Instant::now(),
2392 wal_consumed: 0,
2393 wal_frames_written: 0,
2394 snapshot_ident: None,
2395 open_opts: opts,
2396 holds_lifetime_lock: false,
2397 lock_denied: false,
2398 pinned: false,
2399 }
2400 }
2401
2402 /// Return every field describing stored graph state to its empty value,
2403 /// leaving this handle's own identity alone.
2404 ///
2405 /// Preserved on purpose: the filesystem, open options, lock ownership, the
2406 /// event sink and subscriptions, fsync policy, degraded flag, and the
2407 /// slow-query configuration and log. A caller that registered a sink or a
2408 /// subscription keeps it across a reload.
2409 fn reset_for_reload(&mut self) {
2410 self.ids = Arc::new(IdMap::new());
2411 self.syms = Arc::new(Interner::new());
2412 self.topo = Arc::new(Topology::new());
2413 self.props = Arc::new(ColumnStore::new());
2414 self.labels = Arc::new(Vec::new());
2415 self.ns_names = vec![NS_DEFAULT.to_string()];
2416 self.node_ns = Vec::new();
2417 self.edge_props = Arc::new(EdgeProps::new());
2418 self.engine = RuleEngine::new();
2419 self.view_store = ViewStore::new();
2420 self.fulltext = Arc::new(FulltextIndex::new());
2421 self.prop_index = PropertyIndex::new();
2422 // Cleared like every other declaration: a reload replays the store's own
2423 // WAL, and the opt-in comes back from it or not at all.
2424 self.multiplicity = false;
2425 self.commit_seq = 0;
2426 self.commit_times = core_storage::commit_times::CommitTimes::default();
2427 self.commit_times_poisoned = false;
2428 self.fulltext_rebuild_follows = false;
2429 self.commit_time_override = None;
2430 self.roles = Some(vec![]);
2431 // A fresh cache, not a cleared one: any reader snapshot still holding
2432 // the old `Arc` keeps it to itself, so nothing it memoised against the
2433 // pre-reload store can be read back through this handle.
2434 self.role_masks = Arc::new(crate::mask::RoleMaskCache::new());
2435 // The same move for memos this handle does not own. `commit_seq` is
2436 // zeroed just above and reseeded from `max(last_change)`, which a
2437 // delete-only commit leaves where it was — so a reload can land back on
2438 // a sequence a caller's `Scope` already cached a mask at. A new id is
2439 // what makes that entry stop matching.
2440 self.store_id = crate::mask::StoreId::next();
2441 self.total_wal_commits = 0;
2442 self.base = None;
2443 self.fold_overlay = None;
2444 self.delta_tail = Vec::new();
2445 self.commits_since_fold = 0;
2446 self.deferred_events = Vec::new();
2447 self.v8_sections_loaded
2448 .store(false, std::sync::atomic::Ordering::Release);
2449 self.last_change = HashMap::new();
2450 self.wal_horizon_floor = 0;
2451 self.archive_genesis_chain = false;
2452 // Re-derived by `load_from_disk` from the store it is about to read.
2453 self.snapshot_preserved_history = false;
2454 self.pending_write_authz = None;
2455 self.wal_consumed = 0;
2456 self.wal_frames_written = 0;
2457 self.snapshot_ident = None;
2458 }
2459
2460 /// Load the snapshot base and replay the WAL into an empty handle — the
2461 /// whole of what opening a store does after the struct exists.
2462 ///
2463 /// Split out of the open path so that [`refresh`](GraphDb::refresh) can
2464 /// rebuild a handle in place, without ownership of `F`, when another
2465 /// process replaces the snapshot underneath it.
2466 ///
2467 /// `origin` decides whether the two repair writes this function can make
2468 /// are appropriate; see [`LoadOrigin`].
2469 fn load_from_disk(&mut self, origin: LoadOrigin) -> Result<usize> {
2470 // Both writes below are crash recovery, and only an open is entitled to
2471 // perform them. A read-only handle promises to touch nothing, and a
2472 // reload driven by `refresh` is looking at a store another process is
2473 // actively writing: what looks like a torn tail there is a peer
2474 // mid-append, and what looks like an orphaned archive may be one that
2475 // peer is about to reference.
2476 let may_repair = origin == LoadOrigin::Open && !self.open_opts.read_only;
2477 let repair_wal = self.open_opts.repair_wal && may_repair;
2478 let db = self;
2479 db.wal_horizon_floor = db.fs.read_horizon_floor()?;
2480 db.archive_genesis_chain = db.fs.has_genesis_marker();
2481 // Opening cleanup: remove orphaned archives — archives whose frames all
2482 // fall below the horizon floor. Orphans arise when a crash interrupted
2483 // the retention-prune sequence after the floor was written but before
2484 // all surplus archives were deleted. Safe to delete: floor already
2485 // accounts for their frames.
2486 if may_repair {
2487 db.cleanup_orphaned_archives()?;
2488 }
2489 let _t0 = std::time::Instant::now();
2490 // Peek 6 bytes to determine snapshot version without reading the full
2491 // file. For RealFs this is a true partial read (O(1)); for SimFs the
2492 // default impl reads all bytes and truncates (still correct).
2493 let snap_header = db.fs.read_prefix(FileId::Snapshot, 6)?;
2494 // V8, V9 and V10 share the mmap-able container; V9 only adds section 12
2495 // and V10 adds nothing but its version stamp. A version outside that set
2496 // falls through to the full-read path below, where `snapshot::decode`
2497 // either handles it (V5–V7) or refuses it by name — which is what stops
2498 // an older binary before it reaches the WAL.
2499 let is_v8 = snap_header.len() >= 6
2500 && &snap_header[0..4] == b"GDB1"
2501 && core_storage::snapshot::is_mmap_container(u16::from_le_bytes([
2502 snap_header[4],
2503 snap_header[5],
2504 ]));
2505 // The version this store is stamped with, or `None` when it has never
2506 // been snapshotted. Read from the same six bytes, with no second read.
2507 let snapshot_version = if snap_header.len() >= 6 && &snap_header[0..4] == b"GDB1" {
2508 Some(u16::from_le_bytes([snap_header[4], snap_header[5]]))
2509 } else {
2510 None
2511 };
2512 // No snapshot means no snapshot has ever truncated the WAL, so this
2513 // handle can prove the history is whole. Once a snapshot exists that
2514 // this handle did not take, it cannot: see `snapshot_preserved_history`.
2515 db.snapshot_preserved_history = snap_header.is_empty();
2516 if is_v8 {
2517 // V8: map the file zero-copy (RealFs) or read full bytes (SimFs).
2518 // No 2.4GB heap Vec is allocated on RealFs.
2519 let mapped = Arc::new(
2520 if let Some(snap_path) = db.fs.snapshot_path() {
2521 core_storage::v8::MappedBase::map(&snap_path)
2522 } else {
2523 let snap_bytes = db.fs.read(FileId::Snapshot)?;
2524 core_storage::v8::MappedBase::from_bytes(snap_bytes)
2525 }
2526 .map_err(|e| GraphError::Corrupt {
2527 detail: format!("v8: mmap open: {e:?}"),
2528 })?,
2529 );
2530 db.restore_v8_base(Arc::clone(&mapped))?;
2531 trace_open!("restore_v8_base", _t0);
2532 db.base = Some(mapped);
2533 trace_open!("base assigned", _t0);
2534 } else if !snap_header.is_empty() {
2535 // Legacy V5-V7: full read required for decode.
2536 let snap_bytes = db.fs.read(FileId::Snapshot)?;
2537 if let Some(state) = core_storage::snapshot::decode(&snap_bytes)? {
2538 db.restore_snapshot_state(state)?;
2539 }
2540 }
2541 // else: snap_header is empty = no snapshot file, fresh store.
2542 //
2543 // Seed commit_seq from the highest seq persisted in last_change so that
2544 // WAL-replay frames (which start at commit_seq+1) always exceed any seq
2545 // already stored in the snapshot. Without this, a db with one snapshot
2546 // commit would save last_change["a"]=1, then on reopen the first WAL
2547 // frame would replay at seq=1 again — colliding and making WAL-tail
2548 // mutations indistinguishable from the snapshot baseline.
2549 //
2550 // Safety invariant (seq-recycling):
2551 // Recycled seqs (those below the seeded baseline) were NEVER stored in
2552 // last_change because they belonged to a previous db lifetime — a new
2553 // db starts at commit_seq=0 with an empty last_change. Therefore no
2554 // CAS precondition can carry a recycled seq as its `expected` value
2555 // and accidentally match a live node's last_change entry.
2556 //
2557 // `expected:0` on a deleted-then-reinserted node:
2558 // After deletion, last_changed() returns None; callers that call
2559 // last_changed() and then use NodeUnchangedSince get None.unwrap_or(0)
2560 // = 0. The reinserted node gets seq > 0, so a subsequent CAS with
2561 // expected=0 correctly conflicts. The only way to observe actual=0 in
2562 // a CasConflict would be a caller that invented expected=0 without ever
2563 // calling last_changed() — unreachable via the documented API contract.
2564 if let Some(&max_seq) = db.last_change.values().max() {
2565 db.commit_seq = db.commit_seq.max(max_seq);
2566 }
2567 let bytes = db.fs.read(FileId::Wal)?;
2568 let (records, valid_len) = decode_all(&bytes);
2569 // The valid prefix is replayed either way; `repair_wal` only decides
2570 // whether the truncation is written back. A reader that races a live
2571 // appender must not persist a truncation the writer never asked for.
2572 if valid_len < bytes.len() && repair_wal {
2573 db.fs.write_atomic(FileId::Wal, &bytes[..valid_len])?;
2574 }
2575 // WAL-present path: build indexes eagerly BEFORE replay so that the
2576 // first replayed record does not trigger the lazy-init guard (which
2577 // would call reindex_all_load_state on an empty graph, defeating the
2578 // point of restoring IVF/HNSW blobs from the snapshot).
2579 if !records.is_empty() {
2580 db.ensure_v8_base_sections_loaded();
2581 trace_open!("lazy sections loaded (WAL path)", _t0);
2582 }
2583 // Scoped to this call: `refresh()` also replays frames and is *not*
2584 // followed by a rebuild, so its backfill must still run.
2585 db.fulltext_rebuild_follows = true;
2586 // Every decoded frame occupies an index, replayed or not: markers
2587 // are state no-ops but they are not index no-ops.
2588 let decoded_frames = records.len() as u64;
2589 let replayed = db.apply_frames(records);
2590 db.fulltext_rebuild_follows = false;
2591 let replayed = replayed?;
2592 // ── The multiplicity declaration, recovered from the stamp ───────────
2593 //
2594 // The opt-in is re-emitted into every baseline WAL a snapshot writes, so
2595 // ordinarily the replay above has already found it. But
2596 // `snapshot_with(archive_wal)` renames the live WAL away and writes its
2597 // replacement afterwards, and between those two points the store holds
2598 // no live declaration at all. A crash there — or a single `Err` from any
2599 // call in between — used to opt the store back out on the next open
2600 // (defect #22): it would stop counting and write a **V9** snapshot while
2601 // the archives still carried discriminant 23, which is the exact state
2602 // the V10 stamp exists to prevent.
2603 //
2604 // The V10 stamp is what carries the conclusion. The archive clause is a
2605 // scope restriction, not a second proof — an earlier version of this
2606 // comment, and defect #22, claimed otherwise, and defect #33 corrects
2607 // it. Taking the two in order:
2608 //
2609 // **The stamp.** `snapshot_with` stamps the snapshot from
2610 // `self.multiplicity` *before* it touches the WAL, and nothing rewrites
2611 // a V10 snapshot at V9 while the store believes it is opted in. So a
2612 // V10 stamp says this store reached `enable_multiplicity` far enough to
2613 // write the snapshot — and, decisively, that every older binary already
2614 // refuses this store by name. Opting in here can cost such a reader
2615 // nothing it was not already being told.
2616 //
2617 // **What the archive clause does not prove.** It is *not* evidence that
2618 // the archive was taken while the store was opted in. A store can
2619 // archive at V9 and opt in afterwards, leaving a V10 snapshot standing
2620 // beside an archive whose WAL carries no declaration at all — see
2621 // `a_failed_opt_in_beside_an_archive_comes_back_opted_in`. The inference
2622 // held in the success case by coincidence, not by construction.
2623 //
2624 // **What it does buy: scope.** Without it the recovery would also fire
2625 // on a store that reached the V10 snapshot write and then failed with no
2626 // archive in sight. That store must stay opted out, and can: no WAL was
2627 // renamed away, nothing carries discriminant 23, and its next snapshot
2628 // rewrites at V9, which puts it back within reach of every older reader.
2629 // An archive is the marker for the one state that is not recoverable
2630 // that way — a WAL renamed away that may hold the only copy of the
2631 // declaration. `no_crash_leaves_discriminant_23_unguarded` pins that
2632 // line: it sweeps a workload with no archives at all and refuses a
2633 // V10-implies-enabled rule.
2634 //
2635 // **The invariant, whichever way the clause goes:** the recovery never
2636 // opts in a store whose snapshot is not V10. A V9 store has made no
2637 // promise to an older reader, so opting it in would start writing
2638 // discriminant 23 behind a stamp that does not guard it. Pinned by
2639 // `the_recovery_never_opts_in_a_store_whose_snapshot_is_not_v10` and
2640 // `the_recovery_does_not_opt_a_store_in_by_itself`.
2641 //
2642 // What this recovery cannot do is make the opt-in atomic; it is not,
2643 // and `enable_multiplicity` says so. See defects #32-#34.
2644 if !db.multiplicity
2645 && snapshot_version == Some(core_storage::snapshot::VERSION_10)
2646 && !db.fs.list_archives()?.is_empty()
2647 {
2648 db.multiplicity = true;
2649 }
2650 // The cursor sits at the end of the valid prefix, not the end of the
2651 // file: a torn or still-being-written tail is unconsumed by definition
2652 // and stays visible to `is_stale` until it decodes.
2653 db.wal_consumed = valid_len as u64;
2654 // The frame cursor counts the same sequence `all_frames` returns:
2655 // surviving archives first, then the live WAL, offset by the floor.
2656 // Counting the archives separately rather than calling
2657 // `wal_total_commits` keeps the live WAL from being decoded twice on
2658 // every open, and costs nothing on a store that has never archived.
2659 db.wal_frames_written = db.wal_horizon_floor + db.archive_frame_count()? + decoded_frames;
2660 db.snapshot_ident = db.fs.snapshot_ident().map_err(GraphError::Io)?;
2661 trace_open!("wal replay done", _t0);
2662 // Rebuild view values after WAL replay only when there is no V8 base.
2663 // With a V8 base, view values are correct in the snapshot and are updated
2664 // incrementally during WAL replay (on_edge_changed / on_prop_changed).
2665 // A full rebuild would read overlay-only props (empty after restore_v8_base)
2666 // and overwrite correct base values with wrong results (e.g. NeighborAgg
2667 // Sum reads no "score" in overlay → writes 0.0, shadowing the correct
2668 // base value).
2669 if db.base.is_none() {
2670 let topo_view = TopologyView::owned(&db.topo);
2671 db.view_store.rebuild_all(
2672 Arc::make_mut(&mut db.props),
2673 &topo_view,
2674 &db.ids,
2675 &db.syms,
2676 &db.labels,
2677 );
2678 }
2679 // Rebuild full-text index after WAL replay. Corrects drift from
2680 // per-record incremental apply during replay.
2681 Arc::make_mut(&mut db.fulltext).rebuild_all(
2682 &db.ids,
2683 &db.labels,
2684 &db.syms,
2685 build_props_view(&db.props, &db.base),
2686 );
2687 db.prop_index.rebuild_all(
2688 &db.ids,
2689 &db.labels,
2690 &db.syms,
2691 build_props_view(&db.props, &db.base),
2692 );
2693 // Namespaces: one pass over the `ns` column, after the snapshot is
2694 // restored and the WAL replayed. Replay maintains `node_ns` record by
2695 // record as well; this pass is what makes a snapshot-only open right,
2696 // and it reads nothing on a store with no `ns` column.
2697 db.rebuild_node_ns();
2698 // A mid-build snapshot's HNSW blob carries `complete == false`.
2699 // Register it so `serve`'s ticker sees work without waiting for a write.
2700 db.register_outstanding_index_builds();
2701 // Load roles sidecar. Missing file = no roles (Some(vec![])).
2702 // Corrupt/unparseable = poisoned (None); mask_for_role will fail-loud.
2703 db.roles = Self::load_roles_from_fs(&db.fs)?;
2704 // The time sidecar. Absent is the normal case for any store written
2705 // before v0.6.11 and is not an error; unreadable is recorded so date
2706 // queries can say "damaged" rather than "none recorded".
2707 db.load_commit_times_from_fs();
2708 // Capture the initial MVCC fold so reader() is ready immediately.
2709 db.fold_now();
2710 trace_open!("open_with complete", _t0);
2711 Ok(replayed)
2712 }
2713
2714 /// Apply decoded WAL frames to in-memory state, exactly as the open-path
2715 /// replay does — same `apply` calls, same per-frame delta drain, same
2716 /// commit-seq and last-change bookkeeping. Rules therefore fire and derived
2717 /// edges appear identically whether a frame arrives at open, from a local
2718 /// commit, or from another process by way of [`refresh`](GraphDb::refresh).
2719 ///
2720 /// Returns the number of frames applied.
2721 ///
2722 /// Deltas are drained and discarded per frame: replayed frames are already
2723 /// reflected on disk, so they are not news to a subscriber, and draining
2724 /// inside the loop keeps `pending_deltas` O(1) over a large WAL (I-2).
2725 fn apply_frames(&mut self, records: Vec<WalRecord>) -> Result<usize> {
2726 if records.is_empty() {
2727 return Ok(0);
2728 }
2729 // Materialize any state retained in the mmap base before the first
2730 // frame lands, so a replayed record cannot trip the lazy-init guard and
2731 // rebuild indexes from an empty graph. Both calls are idempotent.
2732 self.ensure_v8_base_sections_loaded();
2733 self.engine.consume_retained_state_eager(
2734 &self.ids,
2735 &self.syms,
2736 &self.labels,
2737 build_props_view(&self.props, &self.base),
2738 );
2739 let applied = records.len();
2740 for rec in records {
2741 self.apply(&rec)?;
2742 let _ = self.engine.drain_deltas();
2743 // Track commit_seq during replay so last_change entries are
2744 // consistent with the seqs assigned by log_then_apply_with on
2745 // subsequent live commits. After N replayed frames, commit_seq=N;
2746 // live commits begin at N+1.
2747 self.commit_seq += 1;
2748 let replay_seq = self.commit_seq;
2749 self.update_last_change_from_rec(&rec, replay_seq);
2750 }
2751 // Enforce I-2: if the per-frame drain above is ever removed or skipped,
2752 // this assert catches the regression in debug builds immediately.
2753 debug_assert_eq!(
2754 self.engine.pending_delta_count(),
2755 0,
2756 "pending_deltas non-empty after replay — \
2757 per-frame drain must run inside the loop to keep memory O(1)"
2758 );
2759 // T2 note: the per-frame drain IS the suppression seam for replay.
2760 // Any future as-of replay path (Plan-15 T2) must drain here to feed
2761 // replaying subscribers; the mechanism is already in place.
2762 let _ = self.engine.drain_deltas(); // belt-and-braces no-op after loop drain
2763 Ok(applied)
2764 }
2765
2766 // ── Multi-process safety: cross-process write lock + WAL tailing ──────────
2767 //
2768 // mushroomdb is many-readers / one-writer across processes. Writers take an
2769 // advisory exclusive lock on the store's `LOCK` file; readers never do.
2770 // Every handle tracks how much of the WAL it has consumed, so it can pick
2771 // up another process's commits by decoding only the new tail rather than
2772 // reopening. See `docs/site/concurrency.md`.
2773
2774 /// Whether the store on disk has moved ahead of (or out from under) this
2775 /// handle's in-memory state.
2776 ///
2777 /// True when the WAL's length differs from this handle's cursor — another
2778 /// process committed, or is mid-append — or when the snapshot file's
2779 /// identity changed. Costs two metadata lookups and reads no file contents,
2780 /// so it is cheap enough for a read path to call.
2781 ///
2782 /// Always false for an as-of view from [`GraphDb::open_at`]: such a view is
2783 /// pinned to one commit and later commits are deliberately invisible to it.
2784 pub fn is_stale(&self) -> Result<bool> {
2785 if self.pinned {
2786 return Ok(false);
2787 }
2788 if self.fs.wal_len().map_err(GraphError::Io)? != self.wal_consumed {
2789 return Ok(true);
2790 }
2791 Ok(self.fs.snapshot_ident().map_err(GraphError::Io)? != self.snapshot_ident)
2792 }
2793
2794 /// Bring this handle up to date with every commit other processes have made,
2795 /// and return how many frames were applied.
2796 ///
2797 /// The WAL tail is decoded from this handle's cursor and applied through the
2798 /// same path the open replay uses, so rules fire and derived edges appear
2799 /// exactly as they would on a fresh open. Interners, id maps and indexes
2800 /// stay valid for the same reason.
2801 ///
2802 /// A frame another process is still writing is left alone: a trailing
2803 /// partial frame is a wait, not a corruption, and the handle stays stale
2804 /// until that frame is complete. Nothing is written to disk, so a read-only
2805 /// handle can refresh freely.
2806 ///
2807 /// When the snapshot file's identity changed, or the WAL is shorter than
2808 /// this handle's cursor, the WAL no longer continues our state — another
2809 /// process snapshotted or archived. The handle is then rebuilt from disk
2810 /// with the options it was opened with, and the return value is the number
2811 /// of frames in the new WAL.
2812 ///
2813 /// Returns 0 for an as-of view, which never follows later commits.
2814 ///
2815 /// # Errors
2816 ///
2817 /// An error here leaves the handle **degraded**: it got partway through
2818 /// applying the tail, or partway through a reload, so its in-memory state
2819 /// no longer matches any point on disk. Further mutations are refused and
2820 /// the handle must be reopened. Nothing on disk was damaged — the store
2821 /// itself is fine, and a fresh open recovers it.
2822 pub fn refresh(&mut self) -> Result<u64> {
2823 if self.pinned {
2824 return Ok(0);
2825 }
2826 let disk_ident = self.fs.snapshot_ident().map_err(GraphError::Io)?;
2827 let wal_len = self.fs.wal_len().map_err(GraphError::Io)?;
2828 if disk_ident != self.snapshot_ident || wal_len < self.wal_consumed {
2829 // The WAL no longer continues our state: rebuild from disk. State
2830 // is cleared first, so a failed load leaves an empty handle — mark
2831 // it degraded rather than let a caller read an empty graph as if
2832 // it were the store's contents.
2833 self.reset_for_reload();
2834 return match self.load_from_disk(LoadOrigin::Reload) {
2835 Ok(frames) => Ok(frames as u64),
2836 Err(e) => {
2837 self.degraded = true;
2838 Err(e)
2839 }
2840 };
2841 }
2842 if wal_len == self.wal_consumed {
2843 return Ok(0);
2844 }
2845 let tail = self
2846 .fs
2847 .read_range(FileId::Wal, self.wal_consumed)
2848 .map_err(GraphError::Io)?;
2849 let (records, valid_len) = decode_all(&tail);
2850 let decoded_frames = records.len() as u64;
2851 let applied = match self.apply_frames(records) {
2852 Ok(n) => n,
2853 Err(e) => {
2854 // Some frames landed and some did not, and the cursor cannot
2855 // say how many. Advancing it would skip the rest; leaving it
2856 // would replay what already applied. Neither is recoverable in
2857 // place, so refuse further writes and require a reopen.
2858 self.degraded = true;
2859 return Err(e);
2860 }
2861 };
2862 // Advance by the bytes actually decoded, never by the file length: an
2863 // incomplete trailing frame stays unconsumed for the next refresh.
2864 self.wal_consumed += valid_len as u64;
2865 self.wal_frames_written += decoded_frames;
2866 // The peer that wrote those frames also stamped them. Absorbing the
2867 // frames without the stamps leaves this handle resolving dates from a
2868 // prefix of the store's history, and — while our own map is still
2869 // empty — one commit away from rewriting the peer's file out of
2870 // existence (`first` below decides on the map, and the map is what we
2871 // just brought up to date).
2872 self.load_commit_times_from_fs();
2873 if applied > 0 {
2874 // Peer commits must reach `reader()` snapshots taken from here on.
2875 // A full fold is what open does; refresh does not build per-commit
2876 // deltas, so there is nothing cheaper that stays correct.
2877 self.fold_now();
2878 }
2879 Ok(applied as u64)
2880 }
2881
2882 /// Byte offset of the WAL prefix this handle has applied.
2883 ///
2884 /// Exposed for tests that assert the cursor tracks appended bytes exactly.
2885 #[doc(hidden)]
2886 pub fn wal_consumed(&self) -> u64 {
2887 self.wal_consumed
2888 }
2889
2890 /// Rewind the WAL cursor after the group-commit drain thread truncated a
2891 /// failed group off the tail, so the cursor still describes the file.
2892 pub(crate) fn set_wal_consumed(&mut self, len: u64) {
2893 self.wal_consumed = len;
2894 }
2895
2896 /// One non-blocking attempt at the cross-process write lock.
2897 ///
2898 /// Takes `&self` so a caller can poll for the lock *before* it acquires the
2899 /// in-process write guard. That ordering is what keeps a busy peer in
2900 /// another process from stalling this process's readers.
2901 ///
2902 /// A handle that owns the lock for its lifetime always succeeds.
2903 pub(crate) fn try_cross_process_lock(&self) -> Result<bool> {
2904 if self.holds_lifetime_lock {
2905 return Ok(true);
2906 }
2907 self.fs.try_lock_exclusive().map_err(GraphError::Io)
2908 }
2909
2910 /// Poll for the cross-process write lock until `wait` elapses.
2911 ///
2912 /// One attempt is always made, so a zero wait is a single try. Returns
2913 /// `false` when the lock is still held elsewhere at the deadline; nothing
2914 /// has been written and retrying later is safe.
2915 ///
2916 /// Only the plain-`GraphDb` open path uses this, where the caller owns the
2917 /// handle outright. [`SharedDb`](crate::SharedDb) polls
2918 /// [`try_cross_process_lock`](GraphDb::try_cross_process_lock) itself so
2919 /// that it holds no in-process guard while it waits.
2920 fn poll_lock(&self, wait: std::time::Duration) -> Result<bool> {
2921 let deadline = std::time::Instant::now() + wait;
2922 loop {
2923 if self.try_cross_process_lock()? {
2924 return Ok(true);
2925 }
2926 let now = std::time::Instant::now();
2927 if now >= deadline {
2928 return Ok(false);
2929 }
2930 std::thread::sleep(LOCK_POLL_INTERVAL.min(deadline.saturating_duration_since(now)));
2931 }
2932 }
2933
2934 /// Open a cross-process write scope, given the outcome of an already-made
2935 /// lock attempt.
2936 ///
2937 /// The caller polls for the lock first — outside any in-process guard — and
2938 /// passes what it got. On success this refreshes, so the writes about to
2939 /// happen land on top of every other process's commits. On failure the
2940 /// handle refuses WAL-appending mutations and `snapshot()` with
2941 /// [`GraphError::Busy`] until [`end_write_lock`](GraphDb::end_write_lock)
2942 /// closes the scope, so a caller holding a guard cannot write behind
2943 /// another process's back.
2944 ///
2945 /// A handle that already owns the lock for its lifetime skips the refresh:
2946 /// no other process can have written, so there is nothing to pick up.
2947 pub(crate) fn enter_write_scope(&mut self, acquired: bool) -> Result<()> {
2948 self.lock_denied = !acquired;
2949 if !acquired || self.holds_lifetime_lock {
2950 return Ok(());
2951 }
2952 if let Err(e) = self.refresh() {
2953 // Do not hold a lock we cannot use: release it and let the caller
2954 // see the underlying failure.
2955 let _ = self.fs.unlock();
2956 self.lock_denied = true;
2957 return Err(e);
2958 }
2959 Ok(())
2960 }
2961
2962 /// Close a cross-process write scope opened by
2963 /// [`enter_write_scope`](GraphDb::enter_write_scope): release the lock and
2964 /// clear the Busy latch. Safe to call when the lock was never taken.
2965 pub(crate) fn end_write_lock(&mut self) {
2966 self.lock_denied = false;
2967 if !self.holds_lifetime_lock {
2968 // Releasing a lock we do not hold is a no-op; a failure to release
2969 // is reported by the OS closing the descriptor at handle drop.
2970 let _ = self.fs.unlock();
2971 }
2972 }
2973
2974 /// As-of replay for [`GraphDb::open_at`]: snapshot base (only when the
2975 /// snapshot truncated the WAL) plus the first `commit + 1` WAL frames;
2976 /// see [`GraphDb::open_at`] for the semantics. The per-frame drain
2977 /// mirrors `open_with` exactly so pending_delta_count is 0 on exit.
2978 /// Restore all persisted state from a decoded snapshot. Shared by
2979 /// `open_with` and (when the snapshot truncated the WAL) `open_at_with`.
2980 fn restore_snapshot_state(
2981 &mut self,
2982 state: core_storage::snapshot::SnapshotState,
2983 ) -> Result<()> {
2984 self.ids = Arc::new(state.ids);
2985 self.syms = Arc::new(state.syms);
2986 self.topo = Arc::new(state.topo);
2987 self.props = Arc::new(state.props);
2988 self.labels = Arc::new(state.labels);
2989 self.edge_props = Arc::new(state.edge_props);
2990 // Cross-section label integrity for V5/V7 snapshots: same invariants as
2991 // restore_v8_base. A crafted bincode snapshot with a short `labels` vec,
2992 // out-of-range sym ids, or a sentinel label on a live node would otherwise
2993 // open successfully and panic later in `NodeRef::label()` or
2994 // `neighborhood_masked()`. Catching it here turns those into typed
2995 // `GraphError::Corrupt` at open time.
2996 {
2997 let ids_len = self.ids.len();
2998 if self.labels.len() != ids_len {
2999 return Err(GraphError::Corrupt {
3000 detail: format!(
3001 "snapshot: labels vec has {} entries but id table has {} total slots",
3002 self.labels.len(),
3003 ids_len,
3004 ),
3005 });
3006 }
3007 let syms_len = self.syms.len() as u32;
3008 for (i, &sym) in self.labels.iter().enumerate() {
3009 let is_tombstoned = self.ids.is_tombstoned(i as u32);
3010 if sym == u32::MAX {
3011 if !is_tombstoned {
3012 return Err(GraphError::Corrupt {
3013 detail: format!(
3014 "snapshot: live node at id slot {i} has sentinel label (u32::MAX)"
3015 ),
3016 });
3017 }
3018 } else if sym >= syms_len {
3019 return Err(GraphError::Corrupt {
3020 detail: format!(
3021 "snapshot: label at id slot {i} references sym {sym} \
3022 which is out of interner range ({syms_len})"
3023 ),
3024 });
3025 }
3026 }
3027 }
3028 let defs: Vec<RuleDef> = state
3029 .rule_defs
3030 .iter()
3031 .map(|b| {
3032 decode_rule_def(b).map_err(|e| GraphError::Corrupt {
3033 detail: format!("snapshot rule_def deserialize: {e}"),
3034 })
3035 })
3036 .collect::<Result<Vec<_>>>()?;
3037 self.engine =
3038 RuleEngine::from_persist(defs, state.provenance, state.rule_tripped, state.rule_fires);
3039 // Candidate indexes are rebuilt lazily on the first mutation (see
3040 // RuleEngine::on_node_changed). HNSW blobs and IVF centroids from the
3041 // snapshot are retained without deserializing so that:
3042 // - clean-open (empty WAL): indexes stay empty; blobs load on first
3043 // ANN query via ensure_hnsw_loaded, or on first mutation via the
3044 // lazy-init guard which calls reindex_all_load_state (the scan
3045 // skips the HNSW build for every side the blob supplies).
3046 // - WAL-present: open_with calls consume_retained_state_eager before
3047 // replay so HNSW/IVF are live before any record fires the hooks.
3048 let ivf_bytes = if state.ivf_state.is_empty() {
3049 Vec::new()
3050 } else {
3051 bincode::serialize(&state.ivf_state).expect("IVF state serialize cannot fail")
3052 };
3053 // Store blobs without eagerly deserializing them.
3054 // `self.ids` is the snapshot's id table at this point — WAL replay has
3055 // not run — so its length is the line an interrupted build is detected
3056 // against.
3057 let snapshot_ids = self.ids.len() as u32;
3058 self.engine
3059 .store_snapshot_state(state.hnsw_state, ivf_bytes, snapshot_ids);
3060 // Restore view defs from snapshot (V5).
3061 // The ColumnStore already contains view values from the snapshot;
3062 // use restore_view (no collision check, no backfill) so the store
3063 // is aware of the definitions. rebuild_all runs after WAL replay.
3064 for def_bytes in &state.view_defs {
3065 let def: ViewDef =
3066 bincode::deserialize(def_bytes).map_err(|e| GraphError::Corrupt {
3067 detail: format!("snapshot view_def deserialize: {e}"),
3068 })?;
3069 self.view_store
3070 .restore_view(def)
3071 .map_err(|e| GraphError::Corrupt {
3072 detail: format!("snapshot view restore: {e}"),
3073 })?;
3074 }
3075 Ok(())
3076 }
3077
3078 /// Restore all persisted state from a V8 `MappedBase` snapshot, **except**
3079 /// topology (`self.topo` stays empty and serves as the WAL-replay overlay).
3080 ///
3081 /// `self.props` IS fully materialised from the base so that HNSW/IVF blob
3082 /// deserialization and view rebuild have access to all column data.
3083 fn restore_v8_base(&mut self, mapped: Arc<core_storage::v8::MappedBase>) -> Result<()> {
3084 self.ids = Arc::new(archived_to_idmap(mapped.ids().map_err(|e| {
3085 GraphError::Corrupt {
3086 detail: format!("v8: ids section: {e:?}"),
3087 }
3088 })?));
3089 self.syms = Arc::new(archived_to_interner(mapped.syms().map_err(|e| {
3090 GraphError::Corrupt {
3091 detail: format!("v8: syms section: {e:?}"),
3092 }
3093 })?));
3094
3095 // C1: self.props is left as an empty overlay. Column reads go through
3096 // props_view() (ColumnsView::with_base), which consults the archived base
3097 // section zero-copy. This avoids the O(columns) heap copy at every open.
3098
3099 // self.topo deliberately left as Topology::new() — overlay path.
3100
3101 let meta = decode_meta(mapped.meta_bytes().map_err(|e| GraphError::Corrupt {
3102 detail: format!("v8: meta section: {e:?}"),
3103 })?)
3104 .map_err(|e| GraphError::Corrupt {
3105 detail: format!("v8: meta decode: {e:?}"),
3106 })?;
3107 self.labels = Arc::new(meta.labels);
3108 // Cross-section label integrity: labels must cover every id slot (live
3109 // and tombstoned), every non-sentinel sym must be within the interner's
3110 // bound, and no live (non-tombstoned) node may carry the u32::MAX
3111 // sentinel label. Without this check, a crafted snapshot where the META
3112 // section (small, CRC-validated) holds a short `labels` vec, out-of-range
3113 // sym ids, or a sentinel label on a live node, would open successfully
3114 // and then panic in `NodeRef::label()`, `neighborhood_masked()`, and
3115 // related read paths. Catching the inconsistency here converts those
3116 // panics into typed `GraphError::Corrupt` at open time.
3117 {
3118 let ids_len = self.ids.len();
3119 if self.labels.len() != ids_len {
3120 return Err(GraphError::Corrupt {
3121 detail: format!(
3122 "v8: labels section has {} entries but id table has {} total slots",
3123 self.labels.len(),
3124 ids_len,
3125 ),
3126 });
3127 }
3128 let syms_len = self.syms.len() as u32;
3129 for (i, &sym) in self.labels.iter().enumerate() {
3130 let is_tombstoned = self.ids.is_tombstoned(i as u32);
3131 if sym == u32::MAX {
3132 // Sentinel is only valid for tombstoned slots.
3133 if !is_tombstoned {
3134 return Err(GraphError::Corrupt {
3135 detail: format!(
3136 "v8: live node at id slot {i} has sentinel label (u32::MAX)"
3137 ),
3138 });
3139 }
3140 } else if sym >= syms_len {
3141 return Err(GraphError::Corrupt {
3142 detail: format!(
3143 "v8: label at id slot {i} references sym {sym} \
3144 which is out of interner range ({syms_len})"
3145 ),
3146 });
3147 }
3148 }
3149 }
3150 // C3: self.edge_props stays as an empty overlay. Reads go through
3151 // edge_props_view() which consults the mmap'd base section zero-copy
3152 // via EdgePropsView::with_base. No heap decode at open time.
3153
3154 // Restore rule engine.
3155 let (rule_def_bytes, rule_tripped, rule_fires) =
3156 archived_rules_meta_to_owned(mapped.rules_meta_section().map_err(|e| {
3157 GraphError::Corrupt {
3158 detail: format!("v8: rules_meta section: {e:?}"),
3159 }
3160 })?);
3161 let defs: Vec<RuleDef> = rule_def_bytes
3162 .iter()
3163 .map(|b| {
3164 decode_rule_def(b).map_err(|e| GraphError::Corrupt {
3165 detail: format!("v8: rule_def deserialize: {e}"),
3166 })
3167 })
3168 .collect::<Result<Vec<_>>>()?;
3169 self.engine = RuleEngine::from_persist(defs, BTreeMap::new(), rule_tripped, rule_fires);
3170 // C4+C5: provenance, HNSW, and IVF sections are NOT read here.
3171 // `ensure_v8_base_sections_loaded` reads them on first use from
3172 // `self.base` (set by the caller immediately after this returns).
3173 // A clean open touches only: header + IDS + SYMS + META + RULES_META.
3174
3175 // Restore view definitions.
3176 let view_defs =
3177 archived_views_to_owned(mapped.views_section().map_err(|e| GraphError::Corrupt {
3178 detail: format!("v8: views section: {e:?}"),
3179 })?);
3180 for def_bytes in &view_defs {
3181 let def: ViewDef =
3182 bincode::deserialize(def_bytes).map_err(|e| GraphError::Corrupt {
3183 detail: format!("v8: view_def deserialize: {e}"),
3184 })?;
3185 self.view_store
3186 .restore_view(def)
3187 .map_err(|e| GraphError::Corrupt {
3188 detail: format!("v8: view restore: {e}"),
3189 })?;
3190 }
3191 // Load the last-change map from section 11 (small section; load eagerly).
3192 // Pre-Task-3 snapshots lack this section; `last_change_bytes` returns &[]
3193 // in that case and `decode_last_change_bytes` returns an empty map.
3194 let last_change_raw = mapped
3195 .last_change_bytes()
3196 .map_err(|e| GraphError::Corrupt {
3197 detail: format!("v8: last_change section: {e:?}"),
3198 })?;
3199 self.last_change = decode_last_change_bytes(last_change_raw);
3200
3201 // Validate that all deferred sections (provenance, HNSW, IVF) fit within
3202 // the file. Pure bounds check — no bytes read, no page faults triggered.
3203 // Catches truncated snapshots at open time before the lazy deferred reads.
3204 mapped.validate_section_bounds().map_err(|e| match e {
3205 GraphError::Corrupt { detail } => GraphError::Corrupt {
3206 detail: format!("v8: section bounds: {detail}"),
3207 },
3208 other => other,
3209 })?;
3210 Ok(())
3211 }
3212
3213 /// Read provenance, HNSW, and IVF sections from the mmap base into the
3214 /// engine's retained fields on first call. Subsequent calls are a no-op
3215 /// (AtomicBool fast-path).
3216 ///
3217 /// Must be called before any code path that reads or mutates engine
3218 /// provenance, HNSW, or IVF state:
3219 /// - WAL replay (before `consume_retained_state_eager`)
3220 /// - First mutation (`log_then_apply_with`)
3221 /// - Read-only paths (`stats`, `explain`, `node_edges`)
3222 /// - Snapshot (`snapshot_with`)
3223 ///
3224 /// No-op for fresh stores and V5-V7 opens (`self.base` is `None`).
3225 fn ensure_v8_base_sections_loaded(&self) {
3226 use std::sync::atomic::Ordering;
3227 if self.v8_sections_loaded.load(Ordering::Acquire) {
3228 return;
3229 }
3230 let _guard = self
3231 .v8_sections_mutex
3232 .lock()
3233 .expect("v8 sections mutex poisoned");
3234 if self.v8_sections_loaded.load(Ordering::Acquire) {
3235 return; // another caller populated while we waited
3236 }
3237 let _t = std::time::Instant::now();
3238 if let Some(base) = &self.base {
3239 // Provenance: raw rkyv bytes; CRC validated inside section_bytes.
3240 // Bounds are already validated at open time (restore_v8_base →
3241 // validate_section_bounds) — unreachable post-validate_section_bounds;
3242 // unwrap_or_default is a safety belt against impossible errors.
3243 let prov_bytes = base
3244 .provenance_raw_bytes()
3245 .map(|b| b.to_vec())
3246 .unwrap_or_default();
3247 self.engine.store_provenance_bytes(prov_bytes);
3248 // HNSW: decode rkyv blobs into owned map.
3249 let hnsw_state = base
3250 .hnsw_section()
3251 .map(archived_hnsw_to_owned)
3252 .unwrap_or_default();
3253 // IVF: raw bincode bytes; deserialized on first mutation/query.
3254 let ivf_bytes = base.ivf_bytes().map(|b| b.to_vec()).unwrap_or_default();
3255 // Called before WAL replay on a WAL-present open (`open_with`) and
3256 // before any write on a clean one, so this is the snapshot's count.
3257 let snapshot_ids = self.ids.len() as u32;
3258 self.engine
3259 .store_snapshot_state(hnsw_state, ivf_bytes, snapshot_ids);
3260 }
3261 self.v8_sections_loaded.store(true, Ordering::Release);
3262 if std::env::var("MUSHROOMDB_TRACE_OPEN").is_ok() {
3263 eprintln!(
3264 "[MUSHROOMDB_TRACE_OPEN] ensure_v8_base_sections_loaded: {:>9.3?}",
3265 _t.elapsed()
3266 );
3267 }
3268 }
3269
3270 /// Return a `TopologyView` that merges the mmap'd base (when present) with
3271 /// the in-memory WAL overlay. Used by all read paths in db.rs that need
3272 /// the full merged topology without going through `self.view()`.
3273 fn topo_view(&self) -> TopologyView<'_> {
3274 match self.base {
3275 None => TopologyView::owned(&self.topo),
3276 Some(ref base) => {
3277 // SAFETY: base lives as long as self; section bounds validated at open.
3278 // topology() uses access_unchecked; all field reads are bounds-checked in seam.rs.
3279 let archived = base
3280 .topology()
3281 .expect("base topology section bounds validated at open");
3282 TopologyView::with_base(&self.topo, archived)
3283 }
3284 }
3285 }
3286
3287 /// Return a `ColumnsView` that merges the mmap'd base columns (when a V8
3288 /// snapshot is open) with the in-memory WAL overlay. Reads consult the
3289 /// overlay first, then fall through to the archived base section zero-copy.
3290 fn props_view(&self) -> core_storage::v8::seam::ColumnsView<'_> {
3291 match self.base {
3292 None => core_storage::v8::seam::ColumnsView::owned(&self.props),
3293 Some(ref base) => {
3294 // columns() uses access_unchecked; field reads are bounds-checked in seam.rs.
3295 let archived = base
3296 .columns()
3297 .expect("base columns section bounds validated at open");
3298 core_storage::v8::seam::ColumnsView::with_base_cached(
3299 &self.props,
3300 archived,
3301 base.mixed_cache(),
3302 )
3303 .with_shared_strings(base_string_table(base))
3304 }
3305 }
3306 }
3307
3308 /// Return an `EdgePropsView` that merges the mmap'd base edge-props section
3309 /// (when a V8 snapshot is open) with the in-memory WAL overlay.
3310 ///
3311 /// Reads consult the overlay first (for post-snapshot mutations), then fall
3312 /// through to the archived base section zero-copy. Tombstones in the
3313 /// overlay mask deleted-from-base entries.
3314 fn edge_props_view(&self) -> EdgePropsView<'_> {
3315 match self.base {
3316 None => EdgePropsView::owned(&self.edge_props),
3317 Some(ref base) => {
3318 // edge_props_section() uses access_unchecked; field reads bounds-checked in seam.rs.
3319 let archived = base
3320 .edge_props_section()
3321 .expect("base edge_props section bounds validated at open");
3322 EdgePropsView::with_base(&self.edge_props, archived)
3323 }
3324 }
3325 }
3326
3327 fn open_at_with(fs: F, commit: u64) -> Result<Self> {
3328 // An as-of view never writes and is pinned to one commit: it takes no
3329 // cross-process lock and does not follow later commits.
3330 let mut db = Self::new_empty(
3331 fs,
3332 OpenOptions {
3333 repair_wal: false,
3334 auto_migrate: false,
3335 read_only: true,
3336 },
3337 );
3338 db.pinned = true; // read_only is set after replay, but pinning is immediate
3339 db.wal_horizon_floor = db.fs.read_horizon_floor()?;
3340 db.archive_genesis_chain = db.fs.has_genesis_marker();
3341 // Same orphaned-archive cleanup as open_with: floor was written first
3342 // during pruning, so a crash may have left stale archives below floor.
3343 db.cleanup_orphaned_archives()?;
3344 // Collect archive frames (oldest-first) and live WAL frames.
3345 // Archives represent pre-snapshot history; the snapshot captures the
3346 // cumulative state at the time of archiving. Crash-window guarantee:
3347 // A: crash before rename → WAL intact, no archive. Reopen: normal.
3348 // B: crash after rename, before new WAL → archive present, WAL
3349 // absent. Reopen: snapshot loaded (full state), no WAL replay.
3350 // C: crash after new baseline WAL written → normal post-archive.
3351 let archive_ns = db.fs.list_archives()?;
3352 let mut archive_frames_all: Vec<WalRecord> = Vec::new();
3353 for n in &archive_ns {
3354 let arc_bytes = db.fs.read_archive(*n)?;
3355 let (arc_frames, _) = decode_all(&arc_bytes);
3356 archive_frames_all.extend(arc_frames);
3357 }
3358 let total_archive_frames = archive_frames_all.len() as u64;
3359
3360 let live_bytes = db.fs.read(FileId::Wal)?;
3361 let (live_records, _valid_len) = decode_all(&live_bytes);
3362 let total_surviving = total_archive_frames + live_records.len() as u64;
3363 // Global total including any pruned history below the horizon floor.
3364 let total = db.wal_horizon_floor + total_surviving;
3365
3366 // Horizon and range check.
3367 if commit < db.wal_horizon_floor {
3368 return Err(GraphError::CommitOutOfRange {
3369 commit,
3370 total,
3371 floor: db.wal_horizon_floor,
3372 });
3373 }
3374 if commit >= total {
3375 return Err(GraphError::CommitOutOfRange {
3376 commit,
3377 total,
3378 floor: db.wal_horizon_floor,
3379 });
3380 }
3381
3382 // Local index into surviving frames (0 = first frame of oldest archive).
3383 let local = commit - db.wal_horizon_floor;
3384
3385 if local < total_archive_frames {
3386 // Target commit is in an archive. Correct replay from empty state
3387 // is only possible when the archive chain is an uninterrupted
3388 // genesis chain (first archive taken from a fresh store, no prior
3389 // WAL truncation) and no archives have been pruned (floor == 0).
3390 //
3391 // If either condition is violated the prefix needed to reconstruct
3392 // the requested state is gone; refuse rather than return wrong data.
3393 if db.wal_horizon_floor > 0 || !db.archive_genesis_chain {
3394 return Err(GraphError::CommitOutOfRange {
3395 commit,
3396 total,
3397 floor: db.wal_horizon_floor,
3398 });
3399 }
3400 // Replay all archive frames up to and including the target commit
3401 // from an empty database state. Archives must be replayed in order
3402 // so that dense-id intern tables are built up correctly.
3403 for rec in archive_frames_all.into_iter().take((local + 1) as usize) {
3404 db.apply(&rec)?;
3405 let _ = db.engine.drain_deltas();
3406 }
3407 } else {
3408 // Target commit is in the live WAL: load snapshot as base, then
3409 // replay the needed live WAL prefix.
3410 //
3411 // Base state: a truncating snapshot (wal_truncated=true) compacts
3412 // all pre-truncation / pre-archive commits. Dense-id records in
3413 // the live WAL reference ids/interns that the snapshot provides.
3414 // Peek 6 bytes (same pattern as open_with).
3415 let snap_header = db.fs.read_prefix(FileId::Snapshot, 6)?;
3416 let is_v8 = snap_header.len() >= 6
3417 && &snap_header[0..4] == b"GDB1"
3418 && core_storage::snapshot::is_mmap_container(u16::from_le_bytes([
3419 snap_header[4],
3420 snap_header[5],
3421 ]));
3422 if is_v8 {
3423 let state = if let Some(snap_path) = db.fs.snapshot_path() {
3424 let mapped = core_storage::v8::MappedBase::map(&snap_path).map_err(|e| {
3425 GraphError::Corrupt {
3426 detail: format!("v8: open_at mmap: {e:?}"),
3427 }
3428 })?;
3429 core_storage::snapshot::decode_v8_from_mapped(&mapped)?
3430 } else {
3431 let snap_bytes = db.fs.read(FileId::Snapshot)?;
3432 core_storage::snapshot::decode(&snap_bytes)?
3433 };
3434 if let Some(state) = state {
3435 if state.wal_truncated {
3436 db.restore_snapshot_state(state)?;
3437 }
3438 }
3439 } else if !snap_header.is_empty() {
3440 let snap_bytes = db.fs.read(FileId::Snapshot)?;
3441 if let Some(state) = core_storage::snapshot::decode(&snap_bytes)? {
3442 if state.wal_truncated {
3443 db.restore_snapshot_state(state)?;
3444 }
3445 }
3446 }
3447 // else: snap_header empty = no snapshot file.
3448 let live_local = local - total_archive_frames;
3449 for rec in live_records.into_iter().take((live_local + 1) as usize) {
3450 db.apply(&rec)?;
3451 let _ = db.engine.drain_deltas();
3452 }
3453 }
3454 // Pin: pending_delta_count must be 0 after as-of replay, mirroring T1's
3455 // post-loop assert in open_with.
3456 debug_assert_eq!(
3457 db.engine.pending_delta_count(),
3458 0,
3459 "pending_deltas non-empty after open_at replay — \
3460 per-frame drain must run inside the loop to keep memory O(1)"
3461 );
3462 let _ = db.engine.drain_deltas(); // belt-and-braces no-op
3463 // Rebuild view values after WAL replay so derived-edge-driven views
3464 // reflect the as-of state. open_at always uses the legacy path (no V8
3465 // base), so topo_view is always owned.
3466 {
3467 let topo_view = TopologyView::owned(&db.topo);
3468 db.view_store.rebuild_all(
3469 Arc::make_mut(&mut db.props),
3470 &topo_view,
3471 &db.ids,
3472 &db.syms,
3473 &db.labels,
3474 );
3475 }
3476 // Rebuild full-text index for as-of view (mirrors open_with pattern).
3477 Arc::make_mut(&mut db.fulltext).rebuild_all(
3478 &db.ids,
3479 &db.labels,
3480 &db.syms,
3481 build_props_view(&db.props, &db.base),
3482 );
3483 db.prop_index.rebuild_all(
3484 &db.ids,
3485 &db.labels,
3486 &db.syms,
3487 build_props_view(&db.props, &db.base),
3488 );
3489 // Namespaces on the temporal handle, built by the same pass the live
3490 // open uses, so an as-of mask narrows by the namespaces of that commit.
3491 db.rebuild_node_ns();
3492 // Load roles sidecar (current roles, not point-in-time).
3493 db.roles = Self::load_roles_from_fs(&db.fs)?;
3494 db.read_only = true;
3495 db.total_wal_commits = total;
3496 // Capture initial fold so reader() is immediately usable.
3497 db.fold_now();
3498 Ok(db)
3499 }
3500
3501 /// Whether this instance is a read-only as-of view.
3502 pub fn is_read_only(&self) -> bool {
3503 self.read_only
3504 }
3505
3506 // ── MVCC epoch reader ─────────────────────────────────────────────────────
3507
3508 /// Clone the current overlay state into a new `FrozenOverlay` and reset
3509 /// the delta tail. Called automatically every `FOLD_EVERY_K` commits and at
3510 /// the end of `open_with` / `open_at_with` to prime the reader.
3511 fn fold_now(&mut self) {
3512 // Eight `Arc::clone`s — refcount bumps, O(1). This used to deep-copy the
3513 // whole overlay: `IdMap` alone is a `HashMap<String, u32>` plus a
3514 // `Vec<String>`, so every node key was copied twice, on every open,
3515 // after every snapshot, every FOLD_EVERY_K commits on the write path,
3516 // and — with no commit threshold — on every `refresh()` that applied a
3517 // peer commit. A refreshing reader now pays nothing for a fold.
3518 //
3519 // `props` and `topo` were already cheap for a different reason: on a
3520 // snapshotted store `restore_v8_base` leaves them as empty overlays over
3521 // the zero-copy mmap. `ids`, `syms` and `fulltext` were not, and that
3522 // inconsistency was the defect.
3523 let frozen = crate::reader::FrozenOverlay {
3524 ids: std::sync::Arc::clone(&self.ids),
3525 syms: std::sync::Arc::clone(&self.syms),
3526 topo: std::sync::Arc::clone(&self.topo),
3527 props: std::sync::Arc::clone(&self.props),
3528 labels: std::sync::Arc::clone(&self.labels),
3529 edge_props: std::sync::Arc::clone(&self.edge_props),
3530 roles: self.roles.clone().map(std::sync::Arc::new),
3531 fulltext: std::sync::Arc::clone(&self.fulltext),
3532 };
3533 self.fold_overlay = Some(Arc::new(frozen));
3534 self.delta_tail.clear();
3535 self.commits_since_fold = 0;
3536 }
3537
3538 /// Capture a lock-free reader snapshot of the current db state.
3539 ///
3540 /// The read lock is held only for the duration of this call (to clone a
3541 /// handful of `Arc` handles). Subsequent query operations run without any
3542 /// lock.
3543 pub fn reader(&self) -> crate::reader::ReaderSnapshot {
3544 crate::reader::ReaderSnapshot::new(
3545 self.fold_overlay
3546 .clone()
3547 .expect("fold_overlay is always Some after open_with; call reader() after open"),
3548 self.base.clone(),
3549 self.delta_tail.clone(),
3550 // The snapshot's effective state is exactly this handle's state at
3551 // this commit, so it shares the memo and its version key.
3552 self.commit_seq,
3553 Arc::clone(&self.role_masks),
3554 )
3555 }
3556
3557 /// Append a delta the reader cannot apply, so that a corrupt overlay is
3558 /// reachable from a test.
3559 ///
3560 /// Compiled only under `test-hooks`, which the server's dev-dependency on
3561 /// this crate turns on. One call permanently corrupts every
3562 /// [`ReaderSnapshot`](crate::reader::ReaderSnapshot) taken from the handle,
3563 /// so it must not be in the published surface: `#[doc(hidden)]` hides it
3564 /// from rustdoc and from nothing else. The feature gate — not
3565 /// `#[cfg(test)]` — because its only callers are in `crates/server/tests`,
3566 /// a different crate, exactly as `core_rules`'s index counters are.
3567 ///
3568 /// [`ReaderSnapshot::effective`](crate::reader::ReaderSnapshot) folds the
3569 /// delta tail into a clone of the frozen overlay and answers
3570 /// [`GraphError::Corrupt`] when a record will not apply. Nothing a caller
3571 /// can do produces that state — `apply_one`'s failures are disagreements
3572 /// between the tail and the fold it is applied to, which the write path
3573 /// cannot create — so the `Corrupt` arm of every scoped reader method was
3574 /// reachable only by inspection until this hook existed. An `Intern` record
3575 /// claiming an id the frozen interner will not hand back is the smallest
3576 /// such disagreement.
3577 ///
3578 /// Only the tail is touched. This handle's own state is untouched and
3579 /// `commit_seq` does not move, so a role mask already memoised at this
3580 /// version stays memoised — which is exactly the state in which the HTTP
3581 /// role branches reach a scoped read with a corrupt overlay under them.
3582 #[cfg(any(test, feature = "test-hooks"))]
3583 #[doc(hidden)]
3584 pub fn push_unapplyable_delta_for_test(&mut self) {
3585 self.delta_tail.push(Arc::new(crate::reader::CommitDelta {
3586 records: vec![WalRecord::Intern {
3587 id: u32::MAX,
3588 text: "delta-tail-corruption".into(),
3589 }],
3590 derived_inserts: Vec::new(),
3591 derived_deletes: Vec::new(),
3592 }));
3593 }
3594
3595 /// Total number of WAL commits at the time [`open_at`] was called.
3596 /// Returns 0 for normal (non-as-of) instances.
3597 pub fn total_wal_commits(&self) -> u64 {
3598 self.total_wal_commits
3599 }
3600
3601 /// Apply a record to in-memory state. Used by both live writes and replay,
3602 /// so replay is definitionally identical to the original execution.
3603 fn apply(&mut self, rec: &WalRecord) -> Result<()> {
3604 // Before the record mutates anything: a store restored from a snapshot
3605 // defers building its candidate indexes until the first write, and that
3606 // build is a full node scan. Left where it used to fire — inside the
3607 // engine hook, after `props.set` and the label assignment — the scan
3608 // read the half-applied record and took the in-flight node's vector for
3609 // one the snapshot should have carried, which read as an interrupted
3610 // vector-index build and cost a full `RebuildRule` on the first
3611 // embedded write after every reopen. Hoisted here the scan sees exactly
3612 // the persisted state; the record's own hook then files its vector
3613 // through the ordinary insert path a line later.
3614 self.populate_indexes_before_write();
3615 match rec {
3616 WalRecord::InsertNode { label, key, props } => {
3617 let id = Arc::make_mut(&mut self.ids).try_insert(key)?;
3618 let sym = Arc::make_mut(&mut self.syms).intern(label);
3619 if self.labels.len() <= id as usize {
3620 // gap slots are sentinels, never valid label symbols
3621 Arc::make_mut(&mut self.labels).resize(id as usize + 1, u32::MAX);
3622 }
3623 Arc::make_mut(&mut self.labels)[id as usize] = sym;
3624 let mut ns_name = NS_DEFAULT.to_string();
3625 for (field, value) in props {
3626 if field == NS_PROP {
3627 ns_name = namespace_of_value(Some(value)).to_string();
3628 }
3629 Arc::make_mut(&mut self.props).set(id, field, value.clone());
3630 }
3631 self.set_node_ns(id, &ns_name);
3632 // Initialize view values for the new node before the engine runs so
3633 // delta-based increments start from a known zero baseline.
3634 self.view_store.init_node_views(
3635 id,
3636 Arc::make_mut(&mut self.props),
3637 &self.syms,
3638 &self.labels,
3639 );
3640 // Fire rules for the newly inserted node.
3641 let cursor = self.engine.pending_delta_count();
3642 let mut eng = std::mem::take(&mut self.engine);
3643 {
3644 let mut gm = make_graph_mut(
3645 &self.ids,
3646 Arc::make_mut(&mut self.syms),
3647 &self.labels,
3648 build_props_view(&self.props, &self.base),
3649 Arc::make_mut(&mut self.topo),
3650 &self.base,
3651 Arc::make_mut(&mut self.edge_props),
3652 );
3653 eng.on_node_changed(id, None, &mut gm);
3654 }
3655 self.engine = eng;
3656 // Process derived-edge deltas for view maintenance.
3657 // Fast path: skip the O(delta_count) allocation when no views exist.
3658 if !self.view_store.is_empty() {
3659 #[cfg(test)]
3660 DELTA_COPY_COUNT.with(|c| c.set(c.get() + 1));
3661 let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
3662 for d in &new_deltas {
3663 self.view_store.on_edge_changed(
3664 d.etype_sym,
3665 d.src_id,
3666 d.dst_id,
3667 d.fired,
3668 Arc::make_mut(&mut self.props),
3669 &build_topo_view(&self.topo, &self.base),
3670 &self.ids,
3671 &self.syms,
3672 &self.labels,
3673 base_columns(&self.base),
3674 );
3675 }
3676 }
3677 // Full-text index maintenance: index enabled fields for this label.
3678 if self.fulltext.has_label(label) {
3679 for (field, value) in props {
3680 if self.fulltext.is_enabled(label, field) {
3681 Arc::make_mut(&mut self.fulltext).add_tokens(id, field, value);
3682 }
3683 }
3684 }
3685 // Property (equality) index maintenance.
3686 if self.prop_index.has_label(label) {
3687 for (field, value) in props {
3688 self.prop_index.set(label, field, id, value);
3689 }
3690 }
3691 }
3692 WalRecord::InsertEdge {
3693 edge_type,
3694 src_key,
3695 dst_key,
3696 } => {
3697 let src = self.ids.get(src_key).ok_or_else(|| GraphError::Corrupt {
3698 detail: format!("wal replay references unknown key {src_key}"),
3699 })?;
3700 let dst = self.ids.get(dst_key).ok_or_else(|| GraphError::Corrupt {
3701 detail: format!("wal replay references unknown key {dst_key}"),
3702 })?;
3703 let etype = Arc::make_mut(&mut self.syms).intern(edge_type);
3704 // Skip if the edge is already visible in the merged base+overlay
3705 // view. This keeps WAL replay idempotent when the WAL contains
3706 // pre-snapshot records that are already encoded in a V8 base
3707 // (keep_wal=true opens and crash-before-truncation scenarios).
3708 if self.base.is_some()
3709 && self
3710 .topo_view()
3711 .neighbors(etype, Direction::Out, src)
3712 .contains(&dst)
3713 {
3714 return Ok(());
3715 }
3716 Arc::make_mut(&mut self.topo).add_edge(etype, src, dst);
3717 // View maintenance for manual edge insert.
3718 self.view_store.on_edge_changed(
3719 etype,
3720 src,
3721 dst,
3722 true,
3723 Arc::make_mut(&mut self.props),
3724 &build_topo_view(&self.topo, &self.base),
3725 &self.ids,
3726 &self.syms,
3727 &self.labels,
3728 base_columns(&self.base),
3729 );
3730 // Rule engine: via-hop rules must update when user edges change.
3731 let cursor = self.engine.pending_delta_count();
3732 let mut eng = std::mem::take(&mut self.engine);
3733 {
3734 let mut gm = make_graph_mut(
3735 &self.ids,
3736 Arc::make_mut(&mut self.syms),
3737 &self.labels,
3738 build_props_view(&self.props, &self.base),
3739 Arc::make_mut(&mut self.topo),
3740 &self.base,
3741 Arc::make_mut(&mut self.edge_props),
3742 );
3743 eng.on_edge_changed(edge_type, src, dst, &mut gm);
3744 }
3745 self.engine = eng;
3746 if !self.view_store.is_empty() {
3747 let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
3748 for d in &new_deltas {
3749 self.view_store.on_edge_changed(
3750 d.etype_sym,
3751 d.src_id,
3752 d.dst_id,
3753 d.fired,
3754 Arc::make_mut(&mut self.props),
3755 &build_topo_view(&self.topo, &self.base),
3756 &self.ids,
3757 &self.syms,
3758 &self.labels,
3759 base_columns(&self.base),
3760 );
3761 }
3762 }
3763 }
3764 WalRecord::SetProp { key, field, value } => {
3765 let id = self.ids.get(key).ok_or_else(|| GraphError::Corrupt {
3766 detail: format!("wal replay references unknown key {key}"),
3767 })?;
3768 let old_value = build_props_view(&self.props, &self.base)
3769 .get(id, field)
3770 .map(|vr| vr.into_value());
3771 Arc::make_mut(&mut self.props).set(id, field, value.clone());
3772 // Fire rules for the changed field.
3773 let cursor = self.engine.pending_delta_count();
3774 let mut eng = std::mem::take(&mut self.engine);
3775 {
3776 let mut gm = make_graph_mut(
3777 &self.ids,
3778 Arc::make_mut(&mut self.syms),
3779 &self.labels,
3780 build_props_view(&self.props, &self.base),
3781 Arc::make_mut(&mut self.topo),
3782 &self.base,
3783 Arc::make_mut(&mut self.edge_props),
3784 );
3785 eng.on_node_changed(id, Some((field, old_value)), &mut gm);
3786 }
3787 self.engine = eng;
3788 // Derived-edge deltas → view updates.
3789 if !self.view_store.is_empty() {
3790 #[cfg(test)]
3791 DELTA_COPY_COUNT.with(|c| c.set(c.get() + 1));
3792 let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
3793 for d in &new_deltas {
3794 self.view_store.on_edge_changed(
3795 d.etype_sym,
3796 d.src_id,
3797 d.dst_id,
3798 d.fired,
3799 Arc::make_mut(&mut self.props),
3800 &build_topo_view(&self.topo, &self.base),
3801 &self.ids,
3802 &self.syms,
3803 &self.labels,
3804 base_columns(&self.base),
3805 );
3806 }
3807 }
3808 // Neighbor-aggregate views that read `field` must also update.
3809 self.view_store.on_prop_changed(
3810 id,
3811 field,
3812 Arc::make_mut(&mut self.props),
3813 &build_topo_view(&self.topo, &self.base),
3814 &self.ids,
3815 &self.syms,
3816 &self.labels,
3817 base_columns(&self.base),
3818 );
3819 // Full-text index maintenance: update tokens for this field if indexed.
3820 if self.fulltext.field_indexed(field) {
3821 let label_opt = self.labels.get(id as usize).and_then(|&sym| {
3822 if sym == u32::MAX {
3823 None
3824 } else {
3825 self.syms.resolve(sym)
3826 }
3827 });
3828 if let Some(label) = label_opt {
3829 if self.fulltext.is_enabled(label, field) {
3830 Arc::make_mut(&mut self.fulltext).remove_node_field(id, field);
3831 Arc::make_mut(&mut self.fulltext).add_tokens(id, field, value);
3832 }
3833 }
3834 }
3835 // Property (equality) index maintenance: re-key this node's value.
3836 if self.prop_index.field_indexed(field) {
3837 let label_opt = self.labels.get(id as usize).and_then(|&sym| {
3838 if sym == u32::MAX {
3839 None
3840 } else {
3841 self.syms.resolve(sym)
3842 }
3843 });
3844 if let Some(label) = label_opt {
3845 self.prop_index.set(label, field, id, value);
3846 }
3847 }
3848 }
3849 WalRecord::Intern { id, text } => {
3850 if let Some(existing) = self.syms.get(text) {
3851 if existing != *id {
3852 return Err(GraphError::Corrupt {
3853 detail: format!(
3854 "wal intern mismatch for {text:?}: have {existing}, record {id}"
3855 ),
3856 });
3857 }
3858 } else {
3859 let got = Arc::make_mut(&mut self.syms).intern(text);
3860 if got != *id {
3861 return Err(GraphError::Corrupt {
3862 detail: format!(
3863 "wal intern assigned {got} for {text:?}, record wanted {id}"
3864 ),
3865 });
3866 }
3867 }
3868 }
3869 WalRecord::InsertNodeId { label, key, props } => {
3870 let id = Arc::make_mut(&mut self.ids).try_insert(key)?;
3871 if self.labels.len() <= id as usize {
3872 Arc::make_mut(&mut self.labels).resize(id as usize + 1, u32::MAX);
3873 }
3874 Arc::make_mut(&mut self.labels)[id as usize] = *label;
3875 let label_str = self
3876 .syms
3877 .resolve(*label)
3878 .ok_or_else(|| GraphError::Corrupt {
3879 detail: format!("wal InsertNodeId unknown label intern {label}"),
3880 })?
3881 .to_string();
3882 let mut ns_name = NS_DEFAULT.to_string();
3883 for (field_sym, value) in props {
3884 let field =
3885 self.syms
3886 .resolve(*field_sym)
3887 .ok_or_else(|| GraphError::Corrupt {
3888 detail: format!(
3889 "wal InsertNodeId unknown field intern {field_sym}"
3890 ),
3891 })?;
3892 if field == NS_PROP {
3893 ns_name = namespace_of_value(Some(value)).to_string();
3894 }
3895 Arc::make_mut(&mut self.props).set(id, field, value.clone());
3896 }
3897 self.set_node_ns(id, &ns_name);
3898 self.view_store.init_node_views(
3899 id,
3900 Arc::make_mut(&mut self.props),
3901 &self.syms,
3902 &self.labels,
3903 );
3904 let cursor = self.engine.pending_delta_count();
3905 let mut eng = std::mem::take(&mut self.engine);
3906 {
3907 let mut gm = make_graph_mut(
3908 &self.ids,
3909 Arc::make_mut(&mut self.syms),
3910 &self.labels,
3911 build_props_view(&self.props, &self.base),
3912 Arc::make_mut(&mut self.topo),
3913 &self.base,
3914 Arc::make_mut(&mut self.edge_props),
3915 );
3916 eng.on_node_changed(id, None, &mut gm);
3917 }
3918 self.engine = eng;
3919 if !self.view_store.is_empty() {
3920 #[cfg(test)]
3921 DELTA_COPY_COUNT.with(|c| c.set(c.get() + 1));
3922 let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
3923 for d in &new_deltas {
3924 self.view_store.on_edge_changed(
3925 d.etype_sym,
3926 d.src_id,
3927 d.dst_id,
3928 d.fired,
3929 Arc::make_mut(&mut self.props),
3930 &build_topo_view(&self.topo, &self.base),
3931 &self.ids,
3932 &self.syms,
3933 &self.labels,
3934 base_columns(&self.base),
3935 );
3936 }
3937 }
3938 if self.fulltext.has_label(&label_str) {
3939 for (field_sym, value) in props {
3940 let Some(field) = self.syms.resolve(*field_sym) else {
3941 continue;
3942 };
3943 if self.fulltext.is_enabled(&label_str, field) {
3944 Arc::make_mut(&mut self.fulltext).add_tokens(id, field, value);
3945 }
3946 }
3947 }
3948 if self.prop_index.has_label(&label_str) {
3949 for (field_sym, value) in props {
3950 let Some(field) = self.syms.resolve(*field_sym) else {
3951 continue;
3952 };
3953 self.prop_index.set(&label_str, field, id, value);
3954 }
3955 }
3956 }
3957 WalRecord::InsertEdgeId { etype, src, dst } => {
3958 // Replay-over-snapshot: dense ids in the pre-snapshot WAL may
3959 // already be tombstoned. Skip rather than attaching edges to
3960 // dead ids (DeleteNode keys the live re-insert, not the old id).
3961 if self.ids.is_tombstoned(*src)
3962 || self.ids.is_tombstoned(*dst)
3963 || self.ids.key_of(*src).is_none()
3964 || self.ids.key_of(*dst).is_none()
3965 {
3966 return Ok(());
3967 }
3968 // Skip if already visible in the merged view (same idempotency
3969 // guard as InsertEdge above: prevents double-counting when
3970 // pre-snapshot WAL records are replayed over a V8 base).
3971 if self.base.is_some()
3972 && self
3973 .topo_view()
3974 .neighbors(*etype, Direction::Out, *src)
3975 .contains(dst)
3976 {
3977 return Ok(());
3978 }
3979 Arc::make_mut(&mut self.topo).add_edge(*etype, *src, *dst);
3980 self.view_store.on_edge_changed(
3981 *etype,
3982 *src,
3983 *dst,
3984 true,
3985 Arc::make_mut(&mut self.props),
3986 &build_topo_view(&self.topo, &self.base),
3987 &self.ids,
3988 &self.syms,
3989 &self.labels,
3990 base_columns(&self.base),
3991 );
3992 // Rule engine: via-hop rules fire when user via-edges are inserted.
3993 // Resolve etype back to string so on_edge_changed can match rules by name.
3994 if let Some(etype_str) = self.syms.resolve(*etype).map(|s| s.to_string()) {
3995 let cursor = self.engine.pending_delta_count();
3996 let mut eng = std::mem::take(&mut self.engine);
3997 {
3998 let mut gm = make_graph_mut(
3999 &self.ids,
4000 Arc::make_mut(&mut self.syms),
4001 &self.labels,
4002 build_props_view(&self.props, &self.base),
4003 Arc::make_mut(&mut self.topo),
4004 &self.base,
4005 Arc::make_mut(&mut self.edge_props),
4006 );
4007 eng.on_edge_changed(&etype_str, *src, *dst, &mut gm);
4008 }
4009 self.engine = eng;
4010 if !self.view_store.is_empty() {
4011 let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
4012 for d in &new_deltas {
4013 self.view_store.on_edge_changed(
4014 d.etype_sym,
4015 d.src_id,
4016 d.dst_id,
4017 d.fired,
4018 Arc::make_mut(&mut self.props),
4019 &build_topo_view(&self.topo, &self.base),
4020 &self.ids,
4021 &self.syms,
4022 &self.labels,
4023 base_columns(&self.base),
4024 );
4025 }
4026 }
4027 }
4028 }
4029 WalRecord::SetPropId { id, field, value } => {
4030 if self.ids.is_tombstoned(*id) || self.ids.key_of(*id).is_none() {
4031 return Ok(());
4032 }
4033 let field_str = self
4034 .syms
4035 .resolve(*field)
4036 .ok_or_else(|| GraphError::Corrupt {
4037 detail: format!("wal SetPropId unknown field intern {field}"),
4038 })?
4039 .to_string();
4040 let old_value = build_props_view(&self.props, &self.base)
4041 .get(*id, &field_str)
4042 .map(|vr| vr.into_value());
4043 Arc::make_mut(&mut self.props).set(*id, &field_str, value.clone());
4044 let cursor = self.engine.pending_delta_count();
4045 let mut eng = std::mem::take(&mut self.engine);
4046 {
4047 let mut gm = make_graph_mut(
4048 &self.ids,
4049 Arc::make_mut(&mut self.syms),
4050 &self.labels,
4051 build_props_view(&self.props, &self.base),
4052 Arc::make_mut(&mut self.topo),
4053 &self.base,
4054 Arc::make_mut(&mut self.edge_props),
4055 );
4056 eng.on_node_changed(*id, Some((field_str.as_str(), old_value)), &mut gm);
4057 }
4058 self.engine = eng;
4059 if !self.view_store.is_empty() {
4060 #[cfg(test)]
4061 DELTA_COPY_COUNT.with(|c| c.set(c.get() + 1));
4062 let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
4063 for d in &new_deltas {
4064 self.view_store.on_edge_changed(
4065 d.etype_sym,
4066 d.src_id,
4067 d.dst_id,
4068 d.fired,
4069 Arc::make_mut(&mut self.props),
4070 &build_topo_view(&self.topo, &self.base),
4071 &self.ids,
4072 &self.syms,
4073 &self.labels,
4074 base_columns(&self.base),
4075 );
4076 }
4077 }
4078 self.view_store.on_prop_changed(
4079 *id,
4080 &field_str,
4081 Arc::make_mut(&mut self.props),
4082 &build_topo_view(&self.topo, &self.base),
4083 &self.ids,
4084 &self.syms,
4085 &self.labels,
4086 base_columns(&self.base),
4087 );
4088 if self.fulltext.field_indexed(&field_str) {
4089 let label_opt = self.labels.get(*id as usize).and_then(|&sym| {
4090 if sym == u32::MAX {
4091 None
4092 } else {
4093 self.syms.resolve(sym)
4094 }
4095 });
4096 if let Some(label) = label_opt {
4097 if self.fulltext.is_enabled(label, &field_str) {
4098 Arc::make_mut(&mut self.fulltext).remove_node_field(*id, &field_str);
4099 Arc::make_mut(&mut self.fulltext).add_tokens(*id, &field_str, value);
4100 }
4101 }
4102 }
4103 if self.prop_index.field_indexed(&field_str) {
4104 let label_opt = self.labels.get(*id as usize).and_then(|&sym| {
4105 if sym == u32::MAX {
4106 None
4107 } else {
4108 self.syms.resolve(sym)
4109 }
4110 });
4111 if let Some(label) = label_opt {
4112 self.prop_index.set(label, &field_str, *id, value);
4113 }
4114 }
4115 }
4116 WalRecord::CreateRule { def_bytes } => {
4117 let def: RuleDef = decode_rule_def(def_bytes).map_err(|e| GraphError::Corrupt {
4118 detail: format!("CreateRule def_bytes deserialize failed: {e}"),
4119 })?;
4120 // Replay-over-snapshot idempotency: the rule was captured in the snapshot
4121 // so the engine already has it; silently skip to avoid a spurious
4122 // RuleInvalid error in the crash window between snapshot write and WAL
4123 // truncation.
4124 if self.engine.rules().any(|r| r.name == def.name) {
4125 return Ok(());
4126 }
4127 let cursor = self.engine.pending_delta_count();
4128 let mut eng = std::mem::take(&mut self.engine);
4129 let result = {
4130 let mut gm = make_graph_mut(
4131 &self.ids,
4132 Arc::make_mut(&mut self.syms),
4133 &self.labels,
4134 build_props_view(&self.props, &self.base),
4135 Arc::make_mut(&mut self.topo),
4136 &self.base,
4137 Arc::make_mut(&mut self.edge_props),
4138 );
4139 eng.create_rule(def, &mut gm)
4140 };
4141 self.engine = eng;
4142 result.map_err(|e| GraphError::RuleInvalid { detail: e })?;
4143 // Derived-edge fires from backfill → view updates.
4144 // Fast path: skip O(edge_count) allocation when no views exist.
4145 if !self.view_store.is_empty() {
4146 #[cfg(test)]
4147 DELTA_COPY_COUNT.with(|c| c.set(c.get() + 1));
4148 let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
4149 for d in &new_deltas {
4150 self.view_store.on_edge_changed(
4151 d.etype_sym,
4152 d.src_id,
4153 d.dst_id,
4154 d.fired,
4155 Arc::make_mut(&mut self.props),
4156 &build_topo_view(&self.topo, &self.base),
4157 &self.ids,
4158 &self.syms,
4159 &self.labels,
4160 base_columns(&self.base),
4161 );
4162 }
4163 }
4164 }
4165 WalRecord::DeleteRule { name } => {
4166 // Replay-over-snapshot idempotency: the snapshot already captured the
4167 // post-delete state so the rule is absent; silently skip to avoid a
4168 // spurious RuleNotFound error in the crash window between snapshot write
4169 // and WAL truncation.
4170 if !self.engine.rules().any(|r| r.name == *name) {
4171 return Ok(());
4172 }
4173 let cursor = self.engine.pending_delta_count();
4174 let mut eng = std::mem::take(&mut self.engine);
4175 let result = {
4176 let mut gm = make_graph_mut(
4177 &self.ids,
4178 Arc::make_mut(&mut self.syms),
4179 &self.labels,
4180 build_props_view(&self.props, &self.base),
4181 Arc::make_mut(&mut self.topo),
4182 &self.base,
4183 Arc::make_mut(&mut self.edge_props),
4184 );
4185 eng.delete_rule(name, &mut gm)
4186 };
4187 self.engine = eng;
4188 result.map_err(|_| GraphError::RuleNotFound { name: name.clone() })?;
4189 // Derived-edge retractions → view updates.
4190 if !self.view_store.is_empty() {
4191 #[cfg(test)]
4192 DELTA_COPY_COUNT.with(|c| c.set(c.get() + 1));
4193 let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
4194 for d in &new_deltas {
4195 self.view_store.on_edge_changed(
4196 d.etype_sym,
4197 d.src_id,
4198 d.dst_id,
4199 d.fired,
4200 Arc::make_mut(&mut self.props),
4201 &build_topo_view(&self.topo, &self.base),
4202 &self.ids,
4203 &self.syms,
4204 &self.labels,
4205 base_columns(&self.base),
4206 );
4207 }
4208 }
4209 }
4210 WalRecord::RemoveProp { key, field } => {
4211 // Recovery-safe: unknown key or already-absent field is a
4212 // clean no-op. Crash-window replay over a snapshot that
4213 // already applied this record must not Err.
4214 let Some(id) = self.ids.get(key) else {
4215 return Ok(());
4216 };
4217 // Read old value through the seam for rule retraction.
4218 let old = build_props_view(&self.props, &self.base)
4219 .get(id, field)
4220 .map(|vr| vr.into_value());
4221 Arc::make_mut(&mut self.props).remove(id, field);
4222 // If the base still supplies the value after the overlay removal,
4223 // record a tombstone so ColumnsView::get does not resurrect it.
4224 // This covers both the base-only case AND the both-resident case:
4225 // base-only (in_overlay=false): old prop was only in base, remove
4226 // is a no-op on overlay, base still visible → tombstone needed.
4227 // both-resident (in_overlay=true): overlay had v2, base has v1;
4228 // removing overlay uncovers v1 → tombstone needed.
4229 // Idempotent on double-replay: second pass sees the tombstone →
4230 // get() returns None → condition is false → no duplicate tombstone.
4231 if build_props_view(&self.props, &self.base)
4232 .get(id, field)
4233 .is_some()
4234 {
4235 Arc::make_mut(&mut self.props).record_prop_tombstone(id, field);
4236 }
4237 let cursor = self.engine.pending_delta_count();
4238 let mut eng = std::mem::take(&mut self.engine);
4239 {
4240 let mut gm = make_graph_mut(
4241 &self.ids,
4242 Arc::make_mut(&mut self.syms),
4243 &self.labels,
4244 build_props_view(&self.props, &self.base),
4245 Arc::make_mut(&mut self.topo),
4246 &self.base,
4247 Arc::make_mut(&mut self.edge_props),
4248 );
4249 eng.on_node_changed(id, Some((field, old)), &mut gm);
4250 }
4251 self.engine = eng;
4252 // Derived-edge deltas → view updates.
4253 if !self.view_store.is_empty() {
4254 #[cfg(test)]
4255 DELTA_COPY_COUNT.with(|c| c.set(c.get() + 1));
4256 let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
4257 for d in &new_deltas {
4258 self.view_store.on_edge_changed(
4259 d.etype_sym,
4260 d.src_id,
4261 d.dst_id,
4262 d.fired,
4263 Arc::make_mut(&mut self.props),
4264 &build_topo_view(&self.topo, &self.base),
4265 &self.ids,
4266 &self.syms,
4267 &self.labels,
4268 base_columns(&self.base),
4269 );
4270 }
4271 }
4272 // Neighbor-aggregate views that read `field` must also update.
4273 self.view_store.on_prop_changed(
4274 id,
4275 field,
4276 Arc::make_mut(&mut self.props),
4277 &build_topo_view(&self.topo, &self.base),
4278 &self.ids,
4279 &self.syms,
4280 &self.labels,
4281 base_columns(&self.base),
4282 );
4283 // Full-text index maintenance: remove tokens for this field.
4284 if self.fulltext.field_indexed(field) {
4285 Arc::make_mut(&mut self.fulltext).remove_node_field(id, field);
4286 }
4287 // Property (equality) index maintenance: drop this node's entry.
4288 if self.prop_index.field_indexed(field) {
4289 if let Some(label) = self.labels.get(id as usize).and_then(|&sym| {
4290 (sym != u32::MAX).then(|| self.syms.resolve(sym)).flatten()
4291 }) {
4292 self.prop_index.remove_node(label, field, id);
4293 }
4294 }
4295 }
4296 WalRecord::DeleteEdge {
4297 edge_type,
4298 src_key,
4299 dst_key,
4300 } => {
4301 // Recovery-safe: unknown keys, unknown etype, or already-
4302 // absent edge is a clean no-op (remove_edge returns false).
4303 let Some(src) = self.ids.get(src_key) else {
4304 return Ok(());
4305 };
4306 let Some(dst) = self.ids.get(dst_key) else {
4307 return Ok(());
4308 };
4309 let Some(etype) = self.syms.get(edge_type) else {
4310 return Ok(());
4311 };
4312 // I3: phantom-tombstone guard. When a V8 base is present, a
4313 // DeleteEdge WAL record for an edge that was already absorbed into
4314 // the new base (i.e. neither in overlay nor in base) must be skipped.
4315 // Without this guard, remove_edge records a tombstone for an edge
4316 // that no longer exists, incorrectly understating edge_count.
4317 if self.base.is_some()
4318 && !self
4319 .topo_view()
4320 .neighbors(etype, core_storage::topology::Direction::Out, src)
4321 .contains(&dst)
4322 {
4323 return Ok(());
4324 }
4325 Arc::make_mut(&mut self.topo).remove_edge(etype, src, dst);
4326 Arc::make_mut(&mut self.edge_props).remove_edge(etype, src, dst);
4327 // View maintenance for manual edge delete (topo already updated above).
4328 self.view_store.on_edge_changed(
4329 etype,
4330 src,
4331 dst,
4332 false,
4333 Arc::make_mut(&mut self.props),
4334 &build_topo_view(&self.topo, &self.base),
4335 &self.ids,
4336 &self.syms,
4337 &self.labels,
4338 base_columns(&self.base),
4339 );
4340 // Rule engine: via-hop rules must retract when user via-edges are deleted.
4341 let cursor = self.engine.pending_delta_count();
4342 let mut eng = std::mem::take(&mut self.engine);
4343 {
4344 let mut gm = make_graph_mut(
4345 &self.ids,
4346 Arc::make_mut(&mut self.syms),
4347 &self.labels,
4348 build_props_view(&self.props, &self.base),
4349 Arc::make_mut(&mut self.topo),
4350 &self.base,
4351 Arc::make_mut(&mut self.edge_props),
4352 );
4353 eng.on_edge_changed(edge_type, src, dst, &mut gm);
4354 }
4355 self.engine = eng;
4356 if !self.view_store.is_empty() {
4357 let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
4358 for d in &new_deltas {
4359 self.view_store.on_edge_changed(
4360 d.etype_sym,
4361 d.src_id,
4362 d.dst_id,
4363 d.fired,
4364 Arc::make_mut(&mut self.props),
4365 &build_topo_view(&self.topo, &self.base),
4366 &self.ids,
4367 &self.syms,
4368 &self.labels,
4369 base_columns(&self.base),
4370 );
4371 }
4372 }
4373 }
4374 WalRecord::DeleteNode { key } => {
4375 // Recovery-safe: already-tombstoned / unknown key is a clean
4376 // no-op. Crash-window replay over a snapshot that already
4377 // applied this record cannot recover the retired id from the
4378 // key (`IdMap::get` is None), so every subsequent step is
4379 // skipped. Each step is independently idempotent if invoked
4380 // twice on a still-live id: retraction is a no-op on empty
4381 // provenance, `remove_edge` returns false, `remove_all` is a
4382 // no-op, `ids.delete` returns None, label sentinel is sticky.
4383 let Some(n) = self.ids.get(key) else {
4384 return Ok(());
4385 };
4386
4387 // (1) Retract derived edges + de-index while props/labels live.
4388 let cursor = self.engine.pending_delta_count();
4389 let mut eng = std::mem::take(&mut self.engine);
4390 {
4391 let mut gm = make_graph_mut(
4392 &self.ids,
4393 Arc::make_mut(&mut self.syms),
4394 &self.labels,
4395 build_props_view(&self.props, &self.base),
4396 Arc::make_mut(&mut self.topo),
4397 &self.base,
4398 Arc::make_mut(&mut self.edge_props),
4399 );
4400 eng.on_node_removed(n, &mut gm);
4401 }
4402 self.engine = eng;
4403 // Derived-edge retractions → view updates for neighbors.
4404 if !self.view_store.is_empty() {
4405 #[cfg(test)]
4406 DELTA_COPY_COUNT.with(|c| c.set(c.get() + 1));
4407 let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
4408 for d in &new_deltas {
4409 self.view_store.on_edge_changed(
4410 d.etype_sym,
4411 d.src_id,
4412 d.dst_id,
4413 d.fired,
4414 Arc::make_mut(&mut self.props),
4415 &build_topo_view(&self.topo, &self.base),
4416 &self.ids,
4417 &self.syms,
4418 &self.labels,
4419 base_columns(&self.base),
4420 );
4421 }
4422 }
4423
4424 // (2) Sweep ALL remaining edges incident to n, both directions,
4425 // every etype. This cascade is intentionally mask-independent:
4426 // topology integrity requires removing every edge touching the
4427 // deleted node regardless of the caller's visibility scope.
4428 // (The mask limits which nodes a role's read phase can return;
4429 // the WAL delete always executes with full storage authority.)
4430 // Collect then remove so neighbor slices stay valid during
4431 // iteration. Remove from topo first, then call view maintenance
4432 // so Avg/Min/Max recompute sees the correct (reduced) neighbor set.
4433 let etypes: Vec<u32> = self.topo.etypes().collect();
4434 let mut doomed = Vec::new();
4435 for et in &etypes {
4436 for &dst in self.topo.neighbors(*et, Direction::Out, n).as_ref() {
4437 doomed.push((*et, n, dst));
4438 }
4439 for &src in self.topo.neighbors(*et, Direction::In, n).as_ref() {
4440 doomed.push((*et, src, n));
4441 }
4442 }
4443 for (et, s, d) in doomed {
4444 Arc::make_mut(&mut self.topo).remove_edge(et, s, d);
4445 Arc::make_mut(&mut self.edge_props).remove_edge(et, s, d);
4446 // View maintenance: n's own view values will be cleared by
4447 // remove_all below; only update surviving neighbors.
4448 self.view_store.on_edge_changed(
4449 et,
4450 s,
4451 d,
4452 false,
4453 Arc::make_mut(&mut self.props),
4454 &build_topo_view(&self.topo, &self.base),
4455 &self.ids,
4456 &self.syms,
4457 &self.labels,
4458 base_columns(&self.base),
4459 );
4460 }
4461
4462 // (3) Drop every remaining prop (`ColumnStore::remove_all`).
4463 Arc::make_mut(&mut self.props).remove_all(n);
4464 // Full-text index maintenance: remove all tokens for this node.
4465 Arc::make_mut(&mut self.fulltext).remove_node(n);
4466 // Property (equality) index maintenance: drop all entries for n.
4467 self.prop_index.remove_node_all(n);
4468
4469 // (4) Retire the dense id and stamp the label sentinel.
4470 Arc::make_mut(&mut self.ids).delete(key);
4471 if let Some(slot) = Arc::make_mut(&mut self.labels).get_mut(n as usize) {
4472 *slot = u32::MAX;
4473 }
4474 }
4475 WalRecord::Batch(inner) => {
4476 // Apply each inner record in order through the same apply path.
4477 // Inner records are validated free of nested Batch by encode_record.
4478 for rec in inner {
4479 self.apply(rec)?;
4480 }
4481 }
4482 WalRecord::RebuildRule { name } => {
4483 // Replay-over-snapshot idempotency: the snapshot may already
4484 // reflect a later delete_rule, so the rule is absent; skip.
4485 if !self.engine.rules().any(|r| r.name == *name) {
4486 return Ok(());
4487 }
4488 let cursor = self.engine.pending_delta_count();
4489 let mut eng = std::mem::take(&mut self.engine);
4490 let result = {
4491 let mut gm = make_graph_mut(
4492 &self.ids,
4493 Arc::make_mut(&mut self.syms),
4494 &self.labels,
4495 build_props_view(&self.props, &self.base),
4496 Arc::make_mut(&mut self.topo),
4497 &self.base,
4498 Arc::make_mut(&mut self.edge_props),
4499 );
4500 eng.rebuild(name, &mut gm)
4501 };
4502 self.engine = eng;
4503 result.map_err(|_| GraphError::RuleNotFound { name: name.clone() })?;
4504 // Derived-edge delta changes → view updates.
4505 if !self.view_store.is_empty() {
4506 #[cfg(test)]
4507 DELTA_COPY_COUNT.with(|c| c.set(c.get() + 1));
4508 let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
4509 for d in &new_deltas {
4510 self.view_store.on_edge_changed(
4511 d.etype_sym,
4512 d.src_id,
4513 d.dst_id,
4514 d.fired,
4515 Arc::make_mut(&mut self.props),
4516 &build_topo_view(&self.topo, &self.base),
4517 &self.ids,
4518 &self.syms,
4519 &self.labels,
4520 base_columns(&self.base),
4521 );
4522 }
4523 }
4524 }
4525 WalRecord::CreateView { def_bytes } => {
4526 let def: ViewDef =
4527 bincode::deserialize(def_bytes).map_err(|e| GraphError::Corrupt {
4528 detail: format!("CreateView def_bytes deserialize failed: {e}"),
4529 })?;
4530 // Replay-over-snapshot idempotency: view already present → skip.
4531 if self.view_store.has_view(&def.name) {
4532 return Ok(());
4533 }
4534 self.view_store
4535 .create_view(
4536 def,
4537 Arc::make_mut(&mut self.props),
4538 &build_topo_view(&self.topo, &self.base),
4539 &self.ids,
4540 &self.syms,
4541 &self.labels,
4542 )
4543 .map_err(|e| GraphError::RuleInvalid { detail: e })?;
4544 }
4545 WalRecord::DeleteView { name } => {
4546 // Replay-over-snapshot idempotency: view already absent → skip.
4547 if !self.view_store.has_view(name) {
4548 return Ok(());
4549 }
4550 self.view_store
4551 .delete_view(
4552 name,
4553 Arc::make_mut(&mut self.props),
4554 &self.ids,
4555 &self.labels,
4556 &self.syms,
4557 )
4558 .map_err(|_| GraphError::RuleNotFound { name: name.clone() })?;
4559 }
4560 WalRecord::EnableFulltext { label, field } => {
4561 // Replay-over-snapshot idempotency: already enabled → skip.
4562 if self.fulltext.is_enabled(label, field) {
4563 return Ok(());
4564 }
4565 Arc::make_mut(&mut self.fulltext).enable(label, field);
4566 if self.fulltext_rebuild_follows {
4567 // The open path rebuilds the whole index after replay, which
4568 // clears every posting this scan would write. Doing it twice
4569 // costs a full tokenise-and-stem pass over the corpus per
4570 // enabled pair: measured at 456 ms against 3.8 ms for the
4571 // same 30,000-entity store with no pair enabled.
4572 return Ok(());
4573 }
4574 // Backfill: index all live nodes of this label that have the field.
4575 let n = self.ids.len() as u32;
4576 for id in 0..n {
4577 let Some(&sym) = self.labels.get(id as usize) else {
4578 continue;
4579 };
4580 if sym == u32::MAX {
4581 continue; // tombstoned
4582 }
4583 let Some(lbl) = self.syms.resolve(sym) else {
4584 continue;
4585 };
4586 if lbl != label {
4587 continue;
4588 }
4589 if let Some(value) = build_props_view(&self.props, &self.base)
4590 .get(id, field)
4591 .map(|vr| vr.into_value())
4592 {
4593 Arc::make_mut(&mut self.fulltext).add_tokens(id, field, &value);
4594 }
4595 }
4596 }
4597 WalRecord::DisableFulltext { label, field } => {
4598 // Replay-over-snapshot idempotency: already disabled → skip.
4599 if !self.fulltext.is_enabled(label, field) {
4600 return Ok(());
4601 }
4602 // If another label still indexes this field, the postings column
4603 // is kept — but it must not contain node_ids from the now-disabled
4604 // label. Remove them before calling disable() so the field_indexed
4605 // guard inside disable() sees the correct post-removal state.
4606 if self.fulltext.field_indexed_by_other(label, field) {
4607 if let Some(label_sym) = self.syms.get(label) {
4608 for (node_id, &lsym) in self.labels.iter().enumerate() {
4609 if lsym == label_sym {
4610 Arc::make_mut(&mut self.fulltext)
4611 .remove_node_field(node_id as u32, field);
4612 }
4613 }
4614 }
4615 }
4616 Arc::make_mut(&mut self.fulltext).disable(label, field);
4617 }
4618 WalRecord::EnableIndex { label, field } => {
4619 // Replay-over-snapshot idempotency: already enabled → skip.
4620 if self.prop_index.is_enabled(label, field) {
4621 return Ok(());
4622 }
4623 self.prop_index.enable(label, field);
4624 // Backfill: index all live nodes of this label that have the field.
4625 let n = self.ids.len() as u32;
4626 for id in 0..n {
4627 let Some(&sym) = self.labels.get(id as usize) else {
4628 continue;
4629 };
4630 if sym == u32::MAX {
4631 continue; // tombstoned
4632 }
4633 let Some(lbl) = self.syms.resolve(sym) else {
4634 continue;
4635 };
4636 if lbl != label {
4637 continue;
4638 }
4639 if let Some(value) = build_props_view(&self.props, &self.base)
4640 .get(id, field)
4641 .map(|vr| vr.into_value())
4642 {
4643 self.prop_index.set(label, field, id, &value);
4644 }
4645 }
4646 }
4647 WalRecord::DisableIndex { label, field } => {
4648 self.prop_index.disable(label, field);
4649 }
4650 // ── insert-count multiplicity (§5.13) ────────────────────────────
4651 //
4652 // Two shapes, told apart by `count`: the opt-in declaration, and an
4653 // absolute count for one triple. Absolute is what makes this
4654 // idempotent over a snapshot base — a pre-snapshot frame replayed
4655 // over a base that already folded it in lands on the same number
4656 // rather than adding to it, which is the failure a delta (or a count
4657 // derived from `InsertEdgeId` records) would have.
4658 WalRecord::SetEdgeCount {
4659 etype,
4660 src,
4661 dst,
4662 count,
4663 } => {
4664 if rec.is_multiplicity_decl() {
4665 self.multiplicity = true;
4666 } else {
4667 Arc::make_mut(&mut self.edge_props).set(
4668 *etype,
4669 *src,
4670 *dst,
4671 EDGE_COUNT_PROP,
4672 Value::Int(*count as i64),
4673 );
4674 }
4675 }
4676 // History markers carry no replay state — rules re-derive edges
4677 // deterministically on open/replay. Skip unconditionally.
4678 WalRecord::DerivedEdgeAdded { .. } | WalRecord::DerivedEdgeRetracted { .. } => {}
4679 // ── rename_node ──────────────────────────────────────────────────
4680 WalRecord::RenameNode { old_key, new_key } => {
4681 // Recovery-safe: if old_key is already gone (key was renamed
4682 // by a snapshot or a prior replay frame), skip cleanly.
4683 if self.ids.get(old_key).is_none() {
4684 return Ok(());
4685 }
4686 // The rename only updates the key-table; the dense id, all
4687 // topo edges, props, labels, and rule state are id-indexed and
4688 // require no change.
4689 Arc::make_mut(&mut self.ids)
4690 .rename(old_key, new_key)
4691 .map_err(|e| GraphError::Corrupt {
4692 detail: format!("wal replay RenameNode {old_key}→{new_key}: {e}"),
4693 })?;
4694 }
4695 }
4696 Ok(())
4697 }
4698
4699 /// Intern `s` in `syms` and emit a WAL `Intern` record so `*Id` records
4700 /// replay on WAL-only `open_at` (no snapshot intern table). Apply is
4701 /// idempotent when the string is already bound. Always emit: after
4702 /// `snapshot()` the WAL is truncated and live intern is not on disk.
4703 fn intern_wal(&mut self, s: &str) -> (u32, WalRecord) {
4704 let id = if let Some(id) = self.syms.get(s) {
4705 id
4706 } else {
4707 Arc::make_mut(&mut self.syms).intern(s)
4708 };
4709 (
4710 id,
4711 WalRecord::Intern {
4712 id,
4713 text: s.to_string(),
4714 },
4715 )
4716 }
4717
4718 /// Rewrite user-facing records into dense-id records. On `Err`, no live
4719 /// state is left mutated: speculative interns made while building the
4720 /// output are rolled back, so a later successful mutation cannot log an
4721 /// `Intern` record whose id replay would never reproduce.
4722 fn rewrite_wal_dense(&mut self, recs: Vec<WalRecord>) -> Result<Vec<WalRecord>> {
4723 self.rewrite_wal_dense_planned(recs.into_iter().map(PlannedRec::Rec).collect())
4724 }
4725
4726 /// [`rewrite_wal_dense`](Self::rewrite_wal_dense) for a frame that still
4727 /// carries [`PlannedRec::DuplicateCount`] entries — the shape a batch
4728 /// produces, where a duplicate's count can only be named once this pass has
4729 /// assigned the frame's own ids.
4730 fn rewrite_wal_dense_planned(&mut self, recs: Vec<PlannedRec>) -> Result<Vec<WalRecord>> {
4731 let syms_checkpoint = self.syms.len();
4732 let result = self.rewrite_wal_dense_inner(recs);
4733 if result.is_err() {
4734 Arc::make_mut(&mut self.syms).truncate(syms_checkpoint);
4735 }
4736 result
4737 }
4738
4739 fn rewrite_wal_dense_inner(&mut self, recs: Vec<PlannedRec>) -> Result<Vec<WalRecord>> {
4740 let mut out = Vec::with_capacity(recs.len());
4741 // Node ids allocated by later apply(InsertNodeId) in this same batch.
4742 let mut pending: std::collections::HashMap<String, u32> = std::collections::HashMap::new();
4743 // Namespace of each node inserted earlier in this same frame, so a SET
4744 // on a node this frame created is measured against the namespace it was
4745 // created in rather than against the store, where it does not exist yet.
4746 let mut pending_ns: std::collections::HashMap<String, String> =
4747 std::collections::HashMap::new();
4748 let mut interned = std::collections::HashSet::<u32>::new();
4749 let mut next = u32::try_from(self.ids.len()).map_err(|_| GraphError::Corrupt {
4750 detail: "id space exhausted".into(),
4751 })?;
4752 // Insert counts this frame has already raised. `edge_insert_count`
4753 // reads committed state, which cannot see a count queued earlier in
4754 // this same frame, so N duplicates of one pair would otherwise all
4755 // compute `committed + 1` and the last would win.
4756 let mut pending_counts: HashMap<(u32, u32, u32), u64> = HashMap::new();
4757 let lookup = |ids: &IdMap,
4758 pending: &std::collections::HashMap<String, u32>,
4759 key: &str|
4760 -> Option<u32> { ids.get(key).or_else(|| pending.get(key).copied()) };
4761 for rec in recs {
4762 // A duplicate insert's count, resolved here and nowhere else.
4763 //
4764 // This is the only pass that knows the frame's own ids: a node
4765 // created earlier in the same frame has no dense id until the
4766 // `InsertNodeId` above allocates one, and an edge type first used in
4767 // this frame is not in `syms` until `intern_wal` puts it there.
4768 // Resolving the count in the batch's validate pass instead — where
4769 // it used to live — meant that a duplicate whose endpoints or type
4770 // were created in the same frame silently produced no count at all,
4771 // which is exactly the shape a mirror rebuild writes (defect #24).
4772 let rec = match rec {
4773 PlannedRec::Rec(rec) => rec,
4774 PlannedRec::DuplicateCount {
4775 edge_type,
4776 src_key,
4777 dst_key,
4778 } => {
4779 let (etype, intern) = self.intern_wal(&edge_type);
4780 if interned.insert(etype) {
4781 out.push(intern);
4782 }
4783 let src = lookup(&self.ids, &pending, &src_key).ok_or_else(|| {
4784 GraphError::Corrupt {
4785 detail: format!("dense WAL rewrite missing src {src_key}"),
4786 }
4787 })?;
4788 let dst = lookup(&self.ids, &pending, &dst_key).ok_or_else(|| {
4789 GraphError::Corrupt {
4790 detail: format!("dense WAL rewrite missing dst {dst_key}"),
4791 }
4792 })?;
4793 let count = pending_counts
4794 .get(&(etype, src, dst))
4795 .copied()
4796 .unwrap_or_else(|| self.edge_insert_count(etype, src, dst))
4797 .saturating_add(1);
4798 pending_counts.insert((etype, src, dst), count);
4799 out.push(WalRecord::SetEdgeCount {
4800 etype,
4801 src,
4802 dst,
4803 count,
4804 });
4805 continue;
4806 }
4807 };
4808 match rec {
4809 WalRecord::InsertNode { label, key, props } => {
4810 // Namespace validation and normalisation, on the one seam
4811 // every user-visible node insert passes through: insert_node,
4812 // a batch, ingest, Cypher CREATE and MERGE all arrive here
4813 // before the WAL append, and replay never does.
4814 let (props, ns_name) = Self::normalise_insert_ns(&key, props)?;
4815 pending_ns.insert(key.clone(), ns_name);
4816 let (label_id, intern) = self.intern_wal(&label);
4817 if interned.insert(label_id) {
4818 out.push(intern);
4819 }
4820 let mut props_id = Vec::with_capacity(props.len());
4821 for (field, value) in props {
4822 let (field_id, intern) = self.intern_wal(&field);
4823 if interned.insert(field_id) {
4824 out.push(intern);
4825 }
4826 props_id.push((field_id, value));
4827 }
4828 if lookup(&self.ids, &pending, &key).is_none() {
4829 pending.insert(key.clone(), next);
4830 next = next.checked_add(1).ok_or_else(|| GraphError::Corrupt {
4831 detail: "id space exhausted".into(),
4832 })?;
4833 }
4834 out.push(WalRecord::InsertNodeId {
4835 label: label_id,
4836 key,
4837 props: props_id,
4838 });
4839 }
4840 WalRecord::SetProp { key, field, value } => {
4841 // A namespace is set at insert and fixed after: the write is
4842 // refused when it would move the node, and dropped when it
4843 // names the namespace the node is already in. Checked here
4844 // so set_prop, a batch, Cypher SET/MERGE and every upsert
4845 // that merges props get the same answer.
4846 if field == NS_PROP {
4847 let Value::Str(ref to) = value else {
4848 return Err(GraphError::RuleInvalid {
4849 detail: format!(
4850 "node {key}: {NS_PROP} must be a string naming a namespace, \
4851 got {value:?}"
4852 ),
4853 });
4854 };
4855 let from = pending_ns
4856 .get(&key)
4857 .cloned()
4858 .or_else(|| self.namespace_of(&key))
4859 .unwrap_or_else(|| NS_DEFAULT.to_string());
4860 let to = to.clone();
4861 if to != from {
4862 return Err(GraphError::NamespaceImmutable {
4863 key: key.clone(),
4864 from,
4865 to,
4866 });
4867 }
4868 continue;
4869 }
4870 let id =
4871 lookup(&self.ids, &pending, &key).ok_or_else(|| GraphError::Corrupt {
4872 detail: format!("dense WAL rewrite missing key {key}"),
4873 })?;
4874 let (field_id, intern) = self.intern_wal(&field);
4875 if interned.insert(field_id) {
4876 out.push(intern);
4877 }
4878 out.push(WalRecord::SetPropId {
4879 id,
4880 field: field_id,
4881 value,
4882 });
4883 }
4884 WalRecord::InsertEdge {
4885 edge_type,
4886 src_key,
4887 dst_key,
4888 } => {
4889 let (etype, intern) = self.intern_wal(&edge_type);
4890 if interned.insert(etype) {
4891 out.push(intern);
4892 }
4893 let src = lookup(&self.ids, &pending, &src_key).ok_or_else(|| {
4894 GraphError::Corrupt {
4895 detail: format!("dense WAL rewrite missing src {src_key}"),
4896 }
4897 })?;
4898 let dst = lookup(&self.ids, &pending, &dst_key).ok_or_else(|| {
4899 GraphError::Corrupt {
4900 detail: format!("dense WAL rewrite missing dst {dst_key}"),
4901 }
4902 })?;
4903 out.push(WalRecord::InsertEdgeId { etype, src, dst });
4904 }
4905 WalRecord::RenameNode {
4906 ref old_key,
4907 ref new_key,
4908 } => {
4909 // Track the rename in `pending` so subsequent InsertEdge /
4910 // SetProp records in this batch can resolve the new key.
4911 let id = lookup(&self.ids, &pending, old_key).ok_or_else(|| {
4912 GraphError::Corrupt {
4913 detail: format!(
4914 "dense WAL rewrite: RenameNode old key {old_key} not found"
4915 ),
4916 }
4917 })?;
4918 pending.remove(old_key.as_str());
4919 pending.insert(new_key.clone(), id);
4920 out.push(rec);
4921 }
4922 // # Symbol-order invariant (load-bearing)
4923 //
4924 // Write-time and replay-time symbol assignment must agree: every
4925 // symbol in a `Batch` frame has to receive the same dense id when
4926 // the frame's records are replayed in order as it received when
4927 // the frame was written.
4928 //
4929 // A rule's backfill interns its `edge_type` lazily
4930 // (`core_rules::engine`, every `g.syms.intern(&def.edge_type)`
4931 // site), and that backfill runs from `apply` — during the
4932 // `CreateRule` record itself, and again from any later
4933 // `InsertNodeId` in the same frame that makes the rule fire. At
4934 // write time the whole batch is rewritten before any of it is
4935 // applied, so a later `InsertEdge` in the same batch would win the
4936 // lower id for its edge type; on replay the rule's lazy intern
4937 // gets there first and steals it, and the `Intern` record fails at
4938 // the `wal intern assigned …` check in `apply`.
4939 //
4940 // Pre-interning the rule's `edge_type` here, and emitting its
4941 // `Intern` record ahead of the `CreateRule` record, makes both
4942 // orders identical. `weight_prop` needs no pre-intern:
4943 // `EdgeProps::set` keys props by `String`, never through the
4944 // interner. `via_edge` needs none either: via-hop rules resolve it
4945 // with `syms.get` and skip when it is absent.
4946 //
4947 // `RebuildRule` and `DeleteRule` need no such handling here:
4948 // `RebuildRule` has no `BatchOp` variant, so it never appears
4949 // inside a `Batch` today — it is only ever issued as its own
4950 // standalone commit (`rebuild_rule`, or the auto-rebuild path
4951 // that logs it as a second commit after the triggering op).
4952 // `DeleteRule` does have a `BatchOp` variant and can appear
4953 // inside a `Batch`, but it carries only a rule `name` — no
4954 // `edge_type` or other symbol that needs pre-interning — so
4955 // only `CreateRule` needs this arm.
4956 WalRecord::CreateRule { ref def_bytes } => {
4957 let def = decode_rule_def(def_bytes).map_err(|e| GraphError::Corrupt {
4958 detail: format!("CreateRule def_bytes deserialize failed: {e}"),
4959 })?;
4960 let (etype, intern) = self.intern_wal(&def.edge_type);
4961 if interned.insert(etype) {
4962 out.push(intern);
4963 }
4964 out.push(rec);
4965 }
4966 other => out.push(other),
4967 }
4968 }
4969 Ok(out)
4970 }
4971
4972 fn log_dense(&mut self, recs: Vec<WalRecord>) -> Result<()> {
4973 let recs = self.rewrite_wal_dense(recs)?;
4974 match recs.len() {
4975 0 => Ok(()),
4976 1 => self.log_then_apply(recs.into_iter().next().unwrap()),
4977 _ => self.log_then_apply(WalRecord::Batch(recs)),
4978 }
4979 }
4980
4981 /// Durable write, then notify the event sink. Replay (`apply` during
4982 /// `open`) never enters this function, so it is the replay-silent seam.
4983 /// Record that the commit occupying `frame_index` happened now, and append
4984 /// those 16 bytes to the sidecar.
4985 ///
4986 /// `frame_index` is the **global 0-based WAL frame index** of the commit's
4987 /// own record — the space every history surface addresses — taken from
4988 /// [`wal_frames_written`](GraphDb::wal_frames_written) before the append
4989 /// that puts the record there.
4990 ///
4991 /// **It is deliberately not derived from `commit_seq`.** A commit is not a
4992 /// frame: one whose rules fire appends a second frame for the derived-edge
4993 /// history marker, so `commit_seq - 1` falls one frame further behind per
4994 /// rule-firing commit and every date resolves to an ever-earlier graph.
4995 /// That was the shipped behaviour through v0.6.11 and it failed silently,
4996 /// because an older graph is a plausible answer rather than an error.
4997 ///
4998 /// Called from exactly one place — `log_then_apply_with`, immediately after
4999 /// `commit_seq` is incremented. Every write path in the engine funnels
5000 /// through that function, and replay deliberately does not: `apply_frames`
5001 /// re-applies commits that already happened, so stamping there would record
5002 /// replay time as commit time.
5003 ///
5004 /// **This is the engine's only wall-clock read.** Everything else uses
5005 /// `Instant`, which is monotonic and not a date.
5006 ///
5007 /// Failure is swallowed on purpose. The sidecar is not part of the WAL or
5008 /// the snapshot, so a failed append must not fail a commit that is already
5009 /// durable — it costs a date, not data. The map is marked poisoned so the
5010 /// gap is reported rather than resolved across.
5011 fn stamp_commit_time(&mut self, frame_index: u64) {
5012 if self.commit_times_poisoned {
5013 return;
5014 }
5015 let unix_ms = match self.commit_time_override {
5016 Some(ms) => ms,
5017 None => {
5018 let Ok(now) = std::time::SystemTime::now().duration_since(std::time::UNIX_EPOCH)
5019 else {
5020 // A clock before 1970. Refuse to invent a timestamp.
5021 self.commit_times_poisoned = true;
5022 return;
5023 };
5024 now.as_millis() as i64
5025 }
5026 };
5027 let first = self.commit_times.is_empty();
5028 self.commit_times.push(frame_index, unix_ms);
5029
5030 let wrote = if first {
5031 self.fs.write_atomic(
5032 FileId::CommitTimes,
5033 &core_storage::commit_times::encode(&self.commit_times),
5034 )
5035 } else {
5036 self.fs.append(
5037 FileId::CommitTimes,
5038 &core_storage::commit_times::encode_entry(frame_index, unix_ms),
5039 )
5040 };
5041 if wrote.is_err() {
5042 self.commit_times_poisoned = true;
5043 }
5044 }
5045
5046 /// Read the time sidecar from disk into this handle.
5047 ///
5048 /// Absent is the normal case for any store written before v0.6.11 and is
5049 /// not an error; unreadable is recorded so date queries can say "damaged"
5050 /// rather than "none recorded".
5051 ///
5052 /// Called at open **and** by `refresh` when a peer's frames are absorbed.
5053 /// Both, because the map is a file another process appends to: a handle
5054 /// that raises its frame cursor to include a peer's commits while holding a
5055 /// stale map would answer dates from a prefix of the truth — and, if its
5056 /// own map were still empty, would rewrite the whole file with one entry
5057 /// and destroy the peer's.
5058 fn load_commit_times_from_fs(&mut self) {
5059 match self.fs.read(FileId::CommitTimes) {
5060 Ok(bytes) if bytes.is_empty() => {}
5061 Ok(bytes) => {
5062 // A map from an older format version is discarded, not
5063 // reported as damage and not reinterpreted. Its entries were
5064 // written correctly against a different meaning of the number
5065 // — see `COMMIT_TIMES_VERSION` — and reading them in this
5066 // build's space would resolve dates onto unrelated commits.
5067 // Leaving the map empty makes the store answer
5068 // `NoRecordedTime`, which is the truth: it records no times
5069 // this build can use, and the next commit starts a usable map.
5070 if core_storage::commit_times::superseded_version(&bytes).is_some() {
5071 return;
5072 }
5073 match core_storage::commit_times::decode(&bytes) {
5074 Ok(t) => self.commit_times = t,
5075 Err(_) => self.commit_times_poisoned = true,
5076 }
5077 }
5078 Err(_) => {}
5079 }
5080 }
5081
5082 /// Rewrite the sidecar from memory. Used after truncation, which is the one
5083 /// operation that cannot be expressed as an append.
5084 fn rewrite_commit_times(&mut self) {
5085 if self.commit_times_poisoned {
5086 return;
5087 }
5088 if self
5089 .fs
5090 .write_atomic(
5091 FileId::CommitTimes,
5092 &core_storage::commit_times::encode(&self.commit_times),
5093 )
5094 .is_err()
5095 {
5096 self.commit_times_poisoned = true;
5097 }
5098 }
5099
5100 /// The greatest commit whose recorded time is at or before `unix_ms`.
5101 ///
5102 /// Errors name what they can answer instead of guessing a commit:
5103 /// `Corrupt` when the sidecar would not decode, `NoRecordedTime` when the
5104 /// store records none, `TimeBeforeFloor` when the instant predates the
5105 /// oldest entry, and `CommitOutOfRange` when the answer falls below the WAL
5106 /// horizon and so cannot be replayed.
5107 ///
5108 /// The answer is a **0-based frame index**, ready to hand to `edges_at` or
5109 /// `was_linked` without adjustment.
5110 pub fn resolve_instant(&self, unix_ms: i64) -> Result<u64> {
5111 if self.commit_times_poisoned {
5112 return Err(GraphError::Corrupt {
5113 detail: "commit_times.bin will not decode; date queries are \
5114 unavailable on this store"
5115 .into(),
5116 });
5117 }
5118 let at = self
5119 .commit_times
5120 .resolve_instant(unix_ms, self.wal_horizon_floor)?;
5121 // The map outlives the history it describes. A truncating snapshot folds
5122 // the WAL and discards it, so entries can name commits the engine can no
5123 // longer replay — the floor check above catches pruning, and this catches
5124 // discarding. Returning an index the caller's next call will reject is a
5125 // two-step error where one will do, and `resolve_date` is public: it
5126 // either hands back a usable index or refuses.
5127 let total = self.wal_total_commits()?;
5128 if at >= total {
5129 return Err(GraphError::CommitOutOfRange {
5130 commit: at,
5131 total,
5132 floor: self.wal_horizon_floor,
5133 });
5134 }
5135 Ok(at)
5136 }
5137
5138 /// Record subsequent commits as having happened at `unix_ms`, or pass
5139 /// `None` to go back to the system clock.
5140 ///
5141 /// For **backfilled history**: a mirror importing rows that already carry
5142 /// their own timestamps, or a replay of events that happened months ago.
5143 /// Without this every such commit is stamped "now", so a store holding a
5144 /// year of imported history answers every date question with
5145 /// `TimeBeforeFloor` — the data is there and no date reaches it.
5146 ///
5147 /// Sticky until changed or cleared, because a day of backfilled rows
5148 /// genuinely shares one instant.
5149 ///
5150 /// **Import in chronological order.** A supplied instant earlier than
5151 /// anything already recorded is refused with
5152 /// [`GraphError::CommitTimeNotMonotonic`], because resolution walks commit
5153 /// order: a later commit carrying an earlier time would silently widen every
5154 /// answer after it. Equal is allowed — that is what a shared day means. The
5155 /// live clock is never held to this, so an NTP step backwards still commits.
5156 ///
5157 /// Deliberately **not** exposed over HTTP or MCP: asserting when a commit
5158 /// happened rewrites the store's apparent history, which is not something a
5159 /// role token models. It is an embedding-caller's operation.
5160 pub fn record_commits_at(&mut self, unix_ms: Option<i64>) -> Result<()> {
5161 if self.read_only {
5162 return Err(GraphError::ReadOnly);
5163 }
5164 if let Some(ms) = unix_ms {
5165 if self.commit_times_poisoned {
5166 return Err(GraphError::Corrupt {
5167 detail: "commit_times.bin will not decode; this store cannot \
5168 record an asserted commit time"
5169 .into(),
5170 });
5171 }
5172 if let Some(newest) = self.commit_times.max_ms() {
5173 if ms < newest {
5174 return Err(GraphError::CommitTimeNotMonotonic {
5175 supplied_ms: ms,
5176 newest_ms: newest,
5177 });
5178 }
5179 }
5180 }
5181 self.commit_time_override = unix_ms;
5182 Ok(())
5183 }
5184
5185 /// The instant subsequent commits are being recorded at, when one is set.
5186 pub fn commit_time_override(&self) -> Option<i64> {
5187 self.commit_time_override
5188 }
5189
5190 /// [`Self::edges_at`] addressed by an instant rather than a commit index.
5191 ///
5192 /// Resolves through [`Self::resolve_instant`] — the last commit at or
5193 /// before the instant — then answers exactly as the commit-indexed call
5194 /// does. A store that records no times refuses by name; it never guesses.
5195 pub fn edges_at_instant(&self, key: &str, unix_ms: i64) -> Result<Vec<EdgeAt>> {
5196 let commit = self.resolve_instant(unix_ms)?;
5197 self.edges_at(key, commit)
5198 }
5199
5200 /// [`Self::was_linked`] addressed by an instant rather than a commit index.
5201 pub fn was_linked_at_instant(
5202 &self,
5203 a: &str,
5204 b: &str,
5205 edge_type: &str,
5206 unix_ms: i64,
5207 ) -> Result<bool> {
5208 let commit = self.resolve_instant(unix_ms)?;
5209 self.was_linked(a, b, edge_type, commit)
5210 }
5211
5212 /// Parse an RFC 3339 instant (or a bare `YYYY-MM-DD`) and resolve it.
5213 ///
5214 /// The one place every caller-facing surface converts a date string, so
5215 /// HTTP, MCP, Python and the CLI cannot drift in what they accept.
5216 pub fn resolve_date(&self, s: &str) -> Result<u64> {
5217 // The **end** of what the string denotes. A bare date is a day, so it
5218 // resolves to the last commit at or before that day's end — resolving to
5219 // the midnight that starts it would exclude everything that happened on
5220 // the date the caller asked about.
5221 let ms = core_storage::commit_times::parse_rfc3339_end_ms(s).ok_or_else(|| {
5222 GraphError::QueryError {
5223 detail: format!(
5224 "could not parse {s:?} as a date; expected RFC 3339 \
5225 (2026-06-19, or 2026-06-19T12:00:00Z)"
5226 ),
5227 }
5228 })?;
5229 self.resolve_instant(ms)
5230 }
5231
5232 /// The recorded wall-clock time of `commit`, when the sidecar holds one.
5233 ///
5234 /// `commit` is a **0-based frame index** — the space `edges_at`,
5235 /// `was_linked` and the history events use, not the 1-based `commit_seq`.
5236 pub fn commit_time_ms(&self, commit: u64) -> Option<i64> {
5237 if self.commit_times_poisoned {
5238 return None;
5239 }
5240 self.commit_times.time_of(commit)
5241 }
5242
5243 fn log_then_apply(&mut self, rec: WalRecord) -> Result<()> {
5244 self.log_then_apply_with(rec, None, self.fsync)
5245 }
5246
5247 /// Whether this frame must fsync under `policy`.
5248 ///
5249 /// Batched contract: user-visible batches (>1 mutation) fsync; single
5250 /// mutations do not. The dense rewrite wraps a single mutation in a
5251 /// `Batch([Intern.., <one *Id record>])`, so `Intern` records are excluded
5252 /// from the count — removing that filter would make every single-op write
5253 /// fsync under Batched (or, if the threshold were raised instead, skip a
5254 /// needed fsync for real two-op batches).
5255 fn wal_needs_sync(policy: FsyncPolicy, rec: &WalRecord) -> bool {
5256 match policy {
5257 FsyncPolicy::Relaxed => false,
5258 FsyncPolicy::Strict => true,
5259 FsyncPolicy::Batched => match rec {
5260 // Intern + one mutation is the single-op rewrite, not a user batch.
5261 WalRecord::Batch(inner) => {
5262 inner
5263 .iter()
5264 .filter(|r| !matches!(r, WalRecord::Intern { .. }))
5265 .count()
5266 > 1
5267 }
5268 _ => false,
5269 },
5270 }
5271 }
5272
5273 /// # Apply-infallibility invariant (load-bearing)
5274 ///
5275 /// The ordering is: WAL append → fsync → apply. If `apply` returned `Err`
5276 /// for a `Batch` frame after a successful WAL write, the WAL would contain
5277 /// the full frame while in-memory state would reflect only the ops before
5278 /// the failure. On reopen, WAL replay would then apply the entire batch —
5279 /// diverging permanently from what the pre-crash process had in memory.
5280 ///
5281 /// For `Batch` frames this situation cannot arise because:
5282 /// - All validation runs via `commit_logged_batch`/`MutPreview` **before**
5283 /// the WAL write. `MutPreview` uses the same `&mut self` that apply will
5284 /// use, with no concurrent mutation between validation exit and apply entry.
5285 /// - Every `apply` arm for a validated op is either infallible by construction
5286 /// (`InsertNode`, `RemoveProp`, `DeleteEdge`, `DeleteNode`), has idempotency
5287 /// guards that return `Ok(())` (`CreateRule`, `DeleteRule`), or is
5288 /// guaranteed-present by validation (`InsertEdge`/`SetProp` key lookups).
5289 /// - `on_node_changed` and `on_node_removed` return `()` — never `Err`.
5290 ///
5291 /// A `debug_assert!` below fires in debug builds if `apply` ever returns
5292 /// `Err` for a `Batch` frame, making any future regression immediately visible
5293 /// in tests rather than silently diverging crash-recovery behaviour.
5294 fn log_then_apply_with(
5295 &mut self,
5296 rec: WalRecord,
5297 ingest: Option<(String, usize)>,
5298 policy: FsyncPolicy,
5299 ) -> Result<()> {
5300 // Read-only guard: as-of instances must never write the WAL.
5301 if self.read_only {
5302 return Err(GraphError::ReadOnly);
5303 }
5304 // Degraded guard: fsync failure left WAL truncated, or a refresh failed
5305 // partway; in-memory state is ahead of (or out of step with) the
5306 // on-disk WAL, so further mutations would deepen the divergence.
5307 // Reopen the database to recover. Checked before the lock guard: this
5308 // is the more serious condition and the more useful error.
5309 if self.degraded {
5310 return Err(GraphError::Io(std::io::Error::other(
5311 "database degraded after group-commit fsync failure; reopen required",
5312 )));
5313 }
5314 // Cross-process guard: this write scope asked for the store's write
5315 // lock and did not get it. Writing anyway would append frames on top of
5316 // a WAL another process is extending, so refuse instead.
5317 if self.lock_denied {
5318 return Err(GraphError::Busy { holder: None });
5319 }
5320 // Ensure retained provenance bytes are decoded into the live mutable
5321 // fields before any mutation touches self.engine.provenance. This is a
5322 // no-op if provenance was never stored (fresh store) or has already been
5323 // consumed (subsequent mutations). WAL replay calls apply() directly
5324 // and is covered by consume_retained_state_eager before replay.
5325 self.ensure_v8_base_sections_loaded();
5326 self.engine.ensure_provenance_loaded_mut();
5327 // Invariant (I-1): no stale deltas may enter from a previous apply.
5328 // If any engine method ever accumulates deltas before erroring, they would
5329 // contaminate the *next* commit's event stream. This assert fires in debug
5330 // builds, making any future regression visible at the earliest point.
5331 debug_assert_eq!(
5332 self.engine.pending_delta_count(),
5333 0,
5334 "stale engine deltas at log_then_apply_with entry — \
5335 a previous apply arm may have accumulated deltas before erroring; \
5336 the caller must drain_deltas() on any error path before returning"
5337 );
5338 let frame = encode_record(&rec);
5339 self.fs.append(FileId::Wal, &frame)?;
5340 // The cursor advances by exactly the bytes appended: these frames are
5341 // ours and already applied, so a later refresh must not replay them.
5342 self.wal_consumed += frame.len() as u64;
5343 self.wal_frames_written += 1;
5344 if Self::wal_needs_sync(policy, &rec) {
5345 self.fs.sync(FileId::Wal)?;
5346 }
5347 // Marker writing always needs the engine deltas, but the engine only
5348 // accumulates them when emit_deltas is true (normally gated on subscribers
5349 // or views being present). Enable emission for this apply if it is
5350 // currently off, then restore the original state unconditionally via an
5351 // RAII guard — this prevents a panic in apply() from leaking the flag.
5352 // The same guard resets the engine's transient chaining state. A panic
5353 // unwinding out of a rule hook would otherwise leave `chain_depth`
5354 // non-zero, which makes every later `begin_chain` decide chaining is
5355 // already running and silently switch it off for good.
5356 struct RestoreEmitDeltas(*mut RuleEngine, bool);
5357 impl Drop for RestoreEmitDeltas {
5358 fn drop(&mut self) {
5359 // SAFETY: pointer into self (GraphDb); guard is dropped within
5360 // this frame before log_then_apply_with returns.
5361 unsafe {
5362 (*self.0).set_emit_deltas(self.1);
5363 (*self.0).reset_chain_state();
5364 }
5365 }
5366 }
5367 let original_emit = self.engine.emit_deltas();
5368 if !original_emit {
5369 self.engine.set_emit_deltas(true);
5370 }
5371 // SAFETY: raw pointer into self; guard dropped within this frame.
5372 let _emit_guard = RestoreEmitDeltas(&mut self.engine as *mut _, original_emit);
5373
5374 let apply_result = self.apply(&rec);
5375 // For Batch frames, post-validation apply must be infallible (see above).
5376 // A debug_assert here catches any future change that makes apply fallible
5377 // before the caller notices via silent WAL/memory divergence.
5378 if matches!(&rec, WalRecord::Batch(_)) {
5379 debug_assert!(
5380 apply_result.is_ok(),
5381 "Batch apply returned Err after successful WAL write — \
5382 the validate-then-apply invariant has been violated; \
5383 see log_then_apply_with invariant doc"
5384 );
5385 }
5386 if apply_result.is_err() {
5387 // Discard any partial deltas accumulated by the failed apply.
5388 // They must not ride the next commit's event stream (I-1).
5389 // _emit_guard restores emit_deltas on drop automatically.
5390 let _ = self.engine.drain_deltas();
5391 let _ = self.engine.take_rebuild_needed();
5392 apply_result?;
5393 }
5394 self.commit_seq += 1;
5395 let seq = self.commit_seq;
5396 // Update per-node last-change map for the committed record.
5397 // Must happen after commit_seq is incremented so the seq is correct.
5398 self.update_last_change_from_rec(&rec, seq);
5399 // Drain engine deltas and distribute to subscribers before the existing
5400 // MutationEvent sink fires — both happen post-fsync, post-apply.
5401 // _emit_guard restores emit_deltas after this line when it drops.
5402 let engine_deltas = self.engine.drain_deltas();
5403
5404 // Append history-marker WAL records for any derived-edge changes so
5405 // that `edge_history` and `was_linked` can surface rule-attributed
5406 // events. Markers are STATE NO-OPS during replay; they are written
5407 // without an additional fsync (the triggering commit's sync already
5408 // happened; the next commit's sync covers these lazily).
5409 if !engine_deltas.is_empty() {
5410 let markers: Vec<WalRecord> = engine_deltas
5411 .iter()
5412 .map(|d| {
5413 if d.fired {
5414 WalRecord::DerivedEdgeAdded {
5415 rule: d.rule.clone(),
5416 edge_type: d.edge_type.clone(),
5417 src_key: d.src_key.clone(),
5418 dst_key: d.dst_key.clone(),
5419 }
5420 } else {
5421 WalRecord::DerivedEdgeRetracted {
5422 rule: d.rule.clone(),
5423 edge_type: d.edge_type.clone(),
5424 src_key: d.src_key.clone(),
5425 dst_key: d.dst_key.clone(),
5426 }
5427 }
5428 })
5429 .collect();
5430 let marker_frame = if markers.len() == 1 {
5431 markers.into_iter().next().unwrap()
5432 } else {
5433 WalRecord::Batch(markers)
5434 };
5435 // Ignore append errors: markers are best-effort history
5436 // annotations. Losing them does not affect state correctness.
5437 // The cursor only advances when the bytes actually landed.
5438 let marker_bytes = encode_record(&marker_frame);
5439 if self.fs.append(FileId::Wal, &marker_bytes).is_ok() {
5440 self.wal_consumed += marker_bytes.len() as u64;
5441 // A marker is a state no-op during replay but it is not an
5442 // index no-op: it occupies a frame that every history surface
5443 // counts. Missing this increment is the whole of defect 1.
5444 self.wal_frames_written += 1;
5445 }
5446 }
5447
5448 // Stamp the commit against the **last** frame it wrote.
5449 //
5450 // `edges_at` and `was_linked` reconstruct derived edges by reading the
5451 // history markers out of the WAL, not by re-running rules over a
5452 // prefix. A commit's complete state — its record *and* the edges its
5453 // rules derived — is therefore only reached at its marker frame, so
5454 // that is the frame a date naming this commit must resolve to. Stamping
5455 // the record's own frame would answer every date with the graph as it
5456 // was one derivation short.
5457 //
5458 // This runs after the marker append for that reason, and it is still
5459 // the single stamping site: one commit, one entry.
5460 self.stamp_commit_time(self.wal_frames_written - 1);
5461
5462 // Record MVCC CommitDelta for the epoch reader. The WAL record is
5463 // stored as-is (including any nested Batch / Intern records); the
5464 // ReaderSnapshot's apply_one function handles all variants.
5465 {
5466 let derived_inserts = engine_deltas
5467 .iter()
5468 .filter(|d| d.fired)
5469 .map(|d| (d.etype_sym, d.src_id, d.dst_id))
5470 .collect();
5471 let derived_deletes = engine_deltas
5472 .iter()
5473 .filter(|d| !d.fired)
5474 .map(|d| (d.etype_sym, d.src_id, d.dst_id))
5475 .collect();
5476 let delta = Arc::new(crate::reader::CommitDelta {
5477 records: vec![rec.clone()],
5478 derived_inserts,
5479 derived_deletes,
5480 });
5481 self.delta_tail.push(delta);
5482 self.commits_since_fold += 1;
5483 if self.commits_since_fold >= crate::reader::FOLD_EVERY_K {
5484 self.fold_now();
5485 }
5486 }
5487
5488 if self.defer_events {
5489 // Group-commit drain thread: hold events until after the group
5490 // fsync so subscribers only observe durable data (R2).
5491 self.deferred_events.push(DeferredEvent {
5492 rec: rec.clone(),
5493 engine_deltas,
5494 seq,
5495 ingest,
5496 });
5497 } else {
5498 self.distribute_events(&rec, &engine_deltas, seq);
5499 self.emit_committed(&rec, ingest);
5500 }
5501 // Drift is only known after apply, so auto-rebuild cannot join the
5502 // triggering op's WAL frame. Issue RebuildRule as a second commit.
5503 // Skip when `rec` is itself RebuildRule: rebuild resets drift, so a
5504 // retrigger loop is impossible if the fit succeeded, but we still
5505 // drain the flag so a leftover cannot re-enter.
5506 // One slice of any outstanding vector-index build rides here too, so a
5507 // store that is being written to finishes its build without anyone
5508 // calling `pump_index_build`. A rule that becomes whole joins the same
5509 // RebuildRule loop below.
5510 let mut rebuilds = self.engine.take_rebuild_needed();
5511 if !matches!(&rec, WalRecord::RebuildRule { .. }) {
5512 // Not after `CreateRule`: that record's own apply already did the
5513 // rule's first slice, and pumping again here would make one
5514 // `create_rule` call do two slices' work under one lock.
5515 // Nothing pending is the overwhelmingly common case and must cost
5516 // a map lookup, not an engine swap: a store being written to has
5517 // long since populated its indexes, so the `pump_index_build`
5518 // entry point owns the not-yet-populated case on its own.
5519 if !is_create_rule_frame(&rec) && !self.engine.builds_in_progress().is_empty() {
5520 rebuilds.extend(self.pump_one_slice().into_iter().map(|b| b.rule));
5521 }
5522 let mut failed = Vec::new();
5523 for name in rebuilds {
5524 if self.engine.rules().any(|r| r.name == name) {
5525 // User op is already durable. A failed second commit must
5526 // not surface as the caller's error.
5527 if let Err(e) =
5528 self.log_then_apply(WalRecord::RebuildRule { name: name.clone() })
5529 {
5530 eprintln!(
5531 "auto-rebuild of rule {name:?} failed after durable user commit: {e}"
5532 );
5533 failed.push(name);
5534 }
5535 }
5536 }
5537 for name in failed {
5538 self.engine.queue_rebuild_needed(name);
5539 }
5540 }
5541 Ok(())
5542 }
5543
5544 /// Install a post-commit hook. Replaces any previous sink.
5545 ///
5546 /// The sink runs inside `log_then_apply` after a successful
5547 /// durable commit, while the caller still holds `&mut self`. When this
5548 /// database is behind a [`crate::SharedDb`], that means the **write
5549 /// guard is held**. The sink must never call `read` / `write` (or any
5550 /// other method) on the same `SharedDb` — the `RwLock` is not
5551 /// re-entrant and doing so deadlocks. The sink is `Send + Sync`;
5552 /// `std::sync::mpsc::Sender` is not `Sync` and will not type-check.
5553 /// Intended examples: `std::sync::mpsc::SyncSender`,
5554 /// `tokio::sync::mpsc::Sender`, `tokio::sync::broadcast::Sender`
5555 /// (non-blocking `send`), or `Arc<Mutex<Vec<MutationEvent>>>`.
5556 pub fn set_event_sink(&mut self, sink: Box<dyn Fn(MutationEvent) + Send + Sync>) {
5557 self.event_sink = Some(sink);
5558 }
5559
5560 /// Whether a post-commit event sink is currently installed.
5561 pub fn has_event_sink(&self) -> bool {
5562 self.event_sink.is_some()
5563 }
5564
5565 /// Set WAL fsync cadence. Default [`FsyncPolicy::Strict`].
5566 pub fn set_fsync_policy(&mut self, p: FsyncPolicy) {
5567 self.fsync = p;
5568 }
5569
5570 /// Return the current WAL fsync cadence.
5571 pub fn fsync_policy(&self) -> FsyncPolicy {
5572 self.fsync
5573 }
5574
5575 // ── Group-commit event deferral ───────────────────────────────────────────
5576
5577 /// Enable or disable deferred event mode.
5578 ///
5579 /// When `true`, event notifications (subscription `DbEvent`s and legacy
5580 /// `MutationEvent` sink calls) are buffered rather than fired immediately.
5581 /// Call [`flush_deferred_events`] after the group fsync to deliver them,
5582 /// or [`discard_deferred_events`] if the fsync failed and the group must
5583 /// be treated as lost.
5584 pub fn set_deferred_events_mode(&mut self, defer: bool) {
5585 self.defer_events = defer;
5586 }
5587
5588 /// Fire all buffered events accumulated since [`set_deferred_events_mode`]
5589 /// was set to true. Clears the buffer.
5590 ///
5591 /// Called by the drain thread AFTER a successful group fsync, so
5592 /// subscribers observe only data that is durably on disk.
5593 pub fn flush_deferred_events(&mut self) {
5594 let events = std::mem::take(&mut self.deferred_events);
5595 for de in events {
5596 self.distribute_events(&de.rec, &de.engine_deltas, de.seq);
5597 self.emit_committed(&de.rec, de.ingest);
5598 }
5599 }
5600
5601 /// Discard all buffered events without firing them.
5602 ///
5603 /// Called by the drain thread when a group fsync fails: the WAL has been
5604 /// truncated back to the pre-group offset, so the committed-but-unsynced
5605 /// ops must not be observable to subscribers.
5606 pub fn discard_deferred_events(&mut self) {
5607 self.deferred_events.clear();
5608 }
5609
5610 // ── Degraded state ────────────────────────────────────────────────────────
5611
5612 /// Mark this database as degraded.
5613 ///
5614 /// Called by the group-commit drain thread after a group fsync failure and
5615 /// WAL truncation: the in-memory state is now ahead of the on-disk WAL, so
5616 /// further mutations would deepen the divergence. All subsequent calls to
5617 /// [`log_then_apply_with`] return `Err` until the database is reopened.
5618 pub fn set_degraded(&mut self) {
5619 self.degraded = true;
5620 }
5621
5622 fn emit(&self, ev: MutationEvent) {
5623 if let Some(sink) = &self.event_sink {
5624 sink(ev);
5625 }
5626 }
5627
5628 fn emit_committed(&self, rec: &WalRecord, ingest: Option<(String, usize)>) {
5629 match rec {
5630 WalRecord::Batch(inner) => {
5631 for r in inner {
5632 if let Some(ev) = event_from_record(r, &self.syms, &self.ids) {
5633 self.emit(ev);
5634 }
5635 }
5636 match ingest {
5637 Some((label, inserted)) => {
5638 self.emit(MutationEvent::Ingested { label, inserted })
5639 }
5640 None => {
5641 let ops = inner
5642 .iter()
5643 .filter(|r| !matches!(r, WalRecord::Intern { .. }))
5644 .count();
5645 if ops > 1 {
5646 self.emit(MutationEvent::BatchApplied { ops });
5647 }
5648 }
5649 }
5650 }
5651 other => {
5652 if let Some(ev) = event_from_record(other, &self.syms, &self.ids) {
5653 self.emit(ev);
5654 }
5655 }
5656 }
5657 }
5658
5659 // -----------------------------------------------------------------------
5660 // Subscription API
5661 // -----------------------------------------------------------------------
5662
5663 /// Distribute post-commit events to all live subscribers.
5664 ///
5665 /// Build a row-key → row-data map from a [`ResultSet`].
5666 ///
5667 /// Each row is serialized to JSON to form its key; a debug fallback is used
5668 /// if serialization fails. Used by both the initial-seed path in
5669 /// [`Self::subscribe_query`] and the per-commit diff path in
5670 /// [`Self::distribute_events`] to keep the two in sync.
5671 fn result_to_row_map(
5672 result: &core_query::ResultSet,
5673 ) -> std::collections::HashMap<String, Vec<Option<Value>>> {
5674 (0..result.len())
5675 .map(|i| {
5676 let row = result.row(i).to_vec();
5677 let key = serde_json::to_string(&row).unwrap_or_else(|_| format!("{row:?}"));
5678 (key, row)
5679 })
5680 .collect()
5681 }
5682
5683 /// Collect the set of label syms touched by a WAL record.
5684 ///
5685 /// Returns `Some(set)` when every record in this commit can be attributed to
5686 /// a known label sym. Returns `None` when the commit must not be skipped:
5687 /// edge records, unresolvable key→label lookups, or any record type not in
5688 /// the explicit handled set.
5689 ///
5690 /// Handled record types and their actions:
5691 /// - `InsertNode` → look up label in interner (fails → None)
5692 /// - `InsertNodeId` → label sym is carried directly
5693 /// - `SetProp` → resolve key→id→label (fails → None)
5694 /// - `DeleteNode` → resolve key→id→label (fails → None)
5695 /// - `Batch` → recurse into every inner record
5696 /// - `InsertEdge`, `DeleteEdge`, `InsertEdgeId` → always None (edge records)
5697 /// - everything else → None (conservative)
5698 fn commit_touched_labels(
5699 rec: &WalRecord,
5700 syms: &Interner,
5701 ids: &IdMap,
5702 labels: &[u32],
5703 ) -> Option<BTreeSet<u32>> {
5704 let mut out = BTreeSet::new();
5705 if Self::collect_touched_labels(rec, syms, ids, labels, &mut out) {
5706 Some(out)
5707 } else {
5708 None
5709 }
5710 }
5711
5712 fn collect_touched_labels(
5713 rec: &WalRecord,
5714 syms: &Interner,
5715 ids: &IdMap,
5716 labels: &[u32],
5717 out: &mut BTreeSet<u32>,
5718 ) -> bool {
5719 match rec {
5720 // String-key insert: the dense rewrite converts this to
5721 // [Intern, InsertNodeId], so this arm fires only for legacy WAL
5722 // records written before the dense path was added.
5723 WalRecord::InsertNode { label, .. } => {
5724 if let Some(sym) = syms.get(label) {
5725 out.insert(sym);
5726 true
5727 } else {
5728 false
5729 }
5730 }
5731 // Dense-id insert (produced by rewrite_wal_dense for every
5732 // insert_node call in the current codebase).
5733 WalRecord::InsertNodeId { label, .. } => {
5734 out.insert(*label);
5735 true
5736 }
5737 // String-key prop set: dense path converts to [Intern, SetPropId].
5738 WalRecord::SetProp { key, .. } => {
5739 if let Some(sym) = Self::resolve_key_label_sym(key, ids, labels) {
5740 out.insert(sym);
5741 true
5742 } else {
5743 false
5744 }
5745 }
5746 // Dense-id prop set (produced by rewrite_wal_dense for set_prop).
5747 WalRecord::SetPropId { id, .. } => {
5748 if let Some(sym) = labels.get(*id as usize).copied().filter(|&s| s != u32::MAX) {
5749 out.insert(sym);
5750 true
5751 } else {
5752 false
5753 }
5754 }
5755 WalRecord::DeleteNode { key } => {
5756 if let Some(sym) = Self::resolve_key_label_sym(key, ids, labels) {
5757 out.insert(sym);
5758 true
5759 } else {
5760 false
5761 }
5762 }
5763 WalRecord::Batch(inner) => inner
5764 .iter()
5765 .all(|r| Self::collect_touched_labels(r, syms, ids, labels, out)),
5766 // Intern is a pure metadata record — it does not touch any node's
5767 // label and is safe to skip for the label-skip predicate.
5768 WalRecord::Intern { .. } => true,
5769 // Edge records: always re-execute (edges can change join results).
5770 WalRecord::InsertEdge { .. }
5771 | WalRecord::DeleteEdge { .. }
5772 | WalRecord::InsertEdgeId { .. } => false,
5773 _ => false,
5774 }
5775 }
5776
5777 /// Resolve a node key to its label sym via the dense id table.
5778 /// Returns `None` if the key is unknown or the label is a tombstone sentinel.
5779 fn resolve_key_label_sym(key: &str, ids: &IdMap, labels: &[u32]) -> Option<u32> {
5780 let id = ids.get(key)?;
5781 let sym = labels.get(id as usize).copied()?;
5782 (sym != u32::MAX).then_some(sym)
5783 }
5784
5785 /// Distribute post-commit events to all live subscribers.
5786 ///
5787 /// Called from `log_then_apply_with` after apply + fsync, before the
5788 /// legacy MutationEvent sink. Prunes dead `Weak` entries in-place.
5789 ///
5790 /// Query subscriptions (subscribe_query) re-execute their plan on every
5791 /// call and diff the result against the previous run. Zero overhead when
5792 /// no query subscriptions are active.
5793 fn distribute_events(&mut self, rec: &WalRecord, engine_deltas: &[EngineEdgeDelta], seq: u64) {
5794 if self.subscriptions.is_empty() && self.query_subscriptions.is_empty() {
5795 return;
5796 }
5797
5798 if !self.subscriptions.is_empty() {
5799 // Build write events from the WAL record.
5800 let write_events: Vec<DbEvent> =
5801 Self::write_events_from_record(rec, seq, &self.syms, &self.ids);
5802
5803 // Build edge events from engine deltas. Weight is looked up from
5804 // edge_props at distribution time (after apply), so it's always fresh.
5805 let edge_events: Vec<DbEvent> = engine_deltas
5806 .iter()
5807 .map(|d| {
5808 if d.fired {
5809 // The score lives under the rule's declared weight_prop,
5810 // which is not always the literal "weight".
5811 let prop = self
5812 .engine
5813 .rules()
5814 .find(|r| r.name == d.rule)
5815 .and_then(|r| r.weight_prop.as_deref());
5816 let weight = prop.and_then(|p| {
5817 self.edge_props
5818 .get(d.etype_sym, d.src_id, d.dst_id, p)
5819 .and_then(|v| {
5820 if let core_storage::Value::Float(f) = v {
5821 Some(*f)
5822 } else {
5823 None
5824 }
5825 })
5826 });
5827 DbEvent::EdgeFired {
5828 rule: d.rule.clone(),
5829 src_key: d.src_key.clone(),
5830 dst_key: d.dst_key.clone(),
5831 edge_type: d.edge_type.clone(),
5832 weight,
5833 commit_seq: seq,
5834 }
5835 } else {
5836 DbEvent::EdgeRetracted {
5837 rule: d.rule.clone(),
5838 src_key: d.src_key.clone(),
5839 dst_key: d.dst_key.clone(),
5840 edge_type: d.edge_type.clone(),
5841 commit_seq: seq,
5842 }
5843 }
5844 })
5845 .collect();
5846
5847 // Prune dead entries; push matching events to live ones.
5848 self.subscriptions.retain(|entry| {
5849 let Some(inner) = entry.inner.upgrade() else {
5850 return false;
5851 };
5852 for ev in &write_events {
5853 if event_matches(ev, &entry.filter) {
5854 inner.push(ev.clone());
5855 }
5856 }
5857 for ev in &edge_events {
5858 if event_matches(ev, &entry.filter) {
5859 inner.push(ev.clone());
5860 }
5861 }
5862 true
5863 });
5864
5865 // Turn off delta accumulation if all subscribers dropped and no views remain.
5866 if self.subscriptions.is_empty() && self.view_store.is_empty() {
5867 self.engine.set_emit_deltas(false);
5868 }
5869 }
5870
5871 // Query subscriptions: full re-run per commit, then diff rows.
5872 // IMPORTANT: full re-execution on every commit — use LIMIT to bound cost.
5873 // Differential evaluation is roadmap / Phase 5.
5874 if !self.query_subscriptions.is_empty() {
5875 // Take the list out so we can call self.view() without borrow conflict.
5876 let mut query_subs = std::mem::take(&mut self.query_subscriptions);
5877 let empty_params = BTreeMap::new();
5878 query_subs.retain_mut(|entry| {
5879 let Some(inner) = entry.inner.upgrade() else {
5880 return false; // subscriber dropped — prune
5881 };
5882 // Label-skip: if the plan has a known scan label and this commit
5883 // can be proven to touch only different labels (and no rule-derived
5884 // edge deltas fired), the result set cannot have changed — skip.
5885 if let Some(scan_sym) = entry.scan_label {
5886 if engine_deltas.is_empty() {
5887 let touched =
5888 Self::commit_touched_labels(rec, &self.syms, &self.ids, &self.labels);
5889 if touched.map(|t| !t.contains(&scan_sym)).unwrap_or(false) {
5890 return true; // safe to skip — result set unchanged
5891 }
5892 }
5893 }
5894 QUERY_SUB_EXECS_TL.with(|c| c.set(c.get() + 1));
5895 let result = match execute(&self.view(), &entry.ops, &Params(&empty_params)) {
5896 Ok(r) => r,
5897 Err(e) => {
5898 // Keep the subscription alive; skip the diff for this commit.
5899 // Re-run errors are transient (e.g., planner change) and
5900 // self-heal when the next commit succeeds.
5901 eprintln!("[mushroomdb] subscribe_query re-run failed: {e}");
5902 return true;
5903 }
5904 };
5905 // Build new row map: serialized-key → row data.
5906 let new_row_map = Self::result_to_row_map(&result);
5907 // Removed rows: in prev but not in new.
5908 for (key, row) in &entry.prev_row_map {
5909 if !new_row_map.contains_key(key) {
5910 inner.push(DbEvent::QueryRowRemoved {
5911 columns: entry.columns.clone(),
5912 row: row.clone(),
5913 });
5914 }
5915 }
5916 // Added rows: in new but not in prev.
5917 for (key, row) in &new_row_map {
5918 if !entry.prev_row_map.contains_key(key) {
5919 inner.push(DbEvent::QueryRowAdded {
5920 columns: entry.columns.clone(),
5921 row: row.clone(),
5922 });
5923 }
5924 }
5925 entry.prev_row_map = new_row_map;
5926 true
5927 });
5928 self.query_subscriptions = query_subs;
5929 }
5930 }
5931
5932 /// Returns `true` if any live subscriber or view definition requires delta
5933 /// accumulation. Used to set `engine.emit_deltas` on subscribe/view DDL.
5934 fn needs_emit_deltas(&self) -> bool {
5935 !self.view_store.is_empty()
5936 || self
5937 .subscriptions
5938 .iter()
5939 .any(|e| e.inner.upgrade().is_some())
5940 }
5941
5942 /// Convert a WAL record into `DbEvent` write events with the given seq.
5943 fn write_events_from_record(
5944 rec: &WalRecord,
5945 seq: u64,
5946 intern: &Interner,
5947 ids: &IdMap,
5948 ) -> Vec<DbEvent> {
5949 match rec {
5950 WalRecord::InsertNode { label, key, .. } => vec![DbEvent::NodeInserted {
5951 label: label.clone(),
5952 key: key.clone(),
5953 commit_seq: seq,
5954 }],
5955 // *Id arms run after a successful apply, so resolution can only
5956 // fail on a programming error. Skip the event rather than emit a
5957 // fabricated "" that clients can't tell from a real empty value
5958 // (mirrors event_from_record returning None).
5959 WalRecord::InsertNodeId { label, key, .. } => intern
5960 .resolve(*label)
5961 .map(|label| DbEvent::NodeInserted {
5962 label: label.to_string(),
5963 key: key.clone(),
5964 commit_seq: seq,
5965 })
5966 .into_iter()
5967 .collect(),
5968 WalRecord::SetProp { key, field, .. } => vec![DbEvent::PropSet {
5969 key: key.clone(),
5970 field: field.clone(),
5971 commit_seq: seq,
5972 }],
5973 WalRecord::SetPropId { id, field, .. } => ids
5974 .key_of(*id)
5975 .zip(intern.resolve(*field))
5976 .map(|(key, field)| DbEvent::PropSet {
5977 key: key.to_string(),
5978 field: field.to_string(),
5979 commit_seq: seq,
5980 })
5981 .into_iter()
5982 .collect(),
5983 WalRecord::RemoveProp { key, field } => vec![DbEvent::PropRemoved {
5984 key: key.clone(),
5985 field: field.clone(),
5986 commit_seq: seq,
5987 }],
5988 WalRecord::InsertEdge {
5989 edge_type,
5990 src_key,
5991 dst_key,
5992 } => vec![DbEvent::EdgeInserted {
5993 edge_type: edge_type.clone(),
5994 src: src_key.clone(),
5995 dst: dst_key.clone(),
5996 commit_seq: seq,
5997 }],
5998 WalRecord::InsertEdgeId { etype, src, dst } => (|| {
5999 Some(DbEvent::EdgeInserted {
6000 edge_type: intern.resolve(*etype)?.to_string(),
6001 src: ids.key_of(*src)?.to_string(),
6002 dst: ids.key_of(*dst)?.to_string(),
6003 commit_seq: seq,
6004 })
6005 })()
6006 .into_iter()
6007 .collect(),
6008 WalRecord::DeleteEdge {
6009 edge_type,
6010 src_key,
6011 dst_key,
6012 } => vec![DbEvent::EdgeDeleted {
6013 edge_type: edge_type.clone(),
6014 src: src_key.clone(),
6015 dst: dst_key.clone(),
6016 commit_seq: seq,
6017 }],
6018 WalRecord::DeleteNode { key } => vec![DbEvent::NodeDeleted {
6019 key: key.clone(),
6020 commit_seq: seq,
6021 }],
6022 WalRecord::Batch(inner) => inner
6023 .iter()
6024 .flat_map(|r| Self::write_events_from_record(r, seq, intern, ids))
6025 .collect(),
6026 WalRecord::CreateRule { .. }
6027 | WalRecord::DeleteRule { .. }
6028 | WalRecord::RebuildRule { .. }
6029 | WalRecord::CreateView { .. }
6030 | WalRecord::DeleteView { .. }
6031 | WalRecord::EnableFulltext { .. }
6032 | WalRecord::DisableFulltext { .. }
6033 | WalRecord::EnableIndex { .. }
6034 | WalRecord::DisableIndex { .. }
6035 | WalRecord::Intern { .. }
6036 // History markers produce no DbEvent — the engine delta already
6037 // fired the EdgeFired/EdgeRetracted subscription events.
6038 | WalRecord::DerivedEdgeAdded { .. }
6039 | WalRecord::DerivedEdgeRetracted { .. }
6040 // A count is not an edge event: the pair it counts already fired one
6041 // when it was first inserted.
6042 | WalRecord::SetEdgeCount { .. }
6043 | WalRecord::RenameNode { .. } => vec![],
6044 }
6045 }
6046
6047 /// Subscribe to edge-fire and edge-retract events for one named rule.
6048 ///
6049 /// Returns `Err(GraphError::RuleNotFound)` if `rule_name` is not
6050 /// currently registered. Dropping the returned [`Subscription`] handle
6051 /// unregisters the subscriber — no further events are queued, no
6052 /// resources leak.
6053 pub fn subscribe_rule(&mut self, rule_name: &str) -> core_storage::Result<Subscription> {
6054 if self.read_only {
6055 return Err(core_storage::GraphError::ReadOnly);
6056 }
6057 if !self.engine.rules().any(|r| r.name == rule_name) {
6058 return Err(core_storage::GraphError::RuleNotFound {
6059 name: rule_name.to_string(),
6060 });
6061 }
6062 let inner = SubInner::new(self.sub_capacity());
6063 self.subscriptions.push(SubEntry {
6064 filter: SubFilter::Rule(rule_name.to_string()),
6065 inner: std::sync::Arc::downgrade(&inner),
6066 });
6067 self.engine.set_emit_deltas(true);
6068 Ok(Subscription(inner))
6069 }
6070
6071 /// Subscribe to edge-fire and edge-retract events for **all** rules.
6072 ///
6073 /// Returns `Err(GraphError::ReadOnly)` if called on an as-of instance —
6074 /// as-of instances never commit, so `distribute_events` never runs and the
6075 /// subscription would never deliver events.
6076 pub fn subscribe_all_rules(&mut self) -> core_storage::Result<Subscription> {
6077 if self.read_only {
6078 return Err(core_storage::GraphError::ReadOnly);
6079 }
6080 let inner = SubInner::new(self.sub_capacity());
6081 self.subscriptions.push(SubEntry {
6082 filter: SubFilter::AllRules,
6083 inner: std::sync::Arc::downgrade(&inner),
6084 });
6085 self.engine.set_emit_deltas(true);
6086 Ok(Subscription(inner))
6087 }
6088
6089 /// Subscribe to write events: node insert/delete, prop set/remove.
6090 ///
6091 /// Does not include edge-fire / edge-retract (rule-derived edge events).
6092 ///
6093 /// Returns `Err(GraphError::ReadOnly)` if called on an as-of instance —
6094 /// as-of instances never commit, so `distribute_events` never runs and the
6095 /// subscription would never deliver events.
6096 pub fn subscribe_writes(&mut self) -> core_storage::Result<Subscription> {
6097 if self.read_only {
6098 return Err(core_storage::GraphError::ReadOnly);
6099 }
6100 let inner = SubInner::new(self.sub_capacity());
6101 self.subscriptions.push(SubEntry {
6102 filter: SubFilter::Writes,
6103 inner: std::sync::Arc::downgrade(&inner),
6104 });
6105 self.engine.set_emit_deltas(true);
6106 Ok(Subscription(inner))
6107 }
6108
6109 /// Subscribe to incremental Cypher query results.
6110 ///
6111 /// Parses and plans `cypher`; rejects the query if the plan is not in the
6112 /// allowlisted subset (see [`core_query::cypher::is_subscribable`]):
6113 /// - `MATCH (n:Label) WHERE … RETURN … [LIMIT n]`
6114 /// - `MATCH (a)-[r:TYPE]->(b) RETURN … [LIMIT n]` (exactly one hop)
6115 ///
6116 /// SKIP is not supported — it shifts the result window on every commit,
6117 /// causing spurious Added/Removed churn for rows whose data never changed.
6118 /// Multi-hop Expand chains are not supported; each additional MATCH clause
6119 /// widens scope beyond the documented single-scan / single-hop subset.
6120 ///
6121 /// After each successful commit, the plan is **fully re-executed** and the
6122 /// result is diffed against the previous run. Added rows produce
6123 /// [`DbEvent::QueryRowAdded`]; removed rows produce
6124 /// [`DbEvent::QueryRowRemoved`].
6125 ///
6126 /// **Full re-run per commit; use LIMIT to bound execution cost.**
6127 /// The existing 1 M intermediate-row cap applies. Differential evaluation
6128 /// is roadmap / Phase 5.
6129 ///
6130 /// Returns `Err(GraphError::ReadOnly)` if called on an as-of instance —
6131 /// as-of instances never commit, so `distribute_events` never runs and the
6132 /// subscription would never deliver events.
6133 ///
6134 /// Returns `Err(GraphError::QueryError)` if the query fails to parse, plan,
6135 /// or if the plan shape is not in the allowlist.
6136 pub fn subscribe_query(&mut self, cypher: &str) -> Result<Subscription> {
6137 if self.read_only {
6138 return Err(GraphError::ReadOnly);
6139 }
6140 let tokens = lex(cypher).map_err(|e| GraphError::QueryError {
6141 detail: format!("lex: {e}"),
6142 })?;
6143 let ast = parse(&tokens).map_err(|e| GraphError::QueryError {
6144 detail: format!("parse: {e}"),
6145 })?;
6146 let ops = plan(&ast).map_err(|e| GraphError::QueryError {
6147 detail: format!("plan: {e}"),
6148 })?;
6149 if !is_subscribable(&ops) {
6150 return Err(GraphError::QueryError {
6151 detail: "subscribe_query only supports allowlisted plan shapes: \
6152 MATCH (n:Label) WHERE … RETURN … [LIMIT n] or \
6153 MATCH (a)-[r:TYPE]->(b) RETURN … [LIMIT n] (exactly one hop). \
6154 Not supported: multi-hop Expand chains, SKIP (creates \
6155 unstable offset windows), ORDER BY, DISTINCT, aggregates, \
6156 variable-length paths, OPTIONAL MATCH, WITH, UNWIND. \
6157 Use LIMIT to bound re-execution cost."
6158 .to_string(),
6159 });
6160 }
6161 // Execute once to capture initial state (initial rows are not emitted as
6162 // events — the subscriber learns the baseline via the first query call).
6163 let empty_params = BTreeMap::new();
6164 let initial = execute(&self.view(), &ops, &Params(&empty_params)).map_err(|e| {
6165 GraphError::QueryError {
6166 detail: format!("execute: {e}"),
6167 }
6168 })?;
6169 let columns = initial.columns().to_vec();
6170 let prev_row_map = Self::result_to_row_map(&initial);
6171 let inner = SubInner::new(self.sub_capacity());
6172 // Derive the scan-label sym for the commit-skip fast-path. Any Expand op
6173 // or unrecognized leading scan → None (always re-execute).
6174 let scan_label = extract_scan_label(&ops, Arc::make_mut(&mut self.syms));
6175 self.query_subscriptions.push(QuerySubEntry {
6176 ops,
6177 columns,
6178 prev_row_map,
6179 inner: std::sync::Arc::downgrade(&inner),
6180 scan_label,
6181 });
6182 Ok(Subscription(inner))
6183 }
6184
6185 /// Queue capacity used for new subscriptions.
6186 fn sub_capacity(&self) -> usize {
6187 self.sub_capacity
6188 }
6189
6190 /// Override per-subscriber queue capacity for subsequently created
6191 /// subscriptions on this db instance.
6192 ///
6193 /// Default is [`DEFAULT_SUB_CAPACITY`] (65,536 events). Use a smaller
6194 /// value in tests to exercise the [`DbEvent::Lagged`] path without
6195 /// generating tens of thousands of events.
6196 ///
6197 /// This is a test-support escape hatch. Calling it in production reduces
6198 /// subscriber reliability (more Lagged events). It is hidden from rustdoc
6199 /// to discourage accidental production use.
6200 #[doc(hidden)]
6201 pub fn set_sub_capacity(&mut self, capacity: usize) {
6202 self.sub_capacity = capacity;
6203 }
6204
6205 // -----------------------------------------------------------------------
6206
6207 /// Start an atomic batch.
6208 ///
6209 /// The returned [`BatchBuilder`] borrows `self` mutably until
6210 /// [`BatchBuilder::commit`]. Builder methods queue ops only — no
6211 /// validation, no WAL I/O. `commit` validates every queued op against
6212 /// live state plus preceding ops in this batch (duplicate key inside
6213 /// the batch is `Err`; an edge between two nodes created earlier in
6214 /// the batch is valid; `delete_node` then insert of the same key is a
6215 /// fresh identity). Validation never mutates the database. Any failure
6216 /// leaves WAL bytes and in-memory state identical to before `commit`.
6217 /// On success, one `WalRecord::Batch` frame is appended (one fsync)
6218 /// and each inner record is applied in order so rules fire per record.
6219 /// An empty batch, or a batch of only no-ops, writes zero WAL bytes.
6220 ///
6221 /// **Rule-window limitation:** batch validation cannot see edges that a
6222 /// rule created earlier in the *same* batch will derive at apply time, so
6223 /// a `delete_edge` / `insert_edge` in that window is silently no-oped
6224 /// where sequential calls would return `Err(RuleOwned)`. State integrity
6225 /// is unaffected (idempotent apply, provenance intact). Create rules in
6226 /// their own batch, or sequentially, when later ops may touch derived
6227 /// edges.
6228 pub fn batch(&mut self) -> BatchBuilder<'_, F> {
6229 BatchBuilder {
6230 db: self,
6231 ops: Vec::new(),
6232 }
6233 }
6234
6235 /// Closure-style atomic write batch.
6236 ///
6237 /// Equivalent to calling [`GraphDb::batch`], invoking `build` to queue ops,
6238 /// then committing. All ops queued inside `build` are validated in order and
6239 /// committed as a single `WalRecord::Batch` frame (one fsync). Rules fire
6240 /// once per inner record, in order, after commit — semantically identical to
6241 /// sequential single-op writes.
6242 ///
6243 /// **Error semantics — validate-then-apply.** `build` queues ops without
6244 /// touching the database. [`BatchBuilder::commit`] validates every op against
6245 /// live state plus earlier ops in this batch before writing anything. If op N
6246 /// fails validation (duplicate key, unknown key, rule-owned edge, …) the
6247 /// entire batch is rejected: no WAL bytes are written and no in-memory state
6248 /// changes. The database is identical to its state before `write_batch` was
6249 /// called.
6250 ///
6251 /// **Atomicity is crash-level, NOT isolation-level.** On replay after a crash,
6252 /// a partial (torn) `Batch` frame applies NONE of its ops — the frame is
6253 /// either fully applied or not at all. However, while applying a committed
6254 /// batch, concurrent readers may observe intermediate states as ops are applied
6255 /// sequentially in memory. There is no interactive transaction isolation in v1.
6256 /// This is documented as "crash-atomic write batches; no interactive
6257 /// transactions or read isolation."
6258 ///
6259 /// **Returns** `(nodes_inserted, edges_inserted)`. An empty or all-noop batch
6260 /// writes zero WAL bytes and returns `(0, 0)`.
6261 ///
6262 /// # Example
6263 ///
6264 /// ```rust,ignore
6265 /// let (nodes, edges) = db.write_batch(|b| {
6266 /// b.insert_node("Person", "alice", vec![("age".into(), Value::Int(30))]);
6267 /// b.insert_node("Person", "bob", vec![]);
6268 /// b.insert_edge("KNOWS", "alice", "bob");
6269 /// b.set_prop("alice", "role", Value::Str("admin".into()));
6270 /// b.delete_node("old_key");
6271 /// })?;
6272 /// // One fsync; on crash replay: all five ops land or none do.
6273 /// ```
6274 pub fn write_batch<C>(&mut self, build: C) -> Result<(usize, usize)>
6275 where
6276 C: FnOnce(&mut BatchBuilder<'_, F>),
6277 {
6278 let mut b = self.batch();
6279 build(&mut b);
6280 b.commit()
6281 }
6282
6283 /// Insert `rows` as nodes of `label`. One call is one atomic batch:
6284 /// auto-declared KeyMatch rules (if any) first, then the accepted node
6285 /// inserts, so incremental fire sees the new rules. Per-row key problems
6286 /// are collected in [`IngestReport::row_errors`] and skipped; a commit
6287 /// `Err` means nothing was applied.
6288 ///
6289 /// Auto-FK rule names are `auto_fk_<src_label_lowercase>_<field>` so
6290 /// distinct source labels sharing an FK field each get their own rule.
6291 pub fn ingest(
6292 &mut self,
6293 label: &str,
6294 rows: Vec<BTreeMap<String, Value>>,
6295 opts: &IngestOptions,
6296 ) -> Result<IngestReport> {
6297 self.ingest_with_edges(label, rows, opts, &[])
6298 }
6299
6300 /// [`ingest`] plus user edges in the **same** previewed WAL batch.
6301 /// A failing edge rejects the whole request; nothing is applied.
6302 pub fn ingest_with_edges(
6303 &mut self,
6304 label: &str,
6305 rows: Vec<BTreeMap<String, Value>>,
6306 opts: &IngestOptions,
6307 edges: &[(String, String, String)],
6308 ) -> Result<IngestReport> {
6309 crate::ingest::run(self, label, rows, opts, edges)
6310 }
6311
6312 /// Parse `json` as an array of objects and ingest via [`GraphDb::ingest`].
6313 ///
6314 /// JSON `null` fields are silently omitted (not stored, not a row error).
6315 /// Nested objects and arrays-of-objects are a per-row error (row skipped).
6316 /// Parse failures and a top-level value that is not an array of objects
6317 /// return [`GraphError::IngestError`].
6318 pub fn ingest_json(
6319 &mut self,
6320 label: &str,
6321 json: &str,
6322 opts: &IngestOptions,
6323 ) -> Result<IngestReport> {
6324 crate::ingest::run_json(self, label, json, opts)
6325 }
6326
6327 fn commit_logged_batch(
6328 &mut self,
6329 ops: Vec<BatchOp>,
6330 ingest: Option<(String, usize)>,
6331 // Two-source rule: write_batch_authz threads authz here directly (never
6332 // touches pending_write_authz); query_write_authz sets the field instead
6333 // and passes None. Only one source is non-None per call.
6334 param_authz: Option<WriteAuthz>,
6335 ) -> Result<BatchOutcome> {
6336 // Read-only guard: catches empty-batch calls before the early-return
6337 // that skips log_then_apply_with, ensuring all mutation entry points fail.
6338 if self.read_only {
6339 return Err(GraphError::ReadOnly);
6340 }
6341 // Ensure provenance is decoded before MutPreview accesses it
6342 // (note_delete_rule / is_rule_owned may call engine.provenance()).
6343 self.engine.ensure_provenance_loaded_mut();
6344
6345 // ── Authz pre-check ──────────────────────────────────────────────────
6346 // Evaluate the decision table per-op BEFORE MutPreview so that a denial
6347 // produces no WAL frame (all-or-nothing at the authz boundary extends
6348 // the existing validate-then-apply contract to role-scope checks).
6349 //
6350 // `batch_created` tracks key→label for nodes created by earlier ops in
6351 // THIS batch, so InsertEdgeUpsert can count same-batch placeholder nodes
6352 // as visible without needing to call `self.ids.get` on not-yet-committed
6353 // keys (they won't be there yet).
6354 //
6355 // Two-source rule: param_authz (write_batch_authz path) takes precedence;
6356 // fall back to self.pending_write_authz (query_write_authz/Cypher path).
6357 // Cloning the field copy avoids a simultaneous borrow of self.ids below.
6358 let authz_opt = param_authz.or_else(|| self.pending_write_authz.clone());
6359 if let Some(ref authz) = authz_opt {
6360 let mut batch_created: BTreeMap<String, String> = BTreeMap::new();
6361 for op in &ops {
6362 self.check_single_op_authz(authz, op, &batch_created)?;
6363 // Update batch_created after a passing authz check so that
6364 // subsequent ops in this batch see the nodes as "about to exist".
6365 match op {
6366 BatchOp::InsertNode { label, key, .. } => {
6367 // Only track genuinely new nodes (absent from the
6368 // snapshot at authz-check time). A pre-existing visible
6369 // key would be a DuplicateKey — not a real creation —
6370 // so MutPreview handles it. Letting it into batch_created
6371 // would allow a later SetProp to bypass update_labels
6372 // via the "batch-created → always updatable" ruling
6373 // (delete+recreate exploit, fix for I1 review round 2).
6374 //
6375 // Accepted edge: for a delete+recreate-with-different-
6376 // label batch, node_status resolves the pre-delete
6377 // (store) label for any subsequent update checks. This
6378 // grants no net-new capability — a role that can delete+
6379 // create can already place arbitrary props via
6380 // InsertNode's own props field.
6381 if self.ids.get(key.as_str()).is_none() {
6382 batch_created.insert(key.clone(), label.clone());
6383 }
6384 }
6385 BatchOp::InsertEdgeUpsert {
6386 placeholder_label,
6387 src_key,
6388 dst_key,
6389 ..
6390 } => {
6391 // Both endpoints will be created if not already in store.
6392 for ep_key in [src_key, dst_key] {
6393 if self.ids.get(ep_key.as_str()).is_none()
6394 && !batch_created.contains_key(ep_key.as_str())
6395 {
6396 batch_created.insert(ep_key.clone(), placeholder_label.clone());
6397 }
6398 }
6399 }
6400 _ => {}
6401 }
6402 }
6403 }
6404
6405 let mut outcome = BatchOutcome::default();
6406 let recs = {
6407 let mut preview = MutPreview::new(self);
6408 let mut recs = Vec::with_capacity(ops.len());
6409 // Which node row we are on, counted over the node-insert ops only.
6410 // A caller that queues its rows in order reads this straight back
6411 // as the index into its own list.
6412 let mut node_row = 0usize;
6413 // Every field name the store knows, which a `Replace` needs to work
6414 // out what it removes. Resolved on the first `Replace` in the frame
6415 // and reused, so N replaces read the field list once, not N times.
6416 let mut store_fields: Option<Vec<String>> = None;
6417 // Duplicate inserts this frame has to count, each paired with the
6418 // position in `recs` it belongs at. The count itself is named in the
6419 // dense rewrite and not here: a duplicate's endpoints and edge type
6420 // may all be created by earlier ops in this same frame, and nothing
6421 // in the frame has a dense id yet. See [`PlannedRec`].
6422 let mut deferred_counts: Vec<(usize, String, String, String)> = Vec::new();
6423 for op in ops {
6424 match op {
6425 BatchOp::InsertNode { label, key, props } => {
6426 node_row += 1;
6427 preview.check_insert_node(&key, &props)?;
6428 preview.note_insert_node(&label, &key, &props);
6429 recs.push(WalRecord::InsertNode { label, key, props });
6430 }
6431 BatchOp::InsertNodeOnConflict {
6432 label,
6433 key,
6434 props,
6435 on_conflict,
6436 } => {
6437 let row = node_row;
6438 node_row += 1;
6439 if !preview.has_key(&key) {
6440 // No conflict: an ordinary insert on any policy —
6441 // except that a supplied view-owned field is the
6442 // same mistake here as on a taken key, and gets the
6443 // same row error rather than a frame error. Without
6444 // this, one op answered one request two ways
6445 // depending on whether the store already had the
6446 // key (defect #19).
6447 if let Some(why) = preview.supplied_view_owned_prop(&key, &props) {
6448 outcome.row_errors.push((row, why));
6449 continue;
6450 }
6451 preview.note_insert_node(&label, &key, &props);
6452 recs.push(WalRecord::InsertNode { label, key, props });
6453 continue;
6454 }
6455 match on_conflict {
6456 OnConflict::Error => {
6457 return Err(GraphError::DuplicateKey { key });
6458 }
6459 OnConflict::Skip => outcome.skipped += 1,
6460 OnConflict::Replace => {
6461 if store_fields.is_none() {
6462 store_fields = Some(preview.db.props_view().field_names());
6463 }
6464 let fields = store_fields.as_deref().unwrap_or_default();
6465 match preview.plan_replace(&label, &key, &props, fields) {
6466 Ok((writes, kept_view_owned)) => {
6467 outcome.kept_view_owned += kept_view_owned;
6468 for (field, value) in writes {
6469 match value {
6470 Some(value) => {
6471 preview.note_set_prop(&key, &field, &value);
6472 recs.push(WalRecord::SetProp {
6473 key: key.clone(),
6474 field,
6475 value,
6476 });
6477 }
6478 None => {
6479 preview.note_remove_prop(&key, &field);
6480 recs.push(WalRecord::RemoveProp {
6481 key: key.clone(),
6482 field,
6483 });
6484 }
6485 }
6486 }
6487 outcome.replaced += 1;
6488 }
6489 Err(why) => outcome.row_errors.push((row, why)),
6490 }
6491 }
6492 }
6493 }
6494 BatchOp::InsertEdge {
6495 edge_type,
6496 src_key,
6497 dst_key,
6498 } => {
6499 if preview.prepare_insert_edge(&edge_type, &src_key, &dst_key)? {
6500 preview.note_insert_edge(&edge_type, &src_key, &dst_key);
6501 recs.push(WalRecord::InsertEdge {
6502 edge_type,
6503 src_key,
6504 dst_key,
6505 });
6506 } else if preview.db.multiplicity {
6507 // A duplicate inside a batch counts the way a
6508 // duplicate through `insert_edge` does: `ingest` and
6509 // Cypher `CREATE` reach this choke-point and not
6510 // that one, and a count only one entry point keeps
6511 // would be worse than no count at all.
6512 //
6513 // This is the one gate on discriminant 23 from the
6514 // batch path: a store that never opted in queues
6515 // nothing here and so writes no such record.
6516 deferred_counts.push((recs.len(), edge_type, src_key, dst_key));
6517 }
6518 }
6519 BatchOp::SetProp { key, field, value } => {
6520 if let Some(view_name) = preview.db.view_store.view_for_prop(&field) {
6521 return Err(GraphError::ViewPropReadOnly {
6522 view_name: view_name.to_string(),
6523 });
6524 }
6525 preview.check_live_key(&key)?;
6526 preview.note_set_prop(&key, &field, &value);
6527 recs.push(WalRecord::SetProp { key, field, value });
6528 }
6529 BatchOp::RemoveProp { key, field } => {
6530 if preview.prepare_remove_prop(&key, &field)? {
6531 preview.note_remove_prop(&key, &field);
6532 recs.push(WalRecord::RemoveProp { key, field });
6533 }
6534 }
6535 BatchOp::DeleteEdge {
6536 edge_type,
6537 src_key,
6538 dst_key,
6539 } => {
6540 if preview.prepare_delete_edge(&edge_type, &src_key, &dst_key)? {
6541 preview.note_delete_edge(&edge_type, &src_key, &dst_key);
6542 recs.push(WalRecord::DeleteEdge {
6543 edge_type,
6544 src_key,
6545 dst_key,
6546 });
6547 }
6548 }
6549 BatchOp::DeleteNode { key } => {
6550 preview.check_live_key(&key)?;
6551 preview.note_delete_node(&key);
6552 recs.push(WalRecord::DeleteNode { key });
6553 }
6554 BatchOp::CreateRule(def) => {
6555 preview.check_create_rule(&def)?;
6556 let def_bytes =
6557 bincode::serialize(&def).map_err(|e| GraphError::Corrupt {
6558 detail: format!("serialize rule: {e}"),
6559 })?;
6560 preview.note_create_rule(&def);
6561 recs.push(WalRecord::CreateRule { def_bytes });
6562 }
6563 BatchOp::DeleteRule { name } => {
6564 preview.check_delete_rule(&name)?;
6565 preview.note_delete_rule(&name);
6566 recs.push(WalRecord::DeleteRule { name });
6567 }
6568 BatchOp::RenameNode { old_key, new_key } => {
6569 preview.check_rename_node(&old_key, &new_key)?;
6570 preview.note_rename_node(&old_key, &new_key);
6571 recs.push(WalRecord::RenameNode { old_key, new_key });
6572 }
6573 BatchOp::InsertEdgeUpsert {
6574 edge_type,
6575 src_key,
6576 dst_key,
6577 placeholder_label,
6578 } => {
6579 // Auto-create any missing endpoints as plain InsertNode ops.
6580 // Rules fire and last-change is updated for each created node.
6581 for key in [&src_key, &dst_key] {
6582 if !preview.has_key(key) {
6583 // A placeholder endpoint carries no props, so
6584 // the view-owned check has nothing to refuse.
6585 preview.check_insert_node(key, &[])?;
6586 preview.note_insert_node(&placeholder_label, key, &[]);
6587 recs.push(WalRecord::InsertNode {
6588 label: placeholder_label.clone(),
6589 key: key.clone(),
6590 props: vec![],
6591 });
6592 }
6593 }
6594 if preview.prepare_insert_edge(&edge_type, &src_key, &dst_key)? {
6595 preview.note_insert_edge(&edge_type, &src_key, &dst_key);
6596 recs.push(WalRecord::InsertEdge {
6597 edge_type,
6598 src_key,
6599 dst_key,
6600 });
6601 } else if preview.db.multiplicity {
6602 // Same choke-point, same gate as `BatchOp::InsertEdge`
6603 // above: an upsert that finds the pair already there
6604 // is a duplicate insert and counts as one.
6605 deferred_counts.push((recs.len(), edge_type, src_key, dst_key));
6606 }
6607 }
6608 }
6609 }
6610 // Splice the deferred counts back into the positions they were
6611 // raised at, so a count still sits exactly where the duplicate did
6612 // — before any later op in the frame that deletes the pair.
6613 let mut planned: Vec<PlannedRec> =
6614 Vec::with_capacity(recs.len() + deferred_counts.len());
6615 let mut deferred = deferred_counts.into_iter().peekable();
6616 for (i, rec) in recs.into_iter().enumerate() {
6617 while deferred.peek().is_some_and(|(at, ..)| *at == i) {
6618 let (_, edge_type, src_key, dst_key) = deferred.next().expect("just peeked");
6619 planned.push(PlannedRec::DuplicateCount {
6620 edge_type,
6621 src_key,
6622 dst_key,
6623 });
6624 }
6625 planned.push(PlannedRec::Rec(rec));
6626 }
6627 for (_, edge_type, src_key, dst_key) in deferred {
6628 planned.push(PlannedRec::DuplicateCount {
6629 edge_type,
6630 src_key,
6631 dst_key,
6632 });
6633 }
6634 planned
6635 };
6636 // A frame that is nothing but skips or refused rows writes no WAL, but
6637 // it still has counts to report, so the early returns carry `outcome`
6638 // rather than zeros.
6639 if recs.is_empty() {
6640 return Ok(outcome);
6641 }
6642 // rewrite_wal_dense converts every InsertNode/InsertEdge into its
6643 // *Id form, so only the dense variants can appear in `recs` here.
6644 let recs = self.rewrite_wal_dense_planned(recs)?;
6645 // The rewrite can empty a non-empty batch: a `SET n.ns` naming the
6646 // namespace the node is already in is a no-op and is dropped there. An
6647 // empty `Batch` frame would still take a commit sequence and a WAL
6648 // record, so a batch that turns out to be nothing writes nothing.
6649 if recs.is_empty() {
6650 return Ok(outcome);
6651 }
6652 outcome.nodes_inserted = recs
6653 .iter()
6654 .filter(|r| matches!(r, WalRecord::InsertNodeId { .. }))
6655 .count();
6656 outcome.edges_inserted = recs
6657 .iter()
6658 .filter(|r| matches!(r, WalRecord::InsertEdgeId { .. }))
6659 .count();
6660 // Ingest / write_batch / query_write: one Batch frame, one fsync per call
6661 // under Strict. Pass self.fsync directly so Strict stays Strict —
6662 // wal_needs_sync(Strict, _) always returns true regardless of op count.
6663 // Mapping Strict → Batched (the prior bug) caused wal_needs_sync to
6664 // short-circuit on single-op batches and silently skip the fsync.
6665 // Batched fsyncs only for multi-op batches; Relaxed always skips.
6666 self.log_then_apply_with(WalRecord::Batch(recs), ingest, self.fsync)?;
6667 Ok(outcome)
6668 }
6669
6670 fn commit_batch(&mut self, ops: Vec<BatchOp>) -> Result<(usize, usize)> {
6671 self.commit_logged_batch(ops, None, None).map(inserted_pair)
6672 }
6673
6674 /// Commit one submission WITHOUT an fsync — for use inside `commit_group`
6675 /// and the group-commit drain thread, which do a single group fsync later.
6676 fn commit_batch_nosync(&mut self, ops: Vec<BatchOp>) -> Result<(usize, usize)> {
6677 // Restore fsync policy even on panic via a raw-pointer drop guard.
6678 // A panic here would poison the RwLock anyway, but the correct policy
6679 // must be in place if the guard is ever unwrapped.
6680 struct RestoreFsync(*mut FsyncPolicy, FsyncPolicy);
6681 impl Drop for RestoreFsync {
6682 fn drop(&mut self) {
6683 // SAFETY: the pointer is valid for the full duration of
6684 // commit_batch_nosync; the guard is dropped before the frame
6685 // returns, and GraphDb outlives this frame.
6686 unsafe {
6687 *self.0 = self.1;
6688 }
6689 }
6690 }
6691 let saved = self.fsync;
6692 // SAFETY: raw pointer into self; guard dropped within this frame.
6693 let _g = RestoreFsync(&mut self.fsync as *mut FsyncPolicy, saved);
6694 self.fsync = FsyncPolicy::Relaxed;
6695 self.commit_logged_batch(ops, None, None).map(inserted_pair)
6696 }
6697
6698 /// Commit multiple op-batches as a **group**: each submission gets its own
6699 /// WAL `Batch` frame, but there is exactly **one** `Fs::sync` for the whole
6700 /// group (under `Strict` / `Batched` policy; `Relaxed` skips all syncs).
6701 ///
6702 /// # Durability semantics
6703 ///
6704 /// A crash before the group fsync may lose **all** submissions in the group.
6705 /// A crash after the group fsync preserves all of them. No submission is
6706 /// ever torn: each WAL frame is either fully applied on replay or dropped
6707 /// in its entirety (CRC-protected frame boundaries).
6708 ///
6709 /// Events and subscription notifications fire per-submission immediately
6710 /// after apply, which may be before the group fsync. From a subscriber's
6711 /// perspective this is equivalent to the `Relaxed` durability window.
6712 /// Submitters using [`SharedDb::submit_batch`] only unblock after the group
6713 /// fsync, so from their perspective durability is fully guaranteed.
6714 ///
6715 /// # MVCC interplay
6716 ///
6717 /// Each submission records its own `CommitDelta`; the fold-every-K counter
6718 /// increments per submission (not per group), preserving existing reader
6719 /// snapshot semantics.
6720 ///
6721 /// # Returns
6722 ///
6723 /// One `Result<(nodes_inserted, edges_inserted)>` per input group element,
6724 /// in order. Failures are per-submission (validation errors); the group
6725 /// fsync error (if any) is returned as the second tuple element.
6726 pub fn commit_group(
6727 &mut self,
6728 groups: Vec<Vec<BatchOp>>,
6729 ) -> (Vec<Result<(usize, usize)>>, Option<GraphError>) {
6730 let mut results = Vec::with_capacity(groups.len());
6731 for ops in groups {
6732 results.push(self.commit_batch_nosync(ops));
6733 }
6734 let any_ok = results.iter().any(|r| r.is_ok());
6735 let sync_err = if self.fsync != FsyncPolicy::Relaxed && any_ok {
6736 self.fs
6737 .sync(core_storage::fs::FileId::Wal)
6738 .map_err(GraphError::Io)
6739 .err()
6740 } else {
6741 None
6742 };
6743 (results, sync_err)
6744 }
6745
6746 /// Like [`commit_group`] but skips the group fsync entirely.
6747 ///
6748 /// Used by the drain thread to apply submissions under the write lock and
6749 /// then perform the single fsync OUTSIDE the lock (via
6750 /// `core_storage::sync_wal_at`), reducing the write-lock hold time visible
6751 /// to concurrent readers.
6752 pub fn commit_group_nosync(
6753 &mut self,
6754 groups: Vec<Vec<BatchOp>>,
6755 ) -> Vec<Result<(usize, usize)>> {
6756 let mut results = Vec::with_capacity(groups.len());
6757 for ops in groups {
6758 results.push(self.commit_batch_nosync(ops));
6759 }
6760 results
6761 }
6762
6763 pub fn insert_node(
6764 &mut self,
6765 label: &str,
6766 key: &str,
6767 props: Vec<(String, Value)>,
6768 ) -> Result<()> {
6769 if self.read_only {
6770 return Err(GraphError::ReadOnly);
6771 }
6772 MutPreview::new(self).check_insert_node(key, &props)?;
6773 self.log_dense(vec![WalRecord::InsertNode {
6774 label: label.into(),
6775 key: key.into(),
6776 props,
6777 }])
6778 }
6779
6780 /// Insert a user edge. `Ok(true)` when the pair was new, `Ok(false)` when it
6781 /// was already there — the question is "was this pair new", and a duplicate
6782 /// does not make it so.
6783 ///
6784 /// On a store that called [`enable_multiplicity`](Self::enable_multiplicity)
6785 /// a duplicate is no longer a total no-op: it raises the pair's insert count
6786 /// (§5.13). Adjacency is still a set, so [`degree`](Self::degree) is
6787 /// unchanged and the return value is still `Ok(false)`; the count is visible
6788 /// only through [`degree_multiplicity`](Self::degree_multiplicity) and the
6789 /// reserved [`EDGE_COUNT_PROP`]. On every other store a duplicate writes
6790 /// nothing at all, as it always has.
6791 pub fn insert_edge(&mut self, edge_type: &str, src_key: &str, dst_key: &str) -> Result<bool> {
6792 if self.read_only {
6793 return Err(GraphError::ReadOnly);
6794 }
6795 if !MutPreview::new(self).prepare_insert_edge(edge_type, src_key, dst_key)? {
6796 // The pair exists. The only thing left to record is that it was
6797 // asked for again, and only where the store asked to be told.
6798 if let Some(rec) = self.edge_count_record(edge_type, src_key, dst_key) {
6799 self.log_then_apply(rec)?;
6800 }
6801 return Ok(false);
6802 }
6803 self.log_dense(vec![WalRecord::InsertEdge {
6804 edge_type: edge_type.into(),
6805 src_key: src_key.into(),
6806 dst_key: dst_key.into(),
6807 }])?;
6808 Ok(true)
6809 }
6810
6811 pub fn set_prop(&mut self, key: &str, field: &str, value: Value) -> Result<()> {
6812 if self.read_only {
6813 return Err(GraphError::ReadOnly);
6814 }
6815 if let Some(view_name) = self.view_store.view_for_prop(field) {
6816 return Err(GraphError::ViewPropReadOnly {
6817 view_name: view_name.to_string(),
6818 });
6819 }
6820 MutPreview::new(self).check_live_key(key)?;
6821 self.log_dense(vec![WalRecord::SetProp {
6822 key: key.into(),
6823 field: field.into(),
6824 value,
6825 }])
6826 }
6827
6828 /// Set several properties on one live node in a single WAL commit.
6829 ///
6830 /// Every per-property check [`set_prop`](Self::set_prop) runs — view-owned
6831 /// names, live key, the `ns` immutability rule and its type — is evaluated
6832 /// for the whole list before any record is logged. The first refusal
6833 /// returns and the node is unchanged. An empty list writes nothing.
6834 pub fn set_props(&mut self, key: &str, props: Vec<(String, Value)>) -> Result<()> {
6835 if self.read_only {
6836 return Err(GraphError::ReadOnly);
6837 }
6838 MutPreview::new(self).check_live_key(key)?;
6839 for (field, _) in &props {
6840 if let Some(view_name) = self.view_store.view_for_prop(field) {
6841 return Err(GraphError::ViewPropReadOnly {
6842 view_name: view_name.to_string(),
6843 });
6844 }
6845 }
6846 if props.is_empty() {
6847 return Ok(());
6848 }
6849 self.write_batch(|b| {
6850 for (field, value) in props {
6851 b.set_prop(key, &field, value);
6852 }
6853 })
6854 .map(|_| ())
6855 }
6856
6857 /// Remove a property. Returns `Ok(false)` (and does not log) if the field
6858 /// is already absent. Unknown or tombstoned keys are `Err(KeyNotFound)`.
6859 /// A field a view owns is `Err(ViewPropReadOnly)` — stated once, in
6860 /// [`MutPreview::prepare_remove_prop`], so that the batch ops reaching that
6861 /// same choke-point cannot miss it.
6862 pub fn remove_prop(&mut self, key: &str, field: &str) -> Result<bool> {
6863 if self.read_only {
6864 return Err(GraphError::ReadOnly);
6865 }
6866 if !MutPreview::new(self).prepare_remove_prop(key, field)? {
6867 return Ok(false);
6868 }
6869 self.log_then_apply(WalRecord::RemoveProp {
6870 key: key.into(),
6871 field: field.into(),
6872 })?;
6873 Ok(true)
6874 }
6875
6876 /// Delete a user edge. Returns `Ok(false)` (and does not log) if the edge
6877 /// is absent. Unknown keys are `Err(KeyNotFound)`. Rule-owned edges — in
6878 /// provenance, or a pair a live rule would derive — are `Err(RuleOwned)`
6879 /// (the rule would just put the edge back; delete or change the rule).
6880 pub fn delete_edge(&mut self, edge_type: &str, src_key: &str, dst_key: &str) -> Result<bool> {
6881 if self.read_only {
6882 return Err(GraphError::ReadOnly);
6883 }
6884 if !MutPreview::new(self).prepare_delete_edge(edge_type, src_key, dst_key)? {
6885 return Ok(false);
6886 }
6887 self.log_then_apply(WalRecord::DeleteEdge {
6888 edge_type: edge_type.into(),
6889 src_key: src_key.into(),
6890 dst_key: dst_key.into(),
6891 })?;
6892 Ok(true)
6893 }
6894
6895 /// Delete a live node. Unknown or already-tombstoned keys are
6896 /// `Err(KeyNotFound)` and are not logged. Validation runs before the WAL
6897 /// write; `apply` of a logged `DeleteNode` for an already-tombstoned key
6898 /// (crash window) is a clean no-op.
6899 ///
6900 /// Returns a [`DeleteReport`] with counts of manual and derived edges
6901 /// removed (computed from live state before the deletion is applied).
6902 pub fn delete_node(&mut self, key: &str) -> Result<DeleteReport> {
6903 if self.read_only {
6904 return Err(GraphError::ReadOnly);
6905 }
6906 // Provenance must be loaded before we query provenance_touching.
6907 self.engine.ensure_provenance_loaded_mut();
6908 let id = self
6909 .ids
6910 .get(key)
6911 .ok_or_else(|| GraphError::KeyNotFound { key: key.into() })?;
6912
6913 // Count edges before the delete is applied so we can report counts.
6914 let derived_set: BTreeSet<(u32, u32, u32)> = self
6915 .engine
6916 .provenance_touching(id)
6917 .map(|(_, etype, src, dst)| (etype, src, dst))
6918 .collect();
6919 let derived_edges = derived_set.len() as u64;
6920
6921 let mut total_topo = 0u64;
6922 let tv = self.topo_view();
6923 for et in tv.etypes() {
6924 total_topo += tv.neighbors(et, Direction::Out, id).len() as u64
6925 + tv.neighbors(et, Direction::In, id).len() as u64;
6926 }
6927 // For symmetric rules (e.g. Overlap), a→b and b→a are two separate directed
6928 // triples in both the topo scan (Out and In from id) and in provenance_touching.
6929 // The subtraction remains correct because both counts include both directions.
6930 let manual_edges = total_topo.saturating_sub(derived_edges);
6931
6932 self.log_then_apply(WalRecord::DeleteNode { key: key.into() })?;
6933 Ok(DeleteReport {
6934 manual_edges,
6935 derived_edges,
6936 })
6937 }
6938
6939 /// Rename a live node's key. The dense id (and therefore all edges,
6940 /// props, history, and last-change tracking) is unaffected.
6941 ///
6942 /// Returns `Err(KeyNotFound)` if `old` is not a live key.
6943 /// Returns `Err(DuplicateKey)` if `new` is already live.
6944 pub fn rename_node(&mut self, old: &str, new: &str) -> Result<()> {
6945 if self.read_only {
6946 return Err(GraphError::ReadOnly);
6947 }
6948 MutPreview::new(self).check_rename_node(old, new)?;
6949 self.log_then_apply(WalRecord::RenameNode {
6950 old_key: old.into(),
6951 new_key: new.into(),
6952 })
6953 }
6954
6955 /// Return the IVF drift counter for the dst-side candidate index of `rule`.
6956 /// `None` if the rule does not exist or is not approximate.
6957 ///
6958 /// The drift counter increments on IVF insert/remove after the last fit.
6959 /// When dst-side drift exceeds [`core_rules::IVF_DRIFT_REBUILD`], apply
6960 /// WAL-logs `RebuildRule` as a second commit (rebuild resets the counter).
6961 pub fn ivf_dst_drift(&self, rule: &str) -> Option<u64> {
6962 // SideIvfExport = (centroids, node→cluster, drift)
6963 self.engine
6964 .export_ivf_state()
6965 .remove(rule)
6966 .map(|(_src, dst)| dst.2)
6967 }
6968
6969 /// Validate and WAL-log a new rule, then backfill derived edges inside apply.
6970 /// Validation and duplicate-name check run before logging so invalid rules
6971 /// never enter the WAL.
6972 pub fn create_rule(&mut self, def: RuleDef) -> Result<()> {
6973 if self.read_only {
6974 return Err(GraphError::ReadOnly);
6975 }
6976 MutPreview::new(self).check_create_rule(&def)?;
6977 let def_bytes = bincode::serialize(&def).map_err(|e| GraphError::Corrupt {
6978 detail: format!("serialize rule: {e}"),
6979 })?;
6980 self.log_dense(vec![WalRecord::CreateRule { def_bytes }])
6981 }
6982
6983 /// Override this handle's HNSW build-slice size, or `None` to restore
6984 /// [`core_rules::HNSW_BUILD_BATCH`].
6985 ///
6986 /// Exposed for tests that need a small slice without a large corpus; not
6987 /// part of the stable surface.
6988 #[doc(hidden)]
6989 pub fn set_hnsw_build_batch(&mut self, batch: Option<usize>) {
6990 self.engine.set_hnsw_build_batch(batch);
6991 }
6992
6993 /// Rules whose vector index is still being built, in name order.
6994 ///
6995 /// The same list [`GraphDb::stats`] reports per rule in `building`.
6996 /// After a clean open this includes a build a snapshot cut short, so
6997 /// `serve`'s ticker can pump it without a write.
6998 pub fn builds_in_progress(&self) -> Vec<BuildProgress> {
6999 self.engine.builds_in_progress()
7000 }
7001
7002 /// Advance any vector index still building and backfill each rule that
7003 /// finishes. Returns what is still outstanding.
7004 ///
7005 /// A map lookup when nothing is pending, so it is cheap to call on a timer.
7006 /// One write lock and at most [`core_rules::HNSW_BUILD_BATCH`] vector
7007 /// inserts per pending rule per call, so a caller can drive a large build
7008 /// to completion without ever holding the lock for more than a slice.
7009 ///
7010 /// A rule that finishes here is backfilled through the same
7011 /// `WalRecord::RebuildRule` second commit that IVF drift already uses, so
7012 /// its derived edges are produced by [`GraphDb::rebuild_rule`]'s code path
7013 /// and appear all at once.
7014 ///
7015 /// Every ordinary write pumps one slice on its own (see the post-commit
7016 /// hook in `log_then_apply_with`), so this is for quiescent stores and for
7017 /// operators who want the build finished before traffic arrives.
7018 pub fn pump_index_build(&mut self) -> Result<Vec<BuildProgress>> {
7019 Ok(self.pump_index_build_reporting()?.1)
7020 }
7021
7022 /// [`GraphDb::pump_index_build`], also reporting the builds that **this**
7023 /// call finished, so a progress display can say so.
7024 ///
7025 /// A build can be registered and completed inside a single call — that is
7026 /// what a mid-build snapshot looks like on reopen, where the index scan
7027 /// finishes the graph and only the backfill is outstanding — and the
7028 /// outstanding list alone cannot show that anything happened.
7029 pub fn pump_index_build_reporting(
7030 &mut self,
7031 ) -> Result<(Vec<BuildProgress>, Vec<BuildProgress>)> {
7032 // A read-only handle cannot issue the `RebuildRule` a finished build
7033 // needs, so it would advance the index and then silently fail to
7034 // produce the edges. Refusing is the honest answer.
7035 if self.read_only {
7036 return Err(GraphError::ReadOnly);
7037 }
7038 let finished = self.pump_one_slice();
7039 for done in &finished {
7040 // The index is whole but the rule still owns no edges. A failed
7041 // second commit must leave the rule re-pumpable rather than
7042 // silently edge-less, so the error is surfaced here — unlike the
7043 // post-commit hook, this call is not riding someone else's commit.
7044 self.log_then_apply(WalRecord::RebuildRule {
7045 name: done.rule.clone(),
7046 })?;
7047 }
7048 Ok((finished, self.engine.builds_in_progress()))
7049 }
7050
7051 /// Run the deferred candidate-index build, if it is still owed, against the
7052 /// graph as it stands *now* — before the caller applies anything.
7053 ///
7054 /// A no-op bool test once the indexes are populated, which is after the
7055 /// first write of the handle's life, and for a store with no rules at all.
7056 fn populate_indexes_before_write(&mut self) {
7057 if !self.engine.needs_index_population() {
7058 return;
7059 }
7060 // The retained snapshot blobs arrive with the V8 base sections; without
7061 // them the scan would rebuild every graph the snapshot already holds.
7062 self.ensure_v8_base_sections_loaded();
7063 if !self.engine.needs_index_population() {
7064 return;
7065 }
7066 let mut eng = std::mem::take(&mut self.engine);
7067 {
7068 let gm = make_graph_mut(
7069 &self.ids,
7070 Arc::make_mut(&mut self.syms),
7071 &self.labels,
7072 build_props_view(&self.props, &self.base),
7073 Arc::make_mut(&mut self.topo),
7074 &self.base,
7075 Arc::make_mut(&mut self.edge_props),
7076 );
7077 eng.populate_indexes(&gm);
7078 }
7079 self.engine = eng;
7080 }
7081
7082 /// One slice of build work for every pending rule. Returns the rules whose
7083 /// index just became whole, which the caller must `RebuildRule`.
7084 ///
7085 /// Goes through the engine even with nothing pending when the indexes have
7086 /// not been populated yet: that call adopts the persisted graphs and, for
7087 /// an incomplete blob already registered at open, leaves the remainder to
7088 /// this slice rather than inserting it inline.
7089 fn pump_one_slice(&mut self) -> Vec<BuildProgress> {
7090 // The retained snapshot blobs — and the id count an interrupted build
7091 // is recognised against — arrive with the V8 base sections, which a
7092 // clean open reads lazily. Without this a freshly opened handle pumps
7093 // against empty retained state and concludes there is nothing to do,
7094 // which is precisely the store `build-index` exists for.
7095 self.ensure_v8_base_sections_loaded();
7096 let mut eng = std::mem::take(&mut self.engine);
7097 let finished = {
7098 let mut gm = make_graph_mut(
7099 &self.ids,
7100 Arc::make_mut(&mut self.syms),
7101 &self.labels,
7102 build_props_view(&self.props, &self.base),
7103 Arc::make_mut(&mut self.topo),
7104 &self.base,
7105 Arc::make_mut(&mut self.edge_props),
7106 );
7107 eng.pump_index_build(&mut gm)
7108 };
7109 self.engine = eng;
7110 finished
7111 }
7112
7113 /// Register a sliced build a snapshot cut short, from blobs with
7114 /// `complete == false`.
7115 ///
7116 /// Peeks the V8 mmap for incomplete entries without copying complete
7117 /// graphs. V5–V7 already hold the blobs in the engine from restore.
7118 fn register_outstanding_index_builds(&mut self) {
7119 if self.engine.indexes_populated() {
7120 return;
7121 }
7122 let extra = self.collect_incomplete_hnsw_blobs();
7123 let mut eng = std::mem::take(&mut self.engine);
7124 {
7125 let gm = make_graph_mut(
7126 &self.ids,
7127 Arc::make_mut(&mut self.syms),
7128 &self.labels,
7129 build_props_view(&self.props, &self.base),
7130 Arc::make_mut(&mut self.topo),
7131 &self.base,
7132 Arc::make_mut(&mut self.edge_props),
7133 );
7134 eng.register_incomplete_hnsw_builds(&extra, &gm);
7135 }
7136 self.engine = eng;
7137 }
7138
7139 /// Incomplete `(src, dst)` HNSW blobs from the V8 mmap, copied only when
7140 /// `complete` is false. Empty when there is no mmap base (V5–V7 uses the
7141 /// engine's retained map instead).
7142 fn collect_incomplete_hnsw_blobs(&self) -> BTreeMap<String, (Vec<u8>, Vec<u8>)> {
7143 let Some(base) = &self.base else {
7144 return BTreeMap::new();
7145 };
7146 let Ok(archived) = base.hnsw_section() else {
7147 return BTreeMap::new();
7148 };
7149 archived
7150 .rules
7151 .iter()
7152 .filter_map(|e| {
7153 let src = e.src_blob.as_slice();
7154 let dst = e.dst_blob.as_slice();
7155 if core_rules::hnsw::hnsw_blob_complete(src) == Some(false)
7156 || core_rules::hnsw::hnsw_blob_complete(dst) == Some(false)
7157 {
7158 Some((e.name.as_str().to_string(), (src.to_vec(), dst.to_vec())))
7159 } else {
7160 None
7161 }
7162 })
7163 .collect()
7164 }
7165
7166 /// WAL-log rule deletion. Returns RuleNotFound if the rule does not exist.
7167 pub fn delete_rule(&mut self, name: &str) -> Result<()> {
7168 if self.read_only {
7169 return Err(GraphError::ReadOnly);
7170 }
7171 MutPreview::new(self).check_delete_rule(name)?;
7172 self.log_then_apply(WalRecord::DeleteRule { name: name.into() })
7173 }
7174
7175 /// Return a snapshot of all registered rules.
7176 pub fn rules(&self) -> Vec<RuleDef> {
7177 self.engine.rules().cloned().collect()
7178 }
7179
7180 // -----------------------------------------------------------------------
7181 // Rule suggestion API
7182 // -----------------------------------------------------------------------
7183
7184 /// Profile the database and suggest linking rules with previewed edge counts.
7185 ///
7186 /// Uses the default seed ([`core_rules::SUGGEST_DEFAULT_SEED`]) for deterministic
7187 /// sampling. Suggestions are sorted by estimated edge count (descending).
7188 /// **NO auto-accept** — call [`GraphDb::create_rule`] explicitly to apply.
7189 pub fn suggest_rules(&self) -> Vec<core_rules::RuleSuggestion> {
7190 self.suggest_rules_seeded(core_rules::SUGGEST_DEFAULT_SEED)
7191 }
7192
7193 /// Like [`suggest_rules`] but with a caller-supplied RNG seed for
7194 /// reproducibility. Same seed + same data = identical output.
7195 pub fn suggest_rules_seeded(&self, seed: u64) -> Vec<core_rules::RuleSuggestion> {
7196 self.suggest_rules_with_config(&core_rules::suggest::SuggestConfig::default(), seed)
7197 .suggestions
7198 }
7199
7200 /// [`suggest_rules_seeded`] with a fully custom [`SuggestConfig`].
7201 ///
7202 /// Returns a [`core_rules::SuggestReport`] that includes both the candidate list
7203 /// and a `truncated` flag indicating whether the global budget fired before all
7204 /// candidates were evaluated.
7205 pub fn suggest_rules_with_config(
7206 &self,
7207 config: &core_rules::suggest::SuggestConfig,
7208 seed: u64,
7209 ) -> core_rules::SuggestReport {
7210 use std::collections::BTreeMap;
7211
7212 // Collect (node_id, key) pairs per label, skipping tombstoned nodes.
7213 let mut label_nodes: BTreeMap<String, Vec<(u32, String)>> = BTreeMap::new();
7214 for id in 0..self.ids.len() as u32 {
7215 let Some(key) = self.ids.key_of(id) else {
7216 continue;
7217 };
7218 let Some(&sym) = self.labels.get(id as usize) else {
7219 continue;
7220 };
7221 if sym == u32::MAX {
7222 continue; // tombstoned
7223 }
7224 let Some(label) = self.syms.resolve(sym) else {
7225 continue;
7226 };
7227 label_nodes
7228 .entry(label.to_string())
7229 .or_default()
7230 .push((id, key.to_string()));
7231 }
7232
7233 let existing = self.rules();
7234 let pv = build_props_view(&self.props, &self.base);
7235 let all_fields: Vec<String> = pv.field_names();
7236
7237 core_rules::suggest::suggest_rules(
7238 &label_nodes,
7239 &|id, field| pv.get(id, field).map(|vr| vr.into_value()),
7240 &all_fields,
7241 &existing,
7242 config,
7243 seed,
7244 )
7245 }
7246
7247 /// Recompute a rule's derived edges from scratch. WAL-logged so un-trip
7248 /// plus later mutations replay identically (rebuild is a pure function
7249 /// of state).
7250 ///
7251 /// Only exit from the tripped latch: if the full desired set fits the
7252 /// budget, it is applied completely and `tripped` clears; if it still
7253 /// exceeds the budget, provenance is left untouched and `tripped` stays
7254 /// true. Counts as a fire evaluation (see [`RuleStats::fires`]).
7255 /// Unknown rule → `RuleNotFound`, nothing logged.
7256 pub fn rebuild_rule(&mut self, name: &str) -> Result<()> {
7257 if self.read_only {
7258 return Err(GraphError::ReadOnly);
7259 }
7260 if !self.engine.rules().any(|r| r.name == name) {
7261 return Err(GraphError::RuleNotFound { name: name.into() });
7262 }
7263 self.log_then_apply(WalRecord::RebuildRule { name: name.into() })
7264 }
7265
7266 // -----------------------------------------------------------------------
7267 // Materialized view API
7268 // -----------------------------------------------------------------------
7269
7270 /// Register a new materialized property view, backfill its values for all
7271 /// existing nodes, and WAL-log the definition.
7272 ///
7273 /// # Errors
7274 /// - `ReadOnly`: called on an as-of instance.
7275 /// - `RuleInvalid`: name collision, view_prop collision, or invalid def.
7276 pub fn create_view(&mut self, def: ViewDef) -> Result<()> {
7277 if self.read_only {
7278 return Err(GraphError::ReadOnly);
7279 }
7280 // Pre-validate before WAL write.
7281 def.validate()
7282 .map_err(|e| GraphError::RuleInvalid { detail: e })?;
7283 if self.view_store.has_view(&def.name) {
7284 return Err(GraphError::RuleInvalid {
7285 detail: format!("view {:?} already exists", def.name),
7286 });
7287 }
7288 if let Some(existing) = self.view_store.view_for_prop(&def.view_prop) {
7289 return Err(GraphError::RuleInvalid {
7290 detail: format!(
7291 "view_prop {:?} is already used by view {:?}",
7292 def.view_prop, existing
7293 ),
7294 });
7295 }
7296 let def_bytes = bincode::serialize(&def).map_err(|e| GraphError::Corrupt {
7297 detail: format!("serialize view: {e}"),
7298 })?;
7299 // Enable delta accumulation before the view is registered so subsequent
7300 // incremental edge events reach view maintenance from this point onward.
7301 // (The backfill inside create_view reads topo directly; it does not rely
7302 // on pending deltas.)
7303 self.engine.set_emit_deltas(true);
7304 self.log_then_apply(WalRecord::CreateView { def_bytes })
7305 }
7306
7307 /// Remove a named view and delete its values from every node.
7308 ///
7309 /// # Errors
7310 /// - `ReadOnly`: called on an as-of instance.
7311 /// - `RuleNotFound`: view does not exist.
7312 pub fn delete_view(&mut self, name: &str) -> Result<()> {
7313 if self.read_only {
7314 return Err(GraphError::ReadOnly);
7315 }
7316 if !self.view_store.has_view(name) {
7317 return Err(GraphError::RuleNotFound { name: name.into() });
7318 }
7319 let result = self.log_then_apply(WalRecord::DeleteView { name: name.into() });
7320 // After deletion, disable accumulation if no listeners remain.
7321 if !self.needs_emit_deltas() {
7322 self.engine.set_emit_deltas(false);
7323 }
7324 result
7325 }
7326
7327 /// Snapshot of all registered view definitions.
7328 pub fn views(&self) -> Vec<ViewDef> {
7329 self.view_store.views().cloned().collect()
7330 }
7331
7332 // -----------------------------------------------------------------------
7333 // Full-text-lite API
7334 // -----------------------------------------------------------------------
7335
7336 /// Enable full-text indexing for all nodes of `label` on property `field`.
7337 ///
7338 /// After this call, every subsequent write to `(label, field)` is reflected
7339 /// in the index incrementally. Existing nodes are backfilled immediately.
7340 /// The declaration is persisted as a WAL record; the index itself is rebuilt
7341 /// from scratch on re-open (no snapshot format changes).
7342 ///
7343 /// # Errors
7344 /// - [`GraphError::ReadOnly`]: called on an as-of instance.
7345 /// - [`GraphError::RuleInvalid`]: `(label, field)` is already indexed.
7346 pub fn enable_fulltext(&mut self, label: &str, field: &str) -> Result<()> {
7347 if self.read_only {
7348 return Err(GraphError::ReadOnly);
7349 }
7350 if self.fulltext.is_enabled(label, field) {
7351 return Err(GraphError::RuleInvalid {
7352 detail: format!("full-text index for ({label:?}, {field:?}) already enabled"),
7353 });
7354 }
7355 self.log_then_apply(WalRecord::EnableFulltext {
7356 label: label.into(),
7357 field: field.into(),
7358 })
7359 }
7360
7361 /// Disable full-text indexing for `(label, field)` and drop its postings.
7362 ///
7363 /// # Errors
7364 /// - [`GraphError::ReadOnly`]: called on an as-of instance.
7365 /// - [`GraphError::RuleNotFound`]: `(label, field)` is not currently indexed.
7366 pub fn disable_fulltext(&mut self, label: &str, field: &str) -> Result<()> {
7367 if self.read_only {
7368 return Err(GraphError::ReadOnly);
7369 }
7370 if !self.fulltext.is_enabled(label, field) {
7371 return Err(GraphError::RuleNotFound {
7372 name: format!("fulltext({label},{field})"),
7373 });
7374 }
7375 self.log_then_apply(WalRecord::DisableFulltext {
7376 label: label.into(),
7377 field: field.into(),
7378 })
7379 }
7380
7381 /// Whether `(label, field)` is currently indexed for full-text search.
7382 pub fn is_fulltext_enabled(&self, label: &str, field: &str) -> bool {
7383 self.fulltext.is_enabled(label, field)
7384 }
7385
7386 /// Every `(label, field)` pair with a live full-text index, sorted.
7387 ///
7388 /// Note that [`GraphDb::search`] is keyed by field alone — a pair only
7389 /// declares which nodes are *indexed*, so callers that want to search
7390 /// everything indexed should query each distinct field once.
7391 pub fn fulltext_pairs(&self) -> Vec<(String, String)> {
7392 let mut v: Vec<(String, String)> = self.fulltext.enabled_pairs().cloned().collect();
7393 v.sort();
7394 v
7395 }
7396
7397 /// Enable an equality index for all nodes of `label` on scalar property
7398 /// `field`. Subsequent `WHERE n.field = value` lookups become O(matches)
7399 /// instead of an O(N_label) scan. Existing nodes are backfilled; the
7400 /// declaration persists via WAL and the postings rebuild on re-open.
7401 ///
7402 /// # Errors
7403 /// - [`GraphError::ReadOnly`]: called on an as-of instance.
7404 /// - [`GraphError::RuleInvalid`]: `(label, field)` is already indexed.
7405 pub fn enable_index(&mut self, label: &str, field: &str) -> Result<()> {
7406 if self.read_only {
7407 return Err(GraphError::ReadOnly);
7408 }
7409 if self.prop_index.is_enabled(label, field) {
7410 return Err(GraphError::RuleInvalid {
7411 detail: format!("property index for ({label:?}, {field:?}) already enabled"),
7412 });
7413 }
7414 self.log_then_apply(WalRecord::EnableIndex {
7415 label: label.into(),
7416 field: field.into(),
7417 })
7418 }
7419
7420 /// Disable the equality index for `(label, field)` and drop its postings.
7421 ///
7422 /// # Errors
7423 /// - [`GraphError::ReadOnly`]: called on an as-of instance.
7424 /// - [`GraphError::RuleNotFound`]: `(label, field)` is not currently indexed.
7425 pub fn disable_index(&mut self, label: &str, field: &str) -> Result<()> {
7426 if self.read_only {
7427 return Err(GraphError::ReadOnly);
7428 }
7429 if !self.prop_index.is_enabled(label, field) {
7430 return Err(GraphError::RuleNotFound {
7431 name: format!("index({label},{field})"),
7432 });
7433 }
7434 self.log_then_apply(WalRecord::DisableIndex {
7435 label: label.into(),
7436 field: field.into(),
7437 })
7438 }
7439
7440 /// Whether `(label, field)` currently has an equality index.
7441 pub fn is_index_enabled(&self, label: &str, field: &str) -> bool {
7442 self.prop_index.is_enabled(label, field)
7443 }
7444
7445 /// Every `(label, field)` pair with a live equality index, sorted.
7446 ///
7447 /// The enumerator [`is_index_enabled`](Self::is_index_enabled) never had:
7448 /// the MCP `schema` tool lists a store's indexes rather than probing them
7449 /// one guess at a time.
7450 pub fn index_pairs(&self) -> Vec<(String, String)> {
7451 self.prop_index.enabled_pairs().cloned().collect()
7452 }
7453
7454 /// The dense id of a live key, `None` for an unknown or deleted one.
7455 ///
7456 /// Ids are allocated in insertion order and never reused, so the lowest
7457 /// live id among a set of keys is the oldest node — which is how identity
7458 /// resolution picks a canonical node without a timestamp field
7459 /// (`memory::identity`). Crate-private: an id is an engine detail, not
7460 /// something a caller should hold.
7461 pub(crate) fn dense_id(&self, key: &str) -> Option<u32> {
7462 self.ids.get(key)
7463 }
7464
7465 /// Start recording insert-count multiplicity on this store (§5.13).
7466 ///
7467 /// Adjacency stays a set and nothing about an existing read changes: a
7468 /// duplicate [`insert_edge`](Self::insert_edge) still returns `Ok(false)`
7469 /// and still leaves [`degree`](Self::degree) alone. What it gains is that
7470 /// the duplicate is *counted*, as the reserved edge property
7471 /// [`EDGE_COUNT_PROP`], readable through
7472 /// [`degree_multiplicity`](Self::degree_multiplicity).
7473 ///
7474 /// # This is a one-way step, and that is why it is a call
7475 ///
7476 /// The count is durable, so it is written to the WAL — as discriminant 23,
7477 /// which no release before v0.6.10 knows. A reader meeting an unknown WAL
7478 /// discriminant cannot know what the record would have changed, so it
7479 /// cannot degrade the way an unreadable index blob can. **After this call
7480 /// the store can no longer be read by an older binary, and there is no call
7481 /// that undoes it.** Gating the record behind this method is what keeps
7482 /// that step a decision an operator makes when they want the feature,
7483 /// rather than one everybody takes by upgrading.
7484 ///
7485 /// # It fails loudly, and that costs a snapshot
7486 ///
7487 /// An older binary does not refuse discriminant 23 — it truncates the WAL
7488 /// at it and, with `repair_wal`, persists the truncation. So this call also
7489 /// writes a **V10 snapshot**, a version no earlier release knows, and it
7490 /// writes it *first*: the snapshot is read before the WAL, so an older
7491 /// binary stops at `snapshot: unsupported version 10` with the WAL
7492 /// untouched. Taking the snapshot before appending the record is what makes
7493 /// the guard unconditional — the store is never, at any interruption point,
7494 /// carrying the record without the stamp that announces it.
7495 ///
7496 /// The snapshot keeps the WAL (`keep_wal: true`): opting in is not a
7497 /// compaction, and history reachable by [`open_at`](Self::open_at) stays
7498 /// reachable. On a large store the call therefore costs one full snapshot
7499 /// write.
7500 ///
7501 /// # What it costs a store that archives
7502 ///
7503 /// Writing `snapshot.bin` is also how the archive path decides whether the
7504 /// store may have a *genesis chain* — whether `open_at` can replay
7505 /// archive-resident commits from empty state. The rule is conservative: a
7506 /// snapshot that was already on disk might have been a truncating one, and
7507 /// once the handle that took it is gone this binary cannot tell. A
7508 /// `keep_wal` snapshot taken by **this** handle is the case where it can, so
7509 /// opting in and then archiving **in the same session** keeps the chain.
7510 ///
7511 /// Opting in, closing the store, and archiving in a *later* session does
7512 /// not — but that is the answer any store with a prior snapshot gets, not
7513 /// something this call causes. A store that wants the chain should take its
7514 /// first archive in the session that opted in.
7515 ///
7516 /// Calling it on a store that has already opted in writes nothing and
7517 /// returns `Ok(())`: an operator should not have to ask first.
7518 ///
7519 /// # This call is not atomic, and an `Err` does not undo it
7520 ///
7521 /// There is no rollback here, and there never was one. An `Err` means this
7522 /// handle stopped believing the store is opted in — `self.multiplicity` is
7523 /// reset, so this handle reports `false` from then on — and nothing more. It
7524 /// says nothing about what reached disk. Two reachable failures leave the
7525 /// opt-in standing:
7526 ///
7527 /// * **The declaration landed and only its fsync failed.** `log_then_apply`
7528 /// appends, then syncs; a failed barrier leaves `MULTIPLICITY_ENABLED`
7529 /// already in `wal.bin`. The next open replays it and the store is opted
7530 /// in. No archive is involved — this one predates the recovery below.
7531 /// * **The declaration never landed, but the V10 snapshot did, on a store
7532 /// that already had an archive.** The open-time recovery in
7533 /// `load_from_disk` reads V10-beside-an-archive as an interrupted archive
7534 /// sequence and opts the store in.
7535 ///
7536 /// So a failed call may leave the opt-in on disk immediately (the first
7537 /// case) or conjure it at the next open (the second), and nothing puts the
7538 /// store back out. Treat `Err` as "the outcome is unknown", not as "nothing
7539 /// happened".
7540 ///
7541 /// **This is safe, and the ordering is the reason.** The V10 stamp is
7542 /// written *before* the declaration, so every one of these intermediate
7543 /// states is one an older binary refuses by name rather than truncates at.
7544 /// The failure direction costs a refusal, never a commit. That ordering is
7545 /// the property worth protecting, not the atomicity this call never had.
7546 ///
7547 /// **To know where the store stands, ask the store.** Reopen it and call
7548 /// [`is_multiplicity_enabled`](Self::is_multiplicity_enabled); that is the
7549 /// only answer that accounts for what reached disk.
7550 ///
7551 /// The one case that really does leave the store opted out is a failure with
7552 /// no archive present and no record written: a stray V10 snapshot remains,
7553 /// costing an older reader a refusal it did not strictly need, and *that*
7554 /// store's next snapshot rewrites at V9.
7555 ///
7556 /// # Errors
7557 /// - [`GraphError::ReadOnly`]: called on an as-of instance.
7558 /// - Anything [`snapshot_with`](Self::snapshot_with) can return, and
7559 /// anything the WAL append or its fsync can return. See the atomicity
7560 /// section above for what the store is left holding.
7561 pub fn enable_multiplicity(&mut self) -> Result<()> {
7562 if self.read_only {
7563 return Err(GraphError::ReadOnly);
7564 }
7565 if self.multiplicity {
7566 return Ok(());
7567 }
7568 // The snapshot goes first, and the order is the guard.
7569 //
7570 // An older reader refuses a V10 snapshot by name and stops; it does not
7571 // refuse discriminant 23, it truncates the WAL at it. So the store must
7572 // never hold the record without the snapshot that announces it — not
7573 // even for the width of one fsync. Writing the snapshot before the
7574 // record makes the only reachable intermediate state "V10 snapshot, no
7575 // record", which is merely conservative: this binary reads it as a
7576 // store that has not opted in, and an older one refuses it.
7577 //
7578 // `keep_wal: true` because opting in is not a compaction: an operator
7579 // asking for multiplicity has not asked to lose the history `open_at`
7580 // can reach.
7581 self.multiplicity = true;
7582 let forced = self
7583 .snapshot_with(SnapshotOptions {
7584 keep_wal: true,
7585 ..SnapshotOptions::default()
7586 })
7587 .and_then(|()| self.log_then_apply(core_storage::wal::MULTIPLICITY_ENABLED));
7588 if forced.is_err() {
7589 // This handle stops believing it is opted in. That is all this line
7590 // does — it is not a rollback, and cannot be one: the declaration
7591 // may already be in `wal.bin` (the append succeeded and only the
7592 // fsync failed), and even when it is not, the V10 snapshot beside an
7593 // existing archive is enough for the open-time recovery to opt the
7594 // store in. See the "not atomic" section on this method.
7595 //
7596 // It fails in the safe direction either way: the V10 stamp reached
7597 // disk before anything a v0.6.9 reader would truncate at, so the
7598 // worst an interruption costs that reader is a refusal by name.
7599 self.multiplicity = false;
7600 }
7601 forced
7602 }
7603
7604 /// Whether this store records insert-count multiplicity.
7605 ///
7606 /// `false` on every store that has not called
7607 /// [`enable_multiplicity`](Self::enable_multiplicity) — which is every
7608 /// store that did not ask for it, including one upgraded from an earlier
7609 /// release.
7610 pub fn is_multiplicity_enabled(&self) -> bool {
7611 self.multiplicity
7612 }
7613
7614 /// How many times `(etype, src, dst)` has been inserted: the reserved
7615 /// `count` edge property, or 1 when it is absent.
7616 ///
7617 /// Answers 1 for a pair on a store that never opted in, which is the truth
7618 /// available there — the pair was inserted at least once, and the store
7619 /// kept no record of any second insert.
7620 fn edge_insert_count(&self, etype: u32, src: u32, dst: u32) -> u64 {
7621 match self.edge_props_view().get(etype, src, dst, EDGE_COUNT_PROP) {
7622 Some(Value::Int(n)) if n > 0 => n as u64,
7623 _ => 1,
7624 }
7625 }
7626
7627 /// The `SetEdgeCount` record a duplicate insert of `(edge_type, src_key,
7628 /// dst_key)` should log, or `None` when nothing should be written.
7629 ///
7630 /// `None` when the store has not opted in, so **no discriminant-23 record
7631 /// is written at all** — the gate the whole feature rests on.
7632 ///
7633 /// The other two `None`s are unreachable from the one caller. This is the
7634 /// single-mutation path, where `prepare_insert_edge` has already refused a
7635 /// missing endpoint and an existing pair's edge type is necessarily
7636 /// interned. A batch is the case where a pair's endpoints and type can all
7637 /// be created by the same frame, and it does not come through here: it
7638 /// queues a [`PlannedRec::DuplicateCount`] and names the count in the dense
7639 /// rewrite, which is the only pass that knows the frame's own ids.
7640 fn edge_count_record(
7641 &self,
7642 edge_type: &str,
7643 src_key: &str,
7644 dst_key: &str,
7645 ) -> Option<WalRecord> {
7646 if !self.multiplicity {
7647 return None;
7648 }
7649 let etype = self.syms.get(edge_type)?;
7650 let src = self.ids.get(src_key)?;
7651 let dst = self.ids.get(dst_key)?;
7652 Some(WalRecord::SetEdgeCount {
7653 etype,
7654 src,
7655 dst,
7656 count: self.edge_insert_count(etype, src, dst).saturating_add(1),
7657 })
7658 }
7659
7660 /// Search a full-text-indexed field.
7661 ///
7662 /// Returns `(node_key, match_count)` pairs sorted by match_count descending,
7663 /// ties broken by key (lexicographic). Tombstoned nodes are excluded.
7664 ///
7665 /// **Query syntax:**
7666 /// - Space-separated terms are AND'd: `"foo bar"` requires both.
7667 /// - `OR` between terms forms disjunction: `"foo OR bar"` matches either.
7668 /// - Trailing `*` on a term is a prefix match: `"rust*"` matches `rustlang`, `rusty`.
7669 /// - `AND` keyword is accepted explicitly and is the default.
7670 /// - Tokenization is unicode-alphanumeric (same as index time); case-insensitive.
7671 ///
7672 /// **Unindexed field:** returns `Ok(vec![])` if `field` is not indexed.
7673 /// Pin: this is the documented, tested, stable behavior for v1.
7674 ///
7675 /// **Memory / performance:** O(postings) lookup; no scan. The index is
7676 /// in-memory and proportional to total indexed text across all enabled fields.
7677 ///
7678 /// **v2 grammar:** supports `"phrase"`, `-negation`, `prefix*`, `OR`, `AND`.
7679 /// Results are BM25-scored (k1=1.2, b=0.75) and sorted by score descending,
7680 /// key ascending for deterministic tiebreaking.
7681 pub fn search(&self, field: &str, query: &str) -> Vec<(String, f64)> {
7682 // Resolve node_ids to keys (excluding tombstones) then re-sort by
7683 // (score DESC, key ASC) to give a deterministic, key-lexicographic
7684 // tiebreak. FulltextIndex::search sorts by (score DESC, node_id ASC)
7685 // which diverges from key order when nodes were not inserted in key-lex order.
7686 let mut results: Vec<(String, f64)> = self
7687 .fulltext
7688 .search(field, query, 0)
7689 .into_iter()
7690 .filter_map(|(id, score)| self.ids.key_of(id).map(|key| (key.to_string(), score)))
7691 .collect();
7692 results.sort_by(|a, b| {
7693 b.1.partial_cmp(&a.1)
7694 .unwrap_or(std::cmp::Ordering::Equal)
7695 .then(a.0.cmp(&b.0))
7696 });
7697 results
7698 }
7699
7700 /// [`search`](Self::search), stopping at the `k` best hits.
7701 ///
7702 /// Same ranking and the same deterministic tiebreak, but the index drops
7703 /// everything past `k` before any key is resolved, so a caller that wants
7704 /// the top few out of a field that matched thousands does not pay to
7705 /// materialise and re-sort the tail. `k == 0` means no limit, exactly as
7706 /// [`search`](Self::search) behaves.
7707 ///
7708 /// The BM25 scoring itself is not bounded by `k` — every candidate is
7709 /// scored either way — so this trims the resolve and the sort, not the
7710 /// search.
7711 pub fn search_top(&self, field: &str, query: &str, k: usize) -> Vec<(String, f64)> {
7712 // A tombstoned id resolves to nothing, so asking the index for exactly
7713 // `k` could return fewer. Over-fetching a little and truncating after
7714 // the filter keeps the count right without unbounding the call.
7715 let want = if k == 0 { 0 } else { k.saturating_mul(2) };
7716 let mut results: Vec<(String, f64)> = self
7717 .fulltext
7718 .search(field, query, want)
7719 .into_iter()
7720 .filter_map(|(id, score)| self.ids.key_of(id).map(|key| (key.to_string(), score)))
7721 .collect();
7722 results.sort_by(|a, b| {
7723 b.1.partial_cmp(&a.1)
7724 .unwrap_or(std::cmp::Ordering::Equal)
7725 .then(a.0.cmp(&b.0))
7726 });
7727 if k > 0 {
7728 results.truncate(k);
7729 }
7730 results
7731 }
7732
7733 /// Hybrid search: Reciprocal Rank Fusion (RRF) over fulltext + vector results.
7734 ///
7735 /// Takes up to `4*k` fulltext hits for `(text_field, query_text)` and up to
7736 /// `4*k` vector hits for `(vector_field, query_vec, min=0.0)`, then fuses
7737 /// them with RRF using a fixed constant of 60.
7738 ///
7739 /// ```text
7740 /// score(d) = Σ 1 / (60 + rank_i(d)) (rank 1-based per list)
7741 /// ```
7742 ///
7743 /// Returns the top `k` nodes by fused score, ties broken by node key
7744 /// ascending (deterministic).
7745 ///
7746 /// # Vector leg fallback
7747 ///
7748 /// When `query_vec` is empty the vector leg is skipped entirely and
7749 /// results are ranked by the text list alone through the same RRF path
7750 /// (each text result scores `1/(60 + rank)` from that single list).
7751 ///
7752 /// When `label` is `None`, the vector leg **always** returns empty results.
7753 /// Internally `label` is mapped to `""`, which does not match any rule-created
7754 /// HNSW index (all such indexes are keyed to a specific non-empty label), and
7755 /// the brute-force fallback finds no nodes with an empty label. The fused
7756 /// ranking is therefore text-only in this case.
7757 pub fn search_hybrid(
7758 &self,
7759 text_field: &str,
7760 query_text: &str,
7761 vector_field: &str,
7762 query_vec: &[f64],
7763 label: Option<&str>,
7764 k: usize,
7765 ) -> Vec<(String, f64)> {
7766 self.search_hybrid_inner(
7767 text_field,
7768 query_text,
7769 vector_field,
7770 query_vec,
7771 label,
7772 k,
7773 None,
7774 )
7775 }
7776
7777 /// [`search_hybrid`](Self::search_hybrid) with **each leg** filtered to the
7778 /// mask before the fusion.
7779 ///
7780 /// Filtering the fused list afterwards would quietly return fewer than `k`.
7781 /// Each leg over-fetches `4*k` candidates, so when the visible nodes rank
7782 /// below `4*k` hidden ones neither leg carries them into the fusion at all
7783 /// and the post-filter has nothing left to keep. Filtering first spends the
7784 /// `4*k` on **visible** hits, so a scoped call is as long as the corpus it
7785 /// can see allows.
7786 ///
7787 /// The ranks that enter RRF are therefore the ranks of the visible corpus,
7788 /// not the visible entries of the store-wide ranking. The constant stays 60
7789 /// and the tiebreak stays key-ascending.
7790 #[allow(clippy::too_many_arguments)]
7791 pub fn search_hybrid_scoped(
7792 &self,
7793 text_field: &str,
7794 query_text: &str,
7795 vector_field: &str,
7796 query_vec: &[f64],
7797 label: Option<&str>,
7798 k: usize,
7799 mask: &crate::mask::NodeMask,
7800 ) -> Vec<(String, f64)> {
7801 self.search_hybrid_inner(
7802 text_field,
7803 query_text,
7804 vector_field,
7805 query_vec,
7806 label,
7807 k,
7808 Some(mask),
7809 )
7810 }
7811
7812 /// The body shared by [`search_hybrid`](Self::search_hybrid) and
7813 /// [`search_hybrid_scoped`](Self::search_hybrid_scoped). `mask = None` is
7814 /// the unscoped contract unchanged: the filter below is then a no-op and
7815 /// the vector leg is the same unmasked call it has always been.
7816 #[allow(clippy::too_many_arguments)]
7817 fn search_hybrid_inner(
7818 &self,
7819 text_field: &str,
7820 query_text: &str,
7821 vector_field: &str,
7822 query_vec: &[f64],
7823 label: Option<&str>,
7824 k: usize,
7825 mask: Option<&crate::mask::NodeMask>,
7826 ) -> Vec<(String, f64)> {
7827 use std::collections::HashMap;
7828
7829 const RRF_K: f64 = 60.0;
7830 let pool = 4 * k;
7831
7832 // Accumulate per-node RRF scores.
7833 let mut scores: HashMap<String, f64> = HashMap::new();
7834
7835 // Text leg. The mask bites on the candidates, before `take(pool)`, so
7836 // the over-fetch is a budget of visible hits rather than one a hidden
7837 // prefix can exhaust.
7838 let text_hits = self.search(text_field, query_text);
7839 let visible_text = text_hits
7840 .into_iter()
7841 .filter(|(key, _count)| mask.is_none_or(|m| m.contains_node(self, key)));
7842 for (rank0, (key, _count)) in visible_text.take(pool).enumerate() {
7843 let rank = (rank0 + 1) as f64;
7844 *scores.entry(key).or_insert(0.0) += 1.0 / (RRF_K + rank);
7845 }
7846
7847 // Vector leg (skipped when query_vec is empty). The masked variant
7848 // applies the mask before its own k-truncation, for the same reason.
7849 if !query_vec.is_empty() {
7850 // `ExactnessCaller::Hybrid`: the leg is the same one
7851 // `find_similar_vector_masked` runs, but the advice its warning
7852 // gives has to fit *this* signature, which has no `exact`.
7853 let vec_hits = self
7854 .find_similar_vector_as(
7855 vector_field,
7856 label,
7857 query_vec,
7858 pool,
7859 0.0,
7860 mask,
7861 None,
7862 false,
7863 ExactnessCaller::Hybrid,
7864 )
7865 .expect("find_similar_vector_as is infallible without where_");
7866 for (rank0, (key, _sim)) in vec_hits.into_iter().enumerate() {
7867 let rank = (rank0 + 1) as f64;
7868 *scores.entry(key).or_insert(0.0) += 1.0 / (RRF_K + rank);
7869 }
7870 }
7871
7872 // Sort: score DESC, then key ASC for deterministic tie-breaking.
7873 let mut ranked: Vec<(String, f64)> = scores.into_iter().collect();
7874 ranked.sort_by(|a, b| {
7875 b.1.partial_cmp(&a.1)
7876 .unwrap_or(std::cmp::Ordering::Equal)
7877 .then(a.0.cmp(&b.0))
7878 });
7879 ranked.truncate(k);
7880 ranked
7881 }
7882
7883 /// For DST/testing: scratch BM25 search over live nodes without the index.
7884 /// Walks every live node, re-stems field tokens, computes corpus stats, and
7885 /// returns BM25-ranked results.
7886 ///
7887 /// The oracle: the ordered key list of `search(field, q)` must equal that of
7888 /// `scratch_search(field, q)` at every quiescent state.
7889 #[doc(hidden)]
7890 pub fn scratch_search(&self, field: &str, query: &str) -> Vec<(String, f64)> {
7891 use core_storage::fulltext::{parse_query, value_tokens_stemmed_with_positions};
7892 use std::collections::BTreeMap;
7893
7894 let groups = parse_query(query);
7895 if groups.is_empty() {
7896 return vec![];
7897 }
7898
7899 // --- Pass 1: collect all live indexed nodes with stemmed token data ---
7900 struct NodeData {
7901 key: String,
7902 /// stemmed_token → positions (sorted)
7903 tokens: BTreeMap<String, Vec<u32>>,
7904 dl: u32,
7905 }
7906
7907 let mut nodes: Vec<NodeData> = Vec::new();
7908 for id in 0..self.ids.len() as u32 {
7909 let Some(key) = self.ids.key_of(id) else {
7910 continue;
7911 };
7912 let Some(&sym) = self.labels.get(id as usize) else {
7913 continue;
7914 };
7915 if sym == u32::MAX {
7916 continue;
7917 }
7918 let label = match self.syms.resolve(sym) {
7919 Some(l) => l,
7920 None => continue,
7921 };
7922 if !self.fulltext.is_enabled(label, field) {
7923 continue;
7924 }
7925 let Some(value) = self.props_view().get(id, field).map(|vr| vr.into_value()) else {
7926 continue;
7927 };
7928 // Use value_tokens_stemmed_with_positions so list elements are
7929 // separated by POSITION_GAP — identical to the index path, which
7930 // prevents phrase queries from matching across element boundaries.
7931 let stemmed_with_pos = match &value {
7932 Value::Str(_) | Value::List(_) => value_tokens_stemmed_with_positions(&value),
7933 _ => continue,
7934 };
7935 let dl = stemmed_with_pos.len() as u32;
7936 let mut tok_map: BTreeMap<String, Vec<u32>> = BTreeMap::new();
7937 for (tok, pos) in stemmed_with_pos {
7938 tok_map.entry(tok).or_default().push(pos);
7939 }
7940 nodes.push(NodeData {
7941 key: key.to_string(),
7942 tokens: tok_map,
7943 dl,
7944 });
7945 }
7946
7947 if nodes.is_empty() {
7948 return vec![];
7949 }
7950
7951 // --- BM25 corpus stats ---
7952 let n = nodes.len() as f64;
7953 let avg_dl: f64 = nodes.iter().map(|nd| nd.dl as f64).sum::<f64>() / n;
7954 // df per stemmed token across all live indexed nodes.
7955 let mut df_map: BTreeMap<&str, f64> = BTreeMap::new();
7956 for nd in &nodes {
7957 for tok in nd.tokens.keys() {
7958 *df_map.entry(tok.as_str()).or_insert(0.0) += 1.0;
7959 }
7960 }
7961
7962 const K1: f64 = 1.2;
7963 const B: f64 = 0.75;
7964
7965 // --- Pass 2: score each node against each OR-group ---
7966 let mut results: Vec<(String, f64)> = Vec::new();
7967 for nd in &nodes {
7968 let dl = nd.dl as f64;
7969 let mut total_score = 0.0f64;
7970
7971 'group: for group in &groups {
7972 let mut group_score = 0.0f64;
7973
7974 for term in group {
7975 if term.negated {
7976 // Negated: if doc has this stemmed token → group fails.
7977 let present = if term.prefix {
7978 nd.tokens.keys().any(|t| t.starts_with(term.token.as_str()))
7979 } else {
7980 nd.tokens.contains_key(term.token.as_str())
7981 };
7982 if present {
7983 continue 'group;
7984 }
7985 continue;
7986 }
7987 if term.prefix {
7988 // Prefix: sum BM25 for all matching stemmed tokens.
7989 let mut prefix_matched = false;
7990 for (tok, positions) in &nd.tokens {
7991 if tok.starts_with(term.token.as_str()) {
7992 let tf = positions.len() as f64;
7993 let df = df_map.get(tok.as_str()).copied().unwrap_or(1.0);
7994 let idf = ((n - df + 0.5) / (df + 0.5) + 1.0).ln();
7995 let tf_norm =
7996 tf * (K1 + 1.0) / (tf + K1 * (1.0 - B + B * dl / avg_dl));
7997 group_score += idf * tf_norm;
7998 prefix_matched = true;
7999 }
8000 }
8001 if !prefix_matched {
8002 continue 'group;
8003 }
8004 } else {
8005 // term.token is already stemmed by parse_query; use directly.
8006 match nd.tokens.get(term.token.as_str()) {
8007 None => continue 'group,
8008 Some(positions) => {
8009 let tf = positions.len() as f64;
8010 let df = df_map.get(term.token.as_str()).copied().unwrap_or(1.0);
8011 let idf = ((n - df + 0.5) / (df + 0.5) + 1.0).ln();
8012 let tf_norm =
8013 tf * (K1 + 1.0) / (tf + K1 * (1.0 - B + B * dl / avg_dl));
8014 group_score += idf * tf_norm;
8015 }
8016 }
8017 }
8018 }
8019
8020 if group_score > 0.0 {
8021 total_score += group_score;
8022 }
8023 }
8024
8025 if total_score > 0.0 {
8026 results.push((nd.key.clone(), total_score));
8027 }
8028 }
8029
8030 results.sort_by(|a, b| {
8031 b.1.partial_cmp(&a.1)
8032 .unwrap_or(std::cmp::Ordering::Equal)
8033 .then(a.0.cmp(&b.0))
8034 });
8035 results
8036 }
8037
8038 /// Return the current view-maintained value of `view_prop` for node `key`.
8039 /// Equivalent to `get_prop` but documents that it reads a view-managed column.
8040 pub fn get_view_prop(&self, key: &str, view_prop: &str) -> Option<Value> {
8041 let id = self.ids.get(key)?;
8042 self.props_view()
8043 .get(id, view_prop)
8044 .map(|vr| vr.into_value())
8045 }
8046
8047 /// For testing / DST oracle: scratch recompute of a view value for one node.
8048 ///
8049 /// Returns `None` if the node does not exist, the view does not exist, or
8050 /// the view has no result for the node (e.g. Avg with no qualifying neighbors).
8051 #[doc(hidden)]
8052 pub fn scratch_view_value(&self, key: &str, view_name: &str) -> Option<Value> {
8053 let node = self.ids.get(key)?;
8054 let def = self.view_store.views().find(|v| v.name == view_name)?;
8055 // Use TopologyView so that NeighborAgg sees base + overlay edges
8056 // without materialising a temporary Topology (I1).
8057 let topo_view = self.topo_view();
8058 core_rules::views::compute_view_value(
8059 def,
8060 node,
8061 self.props_view(),
8062 &topo_view,
8063 &self.ids,
8064 &self.syms,
8065 &self.labels,
8066 )
8067 }
8068
8069 // -----------------------------------------------------------------------
8070 // Graph algorithm API
8071 // -----------------------------------------------------------------------
8072
8073 /// Run PageRank over the unified topology (manual + derived edges).
8074 ///
8075 /// Returns a [`PageRankReport`] with scores sorted descending (ties: key
8076 /// ascending). Set `config.edge_type` to restrict to one edge type.
8077 /// `config.converged` is `true` only when the power iteration converged
8078 /// within `config.max_iters` and within any time budget.
8079 pub fn pagerank(&self, config: &crate::algo::PageRankConfig) -> crate::algo::PageRankReport {
8080 let topo = build_topo_view(&self.topo, &self.base);
8081 let edge_props = self.edge_props_view();
8082 crate::algo::pagerank(
8083 &topo,
8084 &self.ids,
8085 &self.syms,
8086 &self.labels,
8087 &edge_props,
8088 config,
8089 )
8090 }
8091
8092 /// Weakly-connected components over the unified topology (treated as
8093 /// undirected regardless of how edges were inserted).
8094 ///
8095 /// Component IDs are the key of the smallest member in the component
8096 /// (deterministic). Result sorted by (component_id, key).
8097 pub fn connected_components(&self, config: &crate::algo::WccConfig) -> crate::algo::WccReport {
8098 let topo = build_topo_view(&self.topo, &self.base);
8099 let edge_props = self.edge_props_view();
8100 crate::algo::wcc(
8101 &topo,
8102 &self.ids,
8103 &self.syms,
8104 &self.labels,
8105 &edge_props,
8106 config,
8107 )
8108 }
8109
8110 /// Degree centrality for every live node.
8111 ///
8112 /// `direction`: `AlgoDir::Out` = out-degree, `AlgoDir::In` = in-degree,
8113 /// `AlgoDir::Both` = out + in (total directed degree).
8114 ///
8115 /// For one-shot ranking use this; for a live property updated on every
8116 /// write, create a Degree materialized view instead (see `docs/site/algorithms.md`).
8117 pub fn degree_centrality(
8118 &self,
8119 config: &crate::algo::DegreeConfig,
8120 ) -> crate::algo::DegreeReport {
8121 let topo = build_topo_view(&self.topo, &self.base);
8122 let edge_props = self.edge_props_view();
8123 crate::algo::degree_centrality(
8124 &topo,
8125 &self.ids,
8126 &self.syms,
8127 &self.labels,
8128 &edge_props,
8129 config,
8130 )
8131 }
8132
8133 /// Louvain community detection over the unified topology (undirected).
8134 ///
8135 /// See [`crate::algo::LouvainConfig`] for edge-type/weight/label
8136 /// restriction and [`crate::algo::CommunityReport`] for the shape of the
8137 /// result (communities sorted size-desc, then smallest member key asc).
8138 pub fn communities(&self, config: &crate::algo::LouvainConfig) -> crate::algo::CommunityReport {
8139 let topo = build_topo_view(&self.topo, &self.base);
8140 let edge_props = self.edge_props_view();
8141 crate::algo::louvain(
8142 &topo,
8143 &self.ids,
8144 &self.syms,
8145 &self.labels,
8146 &edge_props,
8147 config,
8148 )
8149 }
8150
8151 /// Write a vector of `(node_key, score)` pairs as `prop_name` on each node,
8152 /// atomically via a single write-batch (one WAL frame, one fsync).
8153 ///
8154 /// # Errors
8155 /// - [`GraphError::ReadOnly`]: called on an as-of instance.
8156 /// - [`GraphError::RuleInvalid`]: `prop_name` is managed by an existing view
8157 /// (collision check mirrors `create_view`).
8158 /// - [`GraphError::KeyNotFound`]: a key in `scores` does not exist as a live node.
8159 pub fn write_scores(&mut self, prop_name: &str, scores: &[(String, f64)]) -> Result<()> {
8160 if self.read_only {
8161 return Err(GraphError::ReadOnly);
8162 }
8163 // Collision check: refuse if prop_name is view-managed.
8164 if let Some(view_name) = self.view_store.view_for_prop(prop_name) {
8165 return Err(GraphError::RuleInvalid {
8166 detail: format!(
8167 "prop {:?} is managed by view {:?} and cannot be written as scores",
8168 prop_name, view_name
8169 ),
8170 });
8171 }
8172 // Refuse if prop_name is a view name itself (confusing namespace collision).
8173 if self.view_store.has_view(prop_name) {
8174 return Err(GraphError::RuleInvalid {
8175 detail: format!(
8176 "prop_name {:?} collides with an existing view name",
8177 prop_name
8178 ),
8179 });
8180 }
8181 // Write all scores in a single crash-atomic batch.
8182 self.write_batch(|b| {
8183 for (key, score) in scores {
8184 b.set_prop(key, prop_name, Value::Float(*score));
8185 }
8186 })?;
8187 Ok(())
8188 }
8189
8190 /// Return the value of `field` for the node with key `key`, or `None` if
8191 /// the node or field is absent. Reads through the overlay-over-base
8192 /// `ColumnsView`, materialising base values on demand (zero heap cost for
8193 /// overlay hits; one clone per base hit).
8194 pub fn get_prop(&self, key: &str, field: &str) -> Option<Value> {
8195 let id = self.ids.get(key)?;
8196 self.props_view().get(id, field).map(|vr| vr.into_value())
8197 }
8198
8199 pub fn has_node(&self, key: &str) -> bool {
8200 self.ids.get(key).is_some()
8201 }
8202
8203 /// One human line for a node: the first of `text`, `summary` or `name` it
8204 /// carries, truncated. Used by the recall digest.
8205 pub fn node_summary_line(&self, key: &str) -> Option<String> {
8206 for field in ["text", "summary", "name"] {
8207 if let Some(Value::Str(s)) = self.get_prop(key, field) {
8208 let s = s.trim();
8209 if !s.is_empty() {
8210 return Some(s.chars().take(120).collect());
8211 }
8212 }
8213 }
8214 None
8215 }
8216
8217 /// Borrow the raw id map. Used by `NodeMask::from_keys` to resolve keys.
8218 pub(crate) fn ids(&self) -> &IdMap {
8219 &self.ids
8220 }
8221
8222 // -----------------------------------------------------------------------
8223 // Namespaces
8224 // -----------------------------------------------------------------------
8225
8226 /// The index `name` already has in `ns_names`, if any.
8227 fn ns_index_of(&self, name: &str) -> Option<u32> {
8228 self.ns_names
8229 .iter()
8230 .position(|n| n == name)
8231 .map(|i| i as u32)
8232 }
8233
8234 /// The index for `name`, appending it to `ns_names` when it is new.
8235 ///
8236 /// The table holds one entry per distinct namespace in the store — a
8237 /// tenant count, not a node count — so the linear scan is cheaper than a
8238 /// map and keeps `namespaces()` allocation-free of a second index.
8239 fn ns_index_for(&mut self, name: &str) -> u32 {
8240 match self.ns_index_of(name) {
8241 Some(i) => i,
8242 None => {
8243 self.ns_names.push(name.to_string());
8244 (self.ns_names.len() - 1) as u32
8245 }
8246 }
8247 }
8248
8249 /// The namespace name at `idx`, or [`NS_DEFAULT`] for an index this handle
8250 /// does not know (unreachable; the default is the narrowing answer).
8251 fn ns_name(&self, idx: u32) -> &str {
8252 self.ns_names
8253 .get(idx as usize)
8254 .map(String::as_str)
8255 .unwrap_or(NS_DEFAULT)
8256 }
8257
8258 /// The namespace index of dense node `id`, defaulting for an id with no
8259 /// entry (a node inserted before this handle rebuilt the array cannot
8260 /// exist: every insert path maintains it).
8261 fn node_ns_idx(&self, id: u32) -> u32 {
8262 self.node_ns
8263 .get(id as usize)
8264 .copied()
8265 .unwrap_or(NS_DEFAULT_IDX)
8266 }
8267
8268 /// File node `id` under namespace `name`, growing `node_ns` as `labels`
8269 /// grows. Called from `apply` for every node insert, live and replayed.
8270 fn set_node_ns(&mut self, id: u32, name: &str) {
8271 let idx = if name == NS_DEFAULT {
8272 NS_DEFAULT_IDX
8273 } else {
8274 self.ns_index_for(name)
8275 };
8276 if self.node_ns.len() <= id as usize {
8277 self.node_ns.resize(id as usize + 1, NS_DEFAULT_IDX);
8278 }
8279 self.node_ns[id as usize] = idx;
8280 }
8281
8282 /// Rebuild `node_ns` from the `ns` column — one pass, at the end of an
8283 /// open or a reload, after the snapshot is restored and the WAL replayed.
8284 ///
8285 /// A store with no `ns` column reads nothing: the column-name check fails
8286 /// and the vector is filled with one constant.
8287 fn rebuild_node_ns(&mut self) {
8288 let total = self.ids.len();
8289 self.ns_names.truncate(1);
8290 self.node_ns.clear();
8291 self.node_ns.resize(total, NS_DEFAULT_IDX);
8292 let has_ns_column = {
8293 let cv = self.props_view();
8294 cv.field_names().iter().any(|f| f == NS_PROP)
8295 };
8296 if !has_ns_column {
8297 return;
8298 }
8299 // Collected first so the props view is released before `ns_index_for`
8300 // takes `&mut self`.
8301 let named: Vec<(u32, String)> = {
8302 let cv = self.props_view();
8303 (0..total as u32)
8304 .filter_map(|id| match cv.get(id, NS_PROP).map(|vr| vr.into_value()) {
8305 Some(Value::Str(s)) if s != NS_DEFAULT => Some((id, s)),
8306 _ => None,
8307 })
8308 .collect()
8309 };
8310 for (id, name) in named {
8311 let idx = self.ns_index_for(&name);
8312 self.node_ns[id as usize] = idx;
8313 }
8314 }
8315
8316 /// Every namespace with at least one live node, in name order.
8317 ///
8318 /// `["default"]` on any store that has never named a namespace, including
8319 /// an empty one: a store is always at least its default namespace.
8320 pub fn namespaces(&self) -> Vec<String> {
8321 let mut out: BTreeSet<&str> = BTreeSet::new();
8322 out.insert(NS_DEFAULT);
8323 for (id, &idx) in self.node_ns.iter().enumerate() {
8324 if idx == NS_DEFAULT_IDX || !self.is_live_node(id as u32) {
8325 continue;
8326 }
8327 out.insert(self.ns_name(idx));
8328 }
8329 out.into_iter().map(str::to_string).collect()
8330 }
8331
8332 /// The namespace of `key`, or `None` when the key names no live node.
8333 pub fn namespace_of(&self, key: &str) -> Option<String> {
8334 let id = self.ids.get(key)?;
8335 if !self.is_live_node(id) {
8336 return None;
8337 }
8338 Some(self.ns_name(self.node_ns_idx(id)).to_string())
8339 }
8340
8341 /// Every live node in `namespace`, as a visibility mask.
8342 ///
8343 /// Built off `node_ns` on whichever handle this is, so on a temporal handle
8344 /// it is the namespace's membership at that commit. A name no node uses
8345 /// gives an empty mask — a namespace scope never widens.
8346 pub fn mask_for_namespace(&self, namespace: &str) -> crate::mask::NodeMask {
8347 let Some(idx) = self.ns_index_of(namespace) else {
8348 return crate::mask::NodeMask::from_ids(std::collections::HashSet::new());
8349 };
8350 let visible: std::collections::HashSet<u32> = (0..self.ids.len() as u32)
8351 .filter(|&id| self.node_ns_idx(id) == idx && self.is_live_node(id))
8352 .collect();
8353 crate::mask::NodeMask::from_ids(visible)
8354 }
8355
8356 /// Live-node test used by the namespace accessors: a deleted node keeps its
8357 /// dense id and its `node_ns` slot, and the label sentinel is what marks it
8358 /// gone — the same test `mask_for_role`'s label leg applies implicitly.
8359 fn is_live_node(&self, id: u32) -> bool {
8360 self.labels
8361 .get(id as usize)
8362 .is_some_and(|&sym| sym != u32::MAX)
8363 && self.ids.key_of(id).is_some()
8364 }
8365
8366 /// Per-namespace live node counts for [`Stats`], in name order.
8367 fn namespace_stats(&self) -> Vec<NamespaceStats> {
8368 let mut counts: BTreeMap<&str, usize> = BTreeMap::new();
8369 counts.insert(NS_DEFAULT, 0);
8370 for id in 0..self.ids.len() as u32 {
8371 if !self.is_live_node(id) {
8372 continue;
8373 }
8374 *counts
8375 .entry(self.ns_name(self.node_ns_idx(id)))
8376 .or_insert(0) += 1;
8377 }
8378 counts
8379 .into_iter()
8380 .filter(|&(name, n)| n > 0 || name == NS_DEFAULT)
8381 .map(|(name, nodes_live)| NamespaceStats {
8382 name: name.to_string(),
8383 nodes_live,
8384 })
8385 .collect()
8386 }
8387
8388 /// The namespace a create-class op would put its node in: the `ns` entry of
8389 /// the props it carries, normalised, with absent meaning [`NS_DEFAULT`].
8390 fn created_namespace<'a>(key: &str, props: &'a [(String, Value)]) -> Result<&'a str> {
8391 Ok(namespace_of_value(Self::sole_ns_entry(key, props)?))
8392 }
8393
8394 /// The one `ns` entry in a node's props, or `None` when it carries none.
8395 ///
8396 /// A props list naming `ns` twice is refused. Without that refusal the
8397 /// write path and the authorisation path can read the same list
8398 /// differently — one taking the first entry, the other the last — and
8399 /// `CREATE (n:L {ns: 'mine', ns: 'theirs'})` lands a node in a namespace
8400 /// the role was checked against the other of. One entry is the only shape
8401 /// where "the node's namespace" is a single fact, so it is the only shape
8402 /// accepted, and every reader of it agrees by construction.
8403 fn sole_ns_entry<'a>(key: &str, props: &'a [(String, Value)]) -> Result<Option<&'a Value>> {
8404 let mut found: Option<&'a Value> = None;
8405 for (field, value) in props {
8406 if field != NS_PROP {
8407 continue;
8408 }
8409 if found.is_some() {
8410 return Err(GraphError::RuleInvalid {
8411 detail: format!(
8412 "node {key}: {NS_PROP} is given more than once; a node has exactly \
8413 one namespace"
8414 ),
8415 });
8416 }
8417 found = Some(value);
8418 }
8419 Ok(found)
8420 }
8421
8422 /// The definition of the role a write authorisation names.
8423 ///
8424 /// `None` when `roles.json` was corrupt at open or the role has since been
8425 /// removed — neither can reach a write, because the authorisation carries a
8426 /// mask `mask_for_role` already resolved for that name.
8427 fn role_def_for(&self, role: &str) -> Option<&RoleDef> {
8428 self.roles.as_ref()?.iter().find(|r| r.name == role)
8429 }
8430
8431 /// Validate the `ns` entry of a node's props and drop an explicit default.
8432 ///
8433 /// Runs on the write path only (see `rewrite_wal_dense`), never on replay:
8434 /// a record that reached the WAL was already accepted here.
8435 fn normalise_insert_ns(
8436 key: &str,
8437 props: Vec<(String, Value)>,
8438 ) -> Result<(Vec<(String, Value)>, String)> {
8439 // One `ns` or none: this is where that is enforced, so every later
8440 // reader of the list — the authorisation gate, the two `apply` arms,
8441 // `node_ns` — is looking at a single entry and cannot disagree about
8442 // which one counts.
8443 Self::sole_ns_entry(key, &props)?;
8444 let mut name = NS_DEFAULT.to_string();
8445 let mut out = Vec::with_capacity(props.len());
8446 for (field, value) in props {
8447 if field != NS_PROP {
8448 out.push((field, value));
8449 continue;
8450 }
8451 let Value::Str(ref s) = value else {
8452 return Err(GraphError::RuleInvalid {
8453 detail: format!(
8454 "node {key}: {NS_PROP} must be a string naming a namespace, \
8455 got {value:?}"
8456 ),
8457 });
8458 };
8459 if !valid_namespace(s) {
8460 return Err(GraphError::RuleInvalid {
8461 detail: format!(
8462 "node {key}: {s:?} is not a valid namespace name — 1 to {NS_MAX_LEN} \
8463 characters of [A-Za-z0-9_.-]"
8464 ),
8465 });
8466 }
8467 name = s.clone();
8468 // An explicit default stores nothing, so a single-tenant store
8469 // never grows an `ns` column.
8470 if name != NS_DEFAULT {
8471 out.push((field, value));
8472 }
8473 }
8474 Ok((out, name))
8475 }
8476
8477 // -----------------------------------------------------------------------
8478 // RBAC role resolution
8479 // -----------------------------------------------------------------------
8480
8481 /// Parse `roles.json` bytes from `fs`.
8482 ///
8483 /// Return values:
8484 /// `Ok(Some(roles))` — file absent (returns `vec![]`) **or** file present
8485 /// and valid; in both cases `mask_for_role` uses the
8486 /// list normally (an absent file means no roles defined).
8487 /// `Ok(None)` — file present but corrupt or unrecognised version
8488 /// → poisoned state; `mask_for_role` returns `Err` for
8489 /// any role name until the file is fixed and the DB
8490 /// re-opened (or `apply_schema` is called to repair it).
8491 ///
8492 /// Note: `None` signals corruption, not absence — the opposite of what an
8493 /// optional "file missing" convention would suggest. The open path stores
8494 /// this result on `db.roles` directly.
8495 fn load_roles_from_fs(fs: &F) -> Result<Option<Vec<RoleDef>>> {
8496 let bytes = fs.read(FileId::Roles).map_err(GraphError::Io)?;
8497 if bytes.is_empty() {
8498 // Empty bytes means either the file is absent or zero-byte — both
8499 // are treated identically as "no roles defined". A zero-byte
8500 // roles.json does NOT widen access: an absent file and a zero-byte
8501 // file both resolve to an empty role list (sees nothing by default).
8502 return Ok(Some(vec![]));
8503 }
8504 match serde_json::from_slice::<RolesFile>(&bytes) {
8505 Ok(f) if matches!(f.version, 1..=4) => Ok(Some(f.roles)),
8506 // Corrupt or unrecognised version (>4): poison the roles state.
8507 // Never widen: a version this binary does not know may carry a
8508 // narrowing this binary would not apply.
8509 _ => Ok(None),
8510 }
8511 }
8512
8513 /// Resolve a role to a node-visibility mask against the current graph state.
8514 ///
8515 /// Returns `Err` when:
8516 /// - `roles.json` was present but corrupt at open (poisoned state), or
8517 /// - `role` does not match any defined role name.
8518 ///
8519 /// The mask union is: explicit `keys` (unknown keys silently ignored) plus
8520 /// all live nodes carrying any label in `labels` that also pass the role's
8521 /// [`visible_where`](crate::roles::RoleDef::visible_where) predicate, if it
8522 /// has one. Label resolution is live — new nodes of an allowed label are
8523 /// visible without re-applying the schema, and a property edited out of the
8524 /// predicate takes its node out of the mask on the next read. An empty
8525 /// union = empty mask = sees nothing.
8526 ///
8527 /// This is the one resolver every read path calls, live and as-of alike, so
8528 /// the predicate applies everywhere at once. On an as-of handle the role
8529 /// *definition* is the current one and the graph is the historical one: the
8530 /// predicate is evaluated against the property values at the commit being
8531 /// read.
8532 ///
8533 /// The result is memoised per `(role, commit_seq)`, so a scoped reader
8534 /// between two writes resolves the role once. See
8535 /// [`RoleMaskCache`](crate::mask::RoleMaskCache) for why that cannot go
8536 /// stale.
8537 pub fn mask_for_role(&self, role: &str) -> Result<crate::mask::NodeMask> {
8538 self.role_masks
8539 .get_or_build(role, self.commit_seq, || self.build_mask_for_role(role))
8540 .map(|m| (*m).clone())
8541 }
8542
8543 /// The mask an [`AsOfScope`] names, resolved against this handle.
8544 ///
8545 /// Shared by [`GraphDb::query_at_scoped`] and
8546 /// [`GraphDb::query_at_scoped_in_namespace`] so one scope resolves one way
8547 /// however the namespace leg is added.
8548 fn mask_at_scope(&self, scope: AsOfScope<'_>) -> Result<crate::mask::NodeMask> {
8549 // One resolver answers "what may this role see" — `mask_for_role` — and
8550 // it runs against this handle, so on a temporal one the answer is the
8551 // as-of one.
8552 Ok(match scope {
8553 AsOfScope::Role(role) => self.mask_for_role(role)?,
8554 AsOfScope::Keys(keys) => {
8555 crate::mask::NodeMask::from_keys(self, keys.iter().map(String::as_str))
8556 }
8557 AsOfScope::RoleAndKeys(role, keys) => {
8558 self.mask_for_role(role)?
8559 .intersect(&crate::mask::NodeMask::from_keys(
8560 self,
8561 keys.iter().map(String::as_str),
8562 ))
8563 }
8564 AsOfScope::Namespace(namespace) => self.mask_for_namespace(namespace),
8565 })
8566 }
8567
8568 /// Resolve `role` against the current graph, ignoring the memo.
8569 fn build_mask_for_role(&self, role: &str) -> Result<crate::mask::NodeMask> {
8570 let roles = self.roles.as_ref().ok_or_else(roles_poisoned)?;
8571 let def = roles
8572 .iter()
8573 .find(|r| r.name == role)
8574 .ok_or_else(|| GraphError::KeyNotFound {
8575 key: format!("role:{role}"),
8576 })?;
8577
8578 let mut visible = std::collections::HashSet::new();
8579
8580 // Key leg: resolve explicit keys to dense ids (unknown keys ignored).
8581 // An administrative grant, never narrowed by the predicate.
8582 for key in &def.keys {
8583 if let Some(id) = self.ids.get(key) {
8584 visible.insert(id);
8585 }
8586 }
8587
8588 // Label leg: live scan — iterate labels vec for matching symbol, and
8589 // when the role carries a predicate, test the property as well. The
8590 // property comes from the store's own merged view (overlay over the
8591 // mmap'd base), so an as-of handle reads the values of its own commit.
8592 let props = def.visible_where.as_ref().map(|_| self.props_view());
8593 for label_name in &def.labels {
8594 if let Some(sym) = self.syms.get(label_name) {
8595 for (i, &s) in self.labels.iter().enumerate() {
8596 if s != sym {
8597 continue;
8598 }
8599 let id = i as u32;
8600 match (&def.visible_where, &props) {
8601 (Some(pred), Some(view)) => {
8602 let value = view.get(id, &pred.field).map(|vr| vr.into_value());
8603 if pred.holds(value.as_ref()) {
8604 visible.insert(id);
8605 }
8606 }
8607 _ => {
8608 visible.insert(id);
8609 }
8610 }
8611 }
8612 }
8613 }
8614
8615 // Namespace leg: an intersection over the whole union, the key leg
8616 // included. A namespace is a tenancy boundary, so a key naming a node in
8617 // another tenant's namespace is not an administrative grant — and
8618 // `apply_schema` has already refused that role, so this only has to be
8619 // right about the node that moved into existence afterwards.
8620 if def.namespaces.is_some() {
8621 visible.retain(|&id| def.sees_namespace(self.ns_name(self.node_ns_idx(id))));
8622 }
8623
8624 Ok(crate::mask::NodeMask::from_ids(visible))
8625 }
8626
8627 /// Return the current list of role definitions.
8628 ///
8629 /// Returns an empty list when no roles are defined or when `roles.json`
8630 /// was corrupt at open (check [`mask_for_role`](Self::mask_for_role) for
8631 /// the fail-loud error in that case, or call
8632 /// [`roles_checked`](Self::roles_checked), which is this readout with that
8633 /// error in it).
8634 pub fn roles(&self) -> Vec<RoleDef> {
8635 self.roles.as_deref().unwrap_or(&[]).to_vec()
8636 }
8637
8638 /// The role definitions, or the poison error when `roles.json` was corrupt
8639 /// at open.
8640 ///
8641 /// [`roles`](Self::roles) answers `[]` both for a store that defines no
8642 /// roles and for one whose sidecar did not parse, and a caller validating a
8643 /// role name at boot cannot tell those apart. The wrong reading of the pair
8644 /// is the dangerous one: a store with no roles at all is an unrestricted
8645 /// store, so a poisoned file would read as "nothing is restricted here".
8646 ///
8647 /// This is the same answer, for the same cause, that
8648 /// [`mask_for_role`](Self::mask_for_role) gives on the first read.
8649 pub fn roles_checked(&self) -> Result<Vec<RoleDef>> {
8650 match self.roles.as_deref() {
8651 Some(roles) => Ok(roles.to_vec()),
8652 None => Err(roles_poisoned()),
8653 }
8654 }
8655
8656 // ── Role-scoped write authz ───────────────────────────────────────────────
8657
8658 /// Execute `ops` with optional role-scoped write authorization.
8659 ///
8660 /// - `None` → full authority, identical to [`write_batch`](Self::write_batch)
8661 /// (zero-cost bypass of all authz checks).
8662 /// - `Some(authz)` → the decision table is evaluated per-op BEFORE any WAL
8663 /// record is built. A denial returns an error with no WAL frame written
8664 /// (all-or-nothing at the authz boundary, then at the MutPreview boundary).
8665 ///
8666 /// See the plan's "authz decision table" section for the full semantics.
8667 pub fn write_batch_authz(
8668 &mut self,
8669 authz: Option<&WriteAuthz>,
8670 ops: Vec<BatchOp>,
8671 ) -> Result<(usize, usize)> {
8672 // Thread authz as a direct parameter — never touches pending_write_authz.
8673 self.commit_logged_batch(ops, None, authz.cloned())
8674 .map(inserted_pair)
8675 }
8676
8677 /// Execute a Cypher write statement with role-scoped write authorization.
8678 ///
8679 /// Resolves scope + mask from `self.roles` inside the call (same write-guard
8680 /// lifetime as execution, satisfying §5 lock discipline). The resolved
8681 /// `WriteAuthz` is stored as `pending_write_authz` for the duration of the
8682 /// call so that all inner `batch.commit()` calls are authz-checked.
8683 ///
8684 /// MERGE is handled specially: the MERGE scope precondition (§3.3) is
8685 /// checked in `exec_merge` BEFORE `has_node` to close the §6.2
8686 /// timing-oracle item (hidden ≡ absent for unscoped roles).
8687 ///
8688 /// Roles with `write: None` (v1 behavior) → `RoleWriteDenied` with
8689 /// "this endpoint is not permitted".
8690 pub fn query_write_authz(
8691 &mut self,
8692 role: &str,
8693 cypher: &str,
8694 params: &BTreeMap<String, Value>,
8695 ) -> Result<ResultSet> {
8696 // Resolve scope (fails fast if role has no write scope).
8697 // write:None → byte-identical v1 blanket-403 body (plan §v1-sidecar mandate).
8698 let scope =
8699 {
8700 let roles = self.roles.as_deref().ok_or_else(|| GraphError::Corrupt {
8701 detail: "roles.json was corrupt at open; re-open to restore role access".into(),
8702 })?;
8703 let def = roles.iter().find(|r| r.name == role).ok_or_else(|| {
8704 GraphError::KeyNotFound {
8705 key: format!("role:{role}"),
8706 }
8707 })?;
8708 def.write
8709 .clone()
8710 .ok_or_else(|| GraphError::RoleWriteDenied {
8711 reason: "role-bound token: writes are not permitted".into(),
8712 })?
8713 };
8714 // Resolve mask inside the call (same guard, §5 coherence).
8715 let mask = self.mask_for_role(role)?;
8716 self.pending_write_authz = Some(WriteAuthz {
8717 role: role.into(),
8718 scope,
8719 mask,
8720 });
8721 // RAII guard: always clears pending_write_authz on scope exit, including
8722 // on panic or early-return, mirroring the RestoreEmitDeltas precedent.
8723 struct ClearPendingAuthzOnDrop(*mut Option<WriteAuthz>);
8724 impl Drop for ClearPendingAuthzOnDrop {
8725 fn drop(&mut self) {
8726 // SAFETY: pointer into the owning GraphDb; guard is dropped
8727 // within this function's frame before it returns.
8728 unsafe { *self.0 = None };
8729 }
8730 }
8731 // SAFETY: raw pointer into self; guard dropped before this fn returns.
8732 let _authz_guard = ClearPendingAuthzOnDrop(&mut self.pending_write_authz as *mut _);
8733 let tokens = lex(cypher).map_err(|e| GraphError::QueryError {
8734 detail: format!("lex: {e}"),
8735 })?;
8736 let stmt = parse_write(&tokens).map_err(|e| GraphError::QueryError {
8737 detail: format!("parse: {e}"),
8738 })?;
8739 self.exec_write_stmt(stmt, params)
8740 }
8741
8742 /// Execute `ops` with optional role-scoped write authorization, suppressing
8743 /// fsync (for use inside the group-commit drain thread, which performs one
8744 /// group fsync after releasing the write lock).
8745 ///
8746 /// Identical to [`write_batch_authz`] except the fsync policy is temporarily
8747 /// forced to `Relaxed` for the duration of the call, matching the drain-thread
8748 /// contract established by [`commit_batch_nosync`].
8749 pub(crate) fn write_batch_authz_nosync(
8750 &mut self,
8751 authz: Option<&WriteAuthz>,
8752 ops: Vec<BatchOp>,
8753 ) -> Result<(usize, usize)> {
8754 let saved = self.fsync;
8755 struct RestoreFsync(*mut FsyncPolicy, FsyncPolicy);
8756 impl Drop for RestoreFsync {
8757 fn drop(&mut self) {
8758 // SAFETY: pointer into the owning GraphDb; guard is dropped
8759 // within the enclosing function's frame before it returns.
8760 unsafe { *self.0 = self.1 };
8761 }
8762 }
8763 // SAFETY: raw pointer into self; guard dropped before this fn returns.
8764 let _g = RestoreFsync(&mut self.fsync as *mut FsyncPolicy, saved);
8765 self.fsync = FsyncPolicy::Relaxed;
8766 self.commit_logged_batch(ops, None, authz.cloned())
8767 .map(inserted_pair)
8768 }
8769
8770 /// Execute a `/ingest` request with role-scoped write authorization.
8771 ///
8772 /// Resolves the role's `WriteScope` and `NodeMask` inside this call (same
8773 /// write-guard lifetime as the mutation, satisfying §5 lock discipline).
8774 /// Sets `pending_write_authz` for the duration of the call so that the
8775 /// `commit_ingest` → `commit_logged_batch` path picks up the authz context
8776 /// and evaluates the decision table per-op before any WAL write.
8777 ///
8778 /// §7.3: roles with empty `create_labels` will see every `InsertNode` op
8779 /// denied by the decision table with the appropriate §4.3 scope reason;
8780 /// no special HTTP-layer check is needed.
8781 ///
8782 /// Roles with `write: None` return `RoleWriteDenied` with
8783 /// "writes are not permitted" (byte-identical to v1 blanket 403).
8784 pub fn ingest_with_edges_authz(
8785 &mut self,
8786 role: &str,
8787 label: &str,
8788 rows: Vec<std::collections::BTreeMap<String, Value>>,
8789 opts: &crate::ingest::IngestOptions,
8790 edges: &[(String, String, String)],
8791 ) -> Result<crate::ingest::IngestReport> {
8792 // Resolve scope (fails fast if role has no write scope).
8793 // write:None → byte-identical v1 blanket-403 body (plan §v1-sidecar mandate).
8794 let scope =
8795 {
8796 let roles = self.roles.as_deref().ok_or_else(|| GraphError::Corrupt {
8797 detail: "roles.json was corrupt at open; re-open to restore role access".into(),
8798 })?;
8799 let def = roles.iter().find(|r| r.name == role).ok_or_else(|| {
8800 GraphError::KeyNotFound {
8801 key: format!("role:{role}"),
8802 }
8803 })?;
8804 def.write
8805 .clone()
8806 .ok_or_else(|| GraphError::RoleWriteDenied {
8807 reason: "role-bound token: writes are not permitted".into(),
8808 })?
8809 };
8810 let mask = self.mask_for_role(role)?;
8811 self.pending_write_authz = Some(WriteAuthz {
8812 role: role.into(),
8813 scope,
8814 mask,
8815 });
8816 // RAII guard: always clears pending_write_authz on scope exit, including
8817 // on panic or early-return, mirroring the RestoreEmitDeltas precedent.
8818 struct ClearPendingAuthzOnDrop(*mut Option<WriteAuthz>);
8819 impl Drop for ClearPendingAuthzOnDrop {
8820 fn drop(&mut self) {
8821 // SAFETY: pointer into the owning GraphDb; guard is dropped
8822 // within this function's frame before it returns.
8823 unsafe { *self.0 = None };
8824 }
8825 }
8826 // SAFETY: raw pointer into self; guard dropped before this fn returns.
8827 let _authz_guard = ClearPendingAuthzOnDrop(&mut self.pending_write_authz as *mut _);
8828 self.ingest_with_edges(label, rows, opts, edges)
8829 }
8830
8831 /// Evaluate the write-authz decision table for one `BatchOp`.
8832 ///
8833 /// Called by `commit_logged_batch` for each op when `pending_write_authz`
8834 /// is `Some`, BEFORE MutPreview. A denial returns an error immediately;
8835 /// the remaining ops are not evaluated and no WAL frame is written.
8836 ///
8837 /// `batch_created` carries the key→label pairs of nodes that earlier ops in
8838 /// THIS batch will create. Used by `InsertEdgeUpsert` to count same-batch
8839 /// placeholder nodes as visible (spec: "a placeholder endpoint the SAME
8840 /// batch creates counts as visible if its label passed the create-class gate").
8841 fn check_single_op_authz(
8842 &self,
8843 authz: &WriteAuthz,
8844 op: &BatchOp,
8845 batch_created: &BTreeMap<String, String>,
8846 ) -> Result<()> {
8847 // Helper: 3-way node status under the authz mask.
8848 //
8849 // Batch-created nodes (from earlier InsertNode in THIS batch) are treated
8850 // as Visible with their recorded label — their create gate already passed
8851 // and they are not yet in self.ids (not committed). This fixes the
8852 // MERGE+ON CREATE SET case where InsertNode + SetProp arrive together:
8853 // the SetProp must not see the node as Absent.
8854 let node_status = |key: &str| -> NodeAuthzStatus {
8855 if let Some(label) = batch_created.get(key) {
8856 return NodeAuthzStatus::Visible(label.clone());
8857 }
8858 match self.ids.get(key) {
8859 None => NodeAuthzStatus::Absent,
8860 Some(id) if !authz.mask.contains_id(id) => NodeAuthzStatus::Hidden,
8861 Some(id) => {
8862 let label = self
8863 .labels
8864 .get(id as usize)
8865 .and_then(|&sym| {
8866 if sym == u32::MAX {
8867 None
8868 } else {
8869 self.syms.resolve(sym).map(str::to_string)
8870 }
8871 })
8872 .unwrap_or_default();
8873 NodeAuthzStatus::Visible(label)
8874 }
8875 }
8876 };
8877
8878 // Helper: is an InsertEdgeUpsert endpoint visible?
8879 // A same-batch placeholder counts as visible if its label passed
8880 // the create-class gate (spec "upsert placeholder-counts-as-visible").
8881 let upsert_ep_visible = |ep_key: &str, placeholder_label: &str| -> bool {
8882 // In store and visible?
8883 if let Some(id) = self.ids.get(ep_key) {
8884 return authz.mask.contains_id(id);
8885 }
8886 // Created by an earlier op in this batch?
8887 if let Some(created_label) = batch_created.get(ep_key) {
8888 return authz.scope.create_labels.contains(created_label);
8889 }
8890 // Will be created by THIS InsertEdgeUpsert: placeholder_label
8891 // must pass the create-class gate.
8892 authz
8893 .scope
8894 .create_labels
8895 .contains(&placeholder_label.to_string())
8896 };
8897
8898 match op {
8899 // RenameNode / CreateRule / DeleteRule: defense-in-depth gate.
8900 // These ops are never routed to role-scoped paths by the HTTP layer,
8901 // but we 403 them here to close any future bypass route.
8902 //
8903 // InsertNodeOnConflict joins them: it is reachable only from the
8904 // embedded Python binding, which has no role token, and `Replace`
8905 // is a create and an update at once. Rather than split the decision
8906 // table for an op no role-scoped path constructs, refuse it — a
8907 // role-scoped caller writes through the ops that are already in the
8908 // table.
8909 BatchOp::RenameNode { .. }
8910 | BatchOp::CreateRule(_)
8911 | BatchOp::DeleteRule { .. }
8912 | BatchOp::InsertNodeOnConflict { .. } => {
8913 return Err(GraphError::RoleWriteDenied {
8914 reason: "role-bound token: this endpoint is not permitted".into(),
8915 });
8916 }
8917
8918 // ── CREATE-class: InsertNode ─────────────────────────────────────
8919 //
8920 // Decision table row 1 (scope-before-lookup): check label in
8921 // create_labels BEFORE any key lookup. This is the structural
8922 // closure of the §6.2 timing-oracle item — the denial fires even
8923 // when the store is EMPTY (see test_create_scope_denied_empty_store).
8924 BatchOp::InsertNode { label, key, props } => {
8925 if !authz.scope.create_labels.contains(label) {
8926 return Err(GraphError::RoleWriteDenied {
8927 reason: format!(
8928 "role-bound token: label '{}' not in write scope (create_labels)",
8929 label
8930 ),
8931 });
8932 }
8933 // A role bound to namespaces may only create inside them. The
8934 // never-widen rule is about what a write makes visible to *any*
8935 // party, not only to the writer: a node this role could never
8936 // read back is a write into somebody else's tenancy. Also a
8937 // scope check, so it runs before the key lookup — it discloses
8938 // nothing about the store. Covers Cypher `CREATE` and the node
8939 // `MERGE` creates, both of which arrive as this op.
8940 // Resolved before the role lookup so a props list naming `ns`
8941 // twice is refused for every role, scoped or not: it is the same
8942 // malformed write the seam refuses, and leaving it to the seam
8943 // would mean the gate had already read one of the two.
8944 let target = Self::created_namespace(key, props)?;
8945 if let Some(def) = self.role_def_for(&authz.role) {
8946 if !def.sees_namespace(target) {
8947 return Err(GraphError::RoleWriteDenied {
8948 reason: format!(
8949 "role-bound token: namespace '{target}' not in the role's \
8950 namespaces"
8951 ),
8952 });
8953 }
8954 }
8955 // Row 2/3: key lookup.
8956 match self.ids.get(key.as_str()) {
8957 Some(id) if authz.mask.contains_id(id) => {
8958 // Visible: DuplicateKey — let MutPreview handle this.
8959 }
8960 Some(_) => {
8961 // Hidden: indistinguishable from absent to the role.
8962 return Err(GraphError::RoleWriteDenied {
8963 reason: "role-bound token: target node not visible".into(),
8964 });
8965 }
8966 None => {
8967 // Absent: proceed (create).
8968 }
8969 }
8970 }
8971
8972 // ── UPDATE-class: SetProp, RemoveProp ────────────────────────────
8973 BatchOp::SetProp { key, .. } | BatchOp::RemoveProp { key, .. } => {
8974 if batch_created.contains_key(key.as_str()) {
8975 // Batch-created node: create gate already passed this batch.
8976 // Updating it in the same batch is always allowed, regardless
8977 // of update_labels (ruling §3.5: "writer just created it").
8978 } else {
8979 let label = match node_status(key) {
8980 NodeAuthzStatus::Visible(lbl) => lbl,
8981 _ => {
8982 return Err(GraphError::RoleWriteDenied {
8983 reason: "role-bound token: target node not visible".into(),
8984 });
8985 }
8986 };
8987 if !authz.scope.update_labels.contains(&label) {
8988 return Err(GraphError::RoleWriteDenied {
8989 reason: format!(
8990 "role-bound token: label '{}' not in write scope (update_labels)",
8991 label
8992 ),
8993 });
8994 }
8995 }
8996 }
8997
8998 // ── DELETE-class: DeleteNode ─────────────────────────────────────
8999 BatchOp::DeleteNode { key } => {
9000 let label = match node_status(key) {
9001 NodeAuthzStatus::Visible(lbl) => lbl,
9002 _ => {
9003 return Err(GraphError::RoleWriteDenied {
9004 reason: "role-bound token: target node not visible".into(),
9005 });
9006 }
9007 };
9008 if !authz.scope.delete_labels.contains(&label) {
9009 return Err(GraphError::RoleWriteDenied {
9010 reason: format!(
9011 "role-bound token: label '{}' not in write scope (delete_labels)",
9012 label
9013 ),
9014 });
9015 }
9016 }
9017
9018 // ── DELETE-class: DeleteEdge ─────────────────────────────────────
9019 //
9020 // Derived-edge rejection runs BEFORE the delete_edge_types scope
9021 // check (spec §3.5: "existing derived-edge rejection precedes
9022 // delete_edge_types check").
9023 BatchOp::DeleteEdge {
9024 edge_type,
9025 src_key,
9026 dst_key,
9027 } => {
9028 // Check provenance ownership BEFORE scope (spec §3.5 ordering).
9029 if let (Some(src_id), Some(dst_id), Some(et_sym)) = (
9030 self.ids.get(src_key.as_str()),
9031 self.ids.get(dst_key.as_str()),
9032 self.syms.get(edge_type.as_str()),
9033 ) {
9034 if self.engine.is_owned(et_sym, src_id, dst_id) {
9035 return Err(GraphError::RuleOwned {
9036 detail: format!(
9037 "edge {edge_type} {src_key}→{dst_key} is rule-owned; \
9038 delete or change the owning rule"
9039 ),
9040 });
9041 }
9042 // Also check would_derive via MutPreview (empty overlay, pre-batch).
9043 let preview = MutPreview::new(self);
9044 if preview.would_derive(edge_type, src_key, dst_key) {
9045 return Err(GraphError::RuleOwned {
9046 detail: format!(
9047 "edge {edge_type} {src_key}→{dst_key} is rule-owned; \
9048 delete or change the owning rule, or a live rule would \
9049 re-derive it"
9050 ),
9051 });
9052 }
9053 }
9054 // Scope check (AFTER derived-edge check, BEFORE endpoint visibility).
9055 if !authz.scope.delete_edge_types.contains(edge_type) {
9056 return Err(GraphError::RoleWriteDenied {
9057 reason: format!(
9058 "role-bound token: edge type '{}' not in write scope (delete_edge_types)",
9059 edge_type
9060 ),
9061 });
9062 }
9063 // Both endpoints must be visible.
9064 for ep_key in [src_key.as_str(), dst_key.as_str()] {
9065 match self.ids.get(ep_key) {
9066 None => {
9067 return Err(GraphError::RoleWriteDenied {
9068 reason: "role-bound token: edge endpoint not visible".into(),
9069 });
9070 }
9071 Some(id) if !authz.mask.contains_id(id) => {
9072 return Err(GraphError::RoleWriteDenied {
9073 reason: "role-bound token: edge endpoint not visible".into(),
9074 });
9075 }
9076 _ => {}
9077 }
9078 }
9079 }
9080
9081 // ── EDGE-CREATE: InsertEdge ──────────────────────────────────────
9082 //
9083 // Scope check BEFORE endpoint lookup (preserves timing symmetry).
9084 BatchOp::InsertEdge {
9085 edge_type,
9086 src_key,
9087 dst_key,
9088 } => {
9089 if !authz.scope.create_edge_types.contains(edge_type) {
9090 return Err(GraphError::RoleWriteDenied {
9091 reason: format!(
9092 "role-bound token: edge type '{}' not in write scope (create_edge_types)",
9093 edge_type
9094 ),
9095 });
9096 }
9097 // Both endpoints must be visible. A node created by an earlier
9098 // InsertNode in the same batch (tracked in batch_created) counts
9099 // as visible if its label passed the create-class gate.
9100 for ep_key in [src_key.as_str(), dst_key.as_str()] {
9101 if batch_created.contains_key(ep_key) {
9102 // Created earlier this batch — already scope-checked.
9103 continue;
9104 }
9105 match self.ids.get(ep_key) {
9106 None => {
9107 return Err(GraphError::RoleWriteDenied {
9108 reason: "role-bound token: edge endpoint not visible".into(),
9109 });
9110 }
9111 Some(id) if !authz.mask.contains_id(id) => {
9112 return Err(GraphError::RoleWriteDenied {
9113 reason: "role-bound token: edge endpoint not visible".into(),
9114 });
9115 }
9116 _ => {}
9117 }
9118 }
9119 }
9120
9121 // ── EDGE-CREATE: InsertEdgeUpsert ────────────────────────────────
9122 //
9123 // Scope check first; then endpoint visibility using same-batch
9124 // placeholder awareness (spec: "a placeholder endpoint the SAME
9125 // batch creates counts as visible if its label passed the
9126 // create-class gate").
9127 BatchOp::InsertEdgeUpsert {
9128 edge_type,
9129 src_key,
9130 dst_key,
9131 placeholder_label,
9132 } => {
9133 if !authz.scope.create_edge_types.contains(edge_type) {
9134 return Err(GraphError::RoleWriteDenied {
9135 reason: format!(
9136 "role-bound token: edge type '{}' not in write scope (create_edge_types)",
9137 edge_type
9138 ),
9139 });
9140 }
9141 // Check placeholder label against create_labels (create-class gate).
9142 // This ensures the auto-created endpoints are scope-allowed.
9143 for ep_key in [src_key.as_str(), dst_key.as_str()] {
9144 if !upsert_ep_visible(ep_key, placeholder_label) {
9145 return Err(GraphError::RoleWriteDenied {
9146 reason: "role-bound token: edge endpoint not visible".into(),
9147 });
9148 }
9149 }
9150 // A placeholder is created with no props, so it lands in the
9151 // default namespace. A role that cannot read `default` must not
9152 // create one there, for the same reason it may not create a node
9153 // there outright.
9154 //
9155 // The refusal is byte-identical to the hidden-endpoint one above,
9156 // and deliberately so: this arm fires only for an endpoint that
9157 // does **not** exist, and the one above only for an endpoint that
9158 // does. Two different strings would make the pair an existence
9159 // oracle — ask for an upsert and read off whether the key is
9160 // taken. Hidden ≡ absent is the rule everywhere else in this
9161 // table and it holds here too.
9162 if let Some(def) = self.role_def_for(&authz.role) {
9163 if !def.sees_namespace(NS_DEFAULT) {
9164 for ep_key in [src_key.as_str(), dst_key.as_str()] {
9165 if self.ids.get(ep_key).is_none() && !batch_created.contains_key(ep_key)
9166 {
9167 return Err(GraphError::RoleWriteDenied {
9168 reason: "role-bound token: edge endpoint not visible".into(),
9169 });
9170 }
9171 }
9172 }
9173 }
9174 }
9175 }
9176 Ok(())
9177 }
9178
9179 /// Write `roles` to `roles.json` atomically and update the in-memory list.
9180 ///
9181 /// Called by `apply_schema` when roles change. Never called on unchanged
9182 /// re-apply — this preserves byte-identical idempotency.
9183 pub(crate) fn commit_roles(&mut self, roles: Vec<RoleDef>) -> Result<()> {
9184 let file = RolesFile::new_versioned(roles.clone());
9185 let bytes = serde_json::to_vec(&file).map_err(|e| GraphError::Corrupt {
9186 detail: format!("roles serialization: {e}"),
9187 })?;
9188 self.fs
9189 .write_atomic(FileId::Roles, &bytes)
9190 .map_err(GraphError::Io)?;
9191 self.roles = Some(roles);
9192 // Rewriting the sidecar is not a commit, so `commit_seq` does not move
9193 // and a memoised mask would still match its version. Install a fresh
9194 // cache instead of clearing the shared one: a reader snapshot frozen
9195 // against the old definitions keeps the old `Arc` to itself and can
9196 // never publish an answer this handle would read back.
9197 self.role_masks = Arc::new(crate::mask::RoleMaskCache::new());
9198 // Refresh the MVCC frozen overlay so that reader() immediately sees the
9199 // updated role definitions without waiting for the next K-commit fold.
9200 self.fold_now();
9201 Ok(())
9202 }
9203
9204 fn view(&self) -> GraphView<'_> {
9205 GraphView {
9206 ids: &self.ids,
9207 syms: &self.syms,
9208 labels: &self.labels,
9209 props: self.props_view(),
9210 topo: self.topo_view(),
9211 edge_props: self.edge_props_view(),
9212 mask: None,
9213 prop_index: Some(&self.prop_index),
9214 }
9215 }
9216
9217 fn view_masked<'a>(&'a self, mask: &'a crate::mask::NodeMask) -> GraphView<'a> {
9218 GraphView {
9219 ids: &self.ids,
9220 syms: &self.syms,
9221 labels: &self.labels,
9222 props: self.props_view(),
9223 topo: self.topo_view(),
9224 edge_props: self.edge_props_view(),
9225 mask: Some(&mask.visible),
9226 prop_index: Some(&self.prop_index),
9227 }
9228 }
9229
9230 /// Execute a read-only Cypher query with a node visibility mask.
9231 ///
9232 /// Only nodes whose key is in `mask` are accessible: label scans, key
9233 /// lookups, and neighbor expansions all respect the mask. Edges where
9234 /// either endpoint is hidden are silently dropped.
9235 ///
9236 /// Returns `Err` with a "masked queries are read-only" message when
9237 /// `cypher` is a write statement (CREATE / MERGE / MATCH…SET / DELETE).
9238 pub fn query_masked(
9239 &self,
9240 cypher: &str,
9241 params: &std::collections::BTreeMap<String, Value>,
9242 mask: &crate::mask::NodeMask,
9243 ) -> Result<ResultSet> {
9244 // Reject write statements up front.
9245 let tokens = lex(cypher).map_err(|e| GraphError::QueryError {
9246 detail: format!("lex: {e}"),
9247 })?;
9248 if is_write_tokens(&tokens) {
9249 return Err(GraphError::MaskedReadOnly);
9250 }
9251 let union = parse_read(&tokens).map_err(|e| GraphError::QueryError {
9252 detail: format!("parse: {e}"),
9253 })?;
9254 // Each UNION part executes against the same masked view, so the mask
9255 // applies uniformly across the chain.
9256 execute_union(&self.view_masked(mask), &union, &Params(params)).map_err(|e| {
9257 GraphError::QueryError {
9258 detail: format!("execute: {e}"),
9259 }
9260 })
9261 }
9262
9263 pub fn node_ref(&self, key: &str) -> Option<NodeRef<'_, F>> {
9264 let id = self.ids.get(key)?;
9265 Some(NodeRef { db: self, id })
9266 }
9267
9268 /// BFS neighborhood expansion restricted to visible nodes in `mask`.
9269 ///
9270 /// Hidden nodes are never used as traversal intermediaries in either
9271 /// [`MaskMode::Omit`] or [`MaskMode::Stub`] — a visible node reachable
9272 /// only through a hidden node will not appear in results.
9273 ///
9274 /// In [`MaskMode::Stub`] mode, hidden nodes that are direct neighbours of
9275 /// a visited visible node are appended to the result as stub rows
9276 /// (`label` column is `null`, same key+depth columns as visible rows).
9277 /// They are NOT added to the BFS frontier.
9278 ///
9279 /// Returns `None` when `key` does not exist (caller should 404).
9280 ///
9281 /// **SECURITY**: role-token callers always pass an Omit-mode mask, so
9282 /// stub rows are never produced on the role path.
9283 pub fn neighborhood_masked(
9284 &self,
9285 key: &str,
9286 depth: u32,
9287 edge_types: Option<&[&str]>,
9288 dir: Dir,
9289 mask: &crate::mask::NodeMask,
9290 ) -> Option<ResultSet> {
9291 let start_id = self.ids.get(key)?;
9292 let view = self.view_masked(mask);
9293 let resolved: Option<Vec<u32>> = edge_types.map(|names| {
9294 names
9295 .iter()
9296 .filter_map(|name| view.syms.get(name))
9297 .collect()
9298 });
9299 let nb = neighborhood(&view, start_id, depth, resolved.as_deref(), dir);
9300 let mut rs = ResultSet::new(vec!["key".into(), "label".into(), "depth".into()]);
9301 // Collect visible BFS results (start_id at depth 0, BFS nodes after).
9302 let mut visited: Vec<(u32, u32)> = Vec::with_capacity(nb.nodes.len() + 1);
9303 visited.push((start_id, 0));
9304 for (nid, d) in &nb.nodes {
9305 let k = view.key_of(*nid);
9306 let label = view
9307 .label_of(*nid)
9308 .expect("real nodes always have a label; u32::MAX sentinel cannot occur");
9309 rs.push_row(vec![
9310 Some(Value::Str(k.to_string())),
9311 Some(Value::Str(label.to_string())),
9312 Some(Value::Int(*d as i64)),
9313 ]);
9314 visited.push((*nid, *d));
9315 }
9316 // Stub mode: add hidden direct neighbours of each visited node as stubs.
9317 // Hidden nodes are edge-endpoints only — they are not added to the BFS
9318 // frontier, so the BFS never expands through them.
9319 if mask.mode() == crate::mask::MaskMode::Stub {
9320 let raw_view = self.view();
9321 let mut seen: std::collections::HashSet<u32> =
9322 visited.iter().map(|(id, _)| *id).collect();
9323 for (node_id, node_depth) in &visited {
9324 if *node_depth >= depth {
9325 continue;
9326 }
9327 for e in expand(&raw_view, *node_id, resolved.as_deref(), dir) {
9328 let nbr = if e.src == *node_id { e.dst } else { e.src };
9329 if !mask.contains_id(nbr) && seen.insert(nbr) {
9330 if let Some(k) = self.ids.key_of(nbr) {
9331 rs.push_row(vec![
9332 Some(Value::Str(k.to_string())),
9333 None,
9334 Some(Value::Int((*node_depth + 1) as i64)),
9335 ]);
9336 }
9337 }
9338 }
9339 }
9340 }
9341 Some(rs)
9342 }
9343
9344 /// [`neighborhood_masked`](Self::neighborhood_masked) with the **subject
9345 /// check** a scoped caller needs: a start key the mask hides answers exactly
9346 /// as an absent one does.
9347 ///
9348 /// `neighborhood_masked` expands from any existing key, hidden or not,
9349 /// because a full-token caller supplying a client mask already knows which
9350 /// keys exist. A scoped caller does not, so telling it apart a hidden key
9351 /// from an absent one would be an existence oracle.
9352 ///
9353 /// Expansion itself is unchanged: hidden nodes are neither returned nor used
9354 /// as traversal intermediaries, so a visible node reachable only through a
9355 /// hidden one stays out of the result.
9356 ///
9357 /// Hidden or unknown `key` → [`GraphError::KeyNotFound`].
9358 pub fn neighborhood_scoped(
9359 &self,
9360 key: &str,
9361 depth: u32,
9362 edge_types: Option<&[&str]>,
9363 dir: Dir,
9364 mask: &crate::mask::NodeMask,
9365 ) -> Result<ResultSet> {
9366 if !mask.contains_node(self, key) {
9367 return Err(GraphError::KeyNotFound { key: key.into() });
9368 }
9369 self.neighborhood_masked(key, depth, edge_types, dir, mask)
9370 .ok_or_else(|| GraphError::KeyNotFound { key: key.into() })
9371 }
9372
9373 /// Live node's key, label, and columnar props. Unknown or tombstoned → `None`.
9374 pub fn node_info(&self, key: &str) -> Option<NodeInfo> {
9375 let n = self.node_ref(key)?;
9376 Some(NodeInfo {
9377 key: n.key().to_string(),
9378 label: n.label().to_string(),
9379 props: n.props(),
9380 })
9381 }
9382
9383 /// Look up a node with mask awareness.
9384 ///
9385 /// | Key state | Omit mode | Stub mode |
9386 /// |-------------------|-----------------|------------------------|
9387 /// | does not exist | `None` (→ 404) | `None` (→ 404) |
9388 /// | exists, visible | `Some(Visible)` | `Some(Visible)` |
9389 /// | exists, hidden | `None` (→ 404) | `Some(Restricted)` |
9390 ///
9391 /// **SECURITY**: only call from client-mask (full-token) paths.
9392 /// Role-token paths must use [`node_info`] after an explicit visibility check.
9393 pub fn node_info_masked(
9394 &self,
9395 key: &str,
9396 mask: &crate::mask::NodeMask,
9397 ) -> Option<MaskedNodeResult> {
9398 let id = self.ids.get(key)?;
9399 if mask.contains_id(id) {
9400 Some(MaskedNodeResult::Visible(self.node_info(key)?))
9401 } else {
9402 match mask.mode() {
9403 crate::mask::MaskMode::Stub => Some(MaskedNodeResult::Restricted),
9404 crate::mask::MaskMode::Omit => None,
9405 }
9406 }
9407 }
9408
9409 /// Get edges for `key` with mask-aware hidden-endpoint handling.
9410 ///
9411 /// - Omit mode: edges to hidden endpoints are excluded (same as role-path filtering).
9412 /// - Stub mode: edges to hidden endpoints are included; `src_restricted`/`dst_restricted`
9413 /// is `true` for each hidden endpoint.
9414 ///
9415 /// Unknown key → [`GraphError::KeyNotFound`].
9416 ///
9417 /// **SECURITY**: only call from client-mask (full-token) paths.
9418 pub fn node_edges_masked(
9419 &self,
9420 key: &str,
9421 mask: &crate::mask::NodeMask,
9422 ) -> Result<Vec<MaskedEdge>> {
9423 self.ensure_v8_base_sections_loaded();
9424 let id = self
9425 .ids
9426 .get(key)
9427 .ok_or_else(|| GraphError::KeyNotFound { key: key.into() })?;
9428 let derived: BTreeSet<(u32, u32, u32)> = self
9429 .engine
9430 .provenance_touching(id)
9431 .map(|(_rule, etype, src, dst)| (etype, src, dst))
9432 .collect();
9433 let mut edges = Vec::new();
9434 let tv = self.topo_view();
9435 for etype in tv.etypes() {
9436 // etype comes from the archived CSR (access_unchecked, no eager CRC).
9437 // A bit-flip in the large TOPOLOGY section can produce an etype id
9438 // that is not in the interner. Return Corrupt rather than panic.
9439 let edge_type = self
9440 .syms
9441 .resolve(etype)
9442 .ok_or_else(|| GraphError::Corrupt {
9443 detail: format!("v8: topology etype {etype} not in interner"),
9444 })?
9445 .to_string();
9446 for dir in [Direction::Out, Direction::In] {
9447 for &nbr in tv.neighbors(etype, dir, id).as_ref() {
9448 let nbr_restricted = !mask.contains_id(nbr);
9449 if nbr_restricted && mask.mode() == crate::mask::MaskMode::Omit {
9450 continue;
9451 }
9452 let nbr_key = self
9453 .ids
9454 .key_of(nbr)
9455 .ok_or_else(|| GraphError::Corrupt {
9456 detail: format!("topology id {nbr} has no key"),
9457 })?
9458 .to_string();
9459 let (src_id, dst_id, src_key, dst_key, src_restricted, dst_restricted) =
9460 match dir {
9461 Direction::Out => {
9462 (id, nbr, key.to_string(), nbr_key, false, nbr_restricted)
9463 }
9464 Direction::In => {
9465 (nbr, id, nbr_key, key.to_string(), nbr_restricted, false)
9466 }
9467 };
9468 edges.push(MaskedEdge {
9469 edge_type: edge_type.clone(),
9470 src_key,
9471 src_restricted,
9472 dst_key,
9473 dst_restricted,
9474 derived: derived.contains(&(etype, src_id, dst_id)),
9475 });
9476 }
9477 }
9478 }
9479 edges.sort_by(|a, b| {
9480 a.edge_type
9481 .cmp(&b.edge_type)
9482 .then(a.src_key.cmp(&b.src_key))
9483 .then(a.dst_key.cmp(&b.dst_key))
9484 });
9485 edges.dedup_by(|a, b| {
9486 a.edge_type == b.edge_type && a.src_key == b.src_key && a.dst_key == b.dst_key
9487 });
9488 Ok(edges)
9489 }
9490
9491 /// [`node_edges_masked`](Self::node_edges_masked) with the **subject check**
9492 /// a scoped caller needs, and a plain [`EdgeInfo`] list.
9493 ///
9494 /// `node_edges_masked` raises [`GraphError::KeyNotFound`] only when `key` is
9495 /// unknown; a key that exists but is hidden still yields its (filtered) edge
9496 /// list, which is correct for a full-token client mask and an existence
9497 /// oracle for a scoped one. Here a hidden subject answers exactly as an
9498 /// absent one does.
9499 ///
9500 /// Every edge naming a hidden endpoint is dropped, whatever the mask's
9501 /// [`MaskMode`](crate::mask::MaskMode): a scoped caller never sees a
9502 /// restricted stub, so there is nothing for it to render.
9503 ///
9504 /// Hidden or unknown `key` → [`GraphError::KeyNotFound`].
9505 pub fn node_edges_scoped(
9506 &self,
9507 key: &str,
9508 mask: &crate::mask::NodeMask,
9509 ) -> Result<Vec<EdgeInfo>> {
9510 if !mask.contains_node(self, key) {
9511 return Err(GraphError::KeyNotFound { key: key.into() });
9512 }
9513 Ok(self
9514 .node_edges_masked(key, mask)?
9515 .into_iter()
9516 .filter(|e| !e.src_restricted && !e.dst_restricted)
9517 .map(|e| EdgeInfo {
9518 edge_type: e.edge_type,
9519 src_key: e.src_key,
9520 dst_key: e.dst_key,
9521 derived: e.derived,
9522 })
9523 .collect())
9524 }
9525
9526 /// Every directed edge incident on `key`, both directions, every etype.
9527 ///
9528 /// Walk is `topology.etypes()` × `{Out, In}` × `neighbors()`. `derived` is
9529 /// membership in [`RuleEngine::provenance_touching`] (O(degree) via the
9530 /// Plan-8 `by_node` index). Sorted by `(edge_type, src_key, dst_key)`.
9531 /// Unknown key → [`GraphError::KeyNotFound`].
9532 pub fn node_edges(&self, key: &str) -> Result<Vec<EdgeInfo>> {
9533 self.ensure_v8_base_sections_loaded();
9534 let id = self
9535 .ids
9536 .get(key)
9537 .ok_or_else(|| GraphError::KeyNotFound { key: key.into() })?;
9538 let derived: BTreeSet<(u32, u32, u32)> = self
9539 .engine
9540 .provenance_touching(id)
9541 .map(|(_rule, etype, src, dst)| (etype, src, dst))
9542 .collect();
9543 let mut edges = Vec::new();
9544 let tv = self.topo_view();
9545 for etype in tv.etypes() {
9546 // Same guard as node_edges_masked: etype from unchecked-CRC CSR.
9547 let edge_type = self
9548 .syms
9549 .resolve(etype)
9550 .ok_or_else(|| GraphError::Corrupt {
9551 detail: format!("v8: topology etype {etype} not in interner"),
9552 })?
9553 .to_string();
9554 for dir in [Direction::Out, Direction::In] {
9555 for &nbr in tv.neighbors(etype, dir, id).as_ref() {
9556 let (src, dst, src_key, dst_key) = match dir {
9557 Direction::Out => (
9558 id,
9559 nbr,
9560 key.to_string(),
9561 self.ids
9562 .key_of(nbr)
9563 .ok_or_else(|| GraphError::Corrupt {
9564 detail: format!("topology id {nbr} has no key"),
9565 })?
9566 .to_string(),
9567 ),
9568 Direction::In => (
9569 nbr,
9570 id,
9571 self.ids
9572 .key_of(nbr)
9573 .ok_or_else(|| GraphError::Corrupt {
9574 detail: format!("topology id {nbr} has no key"),
9575 })?
9576 .to_string(),
9577 key.to_string(),
9578 ),
9579 };
9580 edges.push(EdgeInfo {
9581 edge_type: edge_type.clone(),
9582 src_key,
9583 dst_key,
9584 derived: derived.contains(&(etype, src, dst)),
9585 });
9586 }
9587 }
9588 }
9589 edges.sort_by(|a, b| {
9590 a.edge_type
9591 .cmp(&b.edge_type)
9592 .then(a.src_key.cmp(&b.src_key))
9593 .then(a.dst_key.cmp(&b.dst_key))
9594 });
9595 // Self-loops appear in both Out and In; sort makes the pair adjacent
9596 // (sort key matches PartialEq for this case) so one pass drops the dup.
9597 edges.dedup();
9598 Ok(edges)
9599 }
9600
9601 // ── Backup ────────────────────────────────────────────────────────────────
9602
9603 /// Copy this store to `dest` as a consistent, verified snapshot.
9604 ///
9605 /// Copies every durable file in the database directory — `snapshot.bin`,
9606 /// `wal.bin`, all `wal.<N>.archive` files, `wal.floor`, `wal.genesis`, and
9607 /// `roles.json` — into a freshly created `dest` directory using OS-level
9608 /// `copy` calls (no large in-process buffers).
9609 ///
9610 /// # Consistency guarantee
9611 ///
9612 /// The guarantee is **process-local**: the caller holds `&self`, which
9613 /// prevents any concurrent writer in the **same process** from modifying
9614 /// the files during the copy. Running `mushroomdb backup` against a
9615 /// directory that is **concurrently being written by another process** (e.g.
9616 /// `mushroomdb serve`) is **unsafe** — the copy can be torn. The post-copy
9617 /// `verified: true` result reduces but does not eliminate the risk of a
9618 /// silent corrupt backup (CRC catches many bit-flips; it cannot catch a
9619 /// consistent mid-write snapshot).
9620 ///
9621 /// **The safe path for a live-served store is `POST /backup` on the HTTP
9622 /// server.** That handler acquires the read lock on the shared database
9623 /// before calling this method, which is the correct cross-process
9624 /// synchronisation point because the server is the single process writing
9625 /// the files.
9626 ///
9627 /// After copying, opens the destination read-only and runs the CRC section
9628 /// verifier (`verify_snapshot`) to confirm byte-for-byte integrity.
9629 /// `BackupReport::verified` reflects whether both checks passed.
9630 ///
9631 /// Returns `Err` when `self` is not backed by a `RealFs` (e.g. `SimFs`).
9632 pub fn backup_to(&self, dest: &std::path::Path) -> Result<BackupReport> {
9633 // Derive source directory from snapshot_path (RealFs only).
9634 let src_dir = match self.fs.snapshot_path() {
9635 Some(p) => p.parent().map(|d| d.to_path_buf()).ok_or_else(|| {
9636 GraphError::Io(std::io::Error::other("snapshot has no parent dir"))
9637 })?,
9638 None => {
9639 return Err(GraphError::Io(std::io::Error::other(
9640 "backup_to requires a real filesystem (RealFs)",
9641 )))
9642 }
9643 };
9644
9645 std::fs::create_dir_all(dest)?;
9646
9647 let mut files: Vec<String> = Vec::new();
9648 let mut bytes: u64 = 0;
9649
9650 // Helper: copy src_dir/name → dest/name if the file exists.
9651 let mut try_copy = |name: &str| -> std::io::Result<()> {
9652 let src_path = src_dir.join(name);
9653 if src_path.exists() {
9654 let n = std::fs::copy(&src_path, dest.join(name))?;
9655 bytes += n;
9656 files.push(name.to_string());
9657 }
9658 Ok(())
9659 };
9660
9661 try_copy("snapshot.bin")?;
9662 try_copy("snapshot.bin.bak")?;
9663 try_copy("wal.bin")?;
9664 try_copy("wal.floor")?;
9665 try_copy("wal.genesis")?;
9666 try_copy("roles.json")?;
9667
9668 // Copy WAL archives.
9669 let archives = self.fs.list_archives()?;
9670 for n in &archives {
9671 let name = format!("wal.{n}.archive");
9672 let n_bytes = std::fs::copy(src_dir.join(&name), dest.join(&name))?;
9673 bytes += n_bytes;
9674 files.push(name);
9675 }
9676
9677 files.sort();
9678
9679 // Post-copy verification: open dest and run CRC checks.
9680 let snap_in_dest = dest.join("snapshot.bin").exists();
9681 let crc_ok = if snap_in_dest {
9682 crate::verify_snapshot(dest)
9683 .map(|results| results.iter().all(|(_, _, _, r)| r.is_ok()))
9684 .unwrap_or(false)
9685 } else {
9686 true // WAL-only store: nothing to CRC-check in snapshot
9687 };
9688 let opens_ok = GraphDb::<core_storage::fs::RealFs>::open(dest).is_ok();
9689 let verified = crc_ok && opens_ok;
9690
9691 Ok(BackupReport {
9692 files,
9693 bytes,
9694 verified,
9695 })
9696 }
9697
9698 // ── Export helpers ────────────────────────────────────────────────────────
9699
9700 /// All live nodes, sorted by key (deterministic).
9701 ///
9702 /// Reads base + WAL overlay. Tombstoned nodes are excluded.
9703 pub fn all_nodes_for_export(&self) -> Vec<NodeInfo> {
9704 self.ensure_v8_base_sections_loaded();
9705 let pv = self.props_view();
9706 let mut nodes = Vec::new();
9707 for id in 0..self.ids.len() as u32 {
9708 let Some(key) = self.ids.key_of(id) else {
9709 continue;
9710 };
9711 let Some(&sym) = self.labels.get(id as usize) else {
9712 continue;
9713 };
9714 if sym == u32::MAX {
9715 continue; // tombstoned
9716 }
9717 let Some(label) = self.syms.resolve(sym) else {
9718 continue;
9719 };
9720 let mut props = BTreeMap::new();
9721 for field in pv.field_names() {
9722 if let Some(vr) = pv.get(id, &field) {
9723 props.insert(field, vr.into_value());
9724 }
9725 }
9726 nodes.push(NodeInfo {
9727 key: key.to_string(),
9728 label: label.to_string(),
9729 props,
9730 });
9731 }
9732 nodes.sort_by(|a, b| a.key.cmp(&b.key));
9733 nodes
9734 }
9735
9736 /// All directed edges, sorted by `(edge_type, src, dst)`. Each edge appears once.
9737 ///
9738 /// Derived edges carry `derived: true` and the creating rule's name in `rule`.
9739 /// Manual edges carry `derived: false` and `rule: None`.
9740 /// `weight` is the creating rule's `weight_prop` value read off the edge
9741 /// (numeric only), mirroring the convention used by [`GraphDb::explain`]
9742 /// and [`GraphDb::weighted_edges`]. Deterministic across runs on the same
9743 /// store state.
9744 pub fn all_edges_for_export(&self) -> Vec<ExportEdge> {
9745 self.ensure_v8_base_sections_loaded();
9746
9747 // Build (etype_sym, src_id, dst_id) → rule_name for O(1) derivation lookup.
9748 let mut prov: HashMap<(u32, u32, u32), String> = HashMap::new();
9749 for (rule_name, triples) in self.engine.provenance() {
9750 for &(etype, src, dst) in triples {
9751 prov.insert((etype, src, dst), rule_name.clone());
9752 }
9753 }
9754
9755 // rule_name → weight_prop, for O(1) lookup per derived edge.
9756 let weight_props: HashMap<&str, Option<&str>> = self
9757 .engine
9758 .rules()
9759 .map(|r| (r.name.as_str(), r.weight_prop.as_deref()))
9760 .collect();
9761
9762 let tv = self.topo_view();
9763 let ep = self.edge_props_view();
9764 let mut edges = Vec::new();
9765
9766 for id in 0..self.ids.len() as u32 {
9767 let Some(key) = self.ids.key_of(id) else {
9768 continue;
9769 };
9770 let Some(&lsym) = self.labels.get(id as usize) else {
9771 continue;
9772 };
9773 if lsym == u32::MAX {
9774 continue; // tombstoned
9775 }
9776
9777 for etype_sym in tv.etypes() {
9778 // etype from archived CSR (access_unchecked, no eager CRC).
9779 // Skip edges whose etype is not in the interner; this can only
9780 // occur with a corrupt large TOPOLOGY section (bit-flip on an
9781 // etype field in the archived data). The function returns Vec,
9782 // not Result, so we continue rather than propagate.
9783 let Some(edge_type) = self.syms.resolve(etype_sym) else {
9784 continue;
9785 };
9786 let edge_type = edge_type.to_string();
9787 for &nbr in tv.neighbors(etype_sym, Direction::Out, id).as_ref() {
9788 let Some(dst_key) = self.ids.key_of(nbr) else {
9789 continue; // skip corrupt entries
9790 };
9791 let prov_key = (etype_sym, id, nbr);
9792 let rule = prov.get(&prov_key).cloned();
9793 let derived = rule.is_some();
9794 let weight = rule
9795 .as_deref()
9796 .and_then(|rn| weight_props.get(rn).copied().flatten())
9797 .and_then(|prop| match ep.get(etype_sym, id, nbr, prop) {
9798 Some(Value::Float(f)) => Some(f),
9799 Some(Value::Int(i)) => Some(i as f64),
9800 _ => None,
9801 });
9802 edges.push(ExportEdge {
9803 edge_type: edge_type.clone(),
9804 src: key.to_string(),
9805 dst: dst_key.to_string(),
9806 derived,
9807 rule,
9808 weight,
9809 });
9810 }
9811 }
9812 }
9813
9814 edges.sort_by(|a, b| {
9815 a.edge_type
9816 .cmp(&b.edge_type)
9817 .then(a.src.cmp(&b.src))
9818 .then(a.dst.cmp(&b.dst))
9819 });
9820 edges
9821 }
9822
9823 /// What each edge type *is*, without building one record per edge.
9824 ///
9825 /// [`all_edges_for_export`](Self::all_edges_for_export) answers the same
9826 /// question by materialising every edge — three `String`s apiece, a
9827 /// provenance `HashMap` over every derived edge, and a final sort. That is
9828 /// the right shape for an export, and the wrong one for a summary: on a
9829 /// store with 1.3 M derived edges it allocates hundreds of megabytes to
9830 /// produce nine lines. This walks the topology instead, summing neighbour
9831 /// slice lengths and collecting *label symbols* rather than label strings,
9832 /// so the per-edge cost is an integer add and a set insert on a set with
9833 /// as many members as the store has labels.
9834 ///
9835 /// The rule names come off the rule *definitions*, which each declare the
9836 /// `edge_type` they derive, so naming them costs one pass over the rules
9837 /// rather than one provenance lookup per edge. That is also why `rules`
9838 /// is a list: two rules may derive the same type — the association store
9839 /// derives `INDUSTRY_ALIGNMENT` from both a talent→company and a
9840 /// talent→job rule — and naming only one of them would be a half-truth.
9841 /// A type with no rules is one written by hand.
9842 ///
9843 /// `sample` is the first edge of the type in the store's own id order,
9844 /// which is insertion order: deterministic for a given store, and not the
9845 /// same as key order, which cannot be had without resolving a key per
9846 /// edge. Sorted by `edge_type`.
9847 pub fn edge_type_census(&self) -> Vec<EdgeTypeCensus> {
9848 self.ensure_v8_base_sections_loaded();
9849
9850 let mut rules_by_type: BTreeMap<&str, BTreeSet<&str>> = BTreeMap::new();
9851 for r in self.engine.rules() {
9852 rules_by_type
9853 .entry(r.edge_type.as_str())
9854 .or_default()
9855 .insert(r.name.as_str());
9856 }
9857
9858 let tv = self.topo_view();
9859 let node_count = self.ids.len() as u32;
9860 let mut out = Vec::new();
9861 for etype_sym in tv.etypes() {
9862 // An etype the interner cannot resolve means a corrupt TOPOLOGY
9863 // section; skip it rather than name it, as `all_edges_for_export`
9864 // does for the same reason.
9865 let Some(edge_type) = self.syms.resolve(etype_sym) else {
9866 continue;
9867 };
9868 let mut edges: u64 = 0;
9869 let mut src_syms: BTreeSet<u32> = BTreeSet::new();
9870 let mut dst_syms: BTreeSet<u32> = BTreeSet::new();
9871 let mut sample: Option<(u32, u32)> = None;
9872 for id in 0..node_count {
9873 let Some(&lsym) = self.labels.get(id as usize) else {
9874 continue;
9875 };
9876 if lsym == u32::MAX {
9877 continue; // tombstoned
9878 }
9879 let nbrs = tv.neighbors(etype_sym, Direction::Out, id);
9880 let nbrs = nbrs.as_ref();
9881 if nbrs.is_empty() {
9882 continue;
9883 }
9884 edges += nbrs.len() as u64;
9885 src_syms.insert(lsym);
9886 for &nbr in nbrs {
9887 if let Some(&dsym) = self.labels.get(nbr as usize) {
9888 if dsym != u32::MAX {
9889 dst_syms.insert(dsym);
9890 }
9891 }
9892 }
9893 if sample.is_none() {
9894 sample = Some((id, nbrs[0]));
9895 }
9896 }
9897 let resolve = |syms: &BTreeSet<u32>| -> Vec<String> {
9898 syms.iter()
9899 .filter_map(|&s| self.syms.resolve(s))
9900 .map(ToString::to_string)
9901 .collect()
9902 };
9903 out.push(EdgeTypeCensus {
9904 edge_type: edge_type.to_string(),
9905 edges,
9906 src_labels: resolve(&src_syms),
9907 dst_labels: resolve(&dst_syms),
9908 rules: rules_by_type
9909 .get(edge_type)
9910 .map(|rs| rs.iter().map(ToString::to_string).collect())
9911 .unwrap_or_default(),
9912 sample: sample.and_then(|(s, d)| {
9913 Some((
9914 self.ids.key_of(s)?.to_string(),
9915 self.ids.key_of(d)?.to_string(),
9916 ))
9917 }),
9918 });
9919 }
9920 out.sort_by(|a, b| a.edge_type.cmp(&b.edge_type));
9921 out
9922 }
9923
9924 /// All directed edges of `edge_type`, with the raw value of `weight_prop`
9925 /// on each edge when given.
9926 ///
9927 /// `weight` is `Some(f)` only when `weight_prop` is set and the edge
9928 /// carries that property with a numeric (`Int`/`Float`) value; otherwise
9929 /// `None` — callers that want a default weight (e.g. `1.0` for missing
9930 /// props) apply it themselves, matching the convention used internally
9931 /// by [`GraphDb::pagerank`], [`GraphDb::connected_components`],
9932 /// [`GraphDb::degree_centrality`], and [`GraphDb::communities`].
9933 ///
9934 /// Sorted by `(src, dst)` for determinism. Reads the unified topology
9935 /// (manual + rule-derived edges). An unknown `edge_type` returns an
9936 /// empty vec.
9937 pub fn weighted_edges(
9938 &self,
9939 edge_type: &str,
9940 weight_prop: Option<&str>,
9941 ) -> Vec<(String, String, Option<f64>)> {
9942 let Some(etype_sym) = self.syms.get(edge_type) else {
9943 return Vec::new();
9944 };
9945 let tv = self.topo_view();
9946 let ep = self.edge_props_view();
9947 let mut out = Vec::new();
9948 for id in 0..self.ids.len() as u32 {
9949 let Some(key) = self.ids.key_of(id) else {
9950 continue;
9951 };
9952 let Some(&sym) = self.labels.get(id as usize) else {
9953 continue;
9954 };
9955 if sym == u32::MAX {
9956 continue; // tombstoned
9957 }
9958 for &nbr in tv.neighbors(etype_sym, Direction::Out, id).as_ref() {
9959 let Some(dst_key) = self.ids.key_of(nbr) else {
9960 continue;
9961 };
9962 let weight = weight_prop.and_then(|prop| match ep.get(etype_sym, id, nbr, prop) {
9963 Some(Value::Float(f)) => Some(f),
9964 Some(Value::Int(i)) => Some(i as f64),
9965 _ => None,
9966 });
9967 out.push((key.to_string(), dst_key.to_string(), weight));
9968 }
9969 }
9970 out.sort_by(|a, b| a.0.cmp(&b.0).then(a.1.cmp(&b.1)));
9971 out
9972 }
9973
9974 pub fn nodes_with_label(&self, label: &str) -> Vec<NodeRef<'_, F>> {
9975 self.view()
9976 .nodes_with_label(label)
9977 .into_iter()
9978 .map(|id| NodeRef { db: self, id })
9979 .collect()
9980 }
9981
9982 pub fn find_nodes(&self, label: &str, filter: &Filter) -> Vec<NodeRef<'_, F>> {
9983 let view = self.view();
9984 view.nodes_with_label(label)
9985 .into_iter()
9986 .filter(|&id| {
9987 eval_filter(filter, &|field| {
9988 view.prop(id, field).map(|vr| vr.into_value())
9989 })
9990 })
9991 .map(|id| NodeRef { db: self, id })
9992 .collect()
9993 }
9994
9995 /// Returns `true` if any approximate (HNSW) VectorSimilar rule covers
9996 /// `field`. Use as a capability probe: when `true`, `find_similar_vector`
9997 /// with `label = None` will use the native ANN path rather than the O(n)
9998 /// brute-force scan.
9999 pub fn has_vector_rule(&self, field: &str) -> bool {
10000 self.engine.hnsw_has_rule(field)
10001 }
10002
10003 /// How many HNSW graphs this handle has built from scratch since it was
10004 /// opened (one per side of an approximate rule).
10005 ///
10006 /// An open that restored every graph from the snapshot reports `0`.
10007 /// Exposed for tests that assert the open path reuses the persisted index
10008 /// rather than rebuilding it; not part of the stable surface.
10009 #[doc(hidden)]
10010 pub fn hnsw_build_count(&self) -> u64 {
10011 self.engine.hnsw_build_count()
10012 }
10013
10014 /// How many rules this handle still holds a lazily-decoded HNSW graph for.
10015 ///
10016 /// Zero before the first ANN query on a clean open, and again once the
10017 /// live indexes own the graphs. See [`core_rules::RuleEngine::lazy_hnsw_len`].
10018 /// Exposed for tests that assert the lazy copies are released; not part of
10019 /// the stable surface.
10020 #[doc(hidden)]
10021 pub fn lazy_hnsw_len(&self) -> usize {
10022 self.engine.lazy_hnsw_len()
10023 }
10024
10025 /// Find nodes whose `field` vector is most similar to `q` (cosine
10026 /// similarity), returning up to `k` results with similarity ≥ `min`,
10027 /// sorted descending.
10028 ///
10029 /// When `label` is `None` the search spans all labels (via
10030 /// `hnsw_search_any_dst` or a full brute-force scan); when `label` is
10031 /// `Some(lbl)` it restricts to nodes with that label.
10032 ///
10033 /// Uses the HNSW index when one is available (fast path); otherwise falls
10034 /// back to an O(n) brute-force scan.
10035 ///
10036 /// **The index supplies candidates, never scores.** Its own distances are
10037 /// `f32` (accurate to ~1e-6, so an exact duplicate scores 0.9999999), so
10038 /// every candidate is re-scored from the `f64` property vectors by
10039 /// [`exact_vector_similarity`] before `min`, the ordering and the reported
10040 /// score are decided. `k + VECTOR_RESCORE_MARGIN` candidates are fetched so
10041 /// the re-ordering cannot drop a true top-`k` member; see that constant for
10042 /// the rule. The score a caller receives is therefore the same number the
10043 /// brute-force path would have produced, to `f64` precision, and `min = 1.0`
10044 /// finds an exact duplicate.
10045 pub fn find_similar_vector(
10046 &self,
10047 field: &str,
10048 label: Option<&str>,
10049 q: &[f64],
10050 k: usize,
10051 min: f64,
10052 ) -> Vec<(String, f64)> {
10053 self.find_similar_vector_filtered(field, label, q, k, min, None, None, false)
10054 .expect("find_similar_vector_filtered is infallible without where_")
10055 }
10056
10057 /// Like [`find_similar_vector`] but restricts results to nodes visible in
10058 /// `mask`. Hidden nodes never appear in results; the mask is applied
10059 /// **before** k-truncation so a caller still receives up to `k` visible
10060 /// hits.
10061 ///
10062 /// # HNSW path (widening beam)
10063 ///
10064 /// When an HNSW index covers the request, the beam starts at an over-fetch
10065 /// of `k × n / |visible|` (plus the rescore margin) when the mask's
10066 /// selectivity is known from the index length, otherwise at `k` plus that
10067 /// margin. If fewer than `k` visible candidates remain after the mask and
10068 /// `min` filter, the beam doubles — the same ×2 loop exact `VectorSimilar`
10069 /// rules use, capped at `ef_max()` (`EF_MAX` = 4,096). Reaching the cap,
10070 /// or a beam that comes back short of its own width, falls through to the
10071 /// exhaustive masked scan rather than returning a short result.
10072 ///
10073 /// Every surviving candidate is re-scored from the `f64` property vectors,
10074 /// exactly as [`find_similar_vector`] does and for the same reason.
10075 ///
10076 /// # Brute-force path
10077 ///
10078 /// When no HNSW index covers the request, or the beam cannot admit `k`
10079 /// hits, the function builds a masked [`GraphView`] so that `nodes_all` /
10080 /// `nodes_with_label` return only visible nodes, guaranteeing exact `k`
10081 /// results (or all visible nodes if fewer than `k` exist).
10082 pub fn find_similar_vector_masked(
10083 &self,
10084 field: &str,
10085 label: Option<&str>,
10086 q: &[f64],
10087 k: usize,
10088 min: f64,
10089 mask: &crate::mask::NodeMask,
10090 ) -> Vec<(String, f64)> {
10091 self.find_similar_vector_filtered(field, label, q, k, min, Some(mask), None, false)
10092 .expect("find_similar_vector_filtered is infallible without where_")
10093 }
10094
10095 /// Exact or ANN kNN with optional key-list `mask` and property `where_`.
10096 ///
10097 /// `where_` present and failing [`PropPredicate::validate_named`] `"where"`
10098 /// → `QueryError`. `exact=true` or `where_=Some` skip HNSW and GEMM-brute
10099 /// the candidate set (`label ∩ mask ∩ holds(where)`). `mask` alone still
10100 /// uses HNSW when an index covers the field.
10101 #[allow(clippy::too_many_arguments)]
10102 pub fn find_similar_vector_filtered(
10103 &self,
10104 field: &str,
10105 label: Option<&str>,
10106 q: &[f64],
10107 k: usize,
10108 min: f64,
10109 mask: Option<&crate::mask::NodeMask>,
10110 where_: Option<&PropPredicate>,
10111 exact: bool,
10112 ) -> Result<Vec<(String, f64)>> {
10113 self.find_similar_vector_as(
10114 field,
10115 label,
10116 q,
10117 k,
10118 min,
10119 mask,
10120 where_,
10121 exact,
10122 ExactnessCaller::Vector,
10123 )
10124 }
10125
10126 /// [`find_similar_vector_filtered`](Self::find_similar_vector_filtered)
10127 /// with the caller shape named, so the exactness warning can advise the
10128 /// signature that actually reached it. Everything else is identical.
10129 #[allow(clippy::too_many_arguments)]
10130 fn find_similar_vector_as(
10131 &self,
10132 field: &str,
10133 label: Option<&str>,
10134 q: &[f64],
10135 k: usize,
10136 min: f64,
10137 mask: Option<&crate::mask::NodeMask>,
10138 where_: Option<&PropPredicate>,
10139 exact: bool,
10140 caller: ExactnessCaller,
10141 ) -> Result<Vec<(String, f64)>> {
10142 if let Some(pred) = where_ {
10143 pred.validate_named("where")
10144 .map_err(|detail| GraphError::QueryError { detail })?;
10145 }
10146
10147 // Ensure any HNSW blobs retained from the snapshot are deserialized
10148 // before the first ANN query on a clean-open (no-WAL) path. The
10149 // section read has to come first: on a clean open nothing else has
10150 // called it, so without it `retained_hnsw_blobs` is empty,
10151 // `ensure_hnsw_loaded` caches an empty map in its `OnceLock`, and every
10152 // approximate query on the handle runs brute force — correct results,
10153 // silently off the index. Both calls are idempotent and cheap once hot.
10154 self.ensure_v8_base_sections_loaded();
10155 self.engine.ensure_hnsw_loaded();
10156 let norm: f64 = q.iter().map(|x| x * x).sum::<f64>().sqrt();
10157 if norm == 0.0 {
10158 return Ok(vec![]);
10159 }
10160 if let Some(m) = mask {
10161 if k == 0 || m.is_empty() {
10162 return Ok(vec![]);
10163 }
10164 }
10165 let q_unit: Vec<f64> = q.iter().map(|x| x / norm).collect();
10166
10167 // `where` implies exact: a predicate must not ride a silent ANN.
10168 let skip_hnsw = exact || where_.is_some();
10169 if !skip_hnsw {
10170 if let Some(mask) = mask {
10171 if let Some(out) =
10172 self.find_similar_hnsw_masked(field, label, &q_unit, k, min, mask, caller)
10173 {
10174 return Ok(out);
10175 }
10176 } else if let Some(out) = self.find_similar_hnsw(field, label, &q_unit, k, min) {
10177 return Ok(out);
10178 }
10179 }
10180
10181 let view = match mask {
10182 Some(m) => self.view_masked(m),
10183 None => self.view(),
10184 };
10185 let candidate_ids = Self::vector_candidates(&view, label, where_);
10186 Ok(self.brute_vector_hits(&view, candidate_ids, field, &q_unit, k, min))
10187 }
10188
10189 /// Unmasked HNSW path. `None` when no populated index covers the request.
10190 fn find_similar_hnsw(
10191 &self,
10192 field: &str,
10193 label: Option<&str>,
10194 q_unit: &[f64],
10195 k: usize,
10196 min: f64,
10197 ) -> Option<Vec<(String, f64)>> {
10198 // Try HNSW fast path.
10199 // `None` label searches across all VectorSimilar rules covering `field`
10200 // (merging their results); `Some(lbl)` restricts to rules whose
10201 // dst_label matches. Returns `None` when no populated HNSW index
10202 // covers the request — the O(n) brute-force fallback handles that case.
10203 let over_k = k.saturating_add(VECTOR_RESCORE_MARGIN);
10204 let hits = match label {
10205 Some(lbl) => self.engine.hnsw_search_dst(field, lbl, q_unit, over_k)?,
10206 None => self.engine.hnsw_search_any_dst(field, q_unit, over_k)?,
10207 };
10208 // Candidates only: the index's `f32` similarity is discarded and
10209 // each hit is re-scored against the `f64` vectors.
10210 let view = self.view();
10211 let mut out: Vec<(String, f64)> = hits
10212 .into_iter()
10213 .filter_map(|(id, _)| {
10214 let sim = exact_vector_similarity(&view, id, field, q_unit)?;
10215 if sim < min {
10216 return None;
10217 }
10218 Some((self.ids.key_of(id)?.to_string(), sim))
10219 })
10220 .collect();
10221 out.sort_by(|a, b| {
10222 b.1.partial_cmp(&a.1)
10223 .unwrap_or(std::cmp::Ordering::Equal)
10224 .then_with(|| a.0.cmp(&b.0))
10225 });
10226 out.truncate(k);
10227 Some(out)
10228 }
10229
10230 /// Masked HNSW widening beam. `None` when no index covers the request or
10231 /// the beam cannot admit `k` visible hits (caller falls through to brute).
10232 #[allow(clippy::too_many_arguments)]
10233 fn find_similar_hnsw_masked(
10234 &self,
10235 field: &str,
10236 label: Option<&str>,
10237 q_unit: &[f64],
10238 k: usize,
10239 min: f64,
10240 mask: &crate::mask::NodeMask,
10241 caller: ExactnessCaller,
10242 ) -> Option<Vec<(String, f64)>> {
10243 let index_len = match label {
10244 Some(lbl) => self.engine.hnsw_dst_len(field, lbl, q_unit.len()),
10245 None => self.engine.hnsw_any_dst_len(field, q_unit.len()),
10246 };
10247 let n = index_len?;
10248 // The `?` above is the coverage test: past it, an index exists and this
10249 // masked, non-exact call is about to ride it.
10250 self.note_ambiguous_exactness(field, label, caller);
10251 // Same ceiling the exact-rule widening loop in `hnsw_candidates`
10252 // consults — including the `with_ef_max` test hook.
10253 let cap = ef_max();
10254 let visible = mask.len();
10255 let mut ef = k.saturating_add(VECTOR_RESCORE_MARGIN);
10256 if visible > 0 && n > 0 {
10257 let over = k
10258 .saturating_mul(n)
10259 .div_ceil(visible)
10260 .saturating_add(VECTOR_RESCORE_MARGIN);
10261 ef = ef.max(over);
10262 }
10263 loop {
10264 let hits = match label {
10265 Some(lbl) => self
10266 .engine
10267 .hnsw_search_dst_with_ef(field, lbl, q_unit, ef, ef),
10268 None => self
10269 .engine
10270 .hnsw_search_any_dst_with_ef(field, q_unit, ef, ef),
10271 };
10272 let hits = hits?;
10273 let full = hits.len() == ef;
10274 let mut out = self.score_masked_hnsw_hits(&hits, field, q_unit, min, mask);
10275 if out.len() >= k {
10276 out.truncate(k);
10277 return Some(out);
10278 }
10279 // Short of its width (frontier exhausted) or at the ceiling:
10280 // a wider beam reaches nothing new, so the scan answers.
10281 if !full || ef >= cap {
10282 return None;
10283 }
10284 ef = ef.saturating_mul(2);
10285 }
10286 }
10287
10288 /// Say once, per `(field, label)` index and caller shape, that a masked
10289 /// search is answering approximately.
10290 ///
10291 /// A mask narrows *which nodes may be returned*. It does not choose a
10292 /// kernel — `exact=true` and a `where=` predicate do, and nothing else
10293 /// does. A caller who needed exact answers, passed `mask=` alone, and read
10294 /// the mask as a promise of exhaustiveness gets a correct-looking
10295 /// approximate answer and no signal at all; that is a silent wrong answer,
10296 /// and it has cost an integration team real time.
10297 ///
10298 /// The fix is a question, not a behaviour change. Making a mask imply
10299 /// `exact` would turn every existing masked caller's ANN into an O(n) GEMM
10300 /// without asking them, which is a worse trade than the ambiguity.
10301 ///
10302 /// Printed once per index for the reason the dimension-mismatch skip in
10303 /// `core_rules::hnsw` is: a line on every call is a line callers learn to
10304 /// scroll past.
10305 ///
10306 /// `caller` decides the advice. The same leg is reached from two signatures
10307 /// and only one of them has an `exact` argument to pass; see
10308 /// [`ExactnessCaller`].
10309 fn note_ambiguous_exactness(&self, field: &str, label: Option<&str>, caller: ExactnessCaller) {
10310 let entry = (field.to_string(), label.unwrap_or("").to_string(), caller);
10311 let first = match self.warned_ambiguous_exactness.lock() {
10312 Ok(mut seen) => seen.insert(entry),
10313 Err(poisoned) => poisoned.into_inner().insert(entry),
10314 };
10315 if !first {
10316 return;
10317 }
10318 AMBIGUOUS_EXACTNESS_WARNS.with(|c| c.set(c.get().saturating_add(1)));
10319 let which = match label {
10320 Some(lbl) => format!(" (label `{lbl}`)"),
10321 None => String::new(),
10322 };
10323 let subject = caller.subject();
10324 let advice = caller.advice();
10325 let line = format!(
10326 "mushroomdb: {subject} on field `{field}`{which} is answering \
10327 approximately. A mask narrows which nodes may be returned; it does not \
10328 change which kernel runs, and an index covers this field. For an exact \
10329 answer over the same visible candidate set, {advice} Further masked \
10330 searches of this shape on this index are silent."
10331 );
10332 AMBIGUOUS_EXACTNESS_LAST.with(|c| *c.borrow_mut() = Some(line.clone()));
10333 eprintln!("{line}");
10334 }
10335
10336 /// `label ∩ mask ∩ holds(where)`. Index fast path when `label` is `Some`
10337 /// and `(label, where.field)` is enabled; otherwise scan with `visible()`.
10338 fn vector_candidates(
10339 view: &GraphView<'_>,
10340 label: Option<&str>,
10341 where_: Option<&PropPredicate>,
10342 ) -> Vec<u32> {
10343 if let (Some(lbl), Some(pred)) = (label, where_) {
10344 let indexed = view
10345 .prop_index
10346 .is_some_and(|idx| idx.is_enabled(lbl, &pred.field));
10347 if indexed {
10348 match (&pred.eq, &pred.in_) {
10349 (Some(eq), None) => {
10350 if let Some(ids) = view.nodes_with_prop(lbl, &pred.field, eq) {
10351 return ids;
10352 }
10353 }
10354 (None, Some(allowed)) => {
10355 let mut seen = HashSet::new();
10356 let mut out = Vec::new();
10357 for v in allowed {
10358 if let Some(ids) = view.nodes_with_prop(lbl, &pred.field, v) {
10359 for id in ids {
10360 if seen.insert(id) {
10361 out.push(id);
10362 }
10363 }
10364 }
10365 }
10366 return out;
10367 }
10368 _ => {}
10369 }
10370 }
10371 }
10372
10373 let mut ids: Vec<u32> = match label {
10374 Some(lbl) => view
10375 .nodes_with_label(lbl)
10376 .into_iter()
10377 .filter(|&id| view.visible(id))
10378 .collect(),
10379 None => view.nodes_all(),
10380 };
10381 if let Some(pred) = where_ {
10382 ids.retain(|&id| match view.prop(id, &pred.field) {
10383 None => pred.holds(None),
10384 Some(vr) => pred.holds(Some(vr.as_value())),
10385 });
10386 }
10387 ids
10388 }
10389
10390 /// Exact brute kNN: pack candidates at `q_unit`'s dim, GEMV, keep
10391 /// `score >= min`, sort `(sim desc, key asc)`, truncate to `k`.
10392 fn brute_vector_hits(
10393 &self,
10394 view: &GraphView<'_>,
10395 candidate_ids: impl IntoIterator<Item = u32>,
10396 field: &str,
10397 q_unit: &[f64],
10398 k: usize,
10399 min: f64,
10400 ) -> Vec<(String, f64)> {
10401 let rows: Vec<(u32, std::borrow::Cow<'_, [f64]>)> = candidate_ids
10402 .into_iter()
10403 .filter_map(|id| crate::exact_knn::vector_f64(view, id, field).map(|v| (id, v)))
10404 .collect();
10405 let packed =
10406 crate::exact_knn::pack(rows.iter().map(|(id, v)| (*id, v.as_ref())), q_unit.len());
10407 let scores = crate::exact_knn::gemv(&packed, q_unit);
10408 let mut scored: Vec<(String, f64)> = packed
10409 .ids
10410 .iter()
10411 .zip(scores.iter())
10412 .filter_map(|(&id, &sim)| {
10413 if sim < min {
10414 return None;
10415 }
10416 let key = self.ids.key_of(id)?.to_string();
10417 Some((key, sim))
10418 })
10419 .collect();
10420 scored.sort_by(|a, b| {
10421 b.1.partial_cmp(&a.1)
10422 .unwrap_or(std::cmp::Ordering::Equal)
10423 .then_with(|| a.0.cmp(&b.0))
10424 });
10425 scored.truncate(k);
10426 scored
10427 }
10428
10429 /// Exact cosine top-k for each key in `keys`, scored only against `keys`.
10430 ///
10431 /// `min` is cosine similarity in [-1, 1], inclusive (`score >= min`), the
10432 /// same unit and inequality as `find_similar_vector`. Self-matches are
10433 /// excluded. Unknown keys, keys with no `field`, zero-norm or wrong-dim
10434 /// embeddings are omitted as both query and candidate. Duplicate keys are
10435 /// collapsed, first-seen order. Empty `keys` → empty `Ok(vec![])`. Never
10436 /// uses HNSW. `n > PAIRWISE_MAX_N` → `QueryError`.
10437 #[allow(clippy::type_complexity)]
10438 pub fn pairwise_similar(
10439 &self,
10440 keys: &[&str],
10441 field: &str,
10442 k: usize,
10443 min: f64,
10444 ) -> Result<Vec<(String, Vec<(String, f64)>)>> {
10445 let mut seen = HashSet::new();
10446 let mut unique_ids = Vec::new();
10447 for key in keys {
10448 let Some(id) = self.ids.get(key) else {
10449 continue;
10450 };
10451 if seen.insert(id) {
10452 unique_ids.push(id);
10453 }
10454 }
10455 let max_n = crate::exact_knn::pairwise_max_n();
10456 if unique_ids.len() > max_n {
10457 return Err(GraphError::QueryError {
10458 detail: format!(
10459 "pairwise_similar: n={} exceeds PAIRWISE_MAX_N ({max_n})",
10460 unique_ids.len()
10461 ),
10462 });
10463 }
10464 if unique_ids.is_empty() {
10465 return Ok(Vec::new());
10466 }
10467
10468 let view = self.view();
10469 let mut rows: Vec<(u32, std::borrow::Cow<'_, [f64]>)> = Vec::new();
10470 let mut counts: HashMap<usize, usize> = HashMap::new();
10471 for id in unique_ids {
10472 let Some(v) = crate::exact_knn::vector_f64(&view, id, field) else {
10473 continue;
10474 };
10475 let norm: f64 = v.iter().map(|x| x * x).sum::<f64>().sqrt();
10476 if norm == 0.0 {
10477 continue;
10478 }
10479 *counts.entry(v.len()).or_default() += 1;
10480 rows.push((id, v));
10481 }
10482 if rows.is_empty() {
10483 return Ok(Vec::new());
10484 }
10485 let dim = counts
10486 .into_iter()
10487 .max_by_key(|&(d, c)| (c, d))
10488 .map(|(d, _)| d)
10489 .expect("rows non-empty");
10490 let packed = crate::exact_knn::pack(rows.iter().map(|(id, v)| (*id, v.as_ref())), dim);
10491 let n = packed.ids.len();
10492 if n == 0 {
10493 return Ok(Vec::new());
10494 }
10495 let src_keys: Vec<String> = packed
10496 .ids
10497 .iter()
10498 .map(|&id| self.ids.key_of(id).unwrap_or("").to_string())
10499 .collect();
10500
10501 let mut out = Vec::with_capacity(n);
10502 if n <= crate::exact_knn::pairwise_gram_max() {
10503 let sims = crate::exact_knn::gram(&packed);
10504 for i in 0..n {
10505 out.push(Self::topk_from_row(
10506 &src_keys,
10507 i,
10508 &sims[i * n..(i + 1) * n],
10509 k,
10510 min,
10511 ));
10512 }
10513 } else {
10514 for i in 0..n {
10515 let row = &packed.data[i * packed.dim..(i + 1) * packed.dim];
10516 let scores = crate::exact_knn::gemv(&packed, row);
10517 out.push(Self::topk_from_row(&src_keys, i, &scores, k, min));
10518 }
10519 }
10520 Ok(out)
10521 }
10522
10523 /// [`pairwise_similar`](Self::pairwise_similar) over the keys the mask
10524 /// admits — intersected **before** the matmul, never filtered after it.
10525 ///
10526 /// A hidden vector packed into the Gram is a row every visible key is
10527 /// scored against. It can take a visible neighbour's place in the top-`k`,
10528 /// and because the packed dimension is a majority vote over the candidate
10529 /// rows it can decide whether a visible pair is scored at all. Dropping
10530 /// hidden names from the finished answer leaves both effects standing, so
10531 /// the intersection happens first and the answer is byte-for-byte the one
10532 /// `pairwise_similar` gives for the visible keys alone.
10533 ///
10534 /// The caps therefore measure the **post-filter** count: a key set over
10535 /// [`PAIRWISE_MAX_N`](crate::PAIRWISE_MAX_N) unscoped can come under it
10536 /// scoped and succeed, because the work the cap refuses is work this call
10537 /// no longer does. A filtered count still over the cap is still refused.
10538 #[allow(clippy::type_complexity)]
10539 pub fn pairwise_similar_scoped(
10540 &self,
10541 keys: &[&str],
10542 field: &str,
10543 k: usize,
10544 min: f64,
10545 mask: &crate::mask::NodeMask,
10546 ) -> Result<Vec<(String, Vec<(String, f64)>)>> {
10547 let visible: Vec<&str> = keys
10548 .iter()
10549 .copied()
10550 .filter(|key| mask.contains_node(self, key))
10551 .collect();
10552 self.pairwise_similar(&visible, field, k, min)
10553 }
10554
10555 /// Neighbours of packed row `i`: drop self, keep `score >= min`, sort
10556 /// `(sim desc, key asc)`, truncate to `k`. Packed srcs with no survivors
10557 /// still appear as `(src, [])`.
10558 fn topk_from_row(
10559 src_keys: &[String],
10560 i: usize,
10561 scores: &[f64],
10562 k: usize,
10563 min: f64,
10564 ) -> (String, Vec<(String, f64)>) {
10565 let mut neigh: Vec<(String, f64)> = scores
10566 .iter()
10567 .enumerate()
10568 .filter_map(|(j, &sim)| {
10569 if i == j || sim < min {
10570 return None;
10571 }
10572 Some((src_keys[j].clone(), sim))
10573 })
10574 .collect();
10575 neigh.sort_by(|a, b| {
10576 b.1.partial_cmp(&a.1)
10577 .unwrap_or(std::cmp::Ordering::Equal)
10578 .then_with(|| a.0.cmp(&b.0))
10579 });
10580 neigh.truncate(k);
10581 (src_keys[i].clone(), neigh)
10582 }
10583
10584 /// Re-score HNSW candidates from the `f64` vectors, drop hidden / below-`min`
10585 /// hits, order by score then key. The index's own `f32` similarity is discarded.
10586 fn score_masked_hnsw_hits(
10587 &self,
10588 hits: &[(u32, f64)],
10589 field: &str,
10590 q_unit: &[f64],
10591 min: f64,
10592 mask: &crate::mask::NodeMask,
10593 ) -> Vec<(String, f64)> {
10594 let view = self.view_masked(mask);
10595 let mut out: Vec<(String, f64)> = hits
10596 .iter()
10597 .copied()
10598 .filter(|&(id, _)| mask.contains_id(id))
10599 .filter_map(|(id, _)| {
10600 let sim = exact_vector_similarity(&view, id, field, q_unit)?;
10601 if sim < min {
10602 return None;
10603 }
10604 Some((self.ids.key_of(id)?.to_string(), sim))
10605 })
10606 .collect();
10607 out.sort_by(|a, b| {
10608 b.1.partial_cmp(&a.1)
10609 .unwrap_or(std::cmp::Ordering::Equal)
10610 .then_with(|| a.0.cmp(&b.0))
10611 });
10612 out
10613 }
10614
10615 /// Read a single property from an edge.
10616 ///
10617 /// Returns `None` when the edge does not exist, the field is absent, or any
10618 /// of the string keys cannot be resolved to interned ids. Only edge props
10619 /// written by rules (weight fields) are accessible without a `set_edge_prop`
10620 /// binding; topology-only edges (no props set) return `None` for every field.
10621 pub fn get_edge_prop(
10622 &self,
10623 edge_type: &str,
10624 src_key: &str,
10625 dst_key: &str,
10626 field: &str,
10627 ) -> Option<Value> {
10628 let etype = self.syms.get(edge_type)?;
10629 let src = self.ids.get(src_key)?;
10630 let dst = self.ids.get(dst_key)?;
10631 self.edge_props_view().get(etype, src, dst, field)
10632 }
10633
10634 /// Lex → parse → plan → execute `cypher` over a read-only view.
10635 /// Every pipeline `Err(String)` becomes `GraphError::QueryError` with a
10636 /// stage prefix (`lex:` / `parse:` / `plan:` / `execute:`).
10637 pub fn query(&self, cypher: &str, params: &BTreeMap<String, Value>) -> Result<ResultSet> {
10638 let tokens = lex(cypher).map_err(|e| GraphError::QueryError {
10639 detail: format!("lex: {e}"),
10640 })?;
10641 let union = parse_read(&tokens).map_err(|e| GraphError::QueryError {
10642 detail: format!("parse: {e}"),
10643 })?;
10644 let t0 = std::time::Instant::now();
10645 let result = execute_union(&self.view(), &union, &Params(params)).map_err(|e| {
10646 GraphError::QueryError {
10647 detail: format!("execute: {e}"),
10648 }
10649 });
10650 let elapsed_ms = t0.elapsed().as_millis() as u64;
10651 let threshold = self.slow_query_threshold_ms;
10652 if threshold > 0 && elapsed_ms >= threshold {
10653 eprintln!("[mushroomdb] slow query ({elapsed_ms}ms): {cypher}");
10654 let entry = SlowQueryEntry {
10655 ms: elapsed_ms,
10656 query: cypher.to_string(),
10657 at_commit: self.commit_seq,
10658 };
10659 if let Ok(mut log) = self.slow_queries.lock() {
10660 if log.entries.len() == SLOW_QUERY_RING_CAP {
10661 log.entries.pop_front();
10662 }
10663 log.entries.push_back(entry);
10664 log.total += 1;
10665 }
10666 }
10667 result
10668 }
10669
10670 /// Convenience entry-point that accepts a slice of `(name, value)` pairs
10671 /// instead of a pre-built `BTreeMap`. Equivalent to building the map and
10672 /// calling [`GraphDb::query`].
10673 pub fn query_with_params(&self, cypher: &str, params: &[(&str, Value)]) -> Result<ResultSet> {
10674 let map: BTreeMap<String, Value> = params
10675 .iter()
10676 .map(|(k, v)| (k.to_string(), v.clone()))
10677 .collect();
10678 self.query(cypher, &map)
10679 }
10680
10681 /// Execute a Cypher write statement (CREATE / MATCH…SET / MATCH…DELETE / MERGE).
10682 ///
10683 /// All mutations flow through the same `insert_node` / `set_prop` /
10684 /// `delete_edge` / `insert_edge` path as the Rust API so the rule engine
10685 /// fires and the WAL captures everything with one fsync per statement.
10686 ///
10687 /// Returns a one-row [`ResultSet`] with columns `created`, `properties_set`,
10688 /// and `deleted` matching the write-result contract.
10689 ///
10690 /// **Mutation routing**: mutations are collected into a single
10691 /// [`BatchBuilder`] and committed atomically (one WAL `Batch` frame, one
10692 /// fsync). The MATCH phase for SET/DELETE uses a read-only `execute` call
10693 /// over `self.view()` — the borrow is dropped before the batch is opened.
10694 ///
10695 /// **Limitations (v1)**:
10696 /// - SET RHS must be a literal, `$param`, or arithmetic; bare property copy → named error.
10697 /// - `DETACH DELETE n` → calls `delete_node` for each matched node (removes all edges).
10698 /// - Bare `DELETE n` → error if n has any incident edges; succeeds for isolated nodes.
10699 /// - MERGE supports `ON CREATE SET` / `ON MATCH SET` in the same write batch.
10700 /// - Deleting a derived edge → named error "cannot delete derived edge".
10701 pub fn query_write(
10702 &mut self,
10703 cypher: &str,
10704 params: &BTreeMap<String, Value>,
10705 ) -> Result<ResultSet> {
10706 let tokens = lex(cypher).map_err(|e| GraphError::QueryError {
10707 detail: format!("lex: {e}"),
10708 })?;
10709 let stmt = parse_write(&tokens).map_err(|e| GraphError::QueryError {
10710 detail: format!("parse: {e}"),
10711 })?;
10712 self.exec_write_stmt(stmt, params)
10713 }
10714
10715 fn exec_write_stmt(
10716 &mut self,
10717 stmt: WriteStatement,
10718 params: &BTreeMap<String, Value>,
10719 ) -> Result<ResultSet> {
10720 match stmt {
10721 WriteStatement::Create(s) => self.exec_create(s, params),
10722 WriteStatement::MatchSet(s) => self.exec_match_set(s, params),
10723 WriteStatement::MatchDelete(s) => self.exec_match_delete(s, params),
10724 WriteStatement::MatchDeleteNode(s) => self.exec_match_delete_node(s, params),
10725 WriteStatement::Merge(s) => self.exec_merge(s, params),
10726 }
10727 }
10728
10729 fn exec_create(
10730 &mut self,
10731 stmt: core_query::cypher::CreateStmt,
10732 params: &BTreeMap<String, Value>,
10733 ) -> Result<ResultSet> {
10734 // Extract the node key from props: require a string-valued `id` field.
10735 let mut var_to_key: BTreeMap<String, String> = BTreeMap::new();
10736 for node in &stmt.nodes {
10737 let var = node.var.as_deref().unwrap_or("_cn0");
10738 let key = node
10739 .props
10740 .iter()
10741 .find(|(f, _)| f == "id")
10742 .and_then(|(_, v)| {
10743 if let Value::Str(s) = v {
10744 Some(s.clone())
10745 } else {
10746 None
10747 }
10748 })
10749 .ok_or_else(|| GraphError::QueryError {
10750 detail: format!(
10751 "CREATE node ({}:{}) requires a string 'id' property",
10752 var, node.label
10753 ),
10754 })?;
10755 var_to_key.insert(var.to_string(), key);
10756 }
10757
10758 let mut batch = self.batch();
10759 let mut created: usize = 0;
10760 for node in &stmt.nodes {
10761 let var = node.var.as_deref().unwrap_or("_cn0");
10762 let key = &var_to_key[var];
10763 batch.insert_node(&node.label, key, node.props.clone());
10764 created += 1;
10765 }
10766 for edge in &stmt.edges {
10767 let src_key = var_to_key
10768 .get(&edge.src_var)
10769 .ok_or_else(|| GraphError::QueryError {
10770 detail: format!("CREATE edge src variable '{}' is not bound", edge.src_var),
10771 })?;
10772 let dst_key = var_to_key
10773 .get(&edge.dst_var)
10774 .ok_or_else(|| GraphError::QueryError {
10775 detail: format!("CREATE edge dst variable '{}' is not bound", edge.dst_var),
10776 })?;
10777 batch.insert_edge(&edge.etype, src_key, dst_key);
10778 }
10779 batch.commit()?;
10780
10781 // Optional RETURN clause: project created bindings as a read result.
10782 if let Some(returns) = stmt.returns {
10783 // Each created node is looked up by its key via a separate MATCH pattern.
10784 // Multiple single-node patterns cross-join to produce 1 output row with
10785 // all variables bound (each pattern returns exactly 1 row).
10786 let patterns: Vec<Pattern> = stmt
10787 .nodes
10788 .iter()
10789 .map(|node| {
10790 let var = node.var.as_deref().unwrap_or("_cn0");
10791 let key = var_to_key[var].clone();
10792 Pattern {
10793 start: NodePat {
10794 var: Some(var.to_string()),
10795 label: Some(node.label.clone()),
10796 props: vec![("id".to_string(), Operand::Lit(Value::Str(key)))],
10797 },
10798 chain: vec![],
10799 shortest: false,
10800 }
10801 })
10802 .collect();
10803 let q = Query {
10804 matches: patterns,
10805 optional_clauses: vec![],
10806 where_expr: None,
10807 unwinds: vec![],
10808 post_unwind_where: None,
10809 stages: vec![],
10810 returns,
10811 distinct: false,
10812 order_by: vec![],
10813 skip: None,
10814 limit: None,
10815 };
10816 let ops = plan(&q).map_err(|e| GraphError::QueryError {
10817 detail: format!("plan: {e}"),
10818 })?;
10819 return execute(&self.view(), &ops, &Params(params)).map_err(|e| {
10820 GraphError::QueryError {
10821 detail: format!("execute: {e}"),
10822 }
10823 });
10824 }
10825
10826 let mut rs = write_result_set();
10827 rs.push_row(vec![
10828 Some(Value::Int(created as i64)),
10829 Some(Value::Int(0)),
10830 Some(Value::Int(0)),
10831 ]);
10832 Ok(rs)
10833 }
10834
10835 fn exec_match_set(
10836 &mut self,
10837 stmt: core_query::cypher::MatchSetStmt,
10838 params: &BTreeMap<String, Value>,
10839 ) -> Result<ResultSet> {
10840 let project_returns = stmt.returns.clone();
10841 // Collect unique node vars targeted by SET clauses, plus RETURN bindings
10842 // so the post-write projection can look them up by key.
10843 let mut set_vars: Vec<String> = Vec::new();
10844 for s in &stmt.sets {
10845 if !set_vars.contains(&s.var) {
10846 set_vars.push(s.var.clone());
10847 }
10848 }
10849 let rel_vars = pattern_rel_vars(&stmt.matches);
10850 // `count` is the engine's, on an edge: it is the insert-count §5.13
10851 // maintains, and a `SET` that overwrote it would make the number mean
10852 // whatever the last writer said rather than how many times the pair was
10853 // inserted. Refused by name here, before the match runs, so the caller
10854 // is told what is actually wrong instead of meeting the executor's
10855 // generic "did not resolve to a node key" — and so the answer does not
10856 // depend on whether the pattern happened to match a row. The same name
10857 // on a *node* is an ordinary property and is untouched.
10858 for s in &stmt.sets {
10859 if s.field == EDGE_COUNT_PROP && rel_vars.iter().any(|r| r == &s.var) {
10860 return Err(GraphError::QueryError {
10861 detail: format!(
10862 "cannot SET {}.{EDGE_COUNT_PROP}: `{EDGE_COUNT_PROP}` is a reserved edge \
10863 property holding the pair's insert count",
10864 s.var
10865 ),
10866 });
10867 }
10868 }
10869 let mut lookup_vars = set_vars.clone();
10870 for v in pattern_node_vars(&stmt.matches) {
10871 add_var(&mut lookup_vars, &v);
10872 }
10873 if let Some(ref returns) = project_returns {
10874 for v in ret_node_vars(returns) {
10875 if !rel_vars.iter().any(|r| r == &v) {
10876 add_var(&mut lookup_vars, &v);
10877 }
10878 }
10879 }
10880
10881 // Synthesize a read query: MATCH … WHERE … RETURN <lookup_vars>, <set_values…>
10882 // SET values are projected as ScalarExpr items so that arithmetic expressions
10883 // (e.g. `SET n.score = n.score * 1.5`) are evaluated in the matched-row context.
10884 let mut set_returns: Vec<RetItem> = lookup_vars
10885 .iter()
10886 .map(|v| RetItem {
10887 value: RetVal::Var(v.clone()),
10888 alias: None,
10889 })
10890 .collect();
10891 // One computed column per SET clause; alias is `__sv_<i>`.
10892 let set_val_cols: Vec<String> = stmt
10893 .sets
10894 .iter()
10895 .enumerate()
10896 .map(|(i, _)| format!("__sv_{i}"))
10897 .collect();
10898 for (sc, col) in stmt.sets.iter().zip(&set_val_cols) {
10899 set_returns.push(RetItem {
10900 value: RetVal::ScalarExpr(sc.value.clone()),
10901 alias: Some(col.clone()),
10902 });
10903 }
10904 // Capture relationship types while r is bound; SET does not change them.
10905 for r in &rel_vars {
10906 set_returns.push(RetItem {
10907 value: RetVal::FuncCall {
10908 name: "type".into(),
10909 args: vec![Operand::Var(r.clone())],
10910 },
10911 alias: Some(rel_type_alias(r)),
10912 });
10913 }
10914
10915 let read_q = Query {
10916 matches: stmt.matches.clone(),
10917 optional_clauses: vec![],
10918 where_expr: stmt.where_expr.clone(),
10919 unwinds: vec![],
10920 post_unwind_where: None,
10921 stages: vec![],
10922 returns: set_returns,
10923 distinct: false,
10924 order_by: vec![],
10925 skip: None,
10926 limit: None,
10927 };
10928 let ops = plan(&read_q).map_err(|e| GraphError::QueryError {
10929 detail: format!("plan: {e}"),
10930 })?;
10931 // MATCH phase is read-only; borrow ends before batch opens.
10932 //
10933 // When a role-scoped write is in flight, run the MATCH read through
10934 // view_masked so hidden nodes are invisible → hidden ≡ absent ≡
10935 // zero-rows (no SetProp ops generated, no existence-oracle 403).
10936 // Full-authority writes (pending_write_authz=None) keep view().
10937 let match_rs = {
10938 let mask_opt = self.pending_write_authz.as_ref().map(|a| a.mask.clone());
10939 if let Some(ref mask) = mask_opt {
10940 execute(&self.view_masked(mask), &ops, &Params(params))
10941 } else {
10942 execute(&self.view(), &ops, &Params(params))
10943 }
10944 }
10945 .map_err(|e| GraphError::QueryError {
10946 detail: format!("execute: {e}"),
10947 })?;
10948
10949 // Collect (key, field, value) for each matched row × each SET clause.
10950 let mut set_ops: Vec<(String, String, Value)> = Vec::new();
10951 for row_i in 0..match_rs.len() {
10952 for (sc, col) in stmt.sets.iter().zip(&set_val_cols) {
10953 let key = match match_rs.get(row_i, &sc.var) {
10954 Some(Value::Str(k)) => k.clone(),
10955 _ => {
10956 return Err(GraphError::QueryError {
10957 detail: format!(
10958 "SET variable '{}' did not resolve to a node key",
10959 sc.var
10960 ),
10961 })
10962 }
10963 };
10964 // The SET value was already evaluated by the executor.
10965 let value = match match_rs.get(row_i, col) {
10966 Some(v) => v.clone(),
10967 None => {
10968 return Err(GraphError::QueryError {
10969 detail: format!(
10970 "SET value for {}.{} evaluated to null",
10971 sc.var, sc.field
10972 ),
10973 })
10974 }
10975 };
10976 set_ops.push((key, sc.field.clone(), value));
10977 }
10978 }
10979
10980 // Apply as one atomic batch.
10981 let props_set = set_ops.len();
10982 let mut batch = self.batch();
10983 for (key, field, value) in set_ops {
10984 batch.set_prop(&key, &field, value);
10985 }
10986 batch.commit()?;
10987
10988 if let Some(returns) = project_returns {
10989 return project_set_return_rows(self, &rel_vars, &match_rs, &returns, params);
10990 }
10991
10992 let mut rs = write_result_set();
10993 rs.push_row(vec![
10994 Some(Value::Int(0)),
10995 Some(Value::Int(props_set as i64)),
10996 Some(Value::Int(0)),
10997 ]);
10998 Ok(rs)
10999 }
11000
11001 fn exec_match_delete(
11002 &mut self,
11003 stmt: core_query::cypher::MatchDeleteStmt,
11004 params: &BTreeMap<String, Value>,
11005 ) -> Result<ResultSet> {
11006 // Collect unique node vars needed to identify edge endpoints.
11007 let mut node_vars: Vec<String> = Vec::new();
11008 for ed in &stmt.deletes {
11009 if !node_vars.contains(&ed.src_var) {
11010 node_vars.push(ed.src_var.clone());
11011 }
11012 if !node_vars.contains(&ed.dst_var) {
11013 node_vars.push(ed.dst_var.clone());
11014 }
11015 }
11016
11017 // Synthesize read query.
11018 let returns: Vec<RetItem> = node_vars
11019 .iter()
11020 .map(|v| RetItem {
11021 value: RetVal::Var(v.clone()),
11022 alias: None,
11023 })
11024 .collect();
11025 let read_q = Query {
11026 matches: stmt.matches,
11027 optional_clauses: vec![],
11028 where_expr: stmt.where_expr,
11029 unwinds: vec![],
11030 post_unwind_where: None,
11031 stages: vec![],
11032 returns,
11033 distinct: false,
11034 order_by: vec![],
11035 skip: None,
11036 limit: None,
11037 };
11038 let ops = plan(&read_q).map_err(|e| GraphError::QueryError {
11039 detail: format!("plan: {e}"),
11040 })?;
11041 // Role-scoped writes: mask the MATCH read phase so hidden nodes are
11042 // invisible → hidden ≡ absent ≡ zero-rows (spec §3.1, hidden ≡ absent).
11043 let match_rs = {
11044 let mask_opt = self.pending_write_authz.as_ref().map(|a| a.mask.clone());
11045 if let Some(ref mask) = mask_opt {
11046 execute(&self.view_masked(mask), &ops, &Params(params))
11047 } else {
11048 execute(&self.view(), &ops, &Params(params))
11049 }
11050 }
11051 .map_err(|e| GraphError::QueryError {
11052 detail: format!("execute: {e}"),
11053 })?;
11054
11055 // Collect (etype, src_key, dst_key) for each row × each delete target.
11056 let mut del_ops: Vec<(String, String, String)> = Vec::new();
11057 for row_i in 0..match_rs.len() {
11058 for ed in &stmt.deletes {
11059 let src_key = match match_rs.get(row_i, &ed.src_var) {
11060 Some(Value::Str(k)) => k.clone(),
11061 _ => {
11062 return Err(GraphError::QueryError {
11063 detail: format!(
11064 "DELETE src variable '{}' did not resolve to a node key",
11065 ed.src_var
11066 ),
11067 })
11068 }
11069 };
11070 let dst_key = match match_rs.get(row_i, &ed.dst_var) {
11071 Some(Value::Str(k)) => k.clone(),
11072 _ => {
11073 return Err(GraphError::QueryError {
11074 detail: format!(
11075 "DELETE dst variable '{}' did not resolve to a node key",
11076 ed.dst_var
11077 ),
11078 })
11079 }
11080 };
11081 del_ops.push((ed.etype.clone(), src_key, dst_key));
11082 }
11083 }
11084
11085 // Apply as one atomic batch.
11086 let deleted = del_ops.len();
11087 let mut batch = self.batch();
11088 for (etype, src_key, dst_key) in del_ops {
11089 batch.delete_edge(&etype, &src_key, &dst_key);
11090 }
11091 batch.commit().map_err(|e| match e {
11092 GraphError::RuleOwned { .. } => GraphError::QueryError {
11093 detail: "cannot delete derived edge; retract via the rule or change the property"
11094 .to_string(),
11095 },
11096 other => other,
11097 })?;
11098
11099 let mut rs = write_result_set();
11100 rs.push_row(vec![
11101 Some(Value::Int(0)),
11102 Some(Value::Int(0)),
11103 Some(Value::Int(deleted as i64)),
11104 ]);
11105 Ok(rs)
11106 }
11107
11108 /// Execute `MATCH … [DETACH] DELETE <node_var> [, …]`.
11109 ///
11110 /// Collects the matching node keys via an ephemeral read query, then calls
11111 /// `delete_node` on each one. When `stmt.detach` is `false` (bare DELETE)
11112 /// the executor first checks that the node has no incident edges; if any
11113 /// remain it returns a named error matching openCypher semantics.
11114 fn exec_match_delete_node(
11115 &mut self,
11116 stmt: MatchDeleteNodeStmt,
11117 params: &BTreeMap<String, Value>,
11118 ) -> Result<ResultSet> {
11119 // Build a read query returning only the node keys we need.
11120 let returns: Vec<RetItem> = stmt
11121 .node_vars
11122 .iter()
11123 .map(|v| RetItem {
11124 value: RetVal::Var(v.clone()),
11125 alias: None,
11126 })
11127 .collect();
11128 let read_q = Query {
11129 matches: stmt.matches,
11130 optional_clauses: vec![],
11131 where_expr: stmt.where_expr,
11132 unwinds: vec![],
11133 post_unwind_where: None,
11134 stages: vec![],
11135 returns,
11136 distinct: false,
11137 order_by: vec![],
11138 skip: None,
11139 limit: None,
11140 };
11141 let ops = plan(&read_q).map_err(|e| GraphError::QueryError {
11142 detail: format!("plan: {e}"),
11143 })?;
11144 // Role-scoped writes: mask the MATCH read phase so hidden nodes are
11145 // invisible → hidden ≡ absent ≡ zero-rows (spec §3.1, hidden ≡ absent).
11146 let match_rs = {
11147 let mask_opt = self.pending_write_authz.as_ref().map(|a| a.mask.clone());
11148 if let Some(ref mask) = mask_opt {
11149 execute(&self.view_masked(mask), &ops, &Params(params))
11150 } else {
11151 execute(&self.view(), &ops, &Params(params))
11152 }
11153 }
11154 .map_err(|e| GraphError::QueryError {
11155 detail: format!("execute: {e}"),
11156 })?;
11157
11158 // Collect unique node keys to delete (deduplicate across rows × vars).
11159 let mut keys: Vec<String> = Vec::new();
11160 for row_i in 0..match_rs.len() {
11161 for var in &stmt.node_vars {
11162 if let Some(Value::Str(k)) = match_rs.get(row_i, var) {
11163 if !keys.contains(k) {
11164 keys.push(k.clone());
11165 }
11166 }
11167 }
11168 }
11169
11170 if !stmt.detach {
11171 // openCypher bare DELETE: error if any matched node has incident edges.
11172 for key in &keys {
11173 if let Some(id) = self.ids.get(key) {
11174 let tv = self.topo_view();
11175 let has_edges = tv.etypes().any(|et| {
11176 !tv.neighbors(et, Direction::Out, id).is_empty()
11177 || !tv.neighbors(et, Direction::In, id).is_empty()
11178 });
11179 if has_edges {
11180 return Err(GraphError::QueryError {
11181 detail: format!(
11182 "Cannot delete node `{key}` because it still has incident edges. \
11183 Use DETACH DELETE to remove the node and all its edges."
11184 ),
11185 });
11186 }
11187 }
11188 }
11189 }
11190
11191 let mut nodes_deleted = 0i64;
11192 let mut edges_deleted = 0i64;
11193 for key in keys {
11194 match self.delete_node(&key) {
11195 Ok(report) => {
11196 nodes_deleted += 1;
11197 edges_deleted += (report.manual_edges + report.derived_edges) as i64;
11198 }
11199 Err(GraphError::KeyNotFound { .. }) => {
11200 // Node may have been deleted by an earlier iteration (e.g., via
11201 // multiple MATCH rows for the same node). Safe to skip.
11202 }
11203 Err(e) => return Err(e),
11204 }
11205 }
11206
11207 let mut rs = write_result_set();
11208 rs.push_row(vec![
11209 Some(Value::Int(0)),
11210 Some(Value::Int(0)),
11211 Some(Value::Int(nodes_deleted + edges_deleted)),
11212 ]);
11213 Ok(rs)
11214 }
11215
11216 /// Props the MERGE create arm inserts: the identifying key, plus `ns` when
11217 /// the pattern named one, or the executing role's sole namespace when it
11218 /// did not. A role bound to two or more namespaces cannot choose, and is
11219 /// refused with [`MERGE_CREATE_NEEDS_ONE_NAMESPACE`]. The authorizer still
11220 /// refuses a named `ns` the role cannot write.
11221 fn merge_create_props(
11222 &self,
11223 key_field: &str,
11224 key_value: &Value,
11225 named_ns: Option<&Value>,
11226 ) -> Result<Vec<(String, Value)>> {
11227 let mut props = vec![(key_field.to_string(), key_value.clone())];
11228 if let Some(ns) = named_ns {
11229 props.push((NS_PROP.to_string(), ns.clone()));
11230 return Ok(props);
11231 }
11232 if let Some(ns) = self.merge_create_stamp_ns()? {
11233 props.push((NS_PROP.to_string(), Value::Str(ns)));
11234 }
11235 Ok(props)
11236 }
11237
11238 /// The namespace a role-scoped MERGE create stamps when the pattern does
11239 /// not name `ns`. `None` = unscoped / full authority, so the node lands in
11240 /// `default`.
11241 fn merge_create_stamp_ns(&self) -> Result<Option<String>> {
11242 let Some(authz) = self.pending_write_authz.as_ref() else {
11243 return Ok(None);
11244 };
11245 let Some(def) = self.role_def_for(&authz.role) else {
11246 return Ok(None);
11247 };
11248 match def.namespaces.as_deref() {
11249 Some([only]) => Ok(Some(only.clone())),
11250 Some(_) => Err(GraphError::RoleWriteDenied {
11251 reason: MERGE_CREATE_NEEDS_ONE_NAMESPACE.to_string(),
11252 }),
11253 None => Ok(None),
11254 }
11255 }
11256
11257 fn exec_merge(
11258 &mut self,
11259 stmt: core_query::cypher::MergeStmt,
11260 params: &BTreeMap<String, Value>,
11261 ) -> Result<ResultSet> {
11262 // MERGE: check if a node with the given key already exists.
11263 let key = match &stmt.key_value {
11264 Value::Str(s) => s.clone(),
11265 _ => {
11266 return Err(GraphError::QueryError {
11267 detail: format!(
11268 "MERGE key value must be a string (got {:?})",
11269 stmt.key_value
11270 ),
11271 })
11272 }
11273 };
11274
11275 if let Some(var) = stmt.var.as_deref() {
11276 for sc in stmt.on_create.iter().chain(&stmt.on_match) {
11277 if sc.var != var {
11278 return Err(GraphError::QueryError {
11279 detail: format!(
11280 "SET variable '{}' does not match MERGE variable '{var}'",
11281 sc.var
11282 ),
11283 });
11284 }
11285 }
11286 }
11287
11288 // ── MERGE authz pre-check (when role-scoped) ─────────────────────────
11289 //
11290 // MERGE scope precondition: check create OR update scope for the
11291 // declared label BEFORE calling `has_node` (timing-oracle closure,
11292 // spec §6.2 "MERGE visibility oracle" item: hidden ≡ absent for
11293 // unscoped roles — the scope denial fires without touching the key store).
11294 //
11295 // Clone to avoid holding a borrow on `self.pending_write_authz` while
11296 // also calling `self.ids.get(key)`.
11297 let merge_existed: bool = if let Some(authz) = self.pending_write_authz.clone() {
11298 let has_create = authz.scope.create_labels.contains(&stmt.label);
11299 let has_update = authz.scope.update_labels.contains(&stmt.label);
11300 if !has_create && !has_update {
11301 // Scope-before-lookup: 403 without has_node call (timing oracle
11302 // closure — see test_merge_unscoped_no_key_lookup).
11303 return Err(GraphError::RoleWriteDenied {
11304 reason: format!(
11305 "role-bound token: label '{}' not in write scope (create_labels)",
11306 stmt.label
11307 ),
11308 });
11309 }
11310 // Key lookup under mask.
11311 match self.ids.get(key.as_str()) {
11312 Some(id) if authz.mask.contains_id(id) => {
11313 // Visible: must have update scope to proceed to match arm.
11314 if !has_update {
11315 return Err(GraphError::RoleWriteDenied {
11316 reason: format!(
11317 "role-bound token: label '{}' not in write scope (update_labels)",
11318 stmt.label
11319 ),
11320 });
11321 }
11322 true // existed = true → match arm
11323 }
11324 Some(_) => {
11325 // Hidden: same error as absent to the role (spec §3.1/§3.3).
11326 return Err(GraphError::RoleWriteDenied {
11327 reason: "role-bound token: target node not visible".into(),
11328 });
11329 }
11330 None => {
11331 // Absent: must have create scope to proceed to the create arm.
11332 //
11333 // Update-only roles (create_labels empty, update_labels set):
11334 // return the SAME "not visible" error as the hidden-key branch
11335 // so hidden ≡ absent — no distinguishing oracle (spec §6.1
11336 // "confirm existence of hidden nodes: No").
11337 //
11338 // Create-scoped roles (has_create=true): absent → create arm
11339 // as before. The accepted structural key-existence disclosure
11340 // (§THREAT-MODEL) applies only when the role holds create scope.
11341 if !has_create {
11342 return Err(GraphError::RoleWriteDenied {
11343 reason: "role-bound token: target node not visible".into(),
11344 });
11345 }
11346 false // existed = false → create arm
11347 }
11348 }
11349 } else {
11350 // Full authority: use the existing non-masked has_node check.
11351 self.has_node(&key)
11352 };
11353
11354 let existed = merge_existed;
11355 let create_props = if existed {
11356 None
11357 } else {
11358 Some(self.merge_create_props(&stmt.key_field, &stmt.key_value, stmt.ns.as_ref())?)
11359 };
11360 let mut created = 0i64;
11361 if create_props.is_some() || !stmt.on_match.is_empty() {
11362 let mut batch = self.batch();
11363 if let Some(props) = create_props {
11364 batch.insert_node(&stmt.label, &key, props);
11365 for sc in &stmt.on_create {
11366 let value = resolve_merge_set_value(&sc.value, params)?;
11367 batch.set_prop(&key, &sc.field, value);
11368 }
11369 created = 1;
11370 } else {
11371 for sc in &stmt.on_match {
11372 let value = resolve_merge_set_value(&sc.value, params)?;
11373 batch.set_prop(&key, &sc.field, value);
11374 }
11375 }
11376 batch.commit()?;
11377 }
11378
11379 // Refresh the role mask so the just-created node is visible to this
11380 // statement's RETURN (read-after-write). Safe: create_labels ⊆ read labels
11381 // (apply_schema subset rule), so the new node's label is already in the
11382 // role's read scope — this never widens beyond the role's declared labels.
11383 if !existed {
11384 if let Some(role) = self.pending_write_authz.as_ref().map(|a| a.role.clone()) {
11385 let new_mask = self.mask_for_role(&role)?;
11386 if let Some(a) = self.pending_write_authz.as_mut() {
11387 a.mask = new_mask;
11388 }
11389 }
11390 }
11391
11392 // Optional RETURN clause: project the node (created or matched) as a read result.
11393 if let Some(returns) = stmt.returns {
11394 let var = stmt.var.as_deref().unwrap_or("_mn0");
11395 let q = Query {
11396 matches: vec![Pattern {
11397 start: NodePat {
11398 var: Some(var.to_string()),
11399 label: Some(stmt.label.clone()),
11400 props: vec![("id".to_string(), Operand::Lit(stmt.key_value.clone()))],
11401 },
11402 chain: vec![],
11403 shortest: false,
11404 }],
11405 optional_clauses: vec![],
11406 where_expr: None,
11407 unwinds: vec![],
11408 post_unwind_where: None,
11409 stages: vec![],
11410 returns,
11411 distinct: false,
11412 order_by: vec![],
11413 skip: None,
11414 limit: None,
11415 };
11416 let ops = plan(&q).map_err(|e| GraphError::QueryError {
11417 detail: format!("plan: {e}"),
11418 })?;
11419 // Use view_masked when a role-scoped write is in flight so the
11420 // post-merge projection is consistent with the masked read phase.
11421 let mask_opt = self.pending_write_authz.as_ref().map(|a| a.mask.clone());
11422 return (if let Some(ref mask) = mask_opt {
11423 execute(&self.view_masked(mask), &ops, &Params(params))
11424 } else {
11425 execute(&self.view(), &ops, &Params(params))
11426 })
11427 .map_err(|e| GraphError::QueryError {
11428 detail: format!("execute: {e}"),
11429 });
11430 }
11431
11432 let mut rs = write_result_set();
11433 rs.push_row(vec![
11434 Some(Value::Int(created)),
11435 Some(Value::Int(0)),
11436 Some(Value::Int(0)),
11437 ]);
11438 Ok(rs)
11439 }
11440
11441 /// Return all rule-owned edges between `key_a` and `key_b` (either direction),
11442 /// annotated with rule name, edge type, direction, and weight.
11443 /// Results are sorted by (rule, edge_type).
11444 /// Returns `Err(KeyNotFound)` if either key is unknown.
11445 pub fn explain(&self, key_a: &str, key_b: &str) -> Result<Vec<Explanation>> {
11446 self.ensure_v8_base_sections_loaded();
11447 let id_a = self
11448 .ids
11449 .get(key_a)
11450 .ok_or_else(|| GraphError::KeyNotFound { key: key_a.into() })?;
11451 let id_b = self
11452 .ids
11453 .get(key_b)
11454 .ok_or_else(|| GraphError::KeyNotFound { key: key_b.into() })?;
11455
11456 let mut results = Vec::new();
11457
11458 // Walk the smaller incident set so explain is O(min(deg(a), deg(b)))
11459 // rather than O(total provenance).
11460 let scan = if self.engine.provenance_touching_len(id_a)
11461 <= self.engine.provenance_touching_len(id_b)
11462 {
11463 id_a
11464 } else {
11465 id_b
11466 };
11467 for (rule_name, etype, src, dst) in self.engine.provenance_touching(scan) {
11468 if !((src == id_a && dst == id_b) || (src == id_b && dst == id_a)) {
11469 continue;
11470 }
11471 let Some(rule_def) = self.engine.rules().find(|r| r.name == rule_name) else {
11472 continue;
11473 };
11474 let edge_type = match self.syms.resolve(etype) {
11475 Some(s) => s.to_string(),
11476 None => continue,
11477 };
11478 // Provenance (src, dst) ids come from the archived PROVENANCE section
11479 // (large, no eager CRC). A corrupt section can produce ids that are
11480 // out of range; return Corrupt rather than panic.
11481 let src_key = self
11482 .ids
11483 .key_of(src)
11484 .ok_or_else(|| GraphError::Corrupt {
11485 detail: format!("v8: provenance src id {src} not in id table"),
11486 })?
11487 .to_string();
11488 let dst_key = self
11489 .ids
11490 .key_of(dst)
11491 .ok_or_else(|| GraphError::Corrupt {
11492 detail: format!("v8: provenance dst id {dst} not in id table"),
11493 })?
11494 .to_string();
11495 let stored = rule_def.weight_prop.as_deref().and_then(|prop| {
11496 self.edge_props_view()
11497 .get(etype, src, dst, prop)
11498 .and_then(|v| {
11499 if let Value::Float(f) = v {
11500 Some(f)
11501 } else {
11502 None
11503 }
11504 })
11505 });
11506 // Rules that store no weight (KeyMatch/FieldEqual defaults, auto-FK)
11507 // still have a score: recompute it from the predicate so explain
11508 // never reports "no score" for an edge the engine scored. Via-hop
11509 // rules score over their via set, not over (src, dst), so leave
11510 // those None rather than report a number the rule did not produce.
11511 let weight = stored.or_else(|| {
11512 if rule_def.via_edge.is_some() {
11513 return None;
11514 }
11515 let props_view = build_props_view(&self.props, &self.base);
11516 let src_get = |field: &str| props_view.get(src, field).map(|vr| vr.into_value());
11517 let dst_get = |field: &str| props_view.get(dst, field).map(|vr| vr.into_value());
11518 let src_view = NodeView {
11519 key: &src_key,
11520 props: &src_get,
11521 };
11522 let dst_view = NodeView {
11523 key: &dst_key,
11524 props: &dst_get,
11525 };
11526 evaluate(&rule_def.predicate, &src_view, &dst_view)
11527 });
11528 results.push(Explanation {
11529 rule: rule_name.to_string(),
11530 edge_type,
11531 src_key,
11532 dst_key,
11533 weight,
11534 predicate: PredicateSummary {
11535 approximate: rule_def.approximate,
11536 ..PredicateSummary::from(&rule_def.predicate)
11537 },
11538 via_edge: rule_def.via_edge.clone(),
11539 });
11540 }
11541
11542 results.sort_by(|a, b| a.rule.cmp(&b.rule).then(a.edge_type.cmp(&b.edge_type)));
11543 Ok(results)
11544 }
11545
11546 /// [`explain`](Self::explain) with the scoped read contract (§5.3): both
11547 /// endpoints are subject-checked, and any explanation whose evidence runs
11548 /// through a hidden node is **dropped entirely, not redacted**.
11549 ///
11550 /// A plain two-node rule's evidence is the pair itself, so once both
11551 /// subjects are visible there is nothing left to hide. A **via-hop** rule is
11552 /// different: it fires `src → dst` because some node carrying `via_label`
11553 /// sits between them, and [`Explanation`] carries the hop's edge *type*
11554 /// (`via_edge`) and never the hop's key. There is no field to blank, so a
11555 /// redacted explanation would still say "these two are linked through
11556 /// something you cannot see" — which discloses that the something exists.
11557 /// The explanation is therefore kept only when at least one **visible** via
11558 /// node satisfies the rule on its own.
11559 ///
11560 /// The weight is the **visible corpus's** number, not the store's: a via-hop
11561 /// rule stores the max over every via it hopped through, so the stored value
11562 /// can be a score only a hidden via produced. It is recomputed over the
11563 /// visible vias alone.
11564 ///
11565 /// Hidden or unknown `key_a` or `key_b` → [`GraphError::KeyNotFound`].
11566 pub fn explain_scoped(
11567 &self,
11568 key_a: &str,
11569 key_b: &str,
11570 mask: &crate::mask::NodeMask,
11571 ) -> Result<Vec<Explanation>> {
11572 for key in [key_a, key_b] {
11573 if !mask.contains_node(self, key) {
11574 return Err(GraphError::KeyNotFound { key: key.into() });
11575 }
11576 }
11577 Ok(self
11578 .explain(key_a, key_b)?
11579 .into_iter()
11580 .filter_map(|e| self.scoped_explanation(e, mask))
11581 .collect())
11582 }
11583
11584 /// `e` as a caller limited to `mask` may have it, or `None` when it must be
11585 /// dropped entirely.
11586 ///
11587 /// Every non-via-hop explanation passes through untouched: its only nodes
11588 /// are the two subjects, which [`explain_scoped`](Self::explain_scoped) has
11589 /// already checked, and its weight is scored over that pair alone.
11590 ///
11591 /// A via-hop explanation is kept only when some via node the caller may see
11592 /// satisfies the rule on its own — and then its weight is recomputed as the
11593 /// max over exactly those vias. The engine writes the max over **all** of
11594 /// them (`core-rules::engine`, `best = prev.max(score)`), so passing the
11595 /// stored number through would let a hidden node set a figure the caller
11596 /// reads: the same disclosure dropping the explanation exists to prevent.
11597 ///
11598 /// A rule that stores no weight still reports none. The recomputed score is
11599 /// a sanitised version of a number `explain` already returned, never a new
11600 /// one — a scoped read must not say more than the unscoped read it narrows.
11601 fn scoped_explanation(
11602 &self,
11603 e: Explanation,
11604 mask: &crate::mask::NodeMask,
11605 ) -> Option<Explanation> {
11606 let Some(via_edge) = e.via_edge.clone() else {
11607 return Some(e);
11608 };
11609 let Some(rule_def) = self.engine.rules().find(|r| r.name == e.rule) else {
11610 // The rule is gone but its provenance is not; nothing can vouch for
11611 // the hop, so nothing is shown.
11612 return None;
11613 };
11614 let Some(via_label) = rule_def.via_label.as_deref() else {
11615 return Some(e);
11616 };
11617 let (Some(src), Some(dst)) = (self.ids.get(&e.src_key), self.ids.get(&e.dst_key)) else {
11618 return None;
11619 };
11620 let (Some(via_etype), Some(via_sym)) = (self.syms.get(&via_edge), self.syms.get(via_label))
11621 else {
11622 return None;
11623 };
11624 let via_dir = rule_def.via_dir.unwrap_or(Direction::Out);
11625 let props_view = build_props_view(&self.props, &self.base);
11626 let dst_get = |field: &str| props_view.get(dst, field).map(|vr| vr.into_value());
11627 let dst_view = NodeView {
11628 key: &e.dst_key,
11629 props: &dst_get,
11630 };
11631 // The rule's own namespace test, the one the engine applies to each via
11632 // candidate (`core-rules::engine::rule_sees_node`). Without it a
11633 // visible, out-of-namespace via — right label, satisfying predicate —
11634 // vouches for a hop the engine never made, and an explanation whose real
11635 // evidence is a hidden in-namespace node is kept.
11636 let rule_sees = |id: u32| match rule_def.namespace.as_deref() {
11637 None => true,
11638 Some(ns) => {
11639 let value = props_view.get(id, NS_PROP).map(|vr| vr.into_value());
11640 namespace_of_value(value.as_ref()) == ns
11641 }
11642 };
11643 let best = self
11644 .topo_view()
11645 .neighbors(via_etype, via_dir, src)
11646 .iter()
11647 .copied()
11648 .filter_map(|via| {
11649 if !mask.contains_id(via) {
11650 return None;
11651 }
11652 if self.labels.get(via as usize).copied() != Some(via_sym) {
11653 return None;
11654 }
11655 if !rule_sees(via) {
11656 return None;
11657 }
11658 let via_key = self.ids.key_of(via)?;
11659 let via_get = |field: &str| props_view.get(via, field).map(|vr| vr.into_value());
11660 let via_view = NodeView {
11661 key: via_key,
11662 props: &via_get,
11663 };
11664 evaluate(&rule_def.predicate, &via_view, &dst_view)
11665 })
11666 .fold(None::<f64>, |best, score| {
11667 Some(match best {
11668 None => score,
11669 Some(prev) => prev.max(score),
11670 })
11671 })?;
11672 let weight = e.weight.map(|_| best);
11673 Some(Explanation { weight, ..e })
11674 }
11675
11676 pub fn neighbors(&self, key: &str, edge_type: &str, dir: Direction) -> Result<Vec<String>> {
11677 let id = self
11678 .ids
11679 .get(key)
11680 .ok_or_else(|| GraphError::KeyNotFound { key: key.into() })?;
11681 let Some(sym) = self.syms.get(edge_type) else {
11682 return Ok(Vec::new());
11683 };
11684 self.topo_view()
11685 .neighbors(sym, dir, id)
11686 .iter()
11687 .map(|&n| {
11688 self.ids
11689 .key_of(n)
11690 .map(|k| k.to_string())
11691 .ok_or_else(|| GraphError::Corrupt {
11692 detail: format!("topology id {n} has no key"),
11693 })
11694 })
11695 .collect::<Result<Vec<_>>>()
11696 }
11697
11698 /// Unique directed degree of `key`. Unknown key → [`GraphError::KeyNotFound`].
11699 /// Unknown `edge_type` → 0. [`crate::algo::AlgoDir::Both`] is out + in (sum).
11700 pub fn degree(
11701 &self,
11702 key: &str,
11703 edge_type: Option<&str>,
11704 direction: crate::algo::AlgoDir,
11705 ) -> Result<u64> {
11706 let id = self
11707 .ids
11708 .get(key)
11709 .ok_or_else(|| GraphError::KeyNotFound { key: key.into() })?;
11710 let topo = self.topo_view();
11711 Ok(Self::unique_directed_degree(
11712 &topo, &self.syms, id, edge_type, direction,
11713 ))
11714 }
11715
11716 /// [`degree`](Self::degree) summing each pair's **insert count** instead of
11717 /// counting each pair once (§5.13).
11718 ///
11719 /// The unique degree asks how many neighbours there are; this asks how many
11720 /// times they were inserted. A pair with no recorded count contributes 1,
11721 /// so on a store that never called
11722 /// [`enable_multiplicity`](Self::enable_multiplicity) this returns exactly
11723 /// what [`degree`](Self::degree) returns rather than erroring — the
11724 /// distinction is a readout preference, not a demand the store cannot meet.
11725 ///
11726 /// `AlgoDir::Both` still sums out + in, so a pair visible on both sides
11727 /// still contributes twice: multiplicity changes what a pair is worth, never
11728 /// how a direction is counted.
11729 pub fn degree_multiplicity(
11730 &self,
11731 key: &str,
11732 edge_type: Option<&str>,
11733 direction: crate::algo::AlgoDir,
11734 ) -> Result<u64> {
11735 let id = self
11736 .ids
11737 .get(key)
11738 .ok_or_else(|| GraphError::KeyNotFound { key: key.into() })?;
11739 Ok(Self::multiplicity_directed_degree(
11740 &self.topo_view(),
11741 &self.edge_props_view(),
11742 &self.syms,
11743 id,
11744 edge_type,
11745 direction,
11746 None,
11747 ))
11748 }
11749
11750 /// [`degree_multiplicity`](Self::degree_multiplicity) under a scope.
11751 ///
11752 /// The sum covers **visible pairs only**. A hidden neighbour's inserts stay
11753 /// out of it for the reason
11754 /// [`degree_scoped`](Self::degree_scoped) documents, and more sharply: an
11755 /// unscoped multiplicity count discloses not only that a hidden neighbour
11756 /// exists but how often it was written.
11757 ///
11758 /// Hidden or unknown `key` → [`GraphError::KeyNotFound`].
11759 pub fn degree_scoped_multiplicity(
11760 &self,
11761 key: &str,
11762 edge_type: Option<&str>,
11763 direction: crate::algo::AlgoDir,
11764 mask: &crate::mask::NodeMask,
11765 ) -> Result<u64> {
11766 let id = self
11767 .ids
11768 .get(key)
11769 .filter(|&id| mask.contains_id(id))
11770 .ok_or_else(|| GraphError::KeyNotFound { key: key.into() })?;
11771 Ok(Self::multiplicity_directed_degree(
11772 &self.topo_view(),
11773 &self.edge_props_view(),
11774 &self.syms,
11775 id,
11776 edge_type,
11777 direction,
11778 Some(mask),
11779 ))
11780 }
11781
11782 /// [`degree`](Self::degree) counting **only neighbours the mask admits**.
11783 ///
11784 /// The filter is a correctness requirement, not an optimisation: an
11785 /// unfiltered count discloses the existence of a hidden neighbour to a
11786 /// caller who cannot see it, which is the same leak
11787 /// [`node_edges_scoped`](Self::node_edges_scoped) exists to prevent —
11788 /// reached by arithmetic instead of by name.
11789 ///
11790 /// Hidden or unknown `key` → [`GraphError::KeyNotFound`]. Unknown
11791 /// `edge_type` is still 0, as it is unscoped.
11792 pub fn degree_scoped(
11793 &self,
11794 key: &str,
11795 edge_type: Option<&str>,
11796 direction: crate::algo::AlgoDir,
11797 mask: &crate::mask::NodeMask,
11798 ) -> Result<u64> {
11799 let id = self
11800 .ids
11801 .get(key)
11802 .filter(|&id| mask.contains_id(id))
11803 .ok_or_else(|| GraphError::KeyNotFound { key: key.into() })?;
11804 let topo = self.topo_view();
11805 Ok(Self::visible_directed_degree(
11806 &topo, &self.syms, id, edge_type, direction, mask,
11807 ))
11808 }
11809
11810 /// Unique directed degree for a subset or a label scan.
11811 ///
11812 /// Unknown keys in `keys` are omitted (mask-like). `keys = Some(&[])` →
11813 /// empty `Ok(vec![])`. `limit` is applied after sorting degree desc, key
11814 /// asc, and only when `Some`. Invalid `where_` → `QueryError`.
11815 #[allow(clippy::too_many_arguments)]
11816 pub fn degrees(
11817 &self,
11818 keys: Option<&[String]>,
11819 label: Option<&str>,
11820 where_: Option<&PropPredicate>,
11821 edge_type: Option<&str>,
11822 direction: crate::algo::AlgoDir,
11823 limit: Option<usize>,
11824 ) -> Result<Vec<(String, u64)>> {
11825 self.degrees_inner(
11826 keys, label, where_, edge_type, direction, limit, None, false,
11827 )
11828 }
11829
11830 /// [`degrees`](Self::degrees) reporting each row's **insert-count** sum
11831 /// instead of its unique neighbour count, as
11832 /// [`degree_multiplicity`](Self::degree_multiplicity) does for one key.
11833 ///
11834 /// The sort is still degree descending, key ascending — over the counts this
11835 /// reading produces — and `limit` still applies after it.
11836 #[allow(clippy::too_many_arguments)]
11837 pub fn degrees_multiplicity(
11838 &self,
11839 keys: Option<&[String]>,
11840 label: Option<&str>,
11841 where_: Option<&PropPredicate>,
11842 edge_type: Option<&str>,
11843 direction: crate::algo::AlgoDir,
11844 limit: Option<usize>,
11845 ) -> Result<Vec<(String, u64)>> {
11846 self.degrees_inner(keys, label, where_, edge_type, direction, limit, None, true)
11847 }
11848
11849 /// [`degrees_scoped`](Self::degrees_scoped) reporting insert counts.
11850 ///
11851 /// Both filters apply: a hidden key stays out of the result, and every
11852 /// row's sum covers its **visible** pairs only.
11853 #[allow(clippy::too_many_arguments)]
11854 pub fn degrees_scoped_multiplicity(
11855 &self,
11856 keys: Option<&[String]>,
11857 label: Option<&str>,
11858 where_: Option<&PropPredicate>,
11859 edge_type: Option<&str>,
11860 direction: crate::algo::AlgoDir,
11861 limit: Option<usize>,
11862 mask: &crate::mask::NodeMask,
11863 ) -> Result<Vec<(String, u64)>> {
11864 self.degrees_inner(
11865 keys,
11866 label,
11867 where_,
11868 edge_type,
11869 direction,
11870 limit,
11871 Some(mask),
11872 true,
11873 )
11874 }
11875
11876 /// [`degrees`](Self::degrees) with the scope applied on both sides: a hidden
11877 /// key is omitted from the input — whether it arrived in `keys` or came out
11878 /// of the `label`/`where_` scan — and every row's count is the count of its
11879 /// **visible** neighbours, for the reason
11880 /// [`degree_scoped`](Self::degree_scoped) documents.
11881 ///
11882 /// Unlike `degree_scoped`, a hidden key here is not
11883 /// [`GraphError::KeyNotFound`]: `degrees` already drops unknown keys
11884 /// silently, so hidden and absent stay one answer by staying out of the
11885 /// result. `limit` still applies after the sort, and so counts visible rows.
11886 #[allow(clippy::too_many_arguments)]
11887 pub fn degrees_scoped(
11888 &self,
11889 keys: Option<&[String]>,
11890 label: Option<&str>,
11891 where_: Option<&PropPredicate>,
11892 edge_type: Option<&str>,
11893 direction: crate::algo::AlgoDir,
11894 limit: Option<usize>,
11895 mask: &crate::mask::NodeMask,
11896 ) -> Result<Vec<(String, u64)>> {
11897 self.degrees_inner(
11898 keys,
11899 label,
11900 where_,
11901 edge_type,
11902 direction,
11903 limit,
11904 Some(mask),
11905 false,
11906 )
11907 }
11908
11909 /// The body shared by [`degrees`](Self::degrees) and
11910 /// [`degrees_scoped`](Self::degrees_scoped). `mask = None` is the unscoped
11911 /// contract unchanged.
11912 #[allow(clippy::too_many_arguments)]
11913 fn degrees_inner(
11914 &self,
11915 keys: Option<&[String]>,
11916 label: Option<&str>,
11917 where_: Option<&PropPredicate>,
11918 edge_type: Option<&str>,
11919 direction: crate::algo::AlgoDir,
11920 limit: Option<usize>,
11921 mask: Option<&crate::mask::NodeMask>,
11922 multiplicity: bool,
11923 ) -> Result<Vec<(String, u64)>> {
11924 if let Some(pred) = where_ {
11925 pred.validate_named("where")
11926 .map_err(|detail| GraphError::QueryError { detail })?;
11927 }
11928 if matches!(keys, Some(ks) if ks.is_empty()) {
11929 return Ok(Vec::new());
11930 }
11931 let view = self.view();
11932 let ids: Vec<u32> = match keys {
11933 Some(ks) => {
11934 let mut seen = HashSet::new();
11935 let mut out = Vec::new();
11936 for k in ks {
11937 let Some(id) = view.ids.get(k) else {
11938 continue;
11939 };
11940 if !seen.insert(id) {
11941 continue;
11942 }
11943 if let Some(pred) = where_ {
11944 let holds = match view.prop(id, &pred.field) {
11945 None => pred.holds(None),
11946 Some(vr) => pred.holds(Some(vr.as_value())),
11947 };
11948 if !holds {
11949 continue;
11950 }
11951 }
11952 out.push(id);
11953 }
11954 out
11955 }
11956 None => Self::vector_candidates(&view, label, where_),
11957 };
11958 let mut out: Vec<(String, u64)> = ids
11959 .into_iter()
11960 // A hidden candidate leaves as quietly as an unknown key does.
11961 .filter(|&id| mask.is_none_or(|m| m.contains_id(id)))
11962 .filter_map(|id| {
11963 let key = self.ids.key_of(id)?.to_string();
11964 let deg = match (multiplicity, mask) {
11965 // The same `view` the unique arms read, so the per-row
11966 // rebuild F9 measured is gone and all three arms agree on
11967 // the state they are reading.
11968 (true, m) => Self::multiplicity_directed_degree(
11969 &view.topo,
11970 &view.edge_props,
11971 view.syms,
11972 id,
11973 edge_type,
11974 direction,
11975 m,
11976 ),
11977 (false, Some(m)) => Self::visible_directed_degree(
11978 &view.topo, view.syms, id, edge_type, direction, m,
11979 ),
11980 (false, None) => Self::unique_directed_degree(
11981 &view.topo, view.syms, id, edge_type, direction,
11982 ),
11983 };
11984 Some((key, deg))
11985 })
11986 .collect();
11987 out.sort_by(|a, b| b.1.cmp(&a.1).then_with(|| a.0.cmp(&b.0)));
11988 if let Some(lim) = limit {
11989 out.truncate(lim);
11990 }
11991 Ok(out)
11992 }
11993
11994 /// Unique neighbour count for `id` across `edge_type` (or all types) and
11995 /// `direction`. Unknown `edge_type` → 0. `Both` sums out + in.
11996 fn unique_directed_degree(
11997 topo: &TopologyView<'_>,
11998 syms: &Interner,
11999 id: u32,
12000 edge_type: Option<&str>,
12001 direction: crate::algo::AlgoDir,
12002 ) -> u64 {
12003 let dirs: &[Direction] = match direction {
12004 crate::algo::AlgoDir::Out => &[Direction::Out],
12005 crate::algo::AlgoDir::In => &[Direction::In],
12006 crate::algo::AlgoDir::Both => &[Direction::Out, Direction::In],
12007 };
12008 match edge_type {
12009 Some(name) => {
12010 let Some(et) = syms.get(name) else {
12011 return 0;
12012 };
12013 dirs.iter().map(|&d| topo.degree(et, d, id) as u64).sum()
12014 }
12015 None => topo
12016 .etypes()
12017 .map(|et| {
12018 dirs.iter()
12019 .map(|&d| topo.degree(et, d, id) as u64)
12020 .sum::<u64>()
12021 })
12022 .sum(),
12023 }
12024 }
12025
12026 /// [`unique_directed_degree`](Self::unique_directed_degree) counting only
12027 /// neighbours `mask` admits.
12028 ///
12029 /// Same shape, one substitution: `topo.degree` is a length, so it cannot be
12030 /// filtered; the neighbour list it measures can. `Both` still sums out + in,
12031 /// so a node visible on both sides still counts twice — the filter changes
12032 /// which neighbours are counted, never how a degree is defined.
12033 fn visible_directed_degree(
12034 topo: &TopologyView<'_>,
12035 syms: &Interner,
12036 id: u32,
12037 edge_type: Option<&str>,
12038 direction: crate::algo::AlgoDir,
12039 mask: &crate::mask::NodeMask,
12040 ) -> u64 {
12041 let dirs: &[Direction] = match direction {
12042 crate::algo::AlgoDir::Out => &[Direction::Out],
12043 crate::algo::AlgoDir::In => &[Direction::In],
12044 crate::algo::AlgoDir::Both => &[Direction::Out, Direction::In],
12045 };
12046 let visible = |et: u32| -> u64 {
12047 dirs.iter()
12048 .map(|&d| {
12049 topo.neighbors(et, d, id)
12050 .iter()
12051 .filter(|&&n| mask.contains_id(n))
12052 .count() as u64
12053 })
12054 .sum()
12055 };
12056 match edge_type {
12057 Some(name) => syms.get(name).map_or(0, visible),
12058 None => topo.etypes().map(visible).sum(),
12059 }
12060 }
12061
12062 /// Sum of the insert counts of `id`'s pairs (§5.13), over `edge_type` (or
12063 /// all types) and `direction`, restricted to what `mask` admits when one is
12064 /// given.
12065 ///
12066 /// The same neighbour lists the unique reading measures, with each entry
12067 /// worth its pair's count rather than worth 1 — so the filter decides which
12068 /// pairs are in the sum and the count decides what each contributes. A
12069 /// direction decides which way round the pair is addressed: an `In`
12070 /// neighbour `n` of `id` is the pair `(et, n, id)`.
12071 ///
12072 /// Takes its views as parameters, exactly as the unique helpers do, because
12073 /// it is called once per row from a label scan. `edge_props_view()` reaches
12074 /// into the mmap'd base's rkyv section on every call, so building the two
12075 /// views inside made an N-row `degrees(multiplicity=True)` do N section
12076 /// accesses where the unique reading does one: worth 2.57 ms of 16.68 ms
12077 /// over 20 000 rows, about 0.13 us per row (defect #29,
12078 /// `tests/f9_bench.rs`). Most of that call's cost is the per-neighbour
12079 /// count lookup and is inherent, so this is a hoist, not a rescue.
12080 ///
12081 /// The views are exactly `self.view()`'s own `topo` and `edge_props`, so a
12082 /// caller that already has a view passes its halves and reads the same
12083 /// state it reads everything else from.
12084 fn multiplicity_directed_degree(
12085 topo: &TopologyView<'_>,
12086 edge_props: &EdgePropsView<'_>,
12087 syms: &Interner,
12088 id: u32,
12089 edge_type: Option<&str>,
12090 direction: crate::algo::AlgoDir,
12091 mask: Option<&crate::mask::NodeMask>,
12092 ) -> u64 {
12093 let dirs: &[Direction] = match direction {
12094 crate::algo::AlgoDir::Out => &[Direction::Out],
12095 crate::algo::AlgoDir::In => &[Direction::In],
12096 crate::algo::AlgoDir::Both => &[Direction::Out, Direction::In],
12097 };
12098 let count_of = |et: u32, src: u32, dst: u32| -> u64 {
12099 match edge_props.get(et, src, dst, EDGE_COUNT_PROP) {
12100 Some(Value::Int(n)) if n > 0 => n as u64,
12101 _ => 1,
12102 }
12103 };
12104 let per_etype = |et: u32| -> u64 {
12105 dirs.iter()
12106 .map(|&d| {
12107 topo.neighbors(et, d, id)
12108 .iter()
12109 .filter(|&&n| mask.is_none_or(|m| m.contains_id(n)))
12110 .map(|&n| match d {
12111 Direction::Out => count_of(et, id, n),
12112 Direction::In => count_of(et, n, id),
12113 })
12114 .sum::<u64>()
12115 })
12116 .sum()
12117 };
12118 match edge_type {
12119 Some(name) => syms.get(name).map_or(0, per_etype),
12120 None => topo.etypes().map(per_etype).sum(),
12121 }
12122 }
12123
12124 /// Return the last-change commit sequence for `key`, or `None` if the node
12125 /// does not exist or has never been mutated since the last V5-V7 snapshot
12126 /// (horizon-bounded for legacy stores).
12127 ///
12128 /// The returned sequence is a monotonically increasing counter that starts
12129 /// at 1 for the first commit after `open` and increments with every
12130 /// successful write. WAL replay at open also assigns sequences (1..N for N
12131 /// replayed frames), so sequences are consistent across snapshot+WAL cycles.
12132 ///
12133 /// For V5-V7 stores opened without a V8 snapshot, nodes that were present
12134 /// in the snapshot but not touched by any WAL frame will return `None`
12135 /// (horizon-bounded: CAS against such nodes is only safe after the first
12136 /// V8 snapshot or after the node is next mutated).
12137 pub fn last_changed(&self, key: &str) -> Option<u64> {
12138 let id = self.ids.get(key)?;
12139 self.last_change.get(&id).copied()
12140 }
12141
12142 /// Which loaded store this handle is.
12143 ///
12144 /// Paired with [`commit_seq`](GraphDb::commit_seq) it identifies a graph
12145 /// state outright, which `commit_seq` alone does not: two stores of the same
12146 /// age share a sequence, and a reload can return to one. Memos of dense
12147 /// node ids that the handle does not own are stamped with both; see
12148 /// [`StoreStamp`](crate::mask::StoreStamp).
12149 pub(crate) fn store_id(&self) -> crate::mask::StoreId {
12150 self.store_id
12151 }
12152
12153 /// The current commit sequence (number of successful commits since open,
12154 /// including WAL replay frames). Useful for recording a baseline before
12155 /// a read-modify-write cycle.
12156 pub fn commit_seq(&self) -> u64 {
12157 self.commit_seq
12158 }
12159
12160 /// Check that all `preconds` are satisfied against the current db state.
12161 /// Returns `Err(GraphError::CasConflict)` on the first failing precondition.
12162 pub(crate) fn check_preconditions(&self, preconds: &[Precondition]) -> Result<()> {
12163 for precond in preconds {
12164 match precond {
12165 Precondition::NodeUnchangedSince { key, expected } => {
12166 // Missing entry means the node predates the WAL window or
12167 // does not exist; treat as 0 (before any commit).
12168 let actual = self.last_changed(key).unwrap_or_default();
12169 if actual != *expected {
12170 return Err(GraphError::CasConflict {
12171 key: key.clone(),
12172 expected: *expected,
12173 actual,
12174 });
12175 }
12176 }
12177 Precondition::NodeAbsent { key } => {
12178 // Node must not exist (not live).
12179 if self.ids.get(key).is_some() {
12180 let actual = self.last_changed(key).unwrap_or(0);
12181 return Err(GraphError::CasConflict {
12182 key: key.clone(),
12183 expected: u64::MAX,
12184 actual,
12185 });
12186 }
12187 }
12188 }
12189 }
12190 Ok(())
12191 }
12192
12193 /// Apply a batch of mutations with compare-and-set preconditions.
12194 ///
12195 /// All preconditions are checked atomically before any operation is applied.
12196 /// If any precondition fails, the entire batch is rejected with
12197 /// [`GraphError::CasConflict`] and no WAL frame is written.
12198 ///
12199 /// # Returns
12200 /// `(nodes_inserted, edges_inserted)` on success, same as [`write_batch`].
12201 ///
12202 /// # Errors
12203 /// - [`GraphError::CasConflict`] if any precondition is not satisfied.
12204 /// - Any error that [`write_batch`] would return for the ops themselves.
12205 pub fn write_batch_cas(
12206 &mut self,
12207 preconds: Vec<Precondition>,
12208 ops: Vec<BatchOp>,
12209 ) -> Result<(usize, usize)> {
12210 self.check_preconditions(&preconds)?;
12211 self.commit_logged_batch(ops, None, None).map(inserted_pair)
12212 }
12213
12214 /// Update the per-node last-change map for a WAL record at commit `seq`.
12215 ///
12216 /// Called after a successful apply to record which nodes were touched.
12217 /// For replay, called with the WAL-frame's replayed seq.
12218 ///
12219 /// Touch definition (see [`Precondition`] doc):
12220 /// - InsertNode / InsertNodeId / SetProp / SetPropId / RemoveProp → the node.
12221 /// - InsertEdge / InsertEdgeId / DeleteEdge → both src and dst.
12222 /// - DeleteNode → node tombstoned; last_changed() returns None so no update needed.
12223 /// - DerivedEdge markers, Intern, rule/view records → no-ops.
12224 /// - Batch → recurse into inner records.
12225 fn update_last_change_from_rec(&mut self, rec: &WalRecord, seq: u64) {
12226 match rec {
12227 WalRecord::InsertNode { key, .. }
12228 | WalRecord::SetProp { key, .. }
12229 | WalRecord::RemoveProp { key, .. } => {
12230 if let Some(id) = self.ids.get(key) {
12231 self.last_change.insert(id, seq);
12232 }
12233 }
12234 WalRecord::InsertNodeId { key, .. } => {
12235 if let Some(id) = self.ids.get(key) {
12236 self.last_change.insert(id, seq);
12237 }
12238 }
12239 WalRecord::SetPropId { id, .. } => {
12240 self.last_change.insert(*id, seq);
12241 }
12242 WalRecord::InsertEdge {
12243 src_key, dst_key, ..
12244 }
12245 | WalRecord::DeleteEdge {
12246 src_key, dst_key, ..
12247 } => {
12248 if let Some(src_id) = self.ids.get(src_key) {
12249 self.last_change.insert(src_id, seq);
12250 }
12251 if let Some(dst_id) = self.ids.get(dst_key) {
12252 self.last_change.insert(dst_id, seq);
12253 }
12254 }
12255 WalRecord::InsertEdgeId { src, dst, .. } => {
12256 self.last_change.insert(*src, seq);
12257 self.last_change.insert(*dst, seq);
12258 }
12259 // A count record touches the pair, so it touches both endpoints —
12260 // the same reading `InsertEdgeId` gets, because a duplicate insert
12261 // that raises the count *is* a mutation of that pair. The opt-in
12262 // declaration touches nothing.
12263 WalRecord::SetEdgeCount { src, dst, .. } if !rec.is_multiplicity_decl() => {
12264 self.last_change.insert(*src, seq);
12265 self.last_change.insert(*dst, seq);
12266 }
12267 WalRecord::SetEdgeCount { .. } => {}
12268 // DeleteNode: node is tombstoned; last_changed(key) returns None for
12269 // deleted keys (ids.get() returns None post-tombstone), so no update needed.
12270 // History markers: state no-ops; the underlying mutation already
12271 // touched the relevant nodes' last_change entries.
12272 WalRecord::DeleteNode { .. }
12273 | WalRecord::DerivedEdgeAdded { .. }
12274 | WalRecord::DerivedEdgeRetracted { .. }
12275 | WalRecord::Intern { .. }
12276 | WalRecord::CreateRule { .. }
12277 | WalRecord::DeleteRule { .. }
12278 | WalRecord::RebuildRule { .. }
12279 | WalRecord::CreateView { .. }
12280 | WalRecord::DeleteView { .. }
12281 | WalRecord::EnableFulltext { .. }
12282 | WalRecord::DisableFulltext { .. }
12283 | WalRecord::EnableIndex { .. }
12284 | WalRecord::DisableIndex { .. } => {}
12285 // RenameNode: node id is stable; update last_change via the new key.
12286 // Called after apply(), so ids already reflects new_key.
12287 WalRecord::RenameNode { new_key, .. } => {
12288 if let Some(id) = self.ids.get(new_key) {
12289 self.last_change.insert(id, seq);
12290 }
12291 }
12292 WalRecord::Batch(inner) => {
12293 for inner_rec in inner {
12294 self.update_last_change_from_rec(inner_rec, seq);
12295 }
12296 }
12297 }
12298 }
12299
12300 pub fn node_count(&self) -> usize {
12301 self.ids.len()
12302 }
12303
12304 /// Configure archive retention: keep the `N` newest WAL archives at each
12305 /// [`snapshot_with`] call when `archive_wal: true`.
12306 ///
12307 /// `Some(N)` where N > 0 → prune oldest archives keeping the newest N.
12308 /// `Some(0)` or `None` → unlimited (no pruning).
12309 ///
12310 /// Pruning only ever happens inside [`snapshot_with`]; this method only
12311 /// stores the policy. Archives below the retention limit are deleted
12312 /// oldest-first. The horizon floor is updated so that
12313 /// [`was_linked`] / history APIs return `CommitOutOfRange` for commits
12314 /// in pruned archives rather than silently returning wrong data.
12315 pub fn set_wal_archive_retention(&mut self, keep: Option<u32>) {
12316 self.wal_archive_retention = keep;
12317 }
12318
12319 /// Delete any WAL archives that are fully below the current horizon floor.
12320 ///
12321 /// Orphaned archives arise when the floor is written first during retention
12322 /// pruning and then a crash interrupts the archive-delete sequence. The
12323 /// opening cleanup ensures no subsequent read path sees stale data.
12324 ///
12325 /// Under the monotonic naming scheme, the archive name N equals the
12326 /// cumulative end-frame index of the archive in global commit space (i.e.
12327 /// the archive covers global frames `[prev_n, N)`). An archive is
12328 /// fully orphaned when `N <= wal_horizon_floor`: all of its frames fall
12329 /// below the floor and have already been counted in it.
12330 fn cleanup_orphaned_archives(&mut self) -> Result<()> {
12331 if self.wal_horizon_floor == 0 {
12332 // Floor at 0 means no pruning has ever occurred; nothing to clean.
12333 return Ok(());
12334 }
12335 let archive_ns = self.fs.list_archives()?;
12336 for n in archive_ns {
12337 if n <= self.wal_horizon_floor {
12338 // Archive N ends at global frame N; all its frames are below
12339 // the floor (floor already accounts for them) → orphaned.
12340 self.fs.delete_archive(n).map_err(GraphError::Io)?;
12341 } else {
12342 // Archives are sorted ascending; first one above floor stops scan.
12343 break;
12344 }
12345 }
12346 Ok(())
12347 }
12348
12349 /// Collect all WAL frames from surviving archives (oldest-first) then the
12350 /// live WAL into one flat list, and return the total along with the number
12351 /// of archive frames at the front of the list.
12352 ///
12353 /// Commit indices into the returned list are LOCAL (0 = first frame of
12354 /// oldest surviving archive). To obtain the GLOBAL index add
12355 /// `self.wal_horizon_floor`.
12356 /// How many frames the surviving archives hold, without materialising them.
12357 ///
12358 /// The same count `all_frames` puts at the front of its list. Used to seed
12359 /// [`wal_frames_written`](GraphDb::wal_frames_written) at open without
12360 /// decoding the live WAL a second time; free on a store with no archives,
12361 /// which is most of them.
12362 fn archive_frame_count(&self) -> Result<u64> {
12363 let mut n = 0u64;
12364 for a in self.fs.list_archives()? {
12365 let bytes = self.fs.read_archive(a)?;
12366 let (frames, _) = decode_all(&bytes);
12367 n += frames.len() as u64;
12368 }
12369 Ok(n)
12370 }
12371
12372 fn all_frames(&self) -> Result<(Vec<WalRecord>, u64)> {
12373 let archive_ns = self.fs.list_archives()?;
12374 let mut all: Vec<WalRecord> = Vec::new();
12375 for n in archive_ns {
12376 let bytes = self.fs.read_archive(n)?;
12377 let (frames, _) = decode_all(&bytes);
12378 all.extend(frames);
12379 }
12380 let archive_count = all.len() as u64;
12381 let live_bytes = self.fs.read(FileId::Wal)?;
12382 let (live_frames, _) = decode_all(&live_bytes);
12383 all.extend(live_frames);
12384 Ok((all, archive_count))
12385 }
12386
12387 /// Return the total number of committed WAL frames visible in the current
12388 /// horizon window, including frames in surviving WAL archives.
12389 ///
12390 /// This is the exclusive upper bound for valid `at_commit` indices in
12391 /// `was_linked`. Valid indices are `wal_horizon_floor()..wal_total_commits()`.
12392 ///
12393 /// Returns the horizon floor when all surviving history is empty.
12394 pub fn wal_total_commits(&self) -> Result<u64> {
12395 let (frames, _) = self.all_frames()?;
12396 Ok(self.wal_horizon_floor + frames.len() as u64)
12397 }
12398
12399 /// The global frame index of the first commit reachable through surviving
12400 /// archives (0 when no archives have been pruned).
12401 pub fn wal_horizon_floor(&self) -> u64 {
12402 self.wal_horizon_floor
12403 }
12404
12405 /// Return the per-node change history for `key` by scanning the on-disk WAL.
12406 ///
12407 /// ## Horizon
12408 ///
12409 /// History reaches back only to the last WAL-truncating snapshot, exactly like `open_at`.
12410 /// Snapshots written with `keep_wal: true` preserve deeper history. This is the honest,
12411 /// zero-cost contract; a durable history log is out of scope.
12412 ///
12413 /// ## Derived edges
12414 ///
12415 /// Rule-created (derived) edges are **not** in the WAL and therefore do not appear in
12416 /// history. Only edges written directly by the application are recorded.
12417 ///
12418 /// ## Deleted nodes
12419 ///
12420 /// For nodes that have been deleted, dense-id records (SetPropId, InsertEdgeId) that
12421 /// predate the deletion may not resolve (the id is tombstoned in the live map). The
12422 /// string-keyed `DeleteNode` record still matches and produces a `NodeDeleted` entry.
12423 /// Prop/edge history of a deleted node may therefore be partially unresolvable.
12424 ///
12425 /// ## Dense-id edge entries and tombstoned partners
12426 ///
12427 /// Edge entries from dense-id WAL records (`InsertEdgeId`) are omitted when the partner
12428 /// endpoint's dense id is tombstoned. As a result, a live node's history can contain an
12429 /// `EdgeRemoved` (string-keyed, always resolves) without a corresponding `EdgeAdded`.
12430 /// Build commit-bounded alias intervals for `queried_key`.
12431 ///
12432 /// Returns a list of `(key, valid_from_inclusive, valid_until_exclusive)` tuples.
12433 /// A record written under `key` at commit `c` matches the queried identity iff
12434 /// `c >= valid_from && (valid_until.is_none() || c < valid_until)`.
12435 ///
12436 /// Each alias entry carries both a lower and an upper bound so that key-reuse
12437 /// after a rename is handled correctly: if "a" is renamed to "b" at commit 5,
12438 /// then a NEW node is created as "a" at commit 7 and renamed to "c" at commit 10,
12439 /// querying "c" must NOT surface identity-1's events (commits 0–4 under "a");
12440 /// only identity-2's events (commits 7–9 under "a") are in scope.
12441 ///
12442 /// Only **forward aliasing**: querying the *new* key surfaces events written
12443 /// under the *old* key. The reverse direction is not supported.
12444 fn build_key_alias_intervals(
12445 &self,
12446 frames: &[core_storage::wal::WalRecord],
12447 queried_key: &str,
12448 ) -> Vec<(String, u64, Option<u64>)> {
12449 use core_storage::wal::WalRecord;
12450
12451 // Pre-pass: build reverse_rename and key_starts maps.
12452 let mut reverse_rename: HashMap<String, (String, u64)> = HashMap::new();
12453 let mut key_starts: HashMap<String, Vec<u64>> = HashMap::new();
12454
12455 for (local_i, frame) in frames.iter().enumerate() {
12456 let commit = self.wal_horizon_floor + local_i as u64;
12457 let records: &[WalRecord] = match frame {
12458 WalRecord::Batch(inner) => inner.as_slice(),
12459 single => std::slice::from_ref(single),
12460 };
12461 for rec in records {
12462 match rec {
12463 WalRecord::InsertNode { key, .. } | WalRecord::InsertNodeId { key, .. } => {
12464 key_starts.entry(key.clone()).or_default().push(commit);
12465 }
12466 WalRecord::RenameNode { old_key, new_key } => {
12467 // new_key came into existence at this commit.
12468 key_starts.entry(new_key.clone()).or_default().push(commit);
12469 // Record the reverse rename: new_key was introduced by renaming old_key.
12470 reverse_rename.insert(new_key.clone(), (old_key.clone(), commit));
12471 }
12472 _ => {}
12473 }
12474 }
12475 }
12476
12477 // Build alias intervals by following the reverse rename chain.
12478 let mut result: Vec<(String, u64, Option<u64>)> = Vec::new();
12479 let mut current_key = queried_key.to_string();
12480 let mut current_valid_until: Option<u64> = None;
12481
12482 loop {
12483 // valid_from: the most recent commit where current_key was assigned to this
12484 // identity. For aliases (valid_until = Some(vu)), find the last start event
12485 // for the key strictly before vu — this is where the alias's occupancy by
12486 // this identity began, correctly excluding prior identities that reused the key.
12487 let valid_from = if let Some(vu) = current_valid_until {
12488 key_starts
12489 .get(¤t_key)
12490 .and_then(|starts| starts.iter().rev().find(|&&s| s < vu).copied())
12491 .unwrap_or(self.wal_horizon_floor)
12492 } else {
12493 // Queried key — no upper bound; may have been introduced at any commit.
12494 self.wal_horizon_floor
12495 };
12496
12497 result.push((current_key.clone(), valid_from, current_valid_until));
12498
12499 match reverse_rename.get(¤t_key) {
12500 Some((old_key, rename_commit)) => {
12501 current_valid_until = Some(*rename_commit);
12502 current_key = old_key.clone();
12503 }
12504 None => break,
12505 }
12506 }
12507
12508 result
12509 }
12510
12511 /// Returns true if `record_key` matches any alias interval that covers `commit`.
12512 fn aliases_match(
12513 intervals: &[(String, u64, Option<u64>)],
12514 record_key: &str,
12515 commit: u64,
12516 ) -> bool {
12517 intervals
12518 .iter()
12519 .any(|(k, vf, vu)| k == record_key && commit >= *vf && vu.is_none_or(|u| commit < u))
12520 }
12521
12522 /// Return the change history of node `key` by scanning the on-disk WAL.
12523 ///
12524 /// ## Horizon
12525 ///
12526 /// History reaches back only as far as the retained WAL. The returned
12527 /// [`HistoryResult`](crate::history::HistoryResult) carries `total_commits`
12528 /// (the exclusive upper bound for valid commit indices) and `horizon` (the
12529 /// oldest commit still reachable). When `horizon > 0`, older events were
12530 /// pruned and are not in `items`.
12531 pub fn node_history(
12532 &self,
12533 key: &str,
12534 ) -> Result<crate::history::HistoryResult<crate::history::HistoryEntry>> {
12535 use crate::history::{HistoryChange, HistoryEntry, HistoryResult};
12536 use core_storage::wal::WalRecord;
12537
12538 let (frames, _) = self.all_frames()?;
12539 let total_commits = self.wal_horizon_floor + frames.len() as u64;
12540
12541 // Resolve commit-bounded alias intervals for `key` (handles renames in the WAL).
12542 let alias_intervals = self.build_key_alias_intervals(&frames, key);
12543
12544 let mut out: Vec<HistoryEntry> = Vec::new();
12545
12546 for (local_i, frame) in frames.iter().enumerate() {
12547 let commit = self.wal_horizon_floor + local_i as u64;
12548 // Collect the inner records to process — Batch is one commit, single records are one commit.
12549 let records: &[WalRecord] = match frame {
12550 WalRecord::Batch(inner) => inner.as_slice(),
12551 single => std::slice::from_ref(single),
12552 };
12553
12554 for rec in records {
12555 let change = match rec {
12556 WalRecord::InsertNode { label, key: k, .. }
12557 if Self::aliases_match(&alias_intervals, k, commit) =>
12558 {
12559 Some(HistoryChange::NodeInserted {
12560 label: label.clone(),
12561 })
12562 }
12563 WalRecord::InsertNodeId { label, key: k, .. }
12564 if Self::aliases_match(&alias_intervals, k, commit) =>
12565 {
12566 let label_str = match self.syms.resolve(*label) {
12567 Some(s) => s.to_string(),
12568 None => continue,
12569 };
12570 Some(HistoryChange::NodeInserted { label: label_str })
12571 }
12572 WalRecord::SetProp {
12573 key: k,
12574 field,
12575 value,
12576 } if Self::aliases_match(&alias_intervals, k, commit) => {
12577 Some(HistoryChange::PropSet {
12578 field: field.clone(),
12579 value: value.clone(),
12580 })
12581 }
12582 WalRecord::SetPropId { id, field, value } => {
12583 // Use key_of_historical (not key_of) so a node's prop_set
12584 // events remain visible after the node is later deleted:
12585 // key_of returns None for a tombstoned id, which would
12586 // silently drop every PropSet between insert and delete.
12587 // Mirrors the InsertEdgeId arm below and edge_history's
12588 // own id-keyed arms.
12589 match self.ids.key_of_historical(*id) {
12590 // key_of_historical returns the last-known (possibly
12591 // post-rename, possibly post-delete) key; compare to queried key.
12592 Some(resolved) if resolved == key => {
12593 let field_str = match self.syms.resolve(*field) {
12594 Some(s) => s.to_string(),
12595 None => continue,
12596 };
12597 Some(HistoryChange::PropSet {
12598 field: field_str,
12599 value: value.clone(),
12600 })
12601 }
12602 _ => None,
12603 }
12604 }
12605 WalRecord::RemoveProp { key: k, field }
12606 if Self::aliases_match(&alias_intervals, k, commit) =>
12607 {
12608 Some(HistoryChange::PropRemoved {
12609 field: field.clone(),
12610 })
12611 }
12612 WalRecord::InsertEdge {
12613 edge_type,
12614 src_key,
12615 dst_key,
12616 } => {
12617 if Self::aliases_match(&alias_intervals, src_key, commit) {
12618 Some(HistoryChange::EdgeAdded {
12619 edge_type: edge_type.clone(),
12620 other: dst_key.clone(),
12621 outgoing: true,
12622 })
12623 } else if Self::aliases_match(&alias_intervals, dst_key, commit) {
12624 Some(HistoryChange::EdgeAdded {
12625 edge_type: edge_type.clone(),
12626 other: src_key.clone(),
12627 outgoing: false,
12628 })
12629 } else {
12630 None
12631 }
12632 }
12633 WalRecord::InsertEdgeId { etype, src, dst } => {
12634 let etype_str = match self.syms.resolve(*etype) {
12635 Some(s) => s.to_string(),
12636 None => continue,
12637 };
12638 // key_of_historical (not key_of): an edge added before
12639 // either endpoint was later deleted must still resolve —
12640 // see the SetPropId arm above and edge_history's
12641 // InsertEdgeId arm, which use the same lookup for the
12642 // same reason.
12643 let src_key = self.ids.key_of_historical(*src);
12644 let dst_key = self.ids.key_of_historical(*dst);
12645 if src_key == Some(key) {
12646 let other = match dst_key {
12647 Some(s) => s.to_string(),
12648 None => continue,
12649 };
12650 Some(HistoryChange::EdgeAdded {
12651 edge_type: etype_str,
12652 other,
12653 outgoing: true,
12654 })
12655 } else if dst_key == Some(key) {
12656 let other = match src_key {
12657 Some(s) => s.to_string(),
12658 None => continue,
12659 };
12660 Some(HistoryChange::EdgeAdded {
12661 edge_type: etype_str,
12662 other,
12663 outgoing: false,
12664 })
12665 } else {
12666 None
12667 }
12668 }
12669 WalRecord::DeleteEdge {
12670 edge_type,
12671 src_key,
12672 dst_key,
12673 } => {
12674 if Self::aliases_match(&alias_intervals, src_key, commit) {
12675 Some(HistoryChange::EdgeRemoved {
12676 edge_type: edge_type.clone(),
12677 other: dst_key.clone(),
12678 outgoing: true,
12679 })
12680 } else if Self::aliases_match(&alias_intervals, dst_key, commit) {
12681 Some(HistoryChange::EdgeRemoved {
12682 edge_type: edge_type.clone(),
12683 other: src_key.clone(),
12684 outgoing: false,
12685 })
12686 } else {
12687 None
12688 }
12689 }
12690 WalRecord::DeleteNode { key: k }
12691 if Self::aliases_match(&alias_intervals, k, commit) =>
12692 {
12693 Some(HistoryChange::NodeDeleted)
12694 }
12695 // Skip: rule/view/fulltext/intern metadata; Batch wrapper handled above.
12696 _ => None,
12697 };
12698
12699 if let Some(change) = change {
12700 out.push(HistoryEntry { commit, change });
12701 }
12702 }
12703 }
12704
12705 Ok(HistoryResult {
12706 items: out,
12707 total_commits,
12708 horizon: self.wal_horizon_floor,
12709 })
12710 }
12711
12712 /// Return the per-edge change history between nodes `a` and `b` by scanning
12713 /// the on-disk WAL.
12714 ///
12715 /// ## Horizon
12716 ///
12717 /// History reaches back only to the last WAL-truncating snapshot, exactly
12718 /// like `node_history` and `open_at`. The returned [`HistoryResult`] carries
12719 /// `total_commits` (= number of WAL frames), which is the exclusive upper
12720 /// bound for valid commit indices.
12721 ///
12722 /// ## Derived edges
12723 ///
12724 /// Rule-derived edges appear via `DerivedEdgeAdded` / `DerivedEdgeRetracted`
12725 /// WAL markers written by `log_then_apply_with` after each rule-firing
12726 /// mutation. The `rule` field of those events carries the rule name.
12727 ///
12728 /// ## DeleteNode
12729 ///
12730 /// When a node is deleted, its manual incident edges are swept inline without
12731 /// individual `DeleteEdge` WAL records. `edge_history` detects `DeleteNode`
12732 /// events for either endpoint and synthesises `Retracted(rule:None)` events
12733 /// for each manual edge that was active at that point. Derived edges active at
12734 /// the time of deletion are handled by the `DerivedEdgeRetracted` marker that
12735 /// the engine appends immediately after the `DeleteNode` record; those events
12736 /// carry correct rule attribution and are emitted by the marker arm, not the
12737 /// synthetic sweep.
12738 ///
12739 /// ## Masks
12740 ///
12741 /// Like `node_history`, this method has no mask parameter and returns WAL
12742 /// history regardless of any role mask. For masked history semantics, apply
12743 /// the mask at the caller level.
12744 pub fn edge_history(
12745 &self,
12746 a: &str,
12747 b: &str,
12748 ) -> Result<crate::history::HistoryResult<crate::history::EdgeHistoryEvent>> {
12749 use crate::history::{EdgeEvent, EdgeHistoryEvent, HistoryResult};
12750 use core_storage::wal::WalRecord;
12751
12752 let (frames, _) = self.all_frames()?;
12753 let total_commits = self.wal_horizon_floor + frames.len() as u64;
12754
12755 // Resolve all historical names for a and b (handles RenameNode in the WAL).
12756 // Intervals are commit-bounded so recycled keys don't contaminate histories.
12757 let alias_a = self.build_key_alias_intervals(&frames, a);
12758 let alias_b = self.build_key_alias_intervals(&frames, b);
12759
12760 // Active edges between a and b tracked as (edge_type, src_key, dst_key, is_derived).
12761 // The is_derived flag is used by the DeleteNode sweep: manual edges are
12762 // swept with a synthetic Retracted(rule:None); derived edges are skipped
12763 // because the engine writes a DerivedEdgeRetracted marker immediately after
12764 // the DeleteNode record, which carries the correct rule attribution.
12765 let mut active: Vec<(String, String, String, bool)> = Vec::new();
12766 let mut out: Vec<EdgeHistoryEvent> = Vec::new();
12767
12768 for (local_i, frame) in frames.iter().enumerate() {
12769 let commit = self.wal_horizon_floor + local_i as u64;
12770 let records: &[WalRecord] = match frame {
12771 WalRecord::Batch(inner) => inner.as_slice(),
12772 single => std::slice::from_ref(single),
12773 };
12774
12775 for rec in records {
12776 match rec {
12777 WalRecord::InsertEdge {
12778 edge_type,
12779 src_key,
12780 dst_key,
12781 } => {
12782 let is_ab = Self::aliases_match(&alias_a, src_key, commit)
12783 && Self::aliases_match(&alias_b, dst_key, commit);
12784 let is_ba = Self::aliases_match(&alias_b, src_key, commit)
12785 && Self::aliases_match(&alias_a, dst_key, commit);
12786 if is_ab || is_ba {
12787 active.push((
12788 edge_type.clone(),
12789 src_key.clone(),
12790 dst_key.clone(),
12791 false,
12792 ));
12793 out.push(EdgeHistoryEvent {
12794 edge_type: edge_type.clone(),
12795 commit,
12796 event: EdgeEvent::Added,
12797 rule: None,
12798 });
12799 }
12800 }
12801 WalRecord::InsertEdgeId { etype, src, dst } => {
12802 let etype_str = match self.syms.resolve(*etype) {
12803 Some(s) => s.to_string(),
12804 None => continue,
12805 };
12806 // Use key_of_historical so tombstoned nodes (deleted
12807 // later in the WAL) still resolve during the scan.
12808 let src_key = self.ids.key_of_historical(*src);
12809 let dst_key = self.ids.key_of_historical(*dst);
12810 let is_ab = src_key == Some(a) && dst_key == Some(b);
12811 let is_ba = src_key == Some(b) && dst_key == Some(a);
12812 if is_ab || is_ba {
12813 let src_str = src_key.unwrap().to_string();
12814 let dst_str = dst_key.unwrap().to_string();
12815 active.push((etype_str.clone(), src_str, dst_str, false));
12816 out.push(EdgeHistoryEvent {
12817 edge_type: etype_str,
12818 commit,
12819 event: EdgeEvent::Added,
12820 rule: None,
12821 });
12822 }
12823 }
12824 WalRecord::DeleteEdge {
12825 edge_type,
12826 src_key,
12827 dst_key,
12828 } => {
12829 let is_ab = Self::aliases_match(&alias_a, src_key, commit)
12830 && Self::aliases_match(&alias_b, dst_key, commit);
12831 let is_ba = Self::aliases_match(&alias_b, src_key, commit)
12832 && Self::aliases_match(&alias_a, dst_key, commit);
12833 if is_ab || is_ba {
12834 // Remove the first matching active entry (flag ignored).
12835 if let Some(pos) = active.iter().position(|(et, s, d, _)| {
12836 et == edge_type && s == src_key && d == dst_key
12837 }) {
12838 active.remove(pos);
12839 }
12840 out.push(EdgeHistoryEvent {
12841 edge_type: edge_type.clone(),
12842 commit,
12843 event: EdgeEvent::Retracted,
12844 rule: None,
12845 });
12846 }
12847 }
12848 WalRecord::DeleteNode { key: k }
12849 if Self::aliases_match(&alias_a, k, commit)
12850 || Self::aliases_match(&alias_b, k, commit) =>
12851 {
12852 // Sweep: implicitly retract only MANUAL active edges.
12853 // Derived active edges are skipped here because the rule
12854 // engine appends a DerivedEdgeRetracted marker immediately
12855 // after this DeleteNode record; that marker produces the
12856 // single correctly-attributed Retracted event. Derived
12857 // entries are dropped from `active` (the marker arm's
12858 // idempotent retain finds nothing to remove).
12859 for (et, _, _, is_derived) in active.drain(..) {
12860 if !is_derived {
12861 out.push(EdgeHistoryEvent {
12862 edge_type: et,
12863 commit,
12864 event: EdgeEvent::Retracted,
12865 rule: None,
12866 });
12867 }
12868 // Derived: drop silently; marker carries the Retracted event.
12869 }
12870 }
12871 WalRecord::DerivedEdgeAdded {
12872 rule,
12873 edge_type: et,
12874 src_key,
12875 dst_key,
12876 } => {
12877 let is_ab = Self::aliases_match(&alias_a, src_key, commit)
12878 && Self::aliases_match(&alias_b, dst_key, commit);
12879 let is_ba = Self::aliases_match(&alias_b, src_key, commit)
12880 && Self::aliases_match(&alias_a, dst_key, commit);
12881 if is_ab || is_ba {
12882 active.push((et.clone(), src_key.clone(), dst_key.clone(), true));
12883 out.push(EdgeHistoryEvent {
12884 edge_type: et.clone(),
12885 commit,
12886 event: EdgeEvent::Added,
12887 rule: Some(rule.clone()),
12888 });
12889 }
12890 }
12891 WalRecord::DerivedEdgeRetracted {
12892 rule,
12893 edge_type: et,
12894 src_key,
12895 dst_key,
12896 } => {
12897 let is_ab = Self::aliases_match(&alias_a, src_key, commit)
12898 && Self::aliases_match(&alias_b, dst_key, commit);
12899 let is_ba = Self::aliases_match(&alias_b, src_key, commit)
12900 && Self::aliases_match(&alias_a, dst_key, commit);
12901 if is_ab || is_ba {
12902 // Push unconditionally: a derived edge whose Added marker
12903 // predates the history horizon has no `active` entry, but
12904 // the retraction is still a real in-window event.
12905 // Remove from active idempotently if present.
12906 active.retain(|(aet, s, d, _)| {
12907 !(aet == et && s == src_key && d == dst_key)
12908 });
12909 out.push(EdgeHistoryEvent {
12910 edge_type: et.clone(),
12911 commit,
12912 event: EdgeEvent::Retracted,
12913 rule: Some(rule.clone()),
12914 });
12915 }
12916 }
12917 // All other records (InsertNode, SetProp, CreateRule, etc.)
12918 // do not affect edges between a and b.
12919 _ => {}
12920 }
12921 }
12922 }
12923
12924 Ok(HistoryResult {
12925 items: out,
12926 total_commits,
12927 horizon: self.wal_horizon_floor,
12928 })
12929 }
12930
12931 /// Return `true` iff an edge of `edge_type` existed between `a` and `b`
12932 /// (in either direction) at the WAL commit `at_commit`.
12933 ///
12934 /// ## Horizon
12935 ///
12936 /// Valid commit indices are `0..total_commits` where `total_commits` is the
12937 /// number of WAL frames. An `at_commit >= total_commits` is outside the
12938 /// visible horizon and returns [`GraphError::CommitOutOfRange`].
12939 ///
12940 /// ## Derived edges
12941 ///
12942 /// Rule-derived edges are tracked via `DerivedEdgeAdded` / `DerivedEdgeRetracted`
12943 /// WAL markers appended at firing time (Task 1). `was_linked` reads these markers
12944 /// and therefore includes derived edges in its point-in-time evaluation,
12945 /// matching `edge_history`'s fidelity.
12946 pub fn was_linked(&self, a: &str, b: &str, edge_type: &str, at_commit: u64) -> Result<bool> {
12947 use core_storage::wal::WalRecord;
12948
12949 let (frames, _) = self.all_frames()?;
12950 let total_commits = self.wal_horizon_floor + frames.len() as u64;
12951
12952 // Horizon floor: commits in pruned archives are unreachable.
12953 if at_commit < self.wal_horizon_floor {
12954 return Err(GraphError::CommitOutOfRange {
12955 commit: at_commit,
12956 total: total_commits,
12957 floor: self.wal_horizon_floor,
12958 });
12959 }
12960 if at_commit >= total_commits {
12961 return Err(GraphError::CommitOutOfRange {
12962 commit: at_commit,
12963 total: total_commits,
12964 floor: self.wal_horizon_floor,
12965 });
12966 }
12967
12968 // Resolve all historical names for a and b (handles RenameNode in the WAL).
12969 // Intervals are commit-bounded so recycled keys don't contaminate point-in-time reads.
12970 let alias_a = self.build_key_alias_intervals(&frames, a);
12971 let alias_b = self.build_key_alias_intervals(&frames, b);
12972
12973 // Local index into surviving frames (0 = first frame of oldest archive).
12974 let local_commit = at_commit - self.wal_horizon_floor;
12975
12976 // Replay local frames 0..=local_commit, tracking active edges.
12977 let mut active: BTreeSet<(String, String, String)> = BTreeSet::new();
12978
12979 for (local_i, frame) in frames.iter().enumerate().take((local_commit + 1) as usize) {
12980 let commit = self.wal_horizon_floor + local_i as u64;
12981 let records: &[WalRecord] = match frame {
12982 WalRecord::Batch(inner) => inner.as_slice(),
12983 single => std::slice::from_ref(single),
12984 };
12985
12986 for rec in records {
12987 match rec {
12988 WalRecord::InsertEdge {
12989 edge_type: et,
12990 src_key,
12991 dst_key,
12992 } => {
12993 let is_ab = Self::aliases_match(&alias_a, src_key, commit)
12994 && Self::aliases_match(&alias_b, dst_key, commit);
12995 let is_ba = Self::aliases_match(&alias_b, src_key, commit)
12996 && Self::aliases_match(&alias_a, dst_key, commit);
12997 if is_ab || is_ba {
12998 active.insert((et.clone(), src_key.clone(), dst_key.clone()));
12999 }
13000 }
13001 WalRecord::InsertEdgeId { etype, src, dst } => {
13002 let etype_str = match self.syms.resolve(*etype) {
13003 Some(s) => s.to_string(),
13004 None => continue,
13005 };
13006 // Use key_of_historical so tombstoned nodes resolve.
13007 let src_key = self.ids.key_of_historical(*src);
13008 let dst_key = self.ids.key_of_historical(*dst);
13009 let is_ab = src_key == Some(a) && dst_key == Some(b);
13010 let is_ba = src_key == Some(b) && dst_key == Some(a);
13011 if is_ab || is_ba {
13012 active.insert((
13013 etype_str,
13014 src_key.unwrap().to_string(),
13015 dst_key.unwrap().to_string(),
13016 ));
13017 }
13018 }
13019 WalRecord::DeleteEdge {
13020 edge_type: et,
13021 src_key,
13022 dst_key,
13023 } => {
13024 let is_ab = Self::aliases_match(&alias_a, src_key, commit)
13025 && Self::aliases_match(&alias_b, dst_key, commit);
13026 let is_ba = Self::aliases_match(&alias_b, src_key, commit)
13027 && Self::aliases_match(&alias_a, dst_key, commit);
13028 if is_ab || is_ba {
13029 active.remove(&(et.clone(), src_key.clone(), dst_key.clone()));
13030 }
13031 }
13032 WalRecord::DeleteNode { key: k }
13033 if Self::aliases_match(&alias_a, k, commit)
13034 || Self::aliases_match(&alias_b, k, commit) =>
13035 {
13036 // All edges touching the deleted node are gone.
13037 active.retain(|(_, s, d)| s != k && d != k);
13038 }
13039 WalRecord::DerivedEdgeAdded {
13040 edge_type: et,
13041 src_key,
13042 dst_key,
13043 ..
13044 } => {
13045 let is_ab = Self::aliases_match(&alias_a, src_key, commit)
13046 && Self::aliases_match(&alias_b, dst_key, commit);
13047 let is_ba = Self::aliases_match(&alias_b, src_key, commit)
13048 && Self::aliases_match(&alias_a, dst_key, commit);
13049 if is_ab || is_ba {
13050 active.insert((et.clone(), src_key.clone(), dst_key.clone()));
13051 }
13052 }
13053 WalRecord::DerivedEdgeRetracted {
13054 edge_type: et,
13055 src_key,
13056 dst_key,
13057 ..
13058 } => {
13059 let is_ab = Self::aliases_match(&alias_a, src_key, commit)
13060 && Self::aliases_match(&alias_b, dst_key, commit);
13061 let is_ba = Self::aliases_match(&alias_b, src_key, commit)
13062 && Self::aliases_match(&alias_a, dst_key, commit);
13063 if is_ab || is_ba {
13064 active.remove(&(et.clone(), src_key.clone(), dst_key.clone()));
13065 }
13066 }
13067 _ => {}
13068 }
13069 }
13070 }
13071
13072 Ok(active.iter().any(|(et, _, _)| et == edge_type))
13073 }
13074
13075 /// Every edge incident to `key` — either endpoint — that existed at WAL
13076 /// commit `commit`, from ONE scan of the WAL.
13077 ///
13078 /// This is the bulk form of [`was_linked`](GraphDb::was_linked): answering
13079 /// "what did K's relationships look like at commit C" with one call instead
13080 /// of one [`edge_history`](GraphDb::edge_history) per candidate partner.
13081 /// The two agree edge for edge.
13082 ///
13083 /// Results are sorted by `(edge_type, src_key, dst_key)`.
13084 ///
13085 /// ## Horizon
13086 ///
13087 /// Valid commit indices are `wal_horizon_floor()..wal_total_commits()`;
13088 /// anything outside is [`GraphError::CommitOutOfRange`], exactly like
13089 /// `was_linked`. An unknown key is not an error — it simply had no edges.
13090 ///
13091 /// ## Derived edges
13092 ///
13093 /// `DerivedEdgeAdded` / `DerivedEdgeRetracted` markers carry rule
13094 /// attribution, so a rule-owned edge comes back with `derived: true` and
13095 /// `rule: Some(name)`.
13096 ///
13097 /// ## Renames
13098 ///
13099 /// `key` is matched through the same commit-bounded alias intervals
13100 /// `edge_history` uses, so querying a node's *current* key surfaces edges
13101 /// written under an earlier name. Endpoint keys in the result are reported
13102 /// under the name the node carries today, so they can be fed straight back
13103 /// into `node_info`, `explain` or another `edges_at`.
13104 ///
13105 /// ## Masks
13106 ///
13107 /// Like `edge_history` and `node_history`, this reads the WAL regardless of
13108 /// any role mask. Apply masking at the caller level.
13109 pub fn edges_at(&self, key: &str, commit: u64) -> Result<Vec<EdgeAt>> {
13110 use core_storage::wal::WalRecord;
13111
13112 let (frames, _) = self.all_frames()?;
13113 let total_commits = self.wal_horizon_floor + frames.len() as u64;
13114
13115 // Horizon floor: commits in pruned archives are unreachable.
13116 if commit < self.wal_horizon_floor || commit >= total_commits {
13117 return Err(GraphError::CommitOutOfRange {
13118 commit,
13119 total: total_commits,
13120 floor: self.wal_horizon_floor,
13121 });
13122 }
13123
13124 // Commit-bounded historical names of `key` (handles RenameNode).
13125 let alias = self.build_key_alias_intervals(&frames, key);
13126
13127 // Forward rename chain, for reporting endpoints under their current
13128 // names: old key → [(commit, new key)] in ascending commit order.
13129 // Built over the whole WAL, not just the prefix up to `commit`, because
13130 // a rename after `commit` still changes what the node is called today.
13131 let mut renames: HashMap<String, Vec<(u64, String)>> = HashMap::new();
13132 for (local_i, frame) in frames.iter().enumerate() {
13133 let c = self.wal_horizon_floor + local_i as u64;
13134 let records: &[WalRecord] = match frame {
13135 WalRecord::Batch(inner) => inner.as_slice(),
13136 single => std::slice::from_ref(single),
13137 };
13138 for rec in records {
13139 if let WalRecord::RenameNode { old_key, new_key } = rec {
13140 renames
13141 .entry(old_key.clone())
13142 .or_default()
13143 .push((c, new_key.clone()));
13144 }
13145 }
13146 }
13147
13148 // The name a node written as `k` at commit `from` carries today.
13149 // Follows the first rename at or after `from`, then keeps going. The
13150 // iteration cap bounds a rename cycle inside a single batch.
13151 let canon = |k: &str, from: u64| -> String {
13152 if renames.is_empty() {
13153 return k.to_string();
13154 }
13155 let mut cur = k.to_string();
13156 let mut at = from;
13157 for _ in 0..64 {
13158 match renames
13159 .get(&cur)
13160 .and_then(|v| v.iter().find(|(c, _)| *c >= at))
13161 {
13162 Some((c, new)) => {
13163 at = *c;
13164 cur = new.clone();
13165 }
13166 None => break,
13167 }
13168 }
13169 cur
13170 };
13171
13172 let local_commit = commit - self.wal_horizon_floor;
13173 // (edge_type, src_key, dst_key) → (derived, rule)
13174 let mut active: BTreeMap<(String, String, String), (bool, Option<String>)> =
13175 BTreeMap::new();
13176
13177 for (local_i, frame) in frames.iter().enumerate().take((local_commit + 1) as usize) {
13178 let c = self.wal_horizon_floor + local_i as u64;
13179 let records: &[WalRecord] = match frame {
13180 WalRecord::Batch(inner) => inner.as_slice(),
13181 single => std::slice::from_ref(single),
13182 };
13183
13184 for rec in records {
13185 match rec {
13186 WalRecord::InsertEdge {
13187 edge_type,
13188 src_key,
13189 dst_key,
13190 } => {
13191 if Self::aliases_match(&alias, src_key, c)
13192 || Self::aliases_match(&alias, dst_key, c)
13193 {
13194 active.insert(
13195 (edge_type.clone(), canon(src_key, c), canon(dst_key, c)),
13196 (false, None),
13197 );
13198 }
13199 }
13200 WalRecord::InsertEdgeId { etype, src, dst } => {
13201 let Some(etype_str) = self.syms.resolve(*etype) else {
13202 continue;
13203 };
13204 // `key_of_historical` resolves tombstoned ids too, and
13205 // already returns the node's current key — no rename
13206 // canonicalisation needed on this arm.
13207 let (Some(src_key), Some(dst_key)) = (
13208 self.ids.key_of_historical(*src),
13209 self.ids.key_of_historical(*dst),
13210 ) else {
13211 continue;
13212 };
13213 if src_key == key || dst_key == key {
13214 active.insert(
13215 (
13216 etype_str.to_string(),
13217 src_key.to_string(),
13218 dst_key.to_string(),
13219 ),
13220 (false, None),
13221 );
13222 }
13223 }
13224 WalRecord::DeleteEdge {
13225 edge_type,
13226 src_key,
13227 dst_key,
13228 } => {
13229 if Self::aliases_match(&alias, src_key, c)
13230 || Self::aliases_match(&alias, dst_key, c)
13231 {
13232 active.remove(&(
13233 edge_type.clone(),
13234 canon(src_key, c),
13235 canon(dst_key, c),
13236 ));
13237 }
13238 }
13239 WalRecord::DeleteNode { key: k } => {
13240 if active.is_empty() {
13241 continue;
13242 }
13243 if Self::aliases_match(&alias, k, c) {
13244 // Our node is gone; every incident edge goes with it.
13245 active.clear();
13246 } else {
13247 // A partner is gone; its edges to us go with it.
13248 let ck = canon(k, c);
13249 active.retain(|(_, s, d), _| *s != ck && *d != ck);
13250 }
13251 }
13252 WalRecord::DerivedEdgeAdded {
13253 rule,
13254 edge_type,
13255 src_key,
13256 dst_key,
13257 } => {
13258 if Self::aliases_match(&alias, src_key, c)
13259 || Self::aliases_match(&alias, dst_key, c)
13260 {
13261 active.insert(
13262 (edge_type.clone(), canon(src_key, c), canon(dst_key, c)),
13263 (true, Some(rule.clone())),
13264 );
13265 }
13266 }
13267 WalRecord::DerivedEdgeRetracted {
13268 edge_type,
13269 src_key,
13270 dst_key,
13271 ..
13272 } => {
13273 if Self::aliases_match(&alias, src_key, c)
13274 || Self::aliases_match(&alias, dst_key, c)
13275 {
13276 active.remove(&(
13277 edge_type.clone(),
13278 canon(src_key, c),
13279 canon(dst_key, c),
13280 ));
13281 }
13282 }
13283 // InsertNode, SetProp, CreateRule, … do not move edges.
13284 _ => {}
13285 }
13286 }
13287 }
13288
13289 // BTreeMap iteration is already (edge_type, src, dst) order.
13290 Ok(active
13291 .into_iter()
13292 .map(|((edge_type, src_key, dst_key), (derived, rule))| EdgeAt {
13293 edge_type,
13294 src_key,
13295 dst_key,
13296 derived,
13297 rule,
13298 })
13299 .collect())
13300 }
13301
13302 /// The derived edges that would be retracted and derived if `key.field`
13303 /// were set to `value` — computed WITHOUT writing anything.
13304 ///
13305 /// Nothing is committed and nothing on `self` is mutated: the rule engine's
13306 /// provenance, its candidate indexes, the topology and the property columns
13307 /// are all cloned first, the change is applied to the clone, and the real
13308 /// per-node re-derivation (`RuleEngine::on_node_changed` — the same call
13309 /// `set_prop` makes during apply) runs against it. The derived-edge deltas
13310 /// it emits are the answer, so rule semantics — predicates, top-k,
13311 /// via-hops, chaining, weights — are the engine's, not a re-implementation.
13312 ///
13313 /// Works on a read-only handle.
13314 ///
13315 /// **While a rule's vector index is still building** (`RuleStats::building`)
13316 /// the clone carries no pending-build state, so this reports the edges that
13317 /// rule would derive — which the live store will not derive until its
13318 /// backfill runs. Right about the end state, early about the timing.
13319 ///
13320 /// Returns `Err(KeyNotFound)` for an unknown or tombstoned key and
13321 /// `Err(ViewPropReadOnly)` for a field a view owns — matching
13322 /// [`set_prop`](GraphDb::set_prop)'s validation. A change with no effect
13323 /// (the node already holds `value`, or no rule watches `field`) returns
13324 /// empty lists.
13325 ///
13326 /// ## Cost
13327 ///
13328 /// One clone of the property columns, the topology overlay, the symbol
13329 /// interner, the edge properties and the provenance map, plus one candidate
13330 /// re-index (O(nodes × rules)). That is much cheaper than copying the store
13331 /// directory, but it is not free — this is an interactive "what if", not a
13332 /// hot path.
13333 pub fn what_if_set_prop(&self, key: &str, field: &str, value: Value) -> Result<WhatIf> {
13334 // The engine's provenance, HNSW and IVF state live in the mmap'd base
13335 // until something asks for them. On a store opened cold from a snapshot
13336 // this is the first ask, and without it the clone below starts from an
13337 // empty provenance map: nothing to retract, so `lost` comes back empty.
13338 self.ensure_v8_base_sections_loaded();
13339
13340 let empty = WhatIf {
13341 lost: Vec::new(),
13342 gained: Vec::new(),
13343 };
13344
13345 if let Some(view_name) = self.view_store.view_for_prop(field) {
13346 return Err(GraphError::ViewPropReadOnly {
13347 view_name: view_name.to_string(),
13348 });
13349 }
13350 MutPreview::new(self).check_live_key(key)?;
13351 let id = self
13352 .ids
13353 .get(key)
13354 .ok_or_else(|| GraphError::KeyNotFound { key: key.into() })?;
13355
13356 let rules: Vec<RuleDef> = self.engine.rules().cloned().collect();
13357 if rules.is_empty() {
13358 return Ok(empty);
13359 }
13360
13361 // No rule watches this field → no derivation can change.
13362 if !rules.iter().any(|r| r.watched_fields().contains(field)) {
13363 return Ok(empty);
13364 }
13365
13366 let old_value = build_props_view(&self.props, &self.base)
13367 .get(id, field)
13368 .map(|vr| vr.into_value());
13369 if old_value.as_ref() == Some(&value) {
13370 return Ok(empty);
13371 }
13372
13373 // --- Clone every piece of state the re-derivation writes to. ---
13374 // `what_if` mutates its own throwaway copies and writes nothing, so it
13375 // takes owned clones rather than sharing the `Arc`s. Deref-clone, not
13376 // `Arc::clone`: sharing here would make `make_mut` copy on first touch
13377 // anyway, and an owned local keeps the rest of this function unchanged.
13378 let mut props = (*self.props).clone();
13379 let mut topo = (*self.topo).clone();
13380 let mut syms = (*self.syms).clone();
13381 let mut edge_props = (*self.edge_props).clone();
13382
13383 let mut tripped: BTreeMap<String, bool> = BTreeMap::new();
13384 let mut fires: BTreeMap<String, u64> = BTreeMap::new();
13385 for r in &rules {
13386 tripped.insert(r.name.clone(), self.engine.is_tripped(&r.name));
13387 fires.insert(r.name.clone(), self.engine.fire_count(&r.name));
13388 }
13389 // `provenance()` decodes retained snapshot bytes on first use; the
13390 // engine clone needs the real map, not an empty one.
13391 let provenance = self.engine.provenance().clone();
13392 let mut engine = core_rules::RuleEngine::from_persist(rules, provenance, tripped, fires);
13393
13394 // Build the candidate indexes from the state BEFORE the change, exactly
13395 // as apply() sees them: `on_node_changed` withdraws the node under its
13396 // old value and refiles it under the new one, so the index must not
13397 // already reflect the change.
13398 engine.reindex_all_load_state(
13399 &self.ids,
13400 &syms,
13401 &self.labels,
13402 build_props_view(&self.props, &self.base),
13403 self.engine.export_ivf_state(),
13404 self.engine.export_hnsw_state_passthrough(),
13405 );
13406 engine.set_emit_deltas(true);
13407
13408 // --- Apply the hypothetical change and re-derive. ---
13409 props.set(id, field, value);
13410 {
13411 let mut gm = make_graph_mut(
13412 &self.ids,
13413 &mut syms,
13414 &self.labels,
13415 build_props_view(&props, &self.base),
13416 &mut topo,
13417 &self.base,
13418 &mut edge_props,
13419 );
13420 engine.on_node_changed(id, Some((field, old_value)), &mut gm);
13421 }
13422
13423 let mut lost: BTreeSet<EdgeAt> = BTreeSet::new();
13424 let mut gained: BTreeSet<EdgeAt> = BTreeSet::new();
13425 for d in engine.drain_deltas() {
13426 let edge = EdgeAt {
13427 edge_type: d.edge_type,
13428 src_key: d.src_key,
13429 dst_key: d.dst_key,
13430 derived: true,
13431 rule: Some(d.rule),
13432 };
13433 if d.fired {
13434 gained.insert(edge);
13435 } else {
13436 lost.insert(edge);
13437 }
13438 }
13439 // An edge retracted and re-derived within the same re-derivation (top-k
13440 // churn) is not a change the caller would see.
13441 let churn: Vec<EdgeAt> = lost.intersection(&gained).cloned().collect();
13442 for e in churn {
13443 lost.remove(&e);
13444 gained.remove(&e);
13445 }
13446
13447 Ok(WhatIf {
13448 lost: lost.into_iter().collect(),
13449 gained: gained.into_iter().collect(),
13450 })
13451 }
13452
13453 pub fn edge_count(&self) -> u64 {
13454 self.topo_view().edge_count()
13455 }
13456
13457 /// Live/tombstone/edge counts plus per-rule provenance size, trip latch,
13458 /// and fire counter (includes rebuild evaluations). Rules are sorted by name.
13459 pub fn stats(&self) -> Stats {
13460 self.ensure_v8_base_sections_loaded();
13461 let building = self.engine.builds_in_progress();
13462 let rules: Vec<RuleStats> = self
13463 .engine
13464 .rules()
13465 .map(|r| RuleStats {
13466 name: r.name.clone(),
13467 edges: self
13468 .engine
13469 .provenance()
13470 .get(&r.name)
13471 .map(|s| s.len() as u64)
13472 .unwrap_or(0),
13473 tripped: self.engine.is_tripped(&r.name),
13474 fires: self.engine.fire_count(&r.name),
13475 approximate: r.approximate,
13476 building: building.iter().find(|b| b.rule == r.name).cloned(),
13477 })
13478 .collect();
13479 Stats {
13480 nodes_live: self.ids.live_len(),
13481 nodes_tombstoned: self.ids.len() - self.ids.live_len(),
13482 edges: self.topo_view().edge_count(),
13483 rules,
13484 chain_truncations: self.engine.chain_truncations(),
13485 history_floor: self.wal_horizon_floor,
13486 namespaces: self.namespace_stats(),
13487 }
13488 }
13489
13490 /// On-disk size of the WAL file in bytes.
13491 ///
13492 /// Reads file metadata without loading WAL contents. Returns `Err` for
13493 /// in-memory (`SimFs`) databases where no WAL file exists on disk.
13494 pub fn wal_size_bytes(&self) -> std::io::Result<u64> {
13495 let path = self.fs.wal_path().ok_or_else(|| {
13496 std::io::Error::new(
13497 std::io::ErrorKind::Unsupported,
13498 "wal_path not available for this Fs implementation",
13499 )
13500 })?;
13501 Ok(std::fs::metadata(path)?.len())
13502 }
13503
13504 /// Set the slow-query threshold. Queries whose execution time equals or
13505 /// exceeds `ms` milliseconds are logged. Pass `0` to disable.
13506 ///
13507 /// Use this setter in tests — the environment variable
13508 /// `MUSHROOMDB_SLOW_QUERY_MS` is process-global and races parallel test
13509 /// threads.
13510 pub fn set_slow_query_threshold_ms(&mut self, ms: u64) {
13511 self.slow_query_threshold_ms = ms;
13512 }
13513
13514 /// Snapshot of the slow-query ring buffer and lifetime counter.
13515 pub fn slow_query_snapshot(&self) -> SlowQuerySnapshot {
13516 let log = self.slow_queries.lock().unwrap_or_else(|e| e.into_inner());
13517 SlowQuerySnapshot {
13518 threshold_ms: self.slow_query_threshold_ms,
13519 count: log.total,
13520 last: log.entries.iter().cloned().collect(),
13521 }
13522 }
13523
13524 /// Instant the database was opened. Used by consumers (e.g. `/metrics`)
13525 /// to compute uptime.
13526 pub fn started_at(&self) -> std::time::Instant {
13527 self.started_at
13528 }
13529
13530 /// The on-disk snapshot version a store that has opted in to nothing
13531 /// writes — the **floor**, not the whole answer.
13532 ///
13533 /// It is not "the version this binary writes", and it is not "the version
13534 /// this binary reads". Since v0.6.10 this binary writes 9 **or** 10
13535 /// depending on the store — [`snapshot::version_for`] decides, and a store
13536 /// that has called [`enable_multiplicity`](Self::enable_multiplicity)
13537 /// writes 10 — and it reads 5 through 10. A caller comparing a store's
13538 /// stamp against this value must use `>=`, not `==`, or it will report an
13539 /// opted-in store as needing a migration *down*; `cli::run_migrate` is the
13540 /// worked example.
13541 ///
13542 /// The name is kept for compatibility: it is public API reachable from the
13543 /// CLI and from any embedder, and respelling it would break them for a
13544 /// doc-level clarification.
13545 ///
13546 /// [`snapshot::version_for`]: core_storage::snapshot::version_for
13547 pub fn format_version() -> u16 {
13548 core_storage::snapshot::VERSION
13549 }
13550
13551 /// Test-support: total bytes appended (SimFs only usage).
13552 pub fn fs_total_appended(&self) -> usize
13553 where
13554 F: FsIntrospect,
13555 {
13556 self.fs.total_appended()
13557 }
13558
13559 /// Test-support: successful `Fs::sync` calls (SimFs / counting fs).
13560 pub fn fs_sync_count(&self) -> usize
13561 where
13562 F: FsIntrospect,
13563 {
13564 self.fs.sync_count()
13565 }
13566
13567 /// Consume the db, returning its fs (for crash simulation).
13568 pub fn into_fs(self) -> F {
13569 self.fs
13570 }
13571
13572 pub fn snapshot(&mut self) -> Result<()> {
13573 self.snapshot_with(SnapshotOptions::default())
13574 }
13575
13576 /// Snapshot with explicit options.
13577 ///
13578 /// # `keep_wal`
13579 ///
13580 /// When `keep_wal` is `false` (the default, same as [`snapshot`]):
13581 /// - The WAL is replaced with a minimal baseline containing one
13582 /// `EnableFulltext` record per active declaration. All pre-snapshot
13583 /// history is discarded; `open_at` can only reach post-snapshot commits.
13584 ///
13585 /// When `keep_wal` is `true`:
13586 /// - The WAL is left intact. All pre-snapshot commits remain reachable
13587 /// via `open_at`. The existing WAL already contains the original
13588 /// `EnableFulltext` records, so no baseline re-write is needed; the
13589 /// recovery guards in `apply()` silently skip any duplicate records on
13590 /// replay.
13591 /// - Crash window: a crash after the snapshot write but before the next
13592 /// WAL write leaves the full pre-snapshot WAL intact. On reopen the
13593 /// snapshot is loaded and the WAL replayed idempotently over it — safe
13594 /// because every `apply()` arm is idempotent when replayed over an
13595 /// already-current snapshot.
13596 pub fn snapshot_with(&mut self, opts: SnapshotOptions) -> Result<()> {
13597 if self.read_only {
13598 return Err(GraphError::ReadOnly);
13599 }
13600 // A snapshot rewrites `wal.bin` through a tmp+rename, so a peer that is
13601 // appending ends up holding a descriptor on an unlinked inode and loses
13602 // commits it believes durable. Snapshotting therefore requires the
13603 // cross-process write lock, exactly as appending does. Unlike the WAL
13604 // append path this does not go through `log_then_apply_with`, so both
13605 // guards are repeated here.
13606 if self.degraded {
13607 return Err(GraphError::Io(std::io::Error::other(
13608 "database degraded after group-commit fsync failure; reopen required",
13609 )));
13610 }
13611 if self.lock_denied {
13612 return Err(GraphError::Busy { holder: None });
13613 }
13614 // Capture whether snapshot.bin already existed BEFORE this snapshot write.
13615 // Used by the archive path's conservative genesis-chain check: if a prior
13616 // snapshot exists but wal.truncated does not, we cannot distinguish a
13617 // legacy store (may have been truncated in an older code version) from a
13618 // new store that only used keep_wal=true. Conservative: refuse genesis in
13619 // both cases. Must be sampled here, before the snapshot write below.
13620 //
13621 // `snapshot_preserved_history` is the one case where the answer is not a
13622 // guess: a snapshot *this handle* took, on a store that had none when it
13623 // opened, and that kept the WAL. The proxy defers to it, because
13624 // otherwise `enable_multiplicity` — whose forced snapshot is exactly
13625 // that — would permanently disqualify the store from a genesis chain it
13626 // is fully entitled to (defect #23).
13627 let had_prior_snapshot = self.fs.snapshot_path().map(|p| p.exists()).unwrap_or(false)
13628 && !self.snapshot_preserved_history;
13629 // Which version this store writes. V9 unless it has opted in to
13630 // multiplicity, in which case V10 — the stamp that makes a reader which
13631 // does not know WAL discriminant 23 refuse the open instead of
13632 // truncating the WAL at the first such frame. The container is
13633 // identical either way; only these two header bytes move.
13634 let snapshot_version = core_storage::snapshot::version_for(self.multiplicity);
13635 self.ensure_v8_base_sections_loaded();
13636 // Ensure provenance is decoded before to_persist() clones it.
13637 self.engine.ensure_provenance_loaded_mut();
13638 let (rule_defs_typed, provenance, rule_tripped, rule_fires) = self.engine.to_persist();
13639 let rule_defs = rule_defs_typed
13640 .iter()
13641 .map(|r| bincode::serialize(r).expect("RuleDef serialize cannot fail"))
13642 .collect();
13643 // Collect HNSW state and IVF state. When indexes are not yet
13644 // populated (clean open, no mutation since open), pass the retained
13645 // raw bytes through directly so that migrate/snapshot does not
13646 // silently discard fitted approximate-rule indexes.
13647 let hnsw_state = self.engine.export_hnsw_state_passthrough();
13648 let ivf_bytes = if !self.engine.indexes_populated() {
13649 // Pass retained IVF bytes through unchanged (no re-encode).
13650 self.engine.retained_ivf_bytes_clone().unwrap_or_default()
13651 } else {
13652 // Indexes live: encode from current state.
13653 let raw_ivf = self.engine.export_ivf_state();
13654 let ivf_state_map: BTreeMap<String, core_storage::snapshot::PerRuleIvfState> = raw_ivf
13655 .into_iter()
13656 .map(|(name, ((sc, sa, sd), (dc, da, dd)))| {
13657 (
13658 name,
13659 core_storage::snapshot::PerRuleIvfState {
13660 src: core_storage::snapshot::SideIvfState {
13661 centroids: sc,
13662 clusters: sa,
13663 drift: sd,
13664 },
13665 dst: core_storage::snapshot::SideIvfState {
13666 centroids: dc,
13667 clusters: da,
13668 drift: dd,
13669 },
13670 },
13671 )
13672 })
13673 .collect();
13674 if ivf_state_map.is_empty() {
13675 Vec::new()
13676 } else {
13677 bincode::serialize(&ivf_state_map).expect("IVF state serialize cannot fail")
13678 }
13679 };
13680 let view_defs: Vec<Vec<u8>> = self
13681 .view_store
13682 .views()
13683 .map(|v| bincode::serialize(v).expect("ViewDef serialize cannot fail"))
13684 .collect();
13685 if self.base.is_some() {
13686 // V8 merge-snapshot path: encode base+overlay into a new V8 snapshot,
13687 // write it atomically, remap it as the new base, then clear the overlay.
13688 let meta = V8Meta {
13689 labels: (*self.labels).clone(),
13690 edge_props: (*self.edge_props).clone(),
13691 rule_defs,
13692 provenance,
13693 rule_tripped,
13694 rule_fires,
13695 ivf_bytes,
13696 view_defs,
13697 wal_truncated: !opts.keep_wal,
13698 hnsw: hnsw_state,
13699 last_change: self.last_change.clone(),
13700 };
13701 let mut buf: Vec<u8> = Vec::new();
13702 {
13703 // Clone the Arc so the old base stays alive while we encode.
13704 // The borrow of archived_csr (into old_base's mmap) is released
13705 // at the end of this block, before we replace self.base.
13706 let old_base = self.base.clone().expect("is_some checked above");
13707 let archived_csr = old_base.topology().map_err(|e| GraphError::Corrupt {
13708 detail: format!("v8 snapshot: topology section: {e:?}"),
13709 })?;
13710 let archived_cols = old_base.columns().map_err(|e| GraphError::Corrupt {
13711 detail: format!("v8 snapshot: columns section: {e:?}"),
13712 })?;
13713 // `None` when the base predates V9 — the migration path: its
13714 // string columns still carry their own tables and this snapshot
13715 // is the rewrite that collapses them into section 12.
13716 let archived_strings =
13717 old_base
13718 .string_table()
13719 .transpose()
13720 .map_err(|e| GraphError::Corrupt {
13721 detail: format!("v8 snapshot: strings section: {e:?}"),
13722 })?;
13723 let archived_edge_props =
13724 old_base
13725 .edge_props_section()
13726 .map_err(|e| GraphError::Corrupt {
13727 detail: format!("v8 snapshot: edge_props section: {e:?}"),
13728 })?;
13729 let edge_props_raw =
13730 old_base
13731 .edge_props_raw_bytes()
13732 .map_err(|e| GraphError::Corrupt {
13733 detail: format!("v8 snapshot: edge_props raw bytes: {e:?}"),
13734 })?;
13735 let prov_raw =
13736 old_base
13737 .provenance_raw_bytes()
13738 .map_err(|e| GraphError::Corrupt {
13739 detail: format!("v8 snapshot: provenance raw bytes: {e:?}"),
13740 })?;
13741 encode_v8(
13742 Some(archived_csr),
13743 Some(archived_cols),
13744 archived_strings,
13745 Some((archived_edge_props, edge_props_raw)),
13746 Some(prov_raw),
13747 &self.topo,
13748 &self.props,
13749 &self.ids,
13750 &self.syms,
13751 &meta,
13752 &mut buf,
13753 )?;
13754 }
13755 core_storage::snapshot::stamp_container_version(&mut buf, snapshot_version)?;
13756 self.fs.write_atomic(FileId::Snapshot, &buf)?;
13757 // Remap the freshly-written snapshot as the new base.
13758 // C2: use file mmap on RealFs; fall back to from_bytes on SimFs.
13759 let new_base = if let Some(snap_path) = self.fs.snapshot_path() {
13760 core_storage::v8::MappedBase::map(&snap_path)
13761 } else {
13762 core_storage::v8::MappedBase::from_bytes(buf)
13763 }
13764 .map_err(|e| GraphError::Corrupt {
13765 detail: format!("v8 snapshot: remap new base: {e:?}"),
13766 })?;
13767 self.base = Some(Arc::new(new_base));
13768 // Clear the overlay and prop tombstones — all data is now in the new base.
13769 self.topo = Arc::new(Topology::new());
13770 self.props = Arc::new(core_storage::columns::ColumnStore::new());
13771 } else {
13772 // Legacy path (V5–V7 stores without a V8 base).
13773 //
13774 // Memory-diet path: build V8Meta directly from &self — no SnapshotState
13775 // clone and no encode_v8_from_state intermediate clones. The big
13776 // structures (self.topo, self.props) are borrowed, not cloned.
13777 // self.edge_props is moved (not cloned) because we immediately clear it
13778 // when we remap the new V8 snapshot as self.base (see below).
13779 //
13780 // Eliminates from peak RSS vs. the old SnapshotState path:
13781 // • self.topo.clone() (~topology HashMap footprint)
13782 // • self.props.clone() (~column-store footprint)
13783 // • encode_v8_from_state V8Meta secondary clones (labels, edge_props, …)
13784 let meta = V8Meta {
13785 labels: (*self.labels).clone(),
13786 wal_truncated: !opts.keep_wal,
13787 // Move edge_props out so the large overlay is freed when meta
13788 // drops at end of this block (self.edge_props is now empty; reads
13789 // after base assignment go through the mmap'd base section).
13790 edge_props: std::mem::take(Arc::make_mut(&mut self.edge_props)),
13791 rule_defs,
13792 provenance,
13793 rule_tripped,
13794 rule_fires,
13795 ivf_bytes,
13796 view_defs,
13797 hnsw: hnsw_state,
13798 last_change: self.last_change.clone(),
13799 };
13800 let mut buf = Vec::new();
13801 encode_v8(
13802 None,
13803 None,
13804 None,
13805 None,
13806 None,
13807 &self.topo,
13808 &self.props,
13809 &self.ids,
13810 &self.syms,
13811 &meta,
13812 &mut buf,
13813 )?;
13814 // meta (and the moved edge_props inside it) is no longer needed;
13815 // drop it before the write to keep the peak window narrow.
13816 drop(meta);
13817 core_storage::snapshot::stamp_container_version(&mut buf, snapshot_version)?;
13818 self.fs.write_atomic(FileId::Snapshot, &buf)?;
13819 // Remap the freshly-written V8 snapshot as self.base.
13820 // On RealFs: drop the encode buffer before mmap to recover ~1.9 GiB.
13821 // On SimFs (tests): pass buf to from_bytes.
13822 let new_base = if let Some(snap_path) = self.fs.snapshot_path() {
13823 drop(buf);
13824 core_storage::v8::MappedBase::map(&snap_path)
13825 } else {
13826 core_storage::v8::MappedBase::from_bytes(buf)
13827 }
13828 .map_err(|e| GraphError::Corrupt {
13829 detail: format!("v8 snapshot: remap new base (legacy path): {e:?}"),
13830 })?;
13831 self.base = Some(Arc::new(new_base));
13832 // Free the large heap-allocated decoded state — all data is now in the
13833 // mmap'd base. Mirrors the V8 merge-snapshot path (see above).
13834 // self.edge_props was already moved into meta and is effectively empty.
13835 self.topo = Arc::new(Topology::new());
13836 self.props = Arc::new(core_storage::columns::ColumnStore::new());
13837 }
13838
13839 if opts.archive_wal {
13840 // History-preserving snapshot (Task 4):
13841 // 1. Snapshot already written above (write_atomic → fsynced).
13842 // 2. Rename WAL → wal.<commit_seq>.archive (atomic, same fs).
13843 // Crash window B: crash here leaves archive present, WAL
13844 // absent. Reopen: snapshot loaded (full state), no WAL
13845 // replay. Archive is NOT replayed into live state — it is
13846 // pre-snapshot by construction. Safe.
13847 // 3. Optionally write genesis marker (first archive only, no
13848 // prior WAL truncation).
13849 // 4. Prune old archives (retention), update horizon floor.
13850 // Pruning invalidates the genesis chain; delete marker.
13851 // 5. Write new minimal baseline WAL (write_atomic).
13852 // Crash window C: crash here leaves new archive plus no live
13853 // WAL. Same as window B — handled above.
13854 //
13855 // Sample existing archives BEFORE the rename so we can detect
13856 // whether this is the first archive.
13857 let existing_archives = self.fs.list_archives()?;
13858 let is_first_archive = existing_archives.is_empty();
13859
13860 // Compute a globally-monotonic archive name: the name equals the
13861 // cumulative end-frame index of the archive in global commit space.
13862 //
13863 // Using `commit_seq` directly is UNSOUND across sessions: on reopen
13864 // commit_seq is seeded from max(last_change), which underestimates
13865 // the WAL depth when trailing commits (e.g. insert_edge) do not
13866 // update last_change. A session-2 archive could then receive a name
13867 // ≤ the session-1 archive, causing incorrect sort order or collision.
13868 //
13869 // Instead: read and decode the live WAL here (before the rename) to
13870 // get its exact frame count, then add it to the last known global
13871 // end-frame index (the name of the most recent existing archive, or
13872 // wal_horizon_floor if no archives exist). This is O(WAL size) but
13873 // snapshot is already serialising the full graph state, so the cost
13874 // is dominated.
13875 let live_wal_bytes_for_name = self.fs.read(FileId::Wal)?;
13876 let (live_frames_for_name, _) = decode_all(&live_wal_bytes_for_name);
13877 let archive_n = existing_archives
13878 .last()
13879 .copied()
13880 .unwrap_or(self.wal_horizon_floor)
13881 + live_frames_for_name.len() as u64;
13882 self.fs.archive_wal(archive_n)?;
13883
13884 // The replacement WAL goes in **immediately**, with no fallible call
13885 // between it and the rename above.
13886 //
13887 // The rename is what removes the store's live declarations — the
13888 // multiplicity opt-in, and every `EnableFulltext` / `EnableIndex` —
13889 // and this write is what puts them back. Every call that used to sit
13890 // in between (the genesis marker, the retention sweep's reads, the
13891 // floor write, the archive deletes) was a `?` that could leave the
13892 // store with neither, so a single transient `Err` was enough to lose
13893 // a declaration that no rebuild can recover (defect #22).
13894 //
13895 // Ordering alone cannot close the crash window between two
13896 // filesystem calls; for the multiplicity declaration the V10 stamp
13897 // does that on the open path. What ordering does close is the much
13898 // wider window in which an ordinary I/O error did it — and that half
13899 // covers all three declarations, not just the one with a stamp.
13900 let mut baseline_wal: Vec<u8> = Vec::new();
13901 // The multiplicity opt-in is a declaration like the two below it,
13902 // and it is re-emitted for the same reason: truncation must not
13903 // silently opt the store back out and stop counting.
13904 if self.multiplicity {
13905 baseline_wal
13906 .extend_from_slice(&encode_record(&core_storage::wal::MULTIPLICITY_ENABLED));
13907 }
13908 for (label, field) in self.fulltext.enabled_pairs() {
13909 let rec = WalRecord::EnableFulltext {
13910 label: label.clone(),
13911 field: field.clone(),
13912 };
13913 baseline_wal.extend_from_slice(&encode_record(&rec));
13914 }
13915 for (label, field) in self.prop_index.enabled_pairs() {
13916 let rec = WalRecord::EnableIndex {
13917 label: label.clone(),
13918 field: field.clone(),
13919 };
13920 baseline_wal.extend_from_slice(&encode_record(&rec));
13921 }
13922 self.fs.write_atomic(FileId::Wal, &baseline_wal)?;
13923
13924 // Genesis marker: written once when the first archive is taken
13925 // from a store that has never undergone a WAL-truncating snapshot.
13926 // When present, `open_at` may replay archive-resident commits from
13927 // empty state (the archive chain covers from global index 0).
13928 //
13929 // Two conditions must ALL hold:
13930 // 1. This is the first archive (existing_archives was empty).
13931 // 2. No snapshot.bin existed before this operation (had_prior_snapshot=false).
13932 // A WAL-truncating snapshot (keep_wal=false) always writes snapshot.bin
13933 // before truncating the WAL, so if any prior truncating snapshot was taken
13934 // — even in a previous session — snapshot.bin is present and this condition
13935 // is false. This subsumes the cross-session truncation case without
13936 // requiring a separate wal.truncated sidecar file.
13937 // For legacy stores (snapshot.bin written by an older code version that
13938 // may have truncated the WAL), the same conservative refusal applies:
13939 // we cannot prove the chain is complete, so we refuse genesis (cost =
13940 // no as-of-through-archives; never silent wrong data).
13941 // The one exception is a snapshot this handle took itself, on a store
13942 // that had none when it opened, with the WAL kept: there the answer is
13943 // known rather than guessed, and `snapshot_preserved_history` says so.
13944 // Without that exception `enable_multiplicity`'s forced keep_wal
13945 // snapshot would disqualify the store forever (defect #23).
13946 // On SimFs (snapshot_path() == None) had_prior_snapshot is always false,
13947 // so SimFs always passes this check.
13948 if is_first_archive && !had_prior_snapshot {
13949 self.fs.write_genesis_marker()?;
13950 self.archive_genesis_chain = true;
13951 }
13952
13953 // Retention pruning: keep newest `keep` archives; delete oldest.
13954 // Pruning is the ONLY deletion site for archives.
13955 //
13956 // Crash-safety ordering (C1 fix):
13957 // 1. Count frames in surplus archives (reads only — no mutation).
13958 // 2. Advance and PERSIST the horizon floor FIRST via write-then-
13959 // rename (atomic). A crash after this point leaves orphaned
13960 // archives on disk, but the floor is correct. The opening
13961 // cleanup sweep (`cleanup_orphaned_archives`) removes them on
13962 // the next open, so the store is always safe to reopen.
13963 // 3. Delete the genesis marker (floor > 0 already blocks open_at
13964 // via the conjunctive gate; marker cleanup is belt-and-suspenders).
13965 // 4. Delete surplus archives. A crash between any two deletes
13966 // leaves the floor committed and orphaned archives cleaned at
13967 // next open — never a stale floor with a missing archive prefix.
13968 if let Some(keep) = self.wal_archive_retention {
13969 if keep > 0 {
13970 let archives = self.fs.list_archives()?;
13971 // archives is sorted ascending (oldest first)
13972 if archives.len() as u32 > keep {
13973 let surplus = archives.len() - keep as usize;
13974 // Step 1: count pruned frames (reads, no mutation).
13975 let mut pruned_frames = 0u64;
13976 for &n in &archives[..surplus] {
13977 let bytes = self.fs.read_archive(n)?;
13978 let (frames, _) = decode_all(&bytes);
13979 pruned_frames += frames.len() as u64;
13980 }
13981 // Step 2: advance and persist floor FIRST.
13982 self.wal_horizon_floor += pruned_frames;
13983 self.fs.write_horizon_floor(self.wal_horizon_floor)?;
13984 // The time map must not outlive the commits it
13985 // describes: an entry below the new floor would resolve
13986 // a date to a commit the engine can no longer replay,
13987 // which is worse than having no entry at all.
13988 self.commit_times.truncate_below(self.wal_horizon_floor);
13989 self.rewrite_commit_times();
13990 // Step 3: delete genesis marker (floor > 0 already
13991 // blocks open_at; this is belt-and-suspenders cleanup).
13992 if pruned_frames > 0 && self.archive_genesis_chain {
13993 self.fs.delete_genesis_marker()?;
13994 self.archive_genesis_chain = false;
13995 }
13996 // Step 4: delete surplus archives. Crash here →
13997 // orphaned archives; cleaned at next open.
13998 for &n in &archives[..surplus] {
13999 self.fs.delete_archive(n)?;
14000 }
14001 }
14002 }
14003 }
14004 } else if opts.keep_wal {
14005 // keep_wal=true: WAL is left untouched. The existing WAL already
14006 // contains the EnableFulltext records from the original enable calls;
14007 // replay is idempotent (guards in apply() skip already-live entries).
14008 // No baseline re-write is needed or safe here — the full WAL history
14009 // must remain intact for open_at to reach pre-snapshot commits.
14010 } else {
14011 // keep_wal=false (default): truncate by replacing the WAL with a
14012 // minimal baseline of one EnableFulltext record per active pair.
14013 //
14014 // Crash-ordering: write_atomic is atomic.
14015 // • Crash before snapshot write → WAL unchanged. Safe.
14016 // • Crash after snapshot write but before this WAL write → full
14017 // pre-snapshot WAL still present; open_with replays idempotently.
14018 // • Crash after both writes → normal post-snapshot state.
14019 //
14020 // Genesis chain: a WAL-truncating snapshot breaks the archive chain
14021 // for any archives taken AFTER this point (their WAL slices would
14022 // not start at genesis). Delete any existing genesis marker so that
14023 // open_at refuses archive-resident commits. Future sessions are
14024 // covered by had_prior_snapshot: snapshot.bin written here persists
14025 // across sessions and prevents a later archiving session from
14026 // incorrectly claiming a complete genesis chain.
14027 if self.archive_genesis_chain {
14028 self.fs.delete_genesis_marker()?;
14029 self.archive_genesis_chain = false;
14030 }
14031 // And this handle can no longer prove the WAL is whole: it is about
14032 // to truncate it itself. Same-session archives after this point get
14033 // the conservative answer, exactly as cross-session ones do.
14034 self.snapshot_preserved_history = false;
14035 let mut baseline_wal: Vec<u8> = Vec::new();
14036 // The multiplicity opt-in is a declaration like the two below it,
14037 // and it is re-emitted for the same reason: truncation must not
14038 // silently opt the store back out and stop counting.
14039 if self.multiplicity {
14040 baseline_wal
14041 .extend_from_slice(&encode_record(&core_storage::wal::MULTIPLICITY_ENABLED));
14042 }
14043 for (label, field) in self.fulltext.enabled_pairs() {
14044 let rec = WalRecord::EnableFulltext {
14045 label: label.clone(),
14046 field: field.clone(),
14047 };
14048 baseline_wal.extend_from_slice(&encode_record(&rec));
14049 }
14050 for (label, field) in self.prop_index.enabled_pairs() {
14051 let rec = WalRecord::EnableIndex {
14052 label: label.clone(),
14053 field: field.clone(),
14054 };
14055 baseline_wal.extend_from_slice(&encode_record(&rec));
14056 }
14057 self.fs.write_atomic(FileId::Wal, &baseline_wal)?;
14058 // This branch discards history rather than archiving it: the frames
14059 // the map describes are gone and the replacement WAL renumbers from
14060 // the floor, so every surviving entry now names a different frame.
14061 // Keeping them would resolve a date onto an unrelated commit — and
14062 // the stamps written after this point would sit far above the
14063 // store's own frame count, which is how the date surface came to
14064 // refuse commits the index path served perfectly well.
14065 //
14066 // A store that cannot answer a date says so by name
14067 // (`NoRecordedTime`). That is the honest state after discarding the
14068 // history the dates addressed.
14069 self.commit_times = core_storage::commit_times::CommitTimes::default();
14070 self.rewrite_commit_times();
14071 }
14072 // After snapshot the overlay may have changed (V8 merge path clears
14073 // self.topo and self.props). Refresh the MVCC fold so future readers
14074 // see the post-snapshot state rather than stale overlay data.
14075 self.fold_now();
14076 // We wrote the snapshot and (unless keep_wal) replaced the WAL, so both
14077 // markers this handle uses to detect other processes' work must be
14078 // re-taken from disk. Skipping this would make our own snapshot look
14079 // like a peer's on the next staleness check and force a needless
14080 // reload.
14081 self.wal_consumed = self.fs.wal_len().map_err(GraphError::Io)?;
14082 // A snapshot can replace the live WAL with a baseline, which renumbers
14083 // every frame after it. Re-derive the frame cursor from what the store
14084 // now actually holds rather than carrying the pre-snapshot count
14085 // forward — `wal_total_commits` is the same sequence the history
14086 // surfaces index, and the snapshot has already paid a far larger cost
14087 // than one decode.
14088 self.wal_frames_written = self.wal_total_commits()?;
14089 self.snapshot_ident = self.fs.snapshot_ident().map_err(GraphError::Io)?;
14090 Ok(())
14091 }
14092}
14093
14094/// What a batch node insert does when its key is already taken.
14095///
14096/// A mirror rebuild writes a frame onto a store that already has content, so
14097/// "the key exists" is a routine answer rather than a failure. The decision is
14098/// made during the batch's existing validate pass, from one id-map lookup per
14099/// row, so the frame stays atomic and re-ingest stays O(n).
14100#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)]
14101pub enum OnConflict {
14102 /// Refuse the whole frame with [`GraphError::DuplicateKey`]. The default,
14103 /// and the only behaviour before v0.6.10.
14104 #[default]
14105 Error,
14106 /// Leave the stored node exactly as it is — properties, label and edges —
14107 /// and count it in [`BatchOutcome::skipped`].
14108 Skip,
14109 /// Keep the key and make the node's properties **exactly** the supplied
14110 /// props: supplied fields are set, fields absent from the supplied props
14111 /// are removed. A supplied label that differs from the stored one, and a
14112 /// supplied `ns` that would move the node, are row errors — relabelling is
14113 /// [`GraphDb::rename_node`], not a side effect of a rebuild.
14114 ///
14115 /// Two properties are outside "exactly", both because they are not the
14116 /// caller's to supply:
14117 ///
14118 /// - `ns` is immutable, so an omitted `ns` leaves the node where it is
14119 /// rather than moving it to `default`;
14120 /// - a property a **view** owns is kept, not removed. Supplying one is a
14121 /// row error, so omitting it cannot be a request to delete it, and
14122 /// refusing the row instead would make `Replace` impossible for the whole
14123 /// population a view has written to. Each such field kept is counted in
14124 /// [`BatchOutcome::kept_view_owned`] — the row is still `replaced` and
14125 /// still raises no row error, so that count is the only signal a caller
14126 /// gets that the stored node carries a field its frame did not describe.
14127 Replace,
14128}
14129
14130/// What one [`OnConflict::Replace`] row resolves to.
14131///
14132/// The property writes that make the node exactly the supplied props —
14133/// `Some(value)` is a set, `None` a removal — paired with how many view-owned
14134/// fields the row kept instead of removing, which is the one way the result is
14135/// not exactly the supplied props. See [`MutPreview::plan_replace`].
14136type ReplacePlan = (Vec<(String, Option<Value>)>, usize);
14137
14138/// What one committed batch did.
14139///
14140/// [`BatchBuilder::commit`] returns the first two fields as a tuple; the rest
14141/// exist for [`OnConflict`] and are always zero / empty without it.
14142#[derive(Clone, Debug, Default, PartialEq, Eq)]
14143pub struct BatchOutcome {
14144 /// Node records actually written.
14145 pub nodes_inserted: usize,
14146 /// Edge records actually written. A duplicate edge is a silent no-op under
14147 /// every policy — adjacency is a set — and is not counted.
14148 pub edges_inserted: usize,
14149 /// Rows whose key was taken and whose policy was [`OnConflict::Skip`].
14150 pub skipped: usize,
14151 /// Rows whose key was taken and whose policy was [`OnConflict::Replace`].
14152 pub replaced: usize,
14153 /// View-owned properties an [`OnConflict::Replace`] row **kept** although
14154 /// the caller did not supply them — counted per field, so one row that
14155 /// keeps two contributes two.
14156 ///
14157 /// This is the one respect in which `Replace` does not make a node's props
14158 /// exactly the supplied ones (see [`OnConflict::Replace`]). Those rows
14159 /// still count in `replaced` and still raise no `row_errors`, because
14160 /// nothing went wrong: a view's property is not the caller's to supply or
14161 /// to remove. A mirror rebuild that needs its copy to be byte-exact reads
14162 /// this to learn that the store kept fields its frame did not describe.
14163 pub kept_view_owned: usize,
14164 /// `(row, why)` for rows an [`OnConflict::Replace`] refused. `row` counts
14165 /// node-insert ops in this batch from zero, which for a caller that queues
14166 /// its nodes in order is the index of the offending node. The rest of the
14167 /// frame still commits; the refused row changes nothing.
14168 pub row_errors: Vec<(usize, String)>,
14169}
14170
14171/// The `(nodes_inserted, edges_inserted)` pair every pre-0.6.10 commit entry
14172/// point returns. Keeps those signatures unchanged now that the validate pass
14173/// produces a [`BatchOutcome`].
14174fn inserted_pair(outcome: BatchOutcome) -> (usize, usize) {
14175 (outcome.nodes_inserted, outcome.edges_inserted)
14176}
14177
14178/// One entry of a frame the validate pass has decided on, before
14179/// [`GraphDb::rewrite_wal_dense_planned`] turns it into dense-id records.
14180///
14181/// Almost every entry is already a finished [`WalRecord`]. The exception is a
14182/// duplicate edge insert: its count names a dense triple, and on the batch path
14183/// the endpoints and the edge type may all be created by earlier records in the
14184/// *same* frame, so no id for them exists until the dense rewrite allocates it.
14185/// Carrying the keys this far and resolving them there is what lets the count
14186/// survive the shape a mirror rebuild writes (defect #24).
14187enum PlannedRec {
14188 Rec(WalRecord),
14189 DuplicateCount {
14190 edge_type: String,
14191 src_key: String,
14192 dst_key: String,
14193 },
14194}
14195
14196/// Queued mutation for a [`BatchBuilder`] or [`GraphDb::commit_group`].
14197///
14198/// The `submit_batch` / `commit_group` APIs accept `Vec<BatchOp>` so that
14199/// callers can build a set of mutations without holding `&mut GraphDb` and
14200/// hand them off to the group-committing writer for durable, batched I/O.
14201pub enum BatchOp {
14202 InsertNode {
14203 label: String,
14204 key: String,
14205 props: Vec<(String, Value)>,
14206 },
14207 InsertEdge {
14208 edge_type: String,
14209 src_key: String,
14210 dst_key: String,
14211 },
14212 SetProp {
14213 key: String,
14214 field: String,
14215 value: Value,
14216 },
14217 RemoveProp {
14218 key: String,
14219 field: String,
14220 },
14221 DeleteEdge {
14222 edge_type: String,
14223 src_key: String,
14224 dst_key: String,
14225 },
14226 DeleteNode {
14227 key: String,
14228 },
14229 CreateRule(RuleDef),
14230 DeleteRule {
14231 name: String,
14232 },
14233 /// Rename a node's key. Validated: old must exist, new must not.
14234 RenameNode {
14235 old_key: String,
14236 new_key: String,
14237 },
14238 /// Insert an edge, auto-creating any missing endpoint as a plain node with
14239 /// `placeholder_label` and no props. Rules fire and last-change is updated
14240 /// for each created endpoint (normal InsertNode semantics in the batch frame).
14241 InsertEdgeUpsert {
14242 edge_type: String,
14243 src_key: String,
14244 dst_key: String,
14245 placeholder_label: String,
14246 },
14247 /// Insert `key`, or — when the key is already taken — do what `on_conflict`
14248 /// says. Queued by [`BatchBuilder::insert_node_on_conflict`]; `Error`
14249 /// queues a plain [`BatchOp::InsertNode`] instead, so this variant only
14250 /// ever carries `Skip` or `Replace`.
14251 InsertNodeOnConflict {
14252 label: String,
14253 key: String,
14254 props: Vec<(String, Value)>,
14255 on_conflict: OnConflict,
14256 },
14257}
14258
14259/// Three-way node visibility status used by `check_single_op_authz`.
14260enum NodeAuthzStatus {
14261 /// Node exists in the store and is in the role's read mask.
14262 Visible(String), // carries the node's label
14263 /// Node exists in the store but is NOT in the role's read mask.
14264 Hidden,
14265 /// Node does not exist in the store.
14266 Absent,
14267}
14268
14269/// Overlay of ops already accepted earlier in the same batch. Never written
14270/// back to the database — validation only.
14271#[derive(Default)]
14272struct Overlay {
14273 extra_keys: BTreeSet<String>,
14274 /// Label of each node inserted earlier in this batch. The store does not
14275 /// have these keys yet, so `label_of` cannot answer for them, and
14276 /// `OnConflict::Replace` has to compare labels.
14277 extra_labels: BTreeMap<String, String>,
14278 deleted_keys: BTreeSet<String>,
14279 extra_props: BTreeMap<(String, String), Value>,
14280 removed_props: BTreeSet<(String, String)>,
14281 extra_edges: BTreeSet<(String, String, String)>,
14282 deleted_edges: BTreeSet<(String, String, String)>,
14283 extra_rules: BTreeSet<String>,
14284 deleted_rules: BTreeSet<String>,
14285 /// `rule name → (via_edge, edge_type)` for every via-hop rule accepted
14286 /// earlier in this batch. Feeds the rule-chain cycle check, which otherwise
14287 /// sees only the rules already committed to the engine. Keyed by name so a
14288 /// later `DeleteRule` in the same batch drops the arc with the rule.
14289 extra_rule_arcs: BTreeMap<String, (String, String)>,
14290}
14291
14292/// Read-only view of live db state plus a batch overlay. Shared by single-op
14293/// public methods (empty overlay) and `commit_batch`.
14294struct MutPreview<'a, F: Fs> {
14295 db: &'a GraphDb<F>,
14296 overlay: Overlay,
14297}
14298
14299/// Shortest path from `start` to `target` following `arcs` (`from → to`), or
14300/// `None` if `target` is unreachable.
14301///
14302/// Used for rule-chain cycle detection, where an arc is "a rule hops over
14303/// `from` and writes `to`". Breadth-first over BTree-ordered adjacency, so the
14304/// reported path is stable for a given rule set, and iterative so a pathological
14305/// rule graph cannot overflow the stack.
14306fn find_cycle_through(arcs: &[(String, String)], start: &str, target: &str) -> Option<Vec<String>> {
14307 let mut adj: BTreeMap<&str, BTreeSet<&str>> = BTreeMap::new();
14308 for (from, to) in arcs {
14309 adj.entry(from.as_str()).or_default().insert(to.as_str());
14310 }
14311 let mut parent: BTreeMap<&str, &str> = BTreeMap::new();
14312 let mut visited: BTreeSet<&str> = BTreeSet::new();
14313 let mut queue: std::collections::VecDeque<&str> = std::collections::VecDeque::new();
14314 visited.insert(start);
14315 queue.push_back(start);
14316 while let Some(node) = queue.pop_front() {
14317 if node == target {
14318 let mut path = vec![node.to_string()];
14319 let mut cur = node;
14320 while let Some(&p) = parent.get(cur) {
14321 path.push(p.to_string());
14322 cur = p;
14323 }
14324 path.reverse();
14325 return Some(path);
14326 }
14327 for &next in adj.get(node).into_iter().flatten() {
14328 if visited.insert(next) {
14329 parent.insert(next, node);
14330 queue.push_back(next);
14331 }
14332 }
14333 }
14334 None
14335}
14336
14337impl<'a, F: Fs> MutPreview<'a, F> {
14338 fn new(db: &'a GraphDb<F>) -> Self {
14339 Self {
14340 db,
14341 overlay: Overlay::default(),
14342 }
14343 }
14344
14345 fn has_key(&self, key: &str) -> bool {
14346 if self.overlay.extra_keys.contains(key) {
14347 return true;
14348 }
14349 if self.overlay.deleted_keys.contains(key) {
14350 return false;
14351 }
14352 self.db.ids.get(key).is_some()
14353 }
14354
14355 fn has_prop(&self, key: &str, field: &str) -> bool {
14356 if !self.has_key(key) {
14357 return false;
14358 }
14359 let k = (key.to_string(), field.to_string());
14360 if self.overlay.removed_props.contains(&k) {
14361 return false;
14362 }
14363 if self.overlay.extra_props.contains_key(&k) {
14364 return true;
14365 }
14366 // Fresh identity (first insert in this batch, or delete+reinsert):
14367 // ignore props still sitting on the soon-to-be-tombstoned slot.
14368 if self.overlay.extra_keys.contains(key) {
14369 return false;
14370 }
14371 self.db.get_prop(key, field).is_some()
14372 }
14373
14374 fn has_edge(&self, edge_type: &str, src_key: &str, dst_key: &str) -> bool {
14375 let k = (
14376 edge_type.to_string(),
14377 src_key.to_string(),
14378 dst_key.to_string(),
14379 );
14380 if self.overlay.deleted_edges.contains(&k) {
14381 return false;
14382 }
14383 if self.overlay.extra_edges.contains(&k) {
14384 return true;
14385 }
14386 // A key created in this batch (including reinsert) has no db edges.
14387 if self.overlay.extra_keys.contains(src_key) || self.overlay.extra_keys.contains(dst_key) {
14388 return false;
14389 }
14390 if self.overlay.deleted_keys.contains(src_key)
14391 || self.overlay.deleted_keys.contains(dst_key)
14392 {
14393 return false;
14394 }
14395 let Some(src) = self.db.ids.get(src_key) else {
14396 return false;
14397 };
14398 let Some(dst) = self.db.ids.get(dst_key) else {
14399 return false;
14400 };
14401 let Some(sym) = self.db.syms.get(edge_type) else {
14402 return false;
14403 };
14404 self.db
14405 .topo_view()
14406 .neighbors(sym, Direction::Out, src)
14407 .binary_search(&dst)
14408 .is_ok()
14409 }
14410
14411 fn has_rule(&self, name: &str) -> bool {
14412 if self.overlay.extra_rules.contains(name) {
14413 return true;
14414 }
14415 if self.overlay.deleted_rules.contains(name) {
14416 return false;
14417 }
14418 self.db.engine.rules().any(|r| r.name == name)
14419 }
14420
14421 fn is_rule_owned(&self, edge_type: &str, src_key: &str, dst_key: &str) -> bool {
14422 if self.overlay.extra_keys.contains(src_key) || self.overlay.extra_keys.contains(dst_key) {
14423 return false;
14424 }
14425 if self.overlay.deleted_keys.contains(src_key)
14426 || self.overlay.deleted_keys.contains(dst_key)
14427 {
14428 return false;
14429 }
14430 let Some(src) = self.db.ids.get(src_key) else {
14431 return false;
14432 };
14433 let Some(dst) = self.db.ids.get(dst_key) else {
14434 return false;
14435 };
14436 let Some(et) = self.db.syms.get(edge_type) else {
14437 return false;
14438 };
14439 // extra_rules is deliberately not consulted: a CreateRule earlier in
14440 // this batch has not fired, so it contributes no provenance. That is
14441 // the documented rule-window gap (see GraphDb::batch).
14442 if self.overlay.deleted_rules.is_empty() {
14443 return self.db.engine.is_owned(et, src, dst);
14444 }
14445 for (rule, triples) in self.db.engine.provenance() {
14446 if self.overlay.deleted_rules.contains(rule) {
14447 continue;
14448 }
14449 if triples.contains(&(et, src, dst)) {
14450 return true;
14451 }
14452 }
14453 false
14454 }
14455
14456 /// The refusals a node creation makes, in the order it makes them.
14457 ///
14458 /// A view owns its property, and creating a node that carries one is a
14459 /// write to it exactly as `set_prop` is — so it is refused here, at the one
14460 /// choke-point `GraphDb::insert_node`, `BatchOp::InsertNode` and the
14461 /// no-conflict arm of `BatchOp::InsertNodeOnConflict` all pass through.
14462 ///
14463 /// Leaving creation exempt was not harmless. The value was stored and
14464 /// served: a created node the view has no reason to revisit keeps the
14465 /// caller's number for the life of the handle, and the backfill at the next
14466 /// open overwrites it — so the store answered `deg = 777` before a restart
14467 /// and `deg = 0` after, for a property every other surface calls read-only.
14468 /// It also split one op two ways: supplying a view-owned field under
14469 /// `OnConflict::Replace` was already a row error on a taken key while the
14470 /// same field on a fresh key was accepted.
14471 ///
14472 /// Checked before the key, like [`MutPreview::prepare_remove_prop`], so the
14473 /// answer does not depend on whether the key exists.
14474 fn check_insert_node(&self, key: &str, props: &[(String, Value)]) -> Result<()> {
14475 for (field, _) in props {
14476 if let Some(view_name) = self.db.view_store.view_for_prop(field) {
14477 return Err(GraphError::ViewPropReadOnly {
14478 view_name: view_name.to_string(),
14479 });
14480 }
14481 }
14482 if self.has_key(key) {
14483 Err(GraphError::DuplicateKey { key: key.into() })
14484 } else {
14485 Ok(())
14486 }
14487 }
14488
14489 fn check_live_key(&self, key: &str) -> Result<()> {
14490 if self.has_key(key) {
14491 Ok(())
14492 } else {
14493 Err(GraphError::KeyNotFound { key: key.into() })
14494 }
14495 }
14496
14497 fn prepare_insert_edge(&self, edge_type: &str, src_key: &str, dst_key: &str) -> Result<bool> {
14498 for k in [src_key, dst_key] {
14499 if !self.has_key(k) {
14500 return Err(GraphError::KeyNotFound { key: k.into() });
14501 }
14502 }
14503 if self.is_rule_owned(edge_type, src_key, dst_key) {
14504 return Err(GraphError::RuleOwned {
14505 detail: format!("edge {edge_type} {src_key}→{dst_key} is rule-owned"),
14506 });
14507 }
14508 // A user-written edge stays inside one namespace. Derived edges do not
14509 // come through here — the engine adds them directly — and the rule
14510 // scoping check is what keeps those pure.
14511 let src_ns = self.namespace_in_batch(src_key);
14512 let dst_ns = self.namespace_in_batch(dst_key);
14513 if src_ns != dst_ns {
14514 return Err(GraphError::CrossNamespace {
14515 src: src_key.to_string(),
14516 src_ns,
14517 dst: dst_key.to_string(),
14518 dst_ns,
14519 });
14520 }
14521 Ok(!self.has_edge(edge_type, src_key, dst_key))
14522 }
14523
14524 fn prepare_remove_prop(&self, key: &str, field: &str) -> Result<bool> {
14525 // A view owns its property, and the refusal has to live here rather
14526 // than on `GraphDb::remove_prop`: `BatchOp::RemoveProp` never meets
14527 // that one, and it is what the HTTP `DELETE /node/{key}/prop/{field}`
14528 // route, `Batch::remove_prop` and the CLI all submit. This is the one
14529 // choke-point every removal passes, exactly as it is for `ns` below.
14530 // Checked before the key, so the answer does not depend on whether the
14531 // key exists — which is also what `GraphDb::remove_prop` answered when
14532 // it carried the only copy of this guard.
14533 if let Some(view_name) = self.db.view_store.view_for_prop(field) {
14534 return Err(GraphError::ViewPropReadOnly {
14535 view_name: view_name.to_string(),
14536 });
14537 }
14538 self.check_live_key(key)?;
14539 // Removing `ns` is changing the namespace — to `default`, the namespace
14540 // an absent property names. It goes through this one choke-point and NOT
14541 // through `rewrite_wal_dense` (a `RemoveProp` needs no dense rewrite), so
14542 // the immutability rule has to be stated here as well. Without it the
14543 // node silently lands in `default` on the next open: the cross-namespace
14544 // edge guard is defeated and a default-bound role reads a tenant's node.
14545 if field == NS_PROP {
14546 let from = self.namespace_in_batch(key);
14547 if from != NS_DEFAULT {
14548 return Err(GraphError::NamespaceImmutable {
14549 key: key.to_string(),
14550 from,
14551 to: NS_DEFAULT.to_string(),
14552 });
14553 }
14554 // Already in `default`: the removal changes no namespace. It is the
14555 // no-op `set_prop` to the current namespace is, not an error.
14556 return Ok(false);
14557 }
14558 Ok(self.has_prop(key, field))
14559 }
14560
14561 fn prepare_delete_edge(&self, edge_type: &str, src_key: &str, dst_key: &str) -> Result<bool> {
14562 for k in [src_key, dst_key] {
14563 if !self.has_key(k) {
14564 return Err(GraphError::KeyNotFound { key: k.into() });
14565 }
14566 }
14567 // Provenance-owned OR a live rule would derive this pair. User-first
14568 // edges that a later rule matches are not in `owned`, but deleting
14569 // them would leave a hole `rebuild_rule` immediately fills.
14570 if self.is_rule_owned(edge_type, src_key, dst_key) {
14571 return Err(GraphError::RuleOwned {
14572 detail: format!(
14573 "edge {edge_type} {src_key}→{dst_key} is rule-owned; \
14574 delete or change the owning rule"
14575 ),
14576 });
14577 }
14578 if self.would_derive(edge_type, src_key, dst_key) {
14579 return Err(GraphError::RuleOwned {
14580 detail: format!(
14581 "edge {edge_type} {src_key}→{dst_key} is rule-owned; \
14582 delete or change the owning rule, or a live rule would re-derive it"
14583 ),
14584 });
14585 }
14586 Ok(self.has_edge(edge_type, src_key, dst_key))
14587 }
14588
14589 /// True if any live rule (minus overlay-deleted names) would derive
14590 /// `(edge_type, src, dst)` from current overlay-visible props/labels.
14591 /// CreateRule names in `extra_rules` are ignored — same documented
14592 /// same-batch rule-window as [`Self::is_rule_owned`].
14593 ///
14594 /// Mirrored by `guard_matches` in `memory/forget.rs`, which names the
14595 /// rules this refused for: change the two together (ledger row 67).
14596 fn would_derive(&self, edge_type: &str, src_key: &str, dst_key: &str) -> bool {
14597 if src_key == dst_key {
14598 return false;
14599 }
14600 let Some(src_label) = self.label_of(src_key) else {
14601 return false;
14602 };
14603 let Some(dst_label) = self.label_of(dst_key) else {
14604 return false;
14605 };
14606 for rule in self.db.engine.rules() {
14607 if self.overlay.deleted_rules.contains(&rule.name) {
14608 continue;
14609 }
14610 if rule.edge_type != edge_type {
14611 continue;
14612 }
14613 if rule.src_label != src_label || rule.dst_label != dst_label {
14614 continue;
14615 }
14616 let src_props = |f: &str| self.prop_value(src_key, f);
14617 let dst_props = |f: &str| self.prop_value(dst_key, f);
14618 let src_view = NodeView {
14619 key: src_key,
14620 props: &src_props,
14621 };
14622 let dst_view = NodeView {
14623 key: dst_key,
14624 props: &dst_props,
14625 };
14626 if evaluate(&rule.predicate, &src_view, &dst_view).is_some() {
14627 return true;
14628 }
14629 }
14630 false
14631 }
14632
14633 fn label_of(&self, key: &str) -> Option<String> {
14634 if self.overlay.deleted_keys.contains(key) {
14635 return None;
14636 }
14637 // Fresh identities created in this batch have no stored label in the
14638 // overlay; they cannot be provenance-owned yet either.
14639 let id = self.db.ids.get(key)?;
14640 let sym = self.db.labels.get(id as usize).copied()?;
14641 if sym == u32::MAX {
14642 return None;
14643 }
14644 self.db.syms.resolve(sym).map(str::to_string)
14645 }
14646
14647 /// The label `key` carries as this batch sees it — including a node
14648 /// inserted earlier in the same batch, which the store does not have yet.
14649 fn label_in_batch(&self, key: &str) -> Option<String> {
14650 if self.overlay.deleted_keys.contains(key) {
14651 return None;
14652 }
14653 if let Some(label) = self.overlay.extra_labels.get(key) {
14654 return Some(label.clone());
14655 }
14656 self.label_of(key)
14657 }
14658
14659 /// The property writes that make `key`'s props exactly `props`, or why the
14660 /// row is refused.
14661 ///
14662 /// `Some(value)` is a set and `None` is a removal. `store_fields` is every
14663 /// field name the store knows, hoisted by the caller so a frame of N
14664 /// replaces reads the field list once rather than N times.
14665 ///
14666 /// The second half of the pair is how many view-owned fields this row kept
14667 /// rather than removed — the one part of "exactly the supplied props" that
14668 /// does not hold, and the caller's only signal that it did not.
14669 ///
14670 /// The refusals are row errors, not frame errors: a mirror rebuild should
14671 /// learn which of its rows disagree with the store without losing the rows
14672 /// that agree.
14673 fn plan_replace(
14674 &self,
14675 label: &str,
14676 key: &str,
14677 props: &[(String, Value)],
14678 store_fields: &[String],
14679 ) -> std::result::Result<ReplacePlan, String> {
14680 // A different label is a relabel, and a rebuild does not relabel: that
14681 // is `rename_node` or an explicit write, never a side effect here.
14682 let stored = self.label_in_batch(key).unwrap_or_default();
14683 if stored != label {
14684 return Err(format!(
14685 "node {key}: on_conflict=\"replace\" will not relabel {stored:?} to {label:?}; \
14686 relabelling is rename_node or an explicit write"
14687 ));
14688 }
14689 // `ns` is immutable. Replace removes what the supplied props omit, so
14690 // an omitted `ns` is a move to `default` exactly as a different `ns` is
14691 // a move to that one; both are the same refusal.
14692 let from = self.namespace_in_batch(key);
14693 let to = match props.iter().find(|(field, _)| field == NS_PROP) {
14694 Some((_, Value::Str(ns))) => ns.clone(),
14695 Some((_, value)) => {
14696 return Err(format!(
14697 "node {key}: {NS_PROP} must be a string naming a namespace, got {value:?}"
14698 ));
14699 }
14700 None => NS_DEFAULT.to_string(),
14701 };
14702 if to != from {
14703 return Err(format!(
14704 "node {key}: {NS_PROP} is immutable; on_conflict=\"replace\" cannot move it \
14705 from {from:?} to {to:?}"
14706 ));
14707 }
14708
14709 if let Some(why) = self.supplied_view_owned_prop(key, props) {
14710 return Err(why);
14711 }
14712
14713 let supplied: BTreeSet<&str> = props.iter().map(|(field, _)| field.as_str()).collect();
14714 let mut writes = Vec::new();
14715 for (field, value) in props {
14716 // `ns` names the namespace the node is already in, so the write is
14717 // the no-op the dense-rewrite seam would drop anyway.
14718 if field == NS_PROP {
14719 continue;
14720 }
14721 // Already exactly this value: a rebuild of an unchanged row should
14722 // cost no WAL record.
14723 if self.prop_value(key, field).as_ref() == Some(value) {
14724 continue;
14725 }
14726 writes.push((field.clone(), Some(value.clone())));
14727 }
14728 // Everything the node still carries that the supplied props do not.
14729 // `ns` is never removed: it is immutable, and the check above has
14730 // already established the node stays where it is.
14731 let overlay_fields = self
14732 .overlay
14733 .extra_props
14734 .keys()
14735 .filter(|(k, _)| k == key)
14736 .map(|(_, field)| field.as_str());
14737 //
14738 // A view-owned field is filtered out rather than refused. It is not the
14739 // caller's to supply (supplying one is still the row error above) and
14740 // so it is not part of what "exactly the supplied ones" ranges over:
14741 // omitting it is not a request to delete it. Refusing here instead
14742 // would make `replace` impossible for every node a view has written to
14743 // — which on a store carrying a view is the whole population a mirror
14744 // rebuild has to cover.
14745 let omitted: BTreeSet<&str> = store_fields
14746 .iter()
14747 .map(String::as_str)
14748 .chain(overlay_fields)
14749 .filter(|field| {
14750 *field != NS_PROP && !supplied.contains(field) && self.has_prop(key, field)
14751 })
14752 .collect();
14753 // The view-owned half is kept, and counted: the row still commits and
14754 // still reports no error, so without this number a mirror rebuild is
14755 // told it got exactly what it asked for when it did not (defect #18).
14756 let (stale, kept): (Vec<&str>, Vec<&str>) = omitted
14757 .into_iter()
14758 .partition(|field| self.db.view_store.view_for_prop(field).is_none());
14759 writes.extend(stale.into_iter().map(|field| (field.to_string(), None)));
14760 Ok((writes, kept.len()))
14761 }
14762
14763 /// The row error a supplied view-owned field earns, or `None`.
14764 ///
14765 /// Shared by [`MutPreview::plan_replace`] and the no-conflict arm of
14766 /// `BatchOp::InsertNodeOnConflict` so that one op answers a supplied
14767 /// view-owned field the same way whether or not the key was already taken.
14768 fn supplied_view_owned_prop(&self, key: &str, props: &[(String, Value)]) -> Option<String> {
14769 props.iter().find_map(|(field, _)| {
14770 self.db.view_store.view_for_prop(field).map(|view_name| {
14771 format!(
14772 "node {key}: property {field:?} is owned by view {view_name:?} and is \
14773 read-only"
14774 )
14775 })
14776 })
14777 }
14778
14779 /// The namespace `key` is in as this batch sees it — including a node
14780 /// inserted earlier in the same batch, which the store does not have yet.
14781 fn namespace_in_batch(&self, key: &str) -> String {
14782 namespace_of_value(self.prop_value(key, NS_PROP).as_ref()).to_string()
14783 }
14784
14785 fn prop_value(&self, key: &str, field: &str) -> Option<Value> {
14786 if !self.has_key(key) {
14787 return None;
14788 }
14789 let k = (key.to_string(), field.to_string());
14790 if self.overlay.removed_props.contains(&k) {
14791 return None;
14792 }
14793 if let Some(v) = self.overlay.extra_props.get(&k) {
14794 return Some(v.clone());
14795 }
14796 if self.overlay.extra_keys.contains(key) {
14797 return None;
14798 }
14799 self.db.get_prop(key, field)
14800 }
14801
14802 fn check_create_rule(&self, def: &RuleDef) -> Result<()> {
14803 def.validate()
14804 .map_err(|e| GraphError::RuleInvalid { detail: e })?;
14805 if self.has_rule(&def.name) {
14806 return Err(GraphError::RuleInvalid {
14807 detail: format!("rule {:?} already exists", def.name),
14808 });
14809 }
14810 // Rule-chain cycle rejection. Derived edges feed via-hop rules, so a
14811 // rule set forms a graph whose arcs are "hops over `via_edge`, writes
14812 // `edge_type`". A cycle in that graph is a rule set that would re-fire
14813 // itself forever; the engine's depth cap would silently truncate it
14814 // instead, leaving an arbitrary partial result. Reject it here, the one
14815 // place that sees the whole rule set.
14816 //
14817 // Rules accepted earlier in the same batch count too: the overlay
14818 // carries their arcs, so a cycle cannot be assembled one op at a time.
14819 if let Some(via) = def.via_edge.as_deref() {
14820 if via == def.edge_type {
14821 return Err(GraphError::RuleInvalid {
14822 detail: format!("rule chain cycle: {} -> {}", via, def.edge_type),
14823 });
14824 }
14825 let mut arcs: Vec<(String, String)> = self
14826 .db
14827 .engine
14828 .rules()
14829 .filter(|r| !self.overlay.deleted_rules.contains(&r.name))
14830 .filter_map(|r| r.via_edge.clone().map(|v| (v, r.edge_type.clone())))
14831 .collect();
14832 arcs.extend(self.overlay.extra_rule_arcs.values().cloned());
14833 arcs.push((via.to_string(), def.edge_type.clone()));
14834 if let Some(path) = find_cycle_through(&arcs, &def.edge_type, via) {
14835 return Err(GraphError::RuleInvalid {
14836 detail: format!("rule chain cycle: {} -> {}", via, path.join(" -> ")),
14837 });
14838 }
14839 }
14840 Ok(())
14841 }
14842
14843 fn check_delete_rule(&self, name: &str) -> Result<()> {
14844 if self.has_rule(name) {
14845 Ok(())
14846 } else {
14847 Err(GraphError::RuleNotFound { name: name.into() })
14848 }
14849 }
14850
14851 fn note_insert_node(&mut self, label: &str, key: &str, props: &[(String, Value)]) {
14852 self.overlay.deleted_keys.remove(key);
14853 self.overlay.extra_keys.insert(key.to_string());
14854 self.overlay
14855 .extra_labels
14856 .insert(key.to_string(), label.to_string());
14857 self.overlay.extra_props.retain(|(k, _), _| k != key);
14858 self.overlay.removed_props.retain(|(k, _)| k != key);
14859 for (field, value) in props {
14860 self.overlay
14861 .extra_props
14862 .insert((key.to_string(), field.clone()), value.clone());
14863 }
14864 }
14865
14866 fn note_insert_edge(&mut self, edge_type: &str, src_key: &str, dst_key: &str) {
14867 let k = (
14868 edge_type.to_string(),
14869 src_key.to_string(),
14870 dst_key.to_string(),
14871 );
14872 self.overlay.deleted_edges.remove(&k);
14873 self.overlay.extra_edges.insert(k);
14874 }
14875
14876 fn note_set_prop(&mut self, key: &str, field: &str, value: &Value) {
14877 let k = (key.to_string(), field.to_string());
14878 self.overlay.removed_props.remove(&k);
14879 self.overlay.extra_props.insert(k, value.clone());
14880 }
14881
14882 fn note_remove_prop(&mut self, key: &str, field: &str) {
14883 let k = (key.to_string(), field.to_string());
14884 self.overlay.extra_props.remove(&k);
14885 self.overlay.removed_props.insert(k);
14886 }
14887
14888 fn note_delete_edge(&mut self, edge_type: &str, src_key: &str, dst_key: &str) {
14889 let k = (
14890 edge_type.to_string(),
14891 src_key.to_string(),
14892 dst_key.to_string(),
14893 );
14894 self.overlay.extra_edges.remove(&k);
14895 self.overlay.deleted_edges.insert(k);
14896 }
14897
14898 fn note_delete_node(&mut self, key: &str) {
14899 self.overlay.extra_keys.remove(key);
14900 self.overlay.deleted_keys.insert(key.to_string());
14901 self.overlay.extra_props.retain(|(k, _), _| k != key);
14902 self.overlay.removed_props.retain(|(k, _)| k != key);
14903 self.overlay
14904 .extra_edges
14905 .retain(|(_, s, d)| s != key && d != key);
14906 self.overlay
14907 .deleted_edges
14908 .retain(|(_, s, d)| s != key && d != key);
14909 }
14910
14911 fn note_create_rule(&mut self, def: &RuleDef) {
14912 self.overlay.deleted_rules.remove(&def.name);
14913 self.overlay.extra_rules.insert(def.name.clone());
14914 // Rules accepted earlier in this batch are not in the engine yet, so
14915 // the cycle check would not see their arcs. Keep the arc, not just the
14916 // name, so a batch cannot smuggle in a cycle one op at a time.
14917 if let Some(via) = def.via_edge.clone() {
14918 self.overlay
14919 .extra_rule_arcs
14920 .insert(def.name.clone(), (via, def.edge_type.clone()));
14921 }
14922 }
14923
14924 fn check_rename_node(&self, old: &str, new: &str) -> Result<()> {
14925 if !self.has_key(old) {
14926 return Err(GraphError::KeyNotFound { key: old.into() });
14927 }
14928 if self.has_key(new) {
14929 return Err(GraphError::DuplicateKey { key: new.into() });
14930 }
14931 Ok(())
14932 }
14933
14934 fn note_rename_node(&mut self, old: &str, new: &str) {
14935 // Mark old as deleted so subsequent batch ops cannot reference it.
14936 self.overlay.extra_keys.remove(old);
14937 self.overlay.deleted_keys.insert(old.to_string());
14938 // Mark new as extra so subsequent batch ops can reference it.
14939 self.overlay.deleted_keys.remove(new);
14940 self.overlay.extra_keys.insert(new.to_string());
14941 // Migrate any overlay props from old key to new key.
14942 let new_str = new.to_string();
14943 let transferred: Vec<((String, String), Value)> = self
14944 .overlay
14945 .extra_props
14946 .iter()
14947 .filter(|((k, _), _)| k.as_str() == old)
14948 .map(|((_, f), v)| ((new_str.clone(), f.clone()), v.clone()))
14949 .collect();
14950 self.overlay
14951 .extra_props
14952 .retain(|(k, _), _| k.as_str() != old);
14953 for (k, v) in transferred {
14954 self.overlay.extra_props.insert(k, v);
14955 }
14956 // Migrate removed_props.
14957 let transferred_removed: Vec<(String, String)> = self
14958 .overlay
14959 .removed_props
14960 .iter()
14961 .filter(|(k, _)| k.as_str() == old)
14962 .map(|(_, f)| (new_str.clone(), f.clone()))
14963 .collect();
14964 self.overlay
14965 .removed_props
14966 .retain(|(k, _)| k.as_str() != old);
14967 for k in transferred_removed {
14968 self.overlay.removed_props.insert(k);
14969 }
14970 }
14971
14972 fn note_delete_rule(&mut self, name: &str) {
14973 self.overlay.extra_rules.remove(name);
14974 // Drop its chain arc too: a rule created and then deleted in the same
14975 // batch must not make a later, legal rule look like a cycle.
14976 self.overlay.extra_rule_arcs.remove(name);
14977 self.overlay.deleted_rules.insert(name.to_string());
14978 // Treat the deleted rule's current provenance as gone so a later
14979 // delete_edge of those triples is a no-op (matches sequential).
14980 if let Some(triples) = self.db.engine.provenance().get(name) {
14981 for &(et, s, d) in triples {
14982 let Some(etype) = self.db.syms.resolve(et) else {
14983 continue;
14984 };
14985 let Some(src) = self.db.ids.key_of(s) else {
14986 continue;
14987 };
14988 let Some(dst) = self.db.ids.key_of(d) else {
14989 continue;
14990 };
14991 let k = (etype.to_string(), src.to_string(), dst.to_string());
14992 self.overlay.extra_edges.remove(&k);
14993 self.overlay.deleted_edges.insert(k);
14994 }
14995 }
14996 }
14997}
14998
14999/// Collects mutations and commits them as one WAL `Batch` frame.
15000///
15001/// Holds `&mut GraphDb` for its lifetime. Queue with the same method names
15002/// as [`GraphDb`]; call [`commit`](Self::commit) to validate, log, and apply.
15003/// See [`GraphDb::batch`] for validation and atomicity rules.
15004pub struct BatchBuilder<'a, F: Fs> {
15005 db: &'a mut GraphDb<F>,
15006 ops: Vec<BatchOp>,
15007}
15008
15009impl<'a, F: Fs> BatchBuilder<'a, F> {
15010 pub fn insert_node(
15011 &mut self,
15012 label: &str,
15013 key: &str,
15014 props: Vec<(String, Value)>,
15015 ) -> &mut Self {
15016 self.ops.push(BatchOp::InsertNode {
15017 label: label.into(),
15018 key: key.into(),
15019 props,
15020 });
15021 self
15022 }
15023
15024 /// Queue a node insert whose answer to a taken key is `on_conflict`.
15025 ///
15026 /// [`OnConflict::Error`] queues exactly the op [`insert_node`](Self::insert_node)
15027 /// does, so the default path is unchanged.
15028 pub fn insert_node_on_conflict(
15029 &mut self,
15030 label: &str,
15031 key: &str,
15032 props: Vec<(String, Value)>,
15033 on_conflict: OnConflict,
15034 ) -> &mut Self {
15035 self.ops.push(match on_conflict {
15036 OnConflict::Error => BatchOp::InsertNode {
15037 label: label.into(),
15038 key: key.into(),
15039 props,
15040 },
15041 on_conflict => BatchOp::InsertNodeOnConflict {
15042 label: label.into(),
15043 key: key.into(),
15044 props,
15045 on_conflict,
15046 },
15047 });
15048 self
15049 }
15050
15051 pub fn insert_edge(&mut self, edge_type: &str, src_key: &str, dst_key: &str) -> &mut Self {
15052 self.ops.push(BatchOp::InsertEdge {
15053 edge_type: edge_type.into(),
15054 src_key: src_key.into(),
15055 dst_key: dst_key.into(),
15056 });
15057 self
15058 }
15059
15060 pub fn set_prop(&mut self, key: &str, field: &str, value: Value) -> &mut Self {
15061 self.ops.push(BatchOp::SetProp {
15062 key: key.into(),
15063 field: field.into(),
15064 value,
15065 });
15066 self
15067 }
15068
15069 pub fn remove_prop(&mut self, key: &str, field: &str) -> &mut Self {
15070 self.ops.push(BatchOp::RemoveProp {
15071 key: key.into(),
15072 field: field.into(),
15073 });
15074 self
15075 }
15076
15077 pub fn delete_edge(&mut self, edge_type: &str, src_key: &str, dst_key: &str) -> &mut Self {
15078 self.ops.push(BatchOp::DeleteEdge {
15079 edge_type: edge_type.into(),
15080 src_key: src_key.into(),
15081 dst_key: dst_key.into(),
15082 });
15083 self
15084 }
15085
15086 pub fn delete_node(&mut self, key: &str) -> &mut Self {
15087 self.ops.push(BatchOp::DeleteNode { key: key.into() });
15088 self
15089 }
15090
15091 pub fn create_rule(&mut self, def: RuleDef) -> &mut Self {
15092 self.ops.push(BatchOp::CreateRule(def));
15093 self
15094 }
15095
15096 pub fn delete_rule(&mut self, name: &str) -> &mut Self {
15097 self.ops.push(BatchOp::DeleteRule { name: name.into() });
15098 self
15099 }
15100
15101 /// Queue a node-rename in this batch.
15102 ///
15103 /// Validation (old exists, new not taken) runs at commit time.
15104 pub fn rename_node(&mut self, old_key: &str, new_key: &str) -> &mut Self {
15105 self.ops.push(BatchOp::RenameNode {
15106 old_key: old_key.into(),
15107 new_key: new_key.into(),
15108 });
15109 self
15110 }
15111
15112 /// Queue an edge insert with endpoint auto-creation.
15113 ///
15114 /// Any missing endpoint is created as a plain node `{key, label:
15115 /// placeholder_label, no props}` inside this batch frame. Rules fire and
15116 /// last-change is updated for each auto-created node.
15117 pub fn insert_edge_upsert(
15118 &mut self,
15119 edge_type: &str,
15120 src_key: &str,
15121 dst_key: &str,
15122 placeholder_label: &str,
15123 ) -> &mut Self {
15124 self.ops.push(BatchOp::InsertEdgeUpsert {
15125 edge_type: edge_type.into(),
15126 src_key: src_key.into(),
15127 dst_key: dst_key.into(),
15128 placeholder_label: placeholder_label.into(),
15129 });
15130 self
15131 }
15132
15133 /// Validate every queued op, then log one `Batch` frame and apply.
15134 /// Empty / all-noop batches return `Ok(())` without writing the WAL.
15135 /// A second `commit()` after a successful one is an empty-batch no-op
15136 /// (queued ops were taken).
15137 /// Takes `&mut self` so it chains after the queue methods (`b.insert_node(..).commit()`)
15138 /// and also works as `let mut b = db.batch(); b.insert_node(..); b.commit()`.
15139 ///
15140 /// **Rule-window limitation:** batch validation cannot see edges that a
15141 /// rule created earlier in the *same* batch will derive at apply time, so
15142 /// a `delete_edge` / `insert_edge` in that window is silently no-oped
15143 /// where sequential calls would return `Err(RuleOwned)`. State integrity
15144 /// is unaffected (idempotent apply, provenance intact). Create rules in
15145 /// their own batch, or sequentially, when later ops may touch derived
15146 /// edges.
15147 /// Validate every queued op and commit atomically.
15148 ///
15149 /// Returns `(nodes_inserted, edges_inserted)` — the counts of node and edge
15150 /// WAL records actually written (duplicate edges are silent no-ops and are
15151 /// NOT counted). Both are 0 when the batch is empty or all-noop.
15152 pub fn commit(&mut self) -> Result<(usize, usize)> {
15153 let ops = std::mem::take(&mut self.ops);
15154 self.db.commit_batch(ops)
15155 }
15156
15157 /// [`commit`](Self::commit) with the full [`BatchOutcome`] — the counts a
15158 /// caller needs when its rows carry an [`OnConflict`] policy.
15159 pub fn commit_outcome(&mut self) -> Result<BatchOutcome> {
15160 let ops = std::mem::take(&mut self.ops);
15161 self.db.commit_logged_batch(ops, None, None)
15162 }
15163
15164 /// Same as [`commit`](Self::commit) but tail the inner events with
15165 /// [`MutationEvent::Ingested`] instead of [`MutationEvent::BatchApplied`].
15166 pub(crate) fn commit_ingest(&mut self, label: &str, inserted: usize) -> Result<(usize, usize)> {
15167 let ops = std::mem::take(&mut self.ops);
15168 self.db
15169 .commit_logged_batch(ops, Some((label.to_string(), inserted)), None)
15170 .map(inserted_pair)
15171 }
15172}
15173
15174pub struct NodeRef<'a, F: Fs> {
15175 db: &'a GraphDb<F>,
15176 id: u32,
15177}
15178
15179impl<'a, F: Fs> NodeRef<'a, F> {
15180 pub fn key(&self) -> &str {
15181 self.db.ids.key_of(self.id).expect("dense ids")
15182 }
15183
15184 pub fn label(&self) -> &str {
15185 let sym = self
15186 .db
15187 .labels
15188 .get(self.id as usize)
15189 .copied()
15190 .filter(|&s| s != u32::MAX)
15191 .expect("real nodes always have a label; u32::MAX sentinel cannot occur");
15192 self.db.syms.resolve(sym).expect("interned label symbol")
15193 }
15194
15195 pub fn prop(&self, field: &str) -> Option<Value> {
15196 self.db
15197 .props_view()
15198 .get(self.id, field)
15199 .map(|vr| vr.into_value())
15200 }
15201
15202 /// All stored fields for this node, sorted by field name.
15203 ///
15204 /// Reads from the full base+overlay view so that props stored only in the
15205 /// V8 snapshot base (i.e. before any post-snapshot WAL writes) are visible.
15206 pub fn props(&self) -> BTreeMap<String, Value> {
15207 let mut out = BTreeMap::new();
15208 let pv = self.db.props_view();
15209 for field in pv.field_names() {
15210 if let Some(vr) = pv.get(self.id, &field) {
15211 out.insert(field, vr.into_value());
15212 }
15213 }
15214 out
15215 }
15216
15217 /// depth-N BFS as a ResultSet: columns ["key","label","depth"], BFS order.
15218 pub fn neighborhood(&self, depth: u32, edge_types: Option<&[&str]>, dir: Dir) -> ResultSet {
15219 let view = self.db.view();
15220 let resolved: Option<Vec<u32>> = edge_types.map(|names| {
15221 names
15222 .iter()
15223 .filter_map(|name| view.syms.get(name))
15224 .collect()
15225 });
15226 let nb = neighborhood(&view, self.id, depth, resolved.as_deref(), dir);
15227 let mut rs = ResultSet::new(vec!["key".into(), "label".into(), "depth".into()]);
15228 for (nid, d) in nb.nodes {
15229 let key = view.key_of(nid);
15230 let label = view
15231 .label_of(nid)
15232 .expect("real nodes always have a label; u32::MAX sentinel cannot occur");
15233 rs.push_row(vec![
15234 Some(Value::Str(key.to_string())),
15235 Some(Value::Str(label.to_string())),
15236 Some(Value::Int(d as i64)),
15237 ]);
15238 }
15239 rs
15240 }
15241
15242 /// 1-hop, Both directions: edge-type name → sorted unique neighbor keys.
15243 pub fn grouped_by_edge_type(&self) -> BTreeMap<String, Vec<String>> {
15244 let view = self.db.view();
15245 let mut groups: BTreeMap<String, BTreeSet<String>> = BTreeMap::new();
15246 for e in expand(&view, self.id, None, Dir::Both) {
15247 // Skip edges with unknown etypes (only possible from corrupt large
15248 // TOPOLOGY section; function returns BTreeMap not Result).
15249 let Some(etype) = view.syms.resolve(e.etype) else {
15250 continue;
15251 };
15252 let etype = etype.to_string();
15253 let nbr = if e.src == self.id { e.dst } else { e.src };
15254 groups
15255 .entry(etype)
15256 .or_default()
15257 .insert(view.key_of(nbr).to_string());
15258 }
15259 groups
15260 .into_iter()
15261 .map(|(k, v)| (k, v.into_iter().collect()))
15262 .collect()
15263 }
15264}
15265
15266#[cfg(test)]
15267mod tests {
15268 use super::*;
15269 use core_rules::Predicate;
15270
15271 fn tmp_dir(name: &str) -> std::path::PathBuf {
15272 let d =
15273 std::env::temp_dir().join(format!("graphdb-db-unit-{}-{}", name, std::process::id()));
15274 let _ = std::fs::remove_dir_all(&d);
15275 d
15276 }
15277
15278 fn fk_rule() -> RuleDef {
15279 RuleDef {
15280 name: "works_at".into(),
15281 src_label: "Person".into(),
15282 dst_label: "Org".into(),
15283 predicate: Predicate::KeyMatch {
15284 field: "org_id".into(),
15285 },
15286 edge_type: "WORKS_AT".into(),
15287 weight_prop: None,
15288 max_edges: None,
15289 approximate: false,
15290 via_label: None,
15291 via_edge: None,
15292 via_dir: None,
15293 namespace: None,
15294 }
15295 }
15296
15297 /// Regression guard for the no-views delta-copy fast path.
15298 ///
15299 /// When no views are defined, `pending_deltas_since().to_vec()` must never
15300 /// be called — even during a large CreateRule backfill. The DELTA_COPY_COUNT
15301 /// thread-local is incremented inside every `if !view_store.is_empty()` block;
15302 /// a count of 0 after the entire sequence proves the guard fires correctly.
15303 #[test]
15304 fn no_delta_copy_when_no_views() {
15305 DELTA_COPY_COUNT.with(|c| c.set(0));
15306 let dir = tmp_dir("no-delta-copy");
15307 {
15308 let mut db = GraphDb::open(&dir).unwrap();
15309 // Insert 50 Org + 50 Person nodes with FK links.
15310 for i in 0..50u32 {
15311 db.insert_node("Org", &format!("o{i}"), vec![]).unwrap();
15312 }
15313 for i in 0..50u32 {
15314 db.insert_node(
15315 "Person",
15316 &format!("p{i}"),
15317 vec![("org_id".into(), Value::Str(format!("o{i}")))],
15318 )
15319 .unwrap();
15320 }
15321 // CreateRule backfill should NOT invoke to_vec() when no views are defined.
15322 db.create_rule(fk_rule()).unwrap();
15323
15324 // Counter must stay 0 — no views, no copies.
15325 let copies = DELTA_COPY_COUNT.with(|c| c.get());
15326 assert_eq!(
15327 copies, 0,
15328 "pending_deltas_since().to_vec() called despite no views"
15329 );
15330
15331 // Derived edges must still be correct (the guard skips only the
15332 // empty delta propagation loop, not the rule application itself).
15333 let nbrs = db.neighbors("p0", "WORKS_AT", Direction::Out).unwrap();
15334 assert_eq!(
15335 nbrs,
15336 vec!["o0"],
15337 "rule must derive edges even with no views"
15338 );
15339 }
15340 let _ = std::fs::remove_dir_all(&dir);
15341 }
15342
15343 /// Gating regression: subscribe AFTER a backfill must see no stale events.
15344 /// subscribe BEFORE a backfill must see every edge-fire event.
15345 #[test]
15346 fn subscribe_after_backfill_no_stale_events() {
15347 let dir = tmp_dir("sub-after-backfill");
15348 {
15349 let mut db = GraphDb::open(&dir).unwrap();
15350 for i in 0..10u32 {
15351 db.insert_node("Org", &format!("o{i}"), vec![]).unwrap();
15352 db.insert_node(
15353 "Person",
15354 &format!("p{i}"),
15355 vec![("org_id".into(), Value::Str(format!("o{i}")))],
15356 )
15357 .unwrap();
15358 }
15359 // Create rule BEFORE subscribing — emit_deltas is false during backfill.
15360 db.create_rule(fk_rule()).unwrap();
15361
15362 // Subscribe AFTER the backfill — queue must be empty (no stale events).
15363 let sub = db.subscribe_all_rules().unwrap();
15364 // No events should have queued for the prior backfill.
15365 assert!(
15366 sub.try_recv().is_none(),
15367 "subscribe after backfill must see no stale events"
15368 );
15369
15370 // Inserting a new node now should fire an event (emit_deltas is now true).
15371 db.insert_node("Org", "o_new", vec![]).unwrap();
15372 db.insert_node(
15373 "Person",
15374 "p_new",
15375 vec![("org_id".into(), Value::Str("o_new".into()))],
15376 )
15377 .unwrap();
15378 let ev = sub.recv_timeout(std::time::Duration::from_millis(200));
15379 assert!(
15380 ev.is_some(),
15381 "edge-fire event must arrive after subscribe (emit_deltas=true)"
15382 );
15383 }
15384 let _ = std::fs::remove_dir_all(&dir);
15385 }
15386
15387 /// Gating regression: subscribe BEFORE a backfill → events flow.
15388 #[test]
15389 fn subscribe_before_backfill_events_flow() {
15390 let dir = tmp_dir("sub-before-backfill");
15391 {
15392 let mut db = GraphDb::open(&dir).unwrap();
15393 // Subscribe FIRST — emit_deltas becomes true.
15394 let sub = db.subscribe_all_rules().unwrap();
15395
15396 for i in 0..5u32 {
15397 db.insert_node("Org", &format!("o{i}"), vec![]).unwrap();
15398 db.insert_node(
15399 "Person",
15400 &format!("p{i}"),
15401 vec![("org_id".into(), Value::Str(format!("o{i}")))],
15402 )
15403 .unwrap();
15404 }
15405 // Backfill fires with emit_deltas=true → events queued.
15406 db.create_rule(fk_rule()).unwrap();
15407
15408 // Should receive at least one edge-fired event from the backfill.
15409 let mut received = 0usize;
15410 while sub.try_recv().is_some() {
15411 received += 1;
15412 }
15413 assert!(
15414 received > 0,
15415 "subscribe before backfill must receive edge-fire events (got 0)"
15416 );
15417 }
15418 let _ = std::fs::remove_dir_all(&dir);
15419 }
15420
15421 /// Companion: when a view IS defined, the delta path fires and view values update.
15422 #[test]
15423 fn delta_copy_fires_when_view_exists() {
15424 use core_rules::ViewSource;
15425 DELTA_COPY_COUNT.with(|c| c.set(0));
15426 let dir = tmp_dir("delta-copy-with-view");
15427 {
15428 let mut db = GraphDb::open(&dir).unwrap();
15429 db.insert_node("Org", "o1", vec![]).unwrap();
15430 db.insert_node(
15431 "Person",
15432 "p1",
15433 vec![("org_id".into(), Value::Str("o1".into()))],
15434 )
15435 .unwrap();
15436 // Declare a Degree view so is_empty() returns false.
15437 db.create_view(ViewDef {
15438 name: "degree_out".into(),
15439 label: "Person".into(),
15440 view_prop: "degree_out".into(),
15441 source: ViewSource::Degree {
15442 edge_type: "WORKS_AT".into(),
15443 direction: Direction::Out,
15444 },
15445 })
15446 .unwrap();
15447 db.create_rule(fk_rule()).unwrap();
15448
15449 // At least one delta copy should have happened (CreateRule backfill).
15450 let copies = DELTA_COPY_COUNT.with(|c| c.get());
15451 assert!(
15452 copies > 0,
15453 "expected delta copy to fire when a view is defined"
15454 );
15455
15456 // View value should be computed: p1 has one WORKS_AT out-edge.
15457 let info = db.node_info("p1").unwrap();
15458 let degree = info.props.get("degree_out");
15459 assert!(
15460 degree.is_some(),
15461 "view prop should be written to node props"
15462 );
15463 }
15464 let _ = std::fs::remove_dir_all(&dir);
15465 }
15466
15467 /// Regression: `open_at_with` must call `rebuild_all` after WAL replay so
15468 /// derived-edge-driven view values reflect the as-of state rather than just
15469 /// the initial backfill written at `CreateView` time.
15470 ///
15471 /// Base WAL frames (indices 0..=5 before history markers):
15472 /// 0: insert Org "o1"
15473 /// 1: create_view "employee_count" (Degree / WORKS_AT / In) on Org
15474 /// 2: create_rule fk_rule (WORKS_AT, Person→Org via org_id)
15475 /// 3: insert Person "p1" → rule fires WORKS_AT p1→o1 (degree = 1) ← mid
15476 /// 4: insert Person "p2" → rule fires WORKS_AT p2→o1 (degree = 2)
15477 /// 5: insert Person "p3" → rule fires WORKS_AT p3→o1 (degree = 3) ← latest
15478 ///
15479 /// Each rule-fire also appends a DerivedEdgeAdded history-marker frame (state
15480 /// no-op), so the total commit count is higher than the base frame count.
15481 /// The "latest" open_at commit is computed dynamically via `wal_commit_count_at`.
15482 ///
15483 /// Without `rebuild_all`, the as-of instance's "emp" view stays at the
15484 /// initial backfill value (0) instead of reflecting the replayed derived edges.
15485 #[test]
15486 fn open_at_derived_edge_view_values_correct() {
15487 use core_rules::ViewSource;
15488 let dir = tmp_dir("open-at-view-rebuild");
15489 {
15490 let mut db = GraphDb::open(&dir).unwrap();
15491 // frame 0
15492 db.insert_node("Org", "o1", vec![]).unwrap();
15493 // frame 1: create view — initial backfill sees 0 derived edges (none fired yet)
15494 db.create_view(ViewDef {
15495 name: "employee_count".into(),
15496 label: "Org".into(),
15497 view_prop: "emp".into(),
15498 source: ViewSource::Degree {
15499 edge_type: "WORKS_AT".into(),
15500 direction: Direction::In,
15501 },
15502 })
15503 .unwrap();
15504 // frame 2: create rule — no Persons yet; backfill is a no-op
15505 db.create_rule(fk_rule()).unwrap();
15506 // frame 3: p1 — rule fires WORKS_AT p1→o1; degree = 1
15507 db.insert_node(
15508 "Person",
15509 "p1",
15510 vec![("org_id".into(), Value::Str("o1".into()))],
15511 )
15512 .unwrap();
15513 // frame 4: p2 — degree = 2
15514 db.insert_node(
15515 "Person",
15516 "p2",
15517 vec![("org_id".into(), Value::Str("o1".into()))],
15518 )
15519 .unwrap();
15520 // frame 5: p3 — degree = 3
15521 db.insert_node(
15522 "Person",
15523 "p3",
15524 vec![("org_id".into(), Value::Str("o1".into()))],
15525 )
15526 .unwrap();
15527 // Sanity: normal open sees degree = 3.
15528 assert_eq!(
15529 db.get_view_prop("o1", "emp"),
15530 Some(Value::Int(3)),
15531 "normal db must show degree 3 after 3 derived edges"
15532 );
15533 } // WAL flushed
15534
15535 // Re-open normally to get the authoritative reference value.
15536 let normal_db = GraphDb::open(&dir).unwrap();
15537 let normal_emp = normal_db.get_view_prop("o1", "emp");
15538 assert_eq!(
15539 normal_emp,
15540 Some(Value::Int(3)),
15541 "re-opened normal db must show degree 3"
15542 );
15543
15544 // Latest as-of (last WAL commit): must match the normal open.
15545 // History-marker frames are appended after each rule-fire, so the total
15546 // commit count is computed dynamically rather than hardcoded.
15547 let total = crate::wal_commit_count_at(&dir).unwrap();
15548 let aof_latest = GraphDb::open_at(&dir, total - 1).unwrap();
15549 assert_eq!(
15550 aof_latest.get_view_prop("o1", "emp"),
15551 normal_emp,
15552 "open_at latest: derived-edge view must equal normal open (rebuild_all required)"
15553 );
15554
15555 // Mid-history as-of (commit 3 = p1 insert Batch frame): only p1; degree = 1.
15556 // The DerivedEdgeAdded marker for p1 is at frame 4 (state no-op on replay),
15557 // so replaying 0..=3 correctly re-derives only the p1→o1 edge.
15558 let aof_mid = GraphDb::open_at(&dir, 3).unwrap();
15559 assert_eq!(
15560 aof_mid.get_view_prop("o1", "emp"),
15561 Some(Value::Int(1)),
15562 "open_at mid-history: only p1 exists at frame 3, degree must be 1"
15563 );
15564
15565 let _ = std::fs::remove_dir_all(&dir);
15566 }
15567
15568 /// Pin: subscribe_* on an as-of instance must return Err(ReadOnly) —
15569 /// as-of instances never commit, so distribute_events never runs and any
15570 /// subscription would wait forever.
15571 #[test]
15572 fn subscribe_on_as_of_returns_read_only_error() {
15573 let dir = tmp_dir("sub-as-of-read-only");
15574 {
15575 let mut db = GraphDb::open(&dir).unwrap();
15576 db.insert_node("Org", "o1", vec![]).unwrap();
15577 db.create_rule(fk_rule()).unwrap();
15578 }
15579 let mut aof = GraphDb::open_at(&dir, 0).unwrap();
15580
15581 assert!(
15582 matches!(
15583 aof.subscribe_all_rules(),
15584 Err(core_storage::GraphError::ReadOnly)
15585 ),
15586 "subscribe_all_rules on as-of must return ReadOnly"
15587 );
15588 assert!(
15589 matches!(
15590 aof.subscribe_writes(),
15591 Err(core_storage::GraphError::ReadOnly)
15592 ),
15593 "subscribe_writes on as-of must return ReadOnly"
15594 );
15595 assert!(
15596 matches!(
15597 aof.subscribe_rule("works_at"),
15598 Err(core_storage::GraphError::ReadOnly)
15599 ),
15600 "subscribe_rule on as-of must return ReadOnly"
15601 );
15602 let _ = std::fs::remove_dir_all(&dir);
15603 }
15604
15605 /// Regression: a failed dense WAL rewrite must not leave speculative
15606 /// interns in `syms`. If it does, the next successful mutation logs an
15607 /// `Intern` record with an inflated id; replay (which never saw the
15608 /// orphans) assigns a smaller id and the WAL becomes unreplayable.
15609 #[test]
15610 fn dense_rewrite_error_rolls_back_speculative_interns() {
15611 let dir = tmp_dir("dense-rewrite-rollback");
15612 {
15613 let mut db = GraphDb::open(&dir).unwrap();
15614 db.insert_node("Person", "a", vec![]).unwrap();
15615
15616 // Bypass MutPreview validation to hit the rewrite's own error path
15617 // (same shape as an id-exhaustion failure mid-rewrite). The
15618 // InsertEdge arm interns the edge type before it resolves keys.
15619 let err = db.rewrite_wal_dense(vec![WalRecord::InsertEdge {
15620 edge_type: "ORPHAN_TYPE".into(),
15621 src_key: "missing".into(),
15622 dst_key: "a".into(),
15623 }]);
15624 assert!(err.is_err(), "rewrite of a missing src key must fail");
15625 assert_eq!(
15626 db.syms.get("ORPHAN_TYPE"),
15627 None,
15628 "failed rewrite must roll back speculative interns"
15629 );
15630
15631 // A later successful mutation must produce a replayable WAL.
15632 db.set_prop("a", "later_field", Value::Int(2)).unwrap();
15633 }
15634 let db = GraphDb::open(&dir).expect("WAL must replay after failed rewrite");
15635 assert_eq!(db.get_prop("a", "later_field"), Some(Value::Int(2)));
15636 let _ = std::fs::remove_dir_all(&dir);
15637 }
15638}