core_api/db.rs
1use crate::ingest::{IngestOptions, IngestReport};
2use crate::roles::{PropPredicate, RoleDef, RolesFile, WriteScope};
3use crate::subscription::{
4 event_matches, DbEvent, SubEntry, SubFilter, SubInner, Subscription, DEFAULT_SUB_CAPACITY,
5};
6use core_query::cypher::ast::{ret_val_label, ArithOp};
7use core_query::cypher::{
8 execute, execute_union, is_subscribable, is_write_tokens, lex, parse, parse_read, parse_write,
9 plan, Expr, MatchDeleteNodeStmt, NodePat, Operand, Params, Pattern, PlanOp, Query, RetItem,
10 RetVal, WriteStatement,
11};
12use core_query::{eval_cmp, eval_filter, expand, neighborhood, Dir, Filter, GraphView, ResultSet};
13use core_rules::{
14 decode_rule_def, ef_max, evaluate, BuildProgress, EngineEdgeDelta, GraphMut, NodeView,
15 Predicate, RuleDef, RuleEngine, ViewDef, ViewStore,
16};
17use core_storage::fs::{FileId, Fs, FsIntrospect, RealFs};
18use core_storage::fulltext::FulltextIndex;
19use core_storage::property_index::PropertyIndex;
20use core_storage::v8::encode::{
21 archived_hnsw_to_owned, archived_rules_meta_to_owned, archived_to_idmap, archived_to_interner,
22 archived_views_to_owned, decode_last_change_bytes, decode_meta, encode_v8, V8Meta,
23};
24use core_storage::v8::seam::TopologyView;
25use core_storage::wal::{decode_all, encode_record, WalRecord};
26use core_storage::EdgePropsView;
27use core_storage::{
28 namespace_of_value, ColumnStore, Direction, EdgeProps, GraphError, IdMap, Interner, Result,
29 Topology, Value,
30};
31pub use core_storage::{valid_namespace, NS_DEFAULT, NS_MAX_LEN, NS_PROP};
32
33/// Index of [`NS_DEFAULT`] in `GraphDb::ns_names` — always zero, so the
34/// open-time pass over a store with no `ns` column fills `node_ns` with one
35/// constant and allocates no names.
36const NS_DEFAULT_IDX: u32 = 0;
37
38/// The reserved edge property holding a pair's insert count (§5.13).
39///
40/// Absent means 1 — the count is written only from the second insert of a
41/// triple onward, and only on a store that called
42/// [`GraphDb::enable_multiplicity`]. The engine owns the name: Cypher `SET` on
43/// it is refused, as the other reserved names are.
44pub const EDGE_COUNT_PROP: &str = "count";
45use serde::{Deserialize, Serialize};
46use std::collections::{BTreeMap, BTreeSet, HashMap, HashSet};
47use std::sync::Arc;
48
49/// Print a timing checkpoint when MUSHROOMDB_TRACE_OPEN is set.
50/// Zero-cost when the env var is absent (the var check is O(1) after first call).
51macro_rules! trace_open {
52 ($phase:literal, $t:expr) => {
53 if std::env::var("MUSHROOMDB_TRACE_OPEN").is_ok() {
54 eprintln!(
55 "[MUSHROOMDB_TRACE_OPEN] {:40} {:>9.3?}",
56 $phase,
57 $t.elapsed()
58 );
59 }
60 };
61}
62
63/// Print a migration phase checkpoint when MUSHROOMDB_TRACE_MIGRATE is set.
64/// Zero-cost when the env var is absent (the var check is O(1) after first call).
65macro_rules! trace_migrate {
66 ($phase:literal, $t:expr) => {
67 if std::env::var("MUSHROOMDB_TRACE_MIGRATE").is_ok() {
68 eprintln!(
69 "[MUSHROOMDB_TRACE_MIGRATE] {:40} {:>9.3?}",
70 $phase,
71 $t.elapsed()
72 );
73 }
74 };
75}
76
77// Test-only: counts how many times `pending_deltas_since().to_vec()` actually
78// executes (i.e., at least one view is defined). Used to verify the fast-path
79// guard skips the allocation when `view_store.is_empty()`.
80#[cfg(test)]
81thread_local! {
82 static DELTA_COPY_COUNT: std::cell::Cell<usize> = const { std::cell::Cell::new(0) };
83}
84
85// Per-thread count of query-subscription `execute` calls in `distribute_events`.
86//
87// Incremented each time a query subscription actually runs its plan (i.e.,
88// the label-skip fast-path did not fire). Because `distribute_events` is
89// called synchronously on the writer thread, this thread-local correctly
90// isolates each test thread's count even when integration tests run in
91// parallel. Read via [`query_sub_exec_count`].
92thread_local! {
93 static QUERY_SUB_EXECS_TL: std::cell::Cell<usize> = const { std::cell::Cell::new(0) };
94}
95
96/// Return the number of query-subscription re-executions logged on this
97/// thread since the process started (or since last reset via
98/// [`reset_query_sub_exec_count`]).
99///
100/// Primarily for integration tests that verify the label-skip fast-path.
101#[doc(hidden)]
102pub fn query_sub_exec_count() -> usize {
103 QUERY_SUB_EXECS_TL.with(|c| c.get())
104}
105
106/// Reset the per-thread query-subscription execution counter to zero.
107#[doc(hidden)]
108pub fn reset_query_sub_exec_count() {
109 QUERY_SUB_EXECS_TL.with(|c| c.set(0));
110}
111
112// Exact-versus-approximate warnings emitted on this thread. Thread-local for
113// the same reason [`QUERY_SUB_EXECS_TL`] is: integration tests run in parallel
114// and each gets its own thread, so a neighbour's masked search cannot be
115// mistaken for this test's.
116thread_local! {
117 static AMBIGUOUS_EXACTNESS_WARNS: std::cell::Cell<u64> = const { std::cell::Cell::new(0) };
118 static AMBIGUOUS_EXACTNESS_LAST: std::cell::RefCell<Option<String>> =
119 const { std::cell::RefCell::new(None) };
120}
121
122/// How many times a masked, non-exact vector search has explained itself on
123/// this thread since the last [`ambiguous_exactness_warns_reset`].
124///
125/// The line itself is the product; this counter exists so a test can assert it
126/// is printed **once per index** rather than once per call.
127///
128/// **Single-threaded assertions only.** The suppression set this counts is a
129/// `Mutex<HashSet<_>>` on the `GraphDb` — shared by every thread — while the
130/// counter is thread-local. Under a concurrent caller (`serve`, which is the
131/// deployment the warning exists for) the thread that prints the line is not
132/// necessarily the thread that asked, so a zero here does not mean the line was
133/// not printed and a one does not mean it was printed once. It answers
134/// "once per index" only in a test that owns the store.
135#[doc(hidden)]
136pub fn ambiguous_exactness_warns() -> u64 {
137 AMBIGUOUS_EXACTNESS_WARNS.with(|c| c.get())
138}
139
140/// The most recent exactness warning printed on this thread, verbatim.
141///
142/// The advice a caller reads has to be advice that caller can act on, which the
143/// counter alone cannot witness — see
144/// `the_hybrid_path_advises_a_call_the_hybrid_caller_can_make`. Carries the
145/// same single-threaded caveat as [`ambiguous_exactness_warns`].
146#[doc(hidden)]
147pub fn ambiguous_exactness_last_warning() -> Option<String> {
148 AMBIGUOUS_EXACTNESS_LAST.with(|c| c.borrow().clone())
149}
150
151/// Reset this thread's exactness-warning counter and recorded line.
152#[doc(hidden)]
153pub fn ambiguous_exactness_warns_reset() {
154 AMBIGUOUS_EXACTNESS_WARNS.with(|c| c.set(0));
155 AMBIGUOUS_EXACTNESS_LAST.with(|c| *c.borrow_mut() = None);
156}
157
158/// Which signature reached the approximate masked vector leg.
159///
160/// One code path, two entry points, and the advice cannot be the same: a
161/// warning that names an argument the caller's function does not take sends
162/// them looking for a parameter that is not there. `search_hybrid` and
163/// `search_hybrid_scoped` take `(text_field, query_text, vector_field,
164/// query_vec, label, k[, mask])` — no `exact`, no `where`.
165///
166/// The warning is not suppressed on the hybrid path. The approximation is the
167/// same one, and a caller who read `mask=` as a promise of exhaustiveness is
168/// the reader it was written for whichever door they came in by; only the
169/// remedy differs, so only the remedy changes.
170#[derive(Clone, Copy, PartialEq, Eq, Hash, Debug)]
171enum ExactnessCaller {
172 /// `find_similar` / `find_similar_vector_*` — `exact` and `where` are its
173 /// own parameters.
174 Vector,
175 /// `search_hybrid` / `search_hybrid_scoped` — neither argument exists, and
176 /// the leg is one half of a fusion.
177 Hybrid,
178}
179
180impl ExactnessCaller {
181 fn subject(self) -> &'static str {
182 match self {
183 Self::Vector => "a masked vector search",
184 Self::Hybrid => "the vector leg of a masked hybrid search",
185 }
186 }
187
188 fn advice(self) -> &'static str {
189 match self {
190 Self::Vector => "pass exact=True or a where= predicate.",
191 // Names the call that does take the argument, because this one
192 // does not: the caller's own next step, not a parameter hunt.
193 Self::Hybrid => {
194 "run the vector leg on its own with find_similar(field, vector, mask=…, \
195 exact=True) and fuse it with search() yourself — search_hybrid itself \
196 takes no exactness argument."
197 }
198 }
199 }
200}
201
202/// Internal state for a single `subscribe_query` subscription.
203///
204/// On every commit, `distribute_events` re-executes `ops` against the current
205/// graph state, diffs the result against `prev_rows`, and pushes
206/// `DbEvent::QueryRowAdded` / `QueryRowRemoved` events to `inner`.
207///
208/// **Full re-run per commit; use LIMIT to bound execution cost.**
209/// (Differential evaluation is roadmap / Phase 5.)
210pub(crate) struct QuerySubEntry {
211 /// Compiled plan for the subscribed Cypher query.
212 ops: Vec<PlanOp>,
213 /// Column names from the first execution (fixed for the subscription lifetime).
214 columns: Vec<String>,
215 /// Serialized (JSON) row key → row data, representing the result set at
216 /// the end of the last commit. Used to diff against the new result.
217 prev_row_map: std::collections::HashMap<String, Vec<Option<Value>>>,
218 /// Weak pointer to the subscriber queue; dead Weak → subscription dropped.
219 inner: std::sync::Weak<SubInner>,
220 /// Interned label sym captured at subscribe time from the plan's leading scan
221 /// (`ScanLabel`, `IndexScan`, or `IndexIntersect` with a concrete label).
222 ///
223 /// `None` means the plan has an `Expand` op (or no recognizable leading scan
224 /// with a concrete label), and this subscription must re-execute on every
225 /// commit without skipping. This is the conservative v0.4.3 boundary: Expand
226 /// queries are never skipped because edges can alter join results regardless
227 /// of which node labels were written.
228 scan_label: Option<u32>,
229}
230
231/// A post-commit mutation notification.
232///
233/// Emitted from `log_then_apply` after the WAL append, fsync, and
234/// in-memory `apply` all succeed. Never emitted for rejected operations
235/// (validation errors, [`GraphError::RuleOwned`], duplicate keys, no-op
236/// deletes/removes). Event payloads carry user keys and rule names, never
237/// internal ids.
238///
239/// **Replay:** [`GraphDb::open`] / [`GraphDb::open_with`] replay the WAL via
240/// `apply` only. Emission lives exclusively in `log_then_apply`, so
241/// recovery is silent even if a sink were installed (it cannot be: the
242/// sink is in-memory and set after open).
243///
244/// **Ordering:** a `Batch` WAL frame emits one event per inner record, then
245/// [`MutationEvent::BatchApplied`]. An ingest commit emits those same inner
246/// events, then [`MutationEvent::Ingested`] (not `BatchApplied`). An empty
247/// or all-noop batch writes no WAL and emits nothing (including no summary).
248///
249/// **Derived edges:** rule-created or retracted edges are not individually
250/// evented — they are recoverable from the triggering mutation plus the live
251/// rule set. Only the triggering record is emitted.
252///
253/// **Wire form:** externally tagged snake_case JSON
254/// (`{"node_inserted":{"label":"A","key":"k"}}`).
255#[derive(Debug, Clone, PartialEq, Eq, Serialize)]
256#[serde(rename_all = "snake_case")]
257pub enum MutationEvent {
258 NodeInserted {
259 label: String,
260 key: String,
261 },
262 PropSet {
263 key: String,
264 field: String,
265 },
266 PropRemoved {
267 key: String,
268 field: String,
269 },
270 EdgeInserted {
271 edge_type: String,
272 src: String,
273 dst: String,
274 },
275 EdgeDeleted {
276 edge_type: String,
277 src: String,
278 dst: String,
279 },
280 NodeDeleted {
281 key: String,
282 },
283 RuleCreated {
284 name: String,
285 },
286 RuleDeleted {
287 name: String,
288 },
289 RuleRebuilt {
290 name: String,
291 },
292 BatchApplied {
293 ops: usize,
294 },
295 Ingested {
296 label: String,
297 inserted: usize,
298 },
299}
300
301fn event_from_record(rec: &WalRecord, intern: &Interner, ids: &IdMap) -> Option<MutationEvent> {
302 match rec {
303 WalRecord::InsertNode { label, key, .. } => Some(MutationEvent::NodeInserted {
304 label: label.clone(),
305 key: key.clone(),
306 }),
307 WalRecord::InsertNodeId { label, key, .. } => Some(MutationEvent::NodeInserted {
308 label: intern.resolve(*label)?.to_string(),
309 key: key.clone(),
310 }),
311 WalRecord::SetProp { key, field, .. } => Some(MutationEvent::PropSet {
312 key: key.clone(),
313 field: field.clone(),
314 }),
315 WalRecord::SetPropId { id, field, .. } => Some(MutationEvent::PropSet {
316 key: ids.key_of(*id)?.to_string(),
317 field: intern.resolve(*field)?.to_string(),
318 }),
319 WalRecord::RemoveProp { key, field } => Some(MutationEvent::PropRemoved {
320 key: key.clone(),
321 field: field.clone(),
322 }),
323 WalRecord::InsertEdge {
324 edge_type,
325 src_key,
326 dst_key,
327 } => Some(MutationEvent::EdgeInserted {
328 edge_type: edge_type.clone(),
329 src: src_key.clone(),
330 dst: dst_key.clone(),
331 }),
332 WalRecord::InsertEdgeId { etype, src, dst } => Some(MutationEvent::EdgeInserted {
333 edge_type: intern.resolve(*etype)?.to_string(),
334 src: ids.key_of(*src)?.to_string(),
335 dst: ids.key_of(*dst)?.to_string(),
336 }),
337 WalRecord::DeleteEdge {
338 edge_type,
339 src_key,
340 dst_key,
341 } => Some(MutationEvent::EdgeDeleted {
342 edge_type: edge_type.clone(),
343 src: src_key.clone(),
344 dst: dst_key.clone(),
345 }),
346 WalRecord::DeleteNode { key } => Some(MutationEvent::NodeDeleted { key: key.clone() }),
347 WalRecord::CreateRule { def_bytes } => {
348 let def: RuleDef = decode_rule_def(def_bytes).ok()?;
349 Some(MutationEvent::RuleCreated { name: def.name })
350 }
351 WalRecord::DeleteRule { name } => Some(MutationEvent::RuleDeleted { name: name.clone() }),
352 WalRecord::RebuildRule { name } => Some(MutationEvent::RuleRebuilt { name: name.clone() }),
353 WalRecord::Batch(_)
354 | WalRecord::CreateView { .. }
355 | WalRecord::DeleteView { .. }
356 | WalRecord::EnableFulltext { .. }
357 | WalRecord::DisableFulltext { .. }
358 | WalRecord::EnableIndex { .. }
359 | WalRecord::DisableIndex { .. }
360 | WalRecord::Intern { .. }
361 // History markers are no-ops for mutation events — they carry no new
362 // state and rules re-derive deterministically on replay.
363 | WalRecord::DerivedEdgeAdded { .. }
364 | WalRecord::DerivedEdgeRetracted { .. }
365 // A count changes neither the node nor the edge population: the pair it
366 // counts was already there, which is why it is written at all.
367 | WalRecord::SetEdgeCount { .. }
368 // RenameNode carries no node/edge count change; no special event.
369 | WalRecord::RenameNode { .. } => None,
370 }
371}
372
373/// Database-wide counters plus per-rule budget/fire stats.
374#[derive(Debug, Clone, PartialEq, Eq, Serialize)]
375pub struct Stats {
376 pub nodes_live: usize,
377 pub nodes_tombstoned: usize,
378 pub edges: u64,
379 pub rules: Vec<RuleStats>,
380 /// How many writes hit the rule-chaining depth cap with work still pending,
381 /// since this handle was opened. Non-zero means some derived edges beyond
382 /// the cap are stale and no single later write will repair them: split the
383 /// rule chain or shorten it. Never persisted, so it resets on reopen.
384 #[serde(default)]
385 pub chain_truncations: u64,
386 /// The oldest commit index history still reaches (the WAL horizon floor).
387 /// `0` means nothing has been pruned and history is complete; a non-zero
388 /// value means events before that commit were pruned and are gone.
389 #[serde(default)]
390 pub history_floor: u64,
391 /// Live node counts per namespace, in name order. Always carries
392 /// `default` — a store is at least its default namespace — so a
393 /// single-tenant store reads `[{"name":"default", …}]` and a reader can
394 /// tell "no namespaces in use" from one entry.
395 #[serde(default)]
396 pub namespaces: Vec<NamespaceStats>,
397}
398
399/// Live node count for one namespace; one entry of [`Stats::namespaces`].
400#[derive(Debug, Clone, PartialEq, Eq, Serialize)]
401pub struct NamespaceStats {
402 pub name: String,
403 pub nodes_live: usize,
404}
405
406/// One rule's provenance size, trip latch, and fire counter.
407///
408/// `tripped` is a one-way latch: once set, the engine adds no new edges for
409/// that rule until [`GraphDb::rebuild_rule`] (and only if the full desired
410/// set then fits). `fires` counts `on_node_changed` evaluations plus
411/// backfill/rebuild participant ticks (rebuild counts even when it is a
412/// provenance no-op).
413#[derive(Debug, Clone, PartialEq, Eq, Serialize)]
414pub struct RuleStats {
415 pub name: String,
416 pub edges: u64,
417 pub tripped: bool,
418 pub fires: u64,
419 /// Whether this rule uses the approximate IVF-Flat candidate path.
420 pub approximate: bool,
421 /// `Some` while this rule's vector index is still being built.
422 ///
423 /// The rule derives **no** edges until it is `None`: the backfill is one
424 /// commit that runs after the index is whole, so a caller never sees a
425 /// partial edge set. Absent from the JSON when the rule is not building,
426 /// which is every rule created over a corpus at or below
427 /// [`core_rules::HNSW_BUILD_BATCH`] vectors.
428 #[serde(default, skip_serializing_if = "Option::is_none")]
429 pub building: Option<BuildProgress>,
430}
431
432/// One entry in the slow-query ring buffer.
433#[derive(Debug, Clone, Serialize)]
434pub struct SlowQueryEntry {
435 /// Execution time in whole milliseconds.
436 pub ms: u64,
437 /// The Cypher query string that was slow.
438 pub query: String,
439 /// The commit sequence number at the time the query ran.
440 pub at_commit: u64,
441}
442
443/// Snapshot of the slow-query log returned by [`GraphDb::slow_query_snapshot`].
444#[derive(Debug, Clone, Serialize)]
445pub struct SlowQuerySnapshot {
446 /// Current threshold in milliseconds (0 = disabled).
447 pub threshold_ms: u64,
448 /// Total number of slow queries ever recorded (not capped by ring size).
449 pub count: u64,
450 /// Most-recent slow queries (up to 16), oldest first.
451 pub last: Vec<SlowQueryEntry>,
452}
453
454/// Internal ring-buffer state protected by a `Mutex` so `query(&self)` can
455/// write to it without a mutable borrow.
456struct SlowQueryLog {
457 entries: std::collections::VecDeque<SlowQueryEntry>,
458 total: u64,
459}
460
461/// Maximum number of entries kept in the slow-query ring buffer.
462const SLOW_QUERY_RING_CAP: usize = 16;
463
464/// Wire summary of a [`Predicate`]. JSON only — `Explanation` is never
465/// bincode-persisted (WAL/snapshots store `RuleDef` bytes, not this type).
466#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
467pub struct PredicateSummary {
468 pub kind: String,
469 pub fields: Vec<String>,
470 pub min: Option<f64>,
471 pub tolerance: Option<f64>,
472 pub km: Option<f64>,
473 pub parts: Option<Vec<PredicateSummary>>,
474 /// True when the owning rule has `approximate=true` (IVF-Flat candidate path).
475 /// Always false for predicates reported without rule context (sub-predicates in `parts`).
476 #[serde(default)]
477 pub approximate: bool,
478}
479
480impl From<&Predicate> for PredicateSummary {
481 fn from(p: &Predicate) -> Self {
482 match p {
483 Predicate::KeyMatch { field } => PredicateSummary {
484 kind: "key_match".into(),
485 fields: vec![field.clone()],
486 min: None,
487 tolerance: None,
488 km: None,
489 parts: None,
490 approximate: false,
491 },
492 Predicate::FieldEqual { field } => PredicateSummary {
493 kind: "field_equal".into(),
494 fields: vec![field.clone()],
495 min: None,
496 tolerance: None,
497 km: None,
498 parts: None,
499 approximate: false,
500 },
501 Predicate::Overlap { field, min } => PredicateSummary {
502 kind: "overlap".into(),
503 fields: vec![field.clone()],
504 min: Some(*min),
505 tolerance: None,
506 km: None,
507 parts: None,
508 approximate: false,
509 },
510 Predicate::NumericWithin { field, tolerance } => PredicateSummary {
511 kind: "numeric_within".into(),
512 fields: vec![field.clone()],
513 min: None,
514 tolerance: Some(*tolerance),
515 km: None,
516 parts: None,
517 approximate: false,
518 },
519 Predicate::GeoRadius { field, km } => PredicateSummary {
520 kind: "geo_radius".into(),
521 fields: vec![field.clone()],
522 min: None,
523 tolerance: None,
524 km: Some(*km),
525 parts: None,
526 approximate: false,
527 },
528 Predicate::VectorSimilar { field, min } => PredicateSummary {
529 kind: "vector_similar".into(),
530 fields: vec![field.clone()],
531 min: Some(*min),
532 tolerance: None,
533 km: None,
534 parts: None,
535 approximate: false,
536 },
537 Predicate::All(inner) => {
538 let parts: Vec<PredicateSummary> = inner.iter().map(Self::from).collect();
539 let mut fields = Vec::new();
540 for part in &parts {
541 for f in &part.fields {
542 if !fields.contains(f) {
543 fields.push(f.clone());
544 }
545 }
546 }
547 PredicateSummary {
548 kind: "all".into(),
549 fields,
550 min: None,
551 tolerance: None,
552 km: None,
553 parts: Some(parts),
554 approximate: false,
555 }
556 }
557 Predicate::Any(inner) => {
558 let parts: Vec<PredicateSummary> = inner.iter().map(Self::from).collect();
559 let mut fields = Vec::new();
560 for part in &parts {
561 for f in &part.fields {
562 if !fields.contains(f) {
563 fields.push(f.clone());
564 }
565 }
566 }
567 PredicateSummary {
568 kind: "any".into(),
569 fields,
570 min: None,
571 tolerance: None,
572 km: None,
573 parts: Some(parts),
574 approximate: false,
575 }
576 }
577 }
578 }
579}
580
581/// Snapshot of a live node's key, label, and columnar properties.
582///
583/// `props` is a [`BTreeMap`] so field order is deterministic (sorted by name)
584/// regardless of insert order or the columnar store's `HashMap` iteration.
585///
586/// Deliberately does not derive `Serialize`: `Value`'s serde form is
587/// internally tagged. Wire JSON is built by `value_to_json` in the server.
588#[derive(Debug, Clone, PartialEq)]
589pub struct NodeInfo {
590 pub key: String,
591 pub label: String,
592 pub props: BTreeMap<String, Value>,
593}
594
595/// Counts returned by [`GraphDb::delete_node`].
596#[derive(Debug, Clone, PartialEq, Eq, Default)]
597pub struct DeleteReport {
598 /// Number of manual (user-inserted) edges removed.
599 pub manual_edges: u64,
600 /// Number of derived (rule-owned) edges retracted.
601 pub derived_edges: u64,
602}
603
604/// One directed edge incident on a node, with provenance membership.
605///
606/// `derived` is true iff `(edge_type, src, dst)` is in the rule engine's
607/// Plan-8 `by_node` provenance index.
608#[derive(Debug, Clone, PartialEq, Eq, Serialize)]
609pub struct EdgeInfo {
610 pub edge_type: String,
611 pub src_key: String,
612 pub dst_key: String,
613 pub derived: bool,
614}
615
616/// One directed edge incident on a node at a point in WAL history, with the
617/// rule that derived it when it is rule-owned.
618///
619/// Returned by [`GraphDb::edges_at`] (sorted by `(edge_type, src_key, dst_key)`)
620/// and by [`GraphDb::what_if_set_prop`].
621#[derive(Debug, Clone, PartialEq, Eq, PartialOrd, Ord, Serialize)]
622pub struct EdgeAt {
623 pub edge_type: String,
624 pub src_key: String,
625 pub dst_key: String,
626 /// `true` when a rule wrote the edge (`DerivedEdgeAdded` in the WAL, or a
627 /// live provenance entry).
628 pub derived: bool,
629 /// The rule that derived the edge. `None` for a manual edge.
630 pub rule: Option<String>,
631}
632
633/// The derived edges a hypothetical property change would retract and derive.
634///
635/// Returned by [`GraphDb::what_if_set_prop`]. Both lists are sorted by
636/// `(edge_type, src_key, dst_key)` and every entry is rule-derived.
637#[derive(Debug, Clone, PartialEq, Eq, Serialize)]
638pub struct WhatIf {
639 /// Derived edges that exist now and would be retracted.
640 pub lost: Vec<EdgeAt>,
641 /// Derived edges that do not exist now and would be derived.
642 pub gained: Vec<EdgeAt>,
643}
644
645/// An edge with mask-aware endpoint visibility.
646///
647/// Returned by [`GraphDb::node_edges_masked`] in [`crate::mask::MaskMode::Stub`]
648/// mode — hidden endpoints carry `*_restricted: true`.
649#[derive(Debug, Clone, PartialEq, Eq)]
650pub struct MaskedEdge {
651 pub edge_type: String,
652 pub src_key: String,
653 /// `true` when `src_key` is in the DB but hidden from the mask.
654 pub src_restricted: bool,
655 pub dst_key: String,
656 /// `true` when `dst_key` is in the DB but hidden from the mask.
657 pub dst_restricted: bool,
658 pub derived: bool,
659}
660
661/// Result of a mask-aware node lookup via [`GraphDb::node_info_masked`].
662///
663/// `None` from that method means the key does not exist (→ 404).
664/// `Some(Restricted)` is only produced when `mask.mode() == MaskMode::Stub`.
665#[derive(Debug, PartialEq)]
666pub enum MaskedNodeResult {
667 Visible(NodeInfo),
668 /// Node exists in the DB but is hidden from this mask.
669 Restricted,
670}
671
672/// One rule-owned edge between two nodes, with the rule name, edge type,
673/// direction (src_key → dst_key), and weight if the rule stores one.
674#[derive(Debug, Clone, PartialEq, Serialize)]
675pub struct Explanation {
676 pub rule: String,
677 pub edge_type: String,
678 pub src_key: String,
679 pub dst_key: String,
680 pub weight: Option<f64>,
681 pub predicate: PredicateSummary,
682 /// For a via-hop rule, the edge type the rule hops over to reach its
683 /// candidates. `None` for a plain two-node rule. A via-hop rule whose
684 /// `via_edge` is itself rule-derived is the chaining case: the hop edge
685 /// was written by another rule in the same commit.
686 #[serde(default)]
687 pub via_edge: Option<String>,
688}
689
690/// Report returned by [`GraphDb::backup_to`].
691#[derive(Debug, Clone)]
692pub struct BackupReport {
693 /// Filenames copied into the destination directory (sorted ascending).
694 pub files: Vec<String>,
695 /// Total bytes written across all copied files.
696 pub bytes: u64,
697 /// `true` when the destination opened cleanly and passed post-copy checks.
698 ///
699 /// For stores that have a `snapshot.bin` this means: all V8 section CRCs
700 /// matched **and** the destination opened without error.
701 ///
702 /// For WAL-only stores (no `snapshot.bin`) there is no snapshot to
703 /// CRC-check; `verified` is `true` when the destination opened and
704 /// replayed the WAL without error (record-level checksums in the WAL
705 /// provide the integrity signal, not section CRCs).
706 pub verified: bool,
707}
708
709/// One directed edge in export form, with optional rule attribution for derived edges.
710///
711/// Returned by [`GraphDb::all_edges_for_export`].
712///
713/// Does not derive `Eq`/`Ord`: `weight` is an `f64` and NaN breaks a total
714/// order. Callers that need a stable edge ordering already sort by
715/// `(edge_type, src, dst)` explicitly (see `all_edges_for_export`).
716#[derive(Debug, Clone, PartialEq, PartialOrd)]
717pub struct ExportEdge {
718 pub edge_type: String,
719 pub src: String,
720 pub dst: String,
721 pub derived: bool,
722 /// Rule name that created this edge, if derived. `None` for manual edges.
723 pub rule: Option<String>,
724 /// The creating rule's declared `weight_prop`, read off this edge, when
725 /// derived and numeric (`Int`/`Float`). `None` for manual edges, derived
726 /// edges whose rule declares no `weight_prop`, or a non-numeric value.
727 pub weight: Option<f64>,
728}
729
730/// One edge type's shape, as [`GraphDb::edge_type_census`] counts it.
731///
732/// Deliberately per *type* and not per edge: everything here is a summary a
733/// caller can print in one line, and none of it costs a record per edge.
734#[derive(Debug, Clone, PartialEq, Eq)]
735pub struct EdgeTypeCensus {
736 pub edge_type: String,
737 /// Directed edges of this type. Counted the way
738 /// [`GraphDb::edge_count`] counts: each edge once, from its source.
739 pub edges: u64,
740 /// Every label seen on a source of this type, sorted.
741 pub src_labels: Vec<String>,
742 /// Every label seen on a destination of this type, sorted.
743 pub dst_labels: Vec<String>,
744 /// The rules that declare this `edge_type`, sorted. Empty for a type
745 /// written by hand.
746 pub rules: Vec<String>,
747 /// `(src key, dst key)` of the first edge of this type in the store's own
748 /// id order — a real pair to quote in an example.
749 pub sample: Option<(String, String)>,
750}
751
752/// Construct the standard write-query result set (columns: created, properties_set, deleted).
753fn write_result_set() -> ResultSet {
754 ResultSet::new(vec![
755 "created".into(),
756 "properties_set".into(),
757 "deleted".into(),
758 ])
759}
760
761fn resolve_merge_set_value(op: &Operand, params: &BTreeMap<String, Value>) -> Result<Value> {
762 match op {
763 Operand::Lit(v) => Ok(v.clone()),
764 Operand::Param(name) => params
765 .get(name)
766 .cloned()
767 .ok_or_else(|| GraphError::QueryError {
768 detail: format!("missing parameter `{name}`"),
769 }),
770 _ => Err(GraphError::QueryError {
771 detail: "ON CREATE/ON MATCH SET value must be a literal or $parameter".into(),
772 }),
773 }
774}
775
776fn operand_node_vars(op: &Operand, out: &mut Vec<String>) {
777 match op {
778 Operand::Prop { var, .. } | Operand::Var(var) => {
779 if !out.contains(var) {
780 out.push(var.clone());
781 }
782 }
783 Operand::FuncCall { args, .. } => {
784 for arg in args {
785 operand_node_vars(arg, out);
786 }
787 }
788 Operand::BinArith { left, right, .. } => {
789 operand_node_vars(left, out);
790 operand_node_vars(right, out);
791 }
792 Operand::Case { branches, default } => {
793 // Branch conditions reference vars already bound (and mask-filtered)
794 // by the MATCH phase, so collecting from the value operands + ELSE
795 // is sufficient for RETURN-projection var discovery.
796 for (_, value) in branches {
797 operand_node_vars(value, out);
798 }
799 if let Some(d) = default {
800 operand_node_vars(d, out);
801 }
802 }
803 Operand::Index { base, index } => {
804 operand_node_vars(base, out);
805 operand_node_vars(index, out);
806 }
807 Operand::Lit(_) | Operand::Param(_) => {}
808 }
809}
810
811fn ret_node_vars(items: &[RetItem]) -> Vec<String> {
812 let mut out = Vec::new();
813 for item in items {
814 match &item.value {
815 RetVal::Var(v) | RetVal::Prop { var: v, .. } => {
816 if !out.contains(v) {
817 out.push(v.clone());
818 }
819 }
820 RetVal::FuncCall { args, .. } => {
821 for arg in args {
822 operand_node_vars(arg, &mut out);
823 }
824 }
825 RetVal::ScalarExpr(op) => operand_node_vars(op, &mut out),
826 RetVal::Agg { .. } => {}
827 }
828 }
829 out
830}
831
832fn add_var(out: &mut Vec<String>, v: &str) {
833 if !out.iter().any(|x| x == v) {
834 out.push(v.to_string());
835 }
836}
837
838fn pattern_node_vars(pats: &[Pattern]) -> Vec<String> {
839 let mut out = Vec::new();
840 for p in pats {
841 if let Some(v) = &p.start.var {
842 add_var(&mut out, v);
843 }
844 for (_, dest) in &p.chain {
845 if let Some(v) = &dest.var {
846 add_var(&mut out, v);
847 }
848 }
849 }
850 out
851}
852
853fn pattern_rel_vars(pats: &[Pattern]) -> Vec<String> {
854 let mut out = Vec::new();
855 for p in pats {
856 for (rel, _) in &p.chain {
857 if rel.hops.is_none() {
858 if let Some(v) = &rel.var {
859 add_var(&mut out, v);
860 }
861 }
862 }
863 }
864 out
865}
866
867fn rel_type_alias(var: &str) -> String {
868 format!("__rt_{var}")
869}
870
871fn ret_column_name(item: &RetItem) -> String {
872 if let Some(alias) = &item.alias {
873 return alias.clone();
874 }
875 // The same naming rule the planner and the executor use, so a
876 // write-statement RETURN names its columns exactly as a read query does.
877 // An aggregate is not legal in a write-statement RETURN; it keeps the
878 // placeholder it always had.
879 ret_val_label(&item.value).unwrap_or_else(|| "<agg>".to_string())
880}
881
882fn eval_set_return_operand<F: Fs>(
883 db: &GraphDb<F>,
884 match_rs: &ResultSet,
885 row: usize,
886 rel_vars: &[String],
887 op: &Operand,
888 params: &BTreeMap<String, Value>,
889) -> Result<Option<Value>> {
890 match op {
891 Operand::Lit(v) => Ok(Some(v.clone())),
892 Operand::Param(name) => params.get(name).cloned().ok_or_else(|| GraphError::QueryError {
893 detail: format!("missing parameter `{name}`"),
894 }).map(Some),
895 Operand::Var(name) if rel_vars.iter().any(|r| r == name) => Err(GraphError::QueryError {
896 detail: format!(
897 "cannot return relationship variable '{name}' bare; return its properties ({name}.field) instead"
898 ),
899 }),
900 Operand::Var(name) => Ok(match_rs.get(row, name).cloned()),
901 Operand::Prop { var, field } => {
902 if rel_vars.iter().any(|r| r == var) {
903 return Ok(None);
904 }
905 let Some(Value::Str(key)) = match_rs.get(row, var) else {
906 return Ok(None);
907 };
908 if let Some(v) = db.get_prop(key, field) {
909 return Ok(Some(v));
910 }
911 // Same stored-wins identity fallback as the read path:
912 // n.key / n.id / n.label, not only get_prop.
913 Ok(match field.as_str() {
914 "key" | "id" => Some(Value::Str(key.clone())),
915 "label" => db
916 .node_ref(key)
917 .map(|n| Value::Str(n.label().to_owned())),
918 _ => None,
919 })
920 }
921 Operand::FuncCall { name, args } => {
922 eval_set_return_func(db, match_rs, row, rel_vars, name, args, params)
923 }
924 Operand::BinArith { op, left, right } => {
925 let lv = eval_set_return_operand(db, match_rs, row, rel_vars, left, params)?;
926 let rv = eval_set_return_operand(db, match_rs, row, rel_vars, right, params)?;
927 eval_set_return_arith(op, lv, rv)
928 }
929 Operand::Case { branches, default } => {
930 for (cond, value) in branches {
931 if eval_set_return_expr(db, match_rs, row, rel_vars, cond, params, 0)? {
932 return eval_set_return_operand(db, match_rs, row, rel_vars, value, params);
933 }
934 }
935 match default {
936 Some(d) => eval_set_return_operand(db, match_rs, row, rel_vars, d, params),
937 None => Ok(None),
938 }
939 }
940 Operand::Index { base, index } => {
941 let base_val = eval_set_return_operand(db, match_rs, row, rel_vars, base, params)?;
942 let idx_val = eval_set_return_operand(db, match_rs, row, rel_vars, index, params)?;
943 Ok(core_query::value_ops::index_list(base_val, idx_val))
944 }
945 }
946}
947
948fn eval_set_return_expr<F: Fs>(
949 db: &GraphDb<F>,
950 match_rs: &ResultSet,
951 row: usize,
952 rel_vars: &[String],
953 expr: &Expr,
954 params: &BTreeMap<String, Value>,
955 depth: u32,
956) -> Result<bool> {
957 if depth > 256 {
958 return Err(GraphError::QueryError {
959 detail: "expression nesting too deep".into(),
960 });
961 }
962 match expr {
963 Expr::And(lhs, rhs) => {
964 let l = eval_set_return_expr(db, match_rs, row, rel_vars, lhs, params, depth + 1)?;
965 let r = eval_set_return_expr(db, match_rs, row, rel_vars, rhs, params, depth + 1)?;
966 Ok(l && r)
967 }
968 Expr::Or(lhs, rhs) => {
969 let l = eval_set_return_expr(db, match_rs, row, rel_vars, lhs, params, depth + 1)?;
970 let r = eval_set_return_expr(db, match_rs, row, rel_vars, rhs, params, depth + 1)?;
971 Ok(l || r)
972 }
973 Expr::Not(inner) => Ok(!eval_set_return_expr(
974 db,
975 match_rs,
976 row,
977 rel_vars,
978 inner,
979 params,
980 depth + 1,
981 )?),
982 Expr::Cmp { lhs, op, rhs } => {
983 let l = eval_set_return_operand(db, match_rs, row, rel_vars, lhs, params)?;
984 let r = eval_set_return_operand(db, match_rs, row, rel_vars, rhs, params)?;
985 match (l, r) {
986 (Some(a), Some(b)) => Ok(eval_cmp(op, &a, &b)),
987 _ => Ok(false),
988 }
989 }
990 Expr::Truthy(op) => {
991 let val = eval_set_return_operand(db, match_rs, row, rel_vars, op, params)?;
992 Ok(match val {
993 None => false,
994 Some(Value::Bool(b)) => b,
995 Some(Value::Int(n)) => n != 0,
996 Some(Value::Float(f)) => f != 0.0,
997 Some(Value::Str(s)) => !s.is_empty(),
998 Some(Value::List(v)) => !v.is_empty(),
999 Some(Value::Map(m)) => !m.is_empty(),
1000 })
1001 }
1002 Expr::IsNull(op) => {
1003 let val = eval_set_return_operand(db, match_rs, row, rel_vars, op, params)?;
1004 Ok(val.is_none())
1005 }
1006 Expr::IsNotNull(op) => {
1007 let val = eval_set_return_operand(db, match_rs, row, rel_vars, op, params)?;
1008 Ok(val.is_some())
1009 }
1010 Expr::In { expr, list } => {
1011 let Some(needle) = eval_set_return_operand(db, match_rs, row, rel_vars, expr, params)?
1012 else {
1013 return Ok(false);
1014 };
1015 for item_op in list {
1016 match eval_set_return_operand(db, match_rs, row, rel_vars, item_op, params)? {
1017 None => {}
1018 Some(Value::List(items)) => {
1019 for item in items {
1020 if eval_cmp(&core_query::CmpOp::Eq, &needle, &item) {
1021 return Ok(true);
1022 }
1023 }
1024 }
1025 Some(item) if eval_cmp(&core_query::CmpOp::Eq, &needle, &item) => {
1026 return Ok(true);
1027 }
1028 Some(_) => {}
1029 }
1030 }
1031 Ok(false)
1032 }
1033 }
1034}
1035
1036fn eval_set_return_arith(
1037 op: &ArithOp,
1038 lv: Option<Value>,
1039 rv: Option<Value>,
1040) -> Result<Option<Value>> {
1041 match (lv, rv) {
1042 (None, _) | (_, None) => Ok(None),
1043 (Some(Value::Int(a)), Some(Value::Int(b))) => {
1044 let result = match op {
1045 ArithOp::Sub => a.saturating_sub(b),
1046 ArithOp::Mul => a.saturating_mul(b),
1047 ArithOp::Add => a.saturating_add(b),
1048 ArithOp::Div => {
1049 if b == 0 {
1050 return Err(GraphError::QueryError {
1051 detail: "division by zero".into(),
1052 });
1053 }
1054 a.checked_div(b).unwrap_or(i64::MAX)
1055 }
1056 };
1057 Ok(Some(Value::Int(result)))
1058 }
1059 (Some(lv), Some(rv)) => {
1060 let a = match &lv {
1061 Value::Float(f) => *f,
1062 Value::Int(i) => *i as f64,
1063 _ => {
1064 return Err(GraphError::QueryError {
1065 detail: format!("arithmetic operand must be numeric, got {lv:?}"),
1066 })
1067 }
1068 };
1069 let b = match &rv {
1070 Value::Float(f) => *f,
1071 Value::Int(i) => *i as f64,
1072 _ => {
1073 return Err(GraphError::QueryError {
1074 detail: format!("arithmetic operand must be numeric, got {rv:?}"),
1075 })
1076 }
1077 };
1078 let result = match op {
1079 ArithOp::Sub => a - b,
1080 ArithOp::Mul => a * b,
1081 ArithOp::Add => a + b,
1082 ArithOp::Div => {
1083 if b == 0.0 {
1084 return Err(GraphError::QueryError {
1085 detail: "division by zero".into(),
1086 });
1087 }
1088 a / b
1089 }
1090 };
1091 Ok(Some(Value::Float(result)))
1092 }
1093 }
1094}
1095
1096fn eval_set_return_func<F: Fs>(
1097 db: &GraphDb<F>,
1098 match_rs: &ResultSet,
1099 row: usize,
1100 rel_vars: &[String],
1101 name: &str,
1102 args: &[Operand],
1103 params: &BTreeMap<String, Value>,
1104) -> Result<Option<Value>> {
1105 let norm = name.to_ascii_lowercase();
1106 if norm == "type" {
1107 if args.len() != 1 {
1108 return Err(GraphError::QueryError {
1109 detail: format!("type() requires exactly 1 argument, got {}", args.len()),
1110 });
1111 }
1112 let Operand::Var(rel) = &args[0] else {
1113 return Err(GraphError::QueryError {
1114 detail: "type() argument must be a relationship variable (e.g. type(r))".into(),
1115 });
1116 };
1117 return Ok(match_rs.get(row, &rel_type_alias(rel)).cloned());
1118 }
1119 if norm == "key" || norm == "id" {
1120 let fname = if norm == "id" { "id" } else { "key" };
1121 if args.len() != 1 {
1122 return Err(GraphError::QueryError {
1123 detail: format!("{fname}() requires exactly 1 argument, got {}", args.len()),
1124 });
1125 }
1126 let Operand::Var(var) = &args[0] else {
1127 return Err(GraphError::QueryError {
1128 detail: format!("{fname}() argument must be a node variable (e.g. {fname}(n))"),
1129 });
1130 };
1131 if rel_vars.iter().any(|r| r == var) {
1132 return Err(GraphError::QueryError {
1133 detail: format!("{fname}() argument `{var}` is a relationship, not a node"),
1134 });
1135 }
1136 // MATCH rows bind node variables to their key string, so the column
1137 // value *is* the key. `id()` aliases `key()`.
1138 return Ok(match_rs.get(row, var).cloned());
1139 }
1140 let mut vals = Vec::with_capacity(args.len());
1141 for arg in args {
1142 vals.push(eval_set_return_operand(
1143 db, match_rs, row, rel_vars, arg, params,
1144 )?);
1145 }
1146 match norm.as_str() {
1147 "tolower" => {
1148 if vals.len() != 1 {
1149 return Err(GraphError::QueryError {
1150 detail: format!("toLower() requires exactly 1 argument, got {}", vals.len()),
1151 });
1152 }
1153 Ok(vals[0].clone().map(|val| match val {
1154 Value::Str(s) => Value::Str(s.to_ascii_lowercase()),
1155 other => other,
1156 }))
1157 }
1158 "toupper" => {
1159 if vals.len() != 1 {
1160 return Err(GraphError::QueryError {
1161 detail: format!("toUpper() requires exactly 1 argument, got {}", vals.len()),
1162 });
1163 }
1164 Ok(vals[0].clone().map(|val| match val {
1165 Value::Str(s) => Value::Str(s.to_ascii_uppercase()),
1166 other => other,
1167 }))
1168 }
1169 "size" => match vals.first().cloned().flatten() {
1170 None => Ok(None),
1171 Some(Value::Str(s)) => Ok(Some(Value::Int(s.len() as i64))),
1172 Some(Value::List(items)) => Ok(Some(Value::Int(items.len() as i64))),
1173 Some(_) => Ok(None),
1174 },
1175 "coalesce" => Ok(vals.into_iter().flatten().next()),
1176 "abs" => match vals.first().cloned().flatten() {
1177 None => Ok(None),
1178 Some(Value::Int(n)) => Ok(Some(Value::Int(n.saturating_abs()))),
1179 Some(Value::Float(f)) => Ok(Some(Value::Float(f.abs()))),
1180 Some(_) => Ok(None),
1181 },
1182 "round" => match vals.first().cloned().flatten() {
1183 None => Ok(None),
1184 Some(Value::Float(f)) => Ok(Some(Value::Float(f.round()))),
1185 Some(Value::Int(n)) => Ok(Some(Value::Int(n))),
1186 Some(_) => Ok(None),
1187 },
1188 "decay" => {
1189 if vals.len() != 3 {
1190 return Err(GraphError::QueryError {
1191 detail: format!("decay() requires exactly 3 arguments, got {}", vals.len()),
1192 });
1193 }
1194 match (vals[0].clone(), vals[1].clone(), vals[2].clone()) {
1195 (None, _, _) | (_, None, _) | (_, _, None) => Ok(None),
1196 (Some(b), Some(a), Some(h)) => {
1197 let numeric = |v: Value| -> Result<f64> {
1198 match v {
1199 Value::Int(n) => Ok(n as f64),
1200 Value::Float(f) => Ok(f),
1201 other => Err(GraphError::QueryError {
1202 detail: format!(
1203 "decay() requires numeric arguments, got {other:?}"
1204 ),
1205 }),
1206 }
1207 };
1208 let b = numeric(b)?;
1209 let a = numeric(a)?;
1210 let h = numeric(h)?;
1211 if h <= 0.0 {
1212 return Err(GraphError::QueryError {
1213 detail: "decay() requires halflife > 0".into(),
1214 });
1215 }
1216 Ok(Some(Value::Float(b * 0.5f64.powf(a / h))))
1217 }
1218 }
1219 }
1220 _ => Err(GraphError::QueryError {
1221 detail: format!(
1222 "unknown function `{name}`; supported: toLower, toUpper, size, coalesce, type, abs, round, decay, key, id"
1223 ),
1224 }),
1225 }
1226}
1227
1228fn eval_set_return_item<F: Fs>(
1229 db: &GraphDb<F>,
1230 match_rs: &ResultSet,
1231 row: usize,
1232 rel_vars: &[String],
1233 item: &RetItem,
1234 params: &BTreeMap<String, Value>,
1235) -> Result<Option<Value>> {
1236 match &item.value {
1237 RetVal::Var(v) => eval_set_return_operand(
1238 db,
1239 match_rs,
1240 row,
1241 rel_vars,
1242 &Operand::Var(v.clone()),
1243 params,
1244 ),
1245 RetVal::Prop { var, field } => eval_set_return_operand(
1246 db,
1247 match_rs,
1248 row,
1249 rel_vars,
1250 &Operand::Prop {
1251 var: var.clone(),
1252 field: field.clone(),
1253 },
1254 params,
1255 ),
1256 RetVal::FuncCall { name, args } => {
1257 eval_set_return_func(db, match_rs, row, rel_vars, name, args, params)
1258 }
1259 RetVal::ScalarExpr(op) => eval_set_return_operand(db, match_rs, row, rel_vars, op, params),
1260 RetVal::Agg { .. } => Err(GraphError::QueryError {
1261 detail: "aggregates are not supported in MATCH … SET … RETURN".into(),
1262 }),
1263 }
1264}
1265
1266/// Project user RETURN from original MATCH rows after SET. No rematch.
1267fn project_set_return_rows<F: Fs>(
1268 db: &GraphDb<F>,
1269 rel_vars: &[String],
1270 match_rs: &ResultSet,
1271 returns: &[RetItem],
1272 params: &BTreeMap<String, Value>,
1273) -> Result<ResultSet> {
1274 let columns: Vec<String> = returns.iter().map(ret_column_name).collect();
1275 let mut out = ResultSet::new(columns);
1276 for row in 0..match_rs.len() {
1277 let mut cells = Vec::with_capacity(returns.len());
1278 for item in returns {
1279 cells.push(eval_set_return_item(
1280 db, match_rs, row, rel_vars, item, params,
1281 )?);
1282 }
1283 out.push_row(cells);
1284 }
1285 Ok(out)
1286}
1287
1288/// Single construction point for a `GraphMut` view over the split-borrowed graph fields.
1289/// Callers use `std::mem::take` on the engine before calling this, then restore it after.
1290/// Extract a `Vec<f64>` from a `Value::List` whose items are all numeric.
1291/// Returns `None` for non-list values or lists with non-numeric elements.
1292/// Extra candidates pulled from an approximate index before re-scoring, over and
1293/// above the `k` asked for.
1294///
1295/// The index orders candidates by `f32` distances, which agree with the exact
1296/// `f64` cosine to about 1e-6. Re-scoring can therefore only reshuffle
1297/// candidates inside a band that narrow — it cannot move a hit past one that is
1298/// further away by more than 1e-6 — so the only way a true top-`k` member can be
1299/// lost is if the index ranked it just outside `k` on the `f32` order. Fetching
1300/// `k + 16` covers any such band up to 16 members wide, which at 1e-6 means 16
1301/// vectors within a millionth of each other in cosine: a duplicate cluster, and
1302/// then the members are interchangeable anyway. `min` is applied to the exact
1303/// score, never to the index's, so a hit sitting on the threshold is decided
1304/// exactly.
1305const VECTOR_RESCORE_MARGIN: usize = 16;
1306
1307/// Cosine similarity between an already-unit query and node `id`'s `field`
1308/// vector, read from the **`f64`** properties. `None` when the node has no
1309/// numeric-list vector there, or its norm is zero.
1310///
1311/// The single definition of the score this API reports. Both the brute-force
1312/// scan and the re-scoring step that follows an index lookup go through it, so
1313/// the two paths cannot disagree — which is the property
1314/// `index_and_brute_force_agree_on_scores` pins.
1315fn exact_vector_similarity(
1316 view: &GraphView<'_>,
1317 id: u32,
1318 field: &str,
1319 q_unit: &[f64],
1320) -> Option<f64> {
1321 let v = view.prop(id, field)?;
1322 let xs = value_as_float_list(&v.into_value())?;
1323 let v_norm: f64 = xs.iter().map(|x| x * x).sum::<f64>().sqrt();
1324 if v_norm == 0.0 {
1325 return None;
1326 }
1327 Some(
1328 q_unit
1329 .iter()
1330 .zip(xs.iter())
1331 .map(|(a, b)| a * (b / v_norm))
1332 .sum(),
1333 )
1334}
1335
1336fn value_as_float_list(v: &Value) -> Option<Vec<f64>> {
1337 match v {
1338 Value::List(items) => items
1339 .iter()
1340 .map(|item| match item {
1341 Value::Float(f) => Some(*f),
1342 Value::Int(i) => Some(*i as f64),
1343 _ => None,
1344 })
1345 .collect(),
1346 _ => None,
1347 }
1348}
1349
1350fn make_graph_mut<'a>(
1351 ids: &'a IdMap,
1352 syms: &'a mut Interner,
1353 labels: &'a [u32],
1354 props: core_storage::v8::seam::ColumnsView<'a>,
1355 topo: &'a mut Topology,
1356 base: &'a Option<std::sync::Arc<core_storage::v8::MappedBase>>,
1357 edge_props: &'a mut EdgeProps,
1358) -> GraphMut<'a> {
1359 GraphMut {
1360 ids,
1361 syms,
1362 labels,
1363 props,
1364 topo,
1365 base_topo: base_csr(base),
1366 edge_props,
1367 }
1368}
1369
1370/// The archived CSR of an open V8 snapshot, for the rule engine's graph reads.
1371///
1372/// A store opened from a snapshot keeps its edges in the mapping and its
1373/// overlay empty, so a rule that reads the graph's shape has to see both.
1374fn base_csr(
1375 base: &Option<std::sync::Arc<core_storage::v8::MappedBase>>,
1376) -> Option<&core_storage::v8::layout::ArchivedCsr> {
1377 base.as_ref().map(|b| {
1378 b.topology()
1379 .expect("base topology section bounds validated at open")
1380 })
1381}
1382
1383/// Build a `ColumnsView` from the disjoint `props` overlay and optional V8 base.
1384///
1385/// Takes explicit field references rather than `&self` so the caller can hold
1386/// simultaneous mutable borrows of other fields (e.g. `syms`, `topo`).
1387fn build_props_view<'a>(
1388 props: &'a ColumnStore,
1389 base: &'a Option<std::sync::Arc<core_storage::v8::MappedBase>>,
1390) -> core_storage::v8::seam::ColumnsView<'a> {
1391 match base {
1392 None => core_storage::v8::seam::ColumnsView::owned(props),
1393 Some(b) => {
1394 let archived = b
1395 .columns()
1396 .expect("base columns section bounds validated at open");
1397 core_storage::v8::seam::ColumnsView::with_base_cached(props, archived, b.mixed_cache())
1398 .with_shared_strings(base_string_table(b))
1399 }
1400 }
1401}
1402
1403/// The base columns section paired with the string table that resolves its
1404/// string ids — what `ViewStore` needs to read a neighbour's string property
1405/// out of a V9 snapshot.
1406fn base_columns(
1407 base: &Option<std::sync::Arc<core_storage::v8::MappedBase>>,
1408) -> Option<core_storage::v8::seam::BaseColumns<'_>> {
1409 base.as_ref().map(|b| core_storage::v8::seam::BaseColumns {
1410 cols: b
1411 .columns()
1412 .expect("base columns section bounds validated at open"),
1413 strings: base_string_table(b),
1414 })
1415}
1416
1417/// The shared string table of a V9 base, or `None` for a pre-V9 one.
1418///
1419/// Every `ColumnsView` built over a base must carry it: without it a V9
1420/// snapshot's string columns, whose own tables are empty, read back as absent.
1421fn base_string_table(
1422 base: &core_storage::v8::MappedBase,
1423) -> Option<&core_storage::v8::layout::ArchivedStringTable> {
1424 base.string_table()
1425 .transpose()
1426 .expect("base strings section bounds validated at open")
1427}
1428
1429fn build_topo_view<'a>(
1430 overlay: &'a Topology,
1431 base: &'a Option<std::sync::Arc<core_storage::v8::MappedBase>>,
1432) -> core_storage::v8::seam::TopologyView<'a> {
1433 match base {
1434 None => core_storage::v8::seam::TopologyView::owned(overlay),
1435 Some(b) => {
1436 let archived_csr = b
1437 .topology()
1438 .expect("base topology section bounds validated at open");
1439 core_storage::v8::seam::TopologyView::with_base(overlay, archived_csr)
1440 }
1441 }
1442}
1443
1444/// When [`GraphDb`] calls `Fs::sync` after a WAL append.
1445///
1446/// Default is [`Strict`](FsyncPolicy::Strict): every `log_then_apply_with`
1447/// fsyncs (single `insert_node` / `set_prop`). Ingest and `write_batch`
1448/// emit one `WalRecord::Batch` and fsync once at that frame (Batched).
1449/// [`Relaxed`](FsyncPolicy::Relaxed) skips WAL sync; [`GraphDb::snapshot`]
1450/// is still durable via `write_atomic`. Crash-recovery DST stays Strict.
1451#[derive(Clone, Copy, Debug, PartialEq, Eq, Default)]
1452pub enum FsyncPolicy {
1453 /// Every WAL commit calls `fs.sync` (today's behavior).
1454 #[default]
1455 Strict,
1456 /// Sync only at a `Batch` frame end. Single-op path stays Strict unless
1457 /// this policy is set on the database.
1458 Batched,
1459 /// Never call `fs.sync`. [`GraphDb::snapshot`] still syncs via `write_atomic`.
1460 Relaxed,
1461}
1462
1463/// A precondition for a compare-and-set batch write.
1464///
1465/// All preconditions in a [`GraphDb::write_batch_cas`] or
1466/// [`crate::SharedDb::submit_batch_cas`] call are checked atomically before
1467/// any operation in the batch is applied. If any precondition fails, the
1468/// entire batch is rejected with [`GraphError::CasConflict`] and no WAL frame
1469/// is written.
1470///
1471/// # Touch definition
1472///
1473/// A node's last-change commit (`last_changed`) is updated when any of the
1474/// following state-changing WAL records touch it:
1475///
1476/// - `InsertNode` / `InsertNodeId` — the newly-inserted node.
1477/// - `SetProp` / `SetPropId` / `RemoveProp` — the property-bearing node.
1478/// - `InsertEdge` / `InsertEdgeId` / `DeleteEdge` — **both** src and dst
1479/// endpoints (an edge change touches both sides).
1480/// - `DeleteNode` — the node is tombstoned; `last_changed` returns `None`
1481/// for deleted keys so the pre-deletion entry is never observed.
1482///
1483/// History markers (`DerivedEdgeAdded` / `DerivedEdgeRetracted`) are
1484/// state no-ops. The underlying mutation that triggered rule firing already
1485/// updated the relevant nodes' last-change entries. Rule-management records
1486/// (`CreateRule`, `DeleteRule`, `RebuildRule`) and view/full-text declarations
1487/// do not touch any node's last-change.
1488#[derive(Debug, Clone, PartialEq, Eq)]
1489pub enum Precondition {
1490 /// The node's last-change commit must equal `expected`.
1491 ///
1492 /// Fails with [`GraphError::CasConflict`] when:
1493 /// - The node does not exist (`last_changed` returns `None`), or
1494 /// - The recorded commit seq does not match `expected`.
1495 NodeUnchangedSince { key: String, expected: u64 },
1496 /// The node must not exist (not inserted, or already deleted).
1497 ///
1498 /// Fails with [`GraphError::CasConflict`] (expected=`u64::MAX`,
1499 /// actual=`last_changed(key).unwrap_or(0)`) when the node is live.
1500 NodeAbsent { key: String },
1501}
1502
1503pub struct GraphDb<F: Fs> {
1504 fs: F,
1505 ids: IdMap,
1506 syms: Interner,
1507 topo: Topology,
1508 props: ColumnStore,
1509 labels: Vec<u32>, // node id -> label symbol
1510 /// Namespace names by index; index [`NS_DEFAULT_IDX`] is always
1511 /// [`NS_DEFAULT`]. Derived beside [`Self::node_ns`], never persisted.
1512 ///
1513 /// A private table rather than the shared [`Interner`]: interning
1514 /// `"default"` at open would add a symbol to the store's symbol table and
1515 /// change the bytes of the next snapshot of a store that has no namespaces
1516 /// at all.
1517 ns_names: Vec<String>,
1518 /// Namespace index per dense node id, into [`Self::ns_names`];
1519 /// [`NS_DEFAULT_IDX`] for a node with no `ns` property.
1520 ///
1521 /// Derived: built by one pass over the `ns` column at open (which reads
1522 /// nothing when the column does not exist) and maintained at every node
1523 /// insert. Never written to a snapshot or the WAL, because the property it
1524 /// mirrors already is. A namespace cannot change, so no other record shape
1525 /// can move a node between namespaces.
1526 node_ns: Vec<u32>,
1527 edge_props: EdgeProps,
1528 engine: RuleEngine,
1529 view_store: ViewStore,
1530 /// Incremental inverted index for full-text-lite search.
1531 /// Rebuild-on-open: populated from WAL replay + rebuild_all at open end.
1532 fulltext: FulltextIndex,
1533 /// Opt-in equality index over scalar node properties.
1534 /// Rebuild-on-open: declarations replay from the WAL, postings rebuild at
1535 /// open end (mirrors `fulltext`).
1536 prop_index: PropertyIndex,
1537 /// Whether this store records insert-count multiplicity (§5.13).
1538 ///
1539 /// Declared like `prop_index`'s enabled pairs — a WAL record replayed at
1540 /// open, re-emitted into the baseline by a truncating snapshot — but it
1541 /// gates a *format* step rather than an index: `WalRecord::SetEdgeCount`
1542 /// (discriminant 23) is written only when this is `true`, so a store that
1543 /// never opts in stays readable by a binary that predates the record.
1544 multiplicity: bool,
1545 event_sink: Option<Box<dyn Fn(MutationEvent) + Send + Sync>>,
1546 /// WAL fsync cadence. Default [`FsyncPolicy::Strict`].
1547 fsync: FsyncPolicy,
1548 /// Monotonically increasing per-commit counter. A single `log_then_apply_with`
1549 /// call increments this once; all events emitted from that call share the same
1550 /// `commit_seq` value.
1551 commit_seq: u64,
1552 /// RBAC role definitions loaded from `roles.json` at open.
1553 ///
1554 /// `Some(roles)` — loaded successfully (may be empty when no roles are defined).
1555 /// `None` — `roles.json` was present but corrupt; `mask_for_role` returns
1556 /// `Err` for any request (fail-loud, never silently grant empty visibility).
1557 roles: Option<Vec<RoleDef>>,
1558 /// Memo for [`mask_for_role`](GraphDb::mask_for_role), keyed by
1559 /// `(role, commit_seq)` — a scoped reader between two writes resolves once.
1560 ///
1561 /// Shared by `Arc` with every [`ReaderSnapshot`](crate::reader::ReaderSnapshot)
1562 /// taken from this handle. Replaced (not cleared) whenever the role
1563 /// definitions change or the store is reloaded, which `commit_seq` does not
1564 /// record; see [`RoleMaskCache`](crate::mask::RoleMaskCache).
1565 role_masks: Arc<crate::mask::RoleMaskCache>,
1566 /// Which loaded store this handle is, for memos that outlive it.
1567 ///
1568 /// `role_masks` needs no such thing — the handle owns it and replaces it —
1569 /// but a [`Scope`](crate::mask::Scope) is the caller's, so its resolved key
1570 /// leg is stamped with this alongside `commit_seq`. Minted fresh here and
1571 /// again in [`reset_for_reload`](GraphDb::reset_for_reload), at exactly the
1572 /// two points a fresh `RoleMaskCache` is installed; see
1573 /// [`StoreStamp`](crate::mask::StoreStamp) for the invariant.
1574 store_id: crate::mask::StoreId,
1575 /// Live subscriptions. Entries with a dead `Weak` are pruned on the next
1576 /// distribute_events call.
1577 subscriptions: Vec<SubEntry>,
1578 /// Live query subscriptions. Re-executed on every commit when non-empty.
1579 /// Dead `Weak` entries are pruned inside `distribute_events`.
1580 query_subscriptions: Vec<QuerySubEntry>,
1581 /// Queue capacity for new subscriptions created by this db. Default is
1582 /// [`DEFAULT_SUB_CAPACITY`]; can be overridden via [`set_sub_capacity`]
1583 /// to test Lagged behaviour with small queues.
1584 sub_capacity: usize,
1585 /// True for as-of instances opened via [`GraphDb::open_at`].
1586 /// Every mutation method and `snapshot()` returns [`GraphError::ReadOnly`]
1587 /// when this flag is set.
1588 read_only: bool,
1589 /// Total WAL commit count at the time [`open_at`] was called.
1590 /// 0 for normal (non-as-of) instances.
1591 total_wal_commits: u64,
1592 /// Immutable mmap-backed base snapshot (V8). When `Some`, `self.topo` is
1593 /// the WAL-replay overlay (empty at open time, populated by apply()) and
1594 /// reads go through a merged `TopologyView`. `self.props` is always
1595 /// fully materialized (base + WAL replay) for HNSW/IVF and view compat.
1596 base: Option<Arc<core_storage::v8::MappedBase>>,
1597 // ── MVCC epoch reader state ───────────────────────────────────────────────
1598 /// Most-recent full overlay clone. Initialized at end of `open_with` /
1599 /// `open_at_with`; refreshed every `FOLD_EVERY_K` commits.
1600 /// `None` only between struct creation and the first fold.
1601 fold_overlay: Option<Arc<crate::reader::FrozenOverlay>>,
1602 /// Per-commit deltas accumulated since the last fold.
1603 delta_tail: Vec<Arc<crate::reader::CommitDelta>>,
1604 /// How many commits have occurred since the last fold.
1605 commits_since_fold: usize,
1606 /// When true, `log_then_apply_with` buffers event notifications instead of
1607 /// firing them immediately. Used by the group-commit drain thread to defer
1608 /// events until after the group fsync (R2: durability before notification).
1609 /// Cleared to false once the drain thread flushes or discards the buffer.
1610 defer_events: bool,
1611 /// Buffered events accumulated while `defer_events` is true.
1612 deferred_events: Vec<DeferredEvent>,
1613 /// Set to true by the group-commit drain thread when a group fsync fails
1614 /// after WAL truncation. All subsequent mutation attempts return an IO
1615 /// error until the database is reopened.
1616 degraded: bool,
1617 /// Set to `true` after `ensure_v8_base_sections_loaded` has read provenance,
1618 /// HNSW, and IVF sections from the mmap base into the engine's retained
1619 /// fields. `false` on all opens until first use; always `true` for non-V8
1620 /// opens (base is None, fast-path sets flag immediately).
1621 v8_sections_loaded: std::sync::atomic::AtomicBool,
1622 /// Serializes the one-time section population in `ensure_v8_base_sections_loaded`.
1623 v8_sections_mutex: std::sync::Mutex<()>,
1624 /// Per-node last-change commit sequence. `last_change[node_id] = seq` means
1625 /// the node was last modified by commit `seq`.
1626 ///
1627 /// Loaded from V8 section 11 at open; updated on every state-changing commit
1628 /// and WAL replay frame. V5-V7 stores start with an empty map; pre-WAL-horizon
1629 /// nodes return `None` from `last_changed` until they are next mutated.
1630 ///
1631 /// See [`Precondition`] for the full touch definition.
1632 last_change: HashMap<u32, u64>,
1633 /// WAL archive retention policy set by [`set_wal_archive_retention`].
1634 /// `None` = unlimited (keep all archives); `Some(N)` = keep N newest archives,
1635 /// pruning older ones at snapshot time. 0 is treated as unlimited.
1636 wal_archive_retention: Option<u32>,
1637 /// Global frame index of the first commit that is still reachable through
1638 /// surviving archives. Persisted to `wal.floor` sidecar when pruning occurs.
1639 /// Default 0 = all history reachable.
1640 wal_horizon_floor: u64,
1641 /// True when the surviving archive chain forms a continuous WAL history
1642 /// starting from the store's first commit (the genesis chain).
1643 ///
1644 /// `open_at` may replay archive-resident commits from empty state only when
1645 /// this flag is true AND `wal_horizon_floor == 0`. Cleared whenever:
1646 /// - a WAL-truncating snapshot (`keep_wal=false`) is taken after archives
1647 /// already exist (breaks the chain for subsequent archives), or
1648 /// - any archive is pruned (floor advances past zero).
1649 ///
1650 /// Persisted via the `wal.genesis` marker file; loaded from it at open.
1651 archive_genesis_chain: bool,
1652 /// True when this handle can *prove* the live WAL has never been truncated:
1653 /// there was no `snapshot.bin` when it opened the store, and it has taken no
1654 /// truncating snapshot since.
1655 ///
1656 /// The archive path's genesis check asks "did a snapshot exist before this
1657 /// one?" as a proxy for "was the WAL ever truncated". The proxy is sound
1658 /// across sessions — this binary cannot tell a history-preserving snapshot
1659 /// from a truncating one once the handle that took it is gone — but inside
1660 /// one session it is not, and `enable_multiplicity` made that visible: its
1661 /// forced `keep_wal` snapshot left the WAL entirely intact and yet
1662 /// permanently disqualified the store from ever receiving a genesis marker
1663 /// (defect #23). This flag is what the proxy defers to when the answer is
1664 /// actually known.
1665 snapshot_preserved_history: bool,
1666 /// Transient write-authz context set by `write_batch_authz` /
1667 /// `query_write_authz` for the duration of ONE mutation call.
1668 /// Always `None` at rest. Never serialized, never WAL-replayed.
1669 pending_write_authz: Option<WriteAuthz>,
1670 /// Slow-query threshold in milliseconds. 0 = disabled.
1671 /// Seeded from `MUSHROOMDB_SLOW_QUERY_MS` at open; override via
1672 /// [`GraphDb::set_slow_query_threshold_ms`] (tests must use the setter
1673 /// — env vars are process-global and race parallel test threads).
1674 slow_query_threshold_ms: u64,
1675 /// Ring buffer of recent slow queries (interior-mutable so `query(&self)`
1676 /// can record entries without requiring `&mut self`).
1677 slow_queries: std::sync::Mutex<SlowQueryLog>,
1678 /// `(field, label, caller)` triples whose exact-versus-approximate
1679 /// ambiguity this handle has already explained once. See
1680 /// [`note_ambiguous_exactness`](GraphDb::note_ambiguous_exactness).
1681 /// The caller shape is part of the key because the two shapes give
1682 /// different advice — silencing one with the other would leave a caller
1683 /// reading advice meant for a signature it does not have.
1684 /// Advice bookkeeping, not graph state: a reload keeps it, as the
1685 /// slow-query log does.
1686 warned_ambiguous_exactness: std::sync::Mutex<HashSet<(String, String, ExactnessCaller)>>,
1687 /// Instant at which the database was opened (used by `/metrics` uptime).
1688 started_at: std::time::Instant,
1689 // ── Multi-process state (cross-process lock + WAL tailing) ────────────────
1690 /// Byte offset of the WAL prefix already applied to in-memory state.
1691 ///
1692 /// Advanced by exactly the encoded length of every frame this handle
1693 /// appends, and by the decoded byte count of every tail
1694 /// [`refresh`](GraphDb::refresh) absorbs. Rewound by
1695 /// [`set_wal_consumed`](GraphDb::set_wal_consumed) when the group-commit
1696 /// drain thread truncates a failed group. Compared against the WAL's
1697 /// on-disk length to decide staleness.
1698 wal_consumed: u64,
1699 /// Identity of the snapshot this handle's base state came from, as
1700 /// `(len, mtime_nanos)`. A different value means another process replaced
1701 /// the snapshot and the WAL no longer continues our state: refresh reloads.
1702 snapshot_ident: Option<(u64, u64)>,
1703 /// The options this handle was opened with. Replayed verbatim when
1704 /// `refresh` has to rebuild from disk.
1705 open_opts: OpenOptions,
1706 /// True when this handle holds the cross-process write lock for its whole
1707 /// lifetime (a plain read-write open). Per-write lock acquisition is a
1708 /// no-op on such a handle, and never releases the lock.
1709 holds_lifetime_lock: bool,
1710 /// True between a failed lock acquisition and the end of the write scope
1711 /// that failed. Makes every WAL-appending mutation in that scope return
1712 /// [`GraphError::Busy`] instead of writing.
1713 lock_denied: bool,
1714 /// True for an as-of view opened via [`GraphDb::open_at`]. Such a view is
1715 /// pinned to one commit, so it is never stale and never refreshes — later
1716 /// commits by any process are deliberately invisible to it.
1717 pinned: bool,
1718}
1719
1720/// One group of deferred event notifications, held until the group fsync
1721/// completes. Replayed by [`GraphDb::flush_deferred_events`].
1722struct DeferredEvent {
1723 rec: core_storage::WalRecord,
1724 engine_deltas: Vec<EngineEdgeDelta>,
1725 seq: u64,
1726 ingest: Option<(String, usize)>,
1727}
1728
1729/// Options for [`GraphDb::open_with_options`].
1730#[derive(Clone, Copy, Debug)]
1731pub struct OpenOptions {
1732 /// Rewrite an old-format snapshot to the current VERSION after a
1733 /// successful load (default `true`). The old snapshot is kept as
1734 /// `snapshot.bin.bak` until the next clean open at the current version,
1735 /// at which point the `.bak` is deleted.
1736 ///
1737 /// Set to `false` to open a store without touching any on-disk files
1738 /// (useful for read-only inspection of a store at an older format).
1739 pub auto_migrate: bool,
1740
1741 /// Write the valid WAL prefix back over a torn tail on open (default
1742 /// `true`). Truncating a genuinely torn tail is correct crash recovery.
1743 ///
1744 /// Set to `false` for an unattended reader. The valid prefix is still
1745 /// decoded and replayed in memory, but nothing is written: a reader that
1746 /// opens while another process is mid-append would otherwise discard a
1747 /// frame that writer believes durable. `mushroomdb recall`, which runs on
1748 /// every prompt, passes `false` for exactly this reason.
1749 pub repair_wal: bool,
1750
1751 /// Open without ever writing to the store (default `false`).
1752 ///
1753 /// A read-only handle:
1754 /// - returns [`GraphError::ReadOnly`] from every mutation and from
1755 /// `snapshot()`;
1756 /// - performs no disk write at open — no WAL repair write-back and no
1757 /// auto-migration rewrite, whatever the other two flags say;
1758 /// - never takes the cross-process write lock, so it opens immediately even
1759 /// while another process is writing, and never makes a writer wait.
1760 ///
1761 /// [`refresh`](GraphDb::refresh) and [`is_stale`](GraphDb::is_stale) work
1762 /// normally, so a read-only handle can follow another process's commits.
1763 pub read_only: bool,
1764}
1765
1766impl Default for OpenOptions {
1767 fn default() -> Self {
1768 Self {
1769 auto_migrate: true,
1770 repair_wal: true,
1771 read_only: false,
1772 }
1773 }
1774}
1775
1776/// How long a writer polls for the cross-process write lock before giving up
1777/// with [`GraphError::Busy`].
1778///
1779/// Long enough to ride out another process's commit (a batch apply plus one
1780/// fsync), short enough that a stuck peer surfaces as an error rather than a
1781/// hang.
1782pub const WRITE_LOCK_WAIT: std::time::Duration = std::time::Duration::from_secs(2);
1783
1784/// Refusal when a `MERGE` create cannot choose a namespace.
1785///
1786/// A role bound to two or more namespaces cannot have its create arm land in
1787/// `default`, and the statement did not name `ns`. The role must name one.
1788pub const MERGE_CREATE_NEEDS_ONE_NAMESPACE: &str =
1789 "role-bound token: MERGE create requires the role to name one namespace";
1790
1791/// Interval between poll attempts while waiting for the cross-process lock.
1792pub(crate) const LOCK_POLL_INTERVAL: std::time::Duration = std::time::Duration::from_millis(10);
1793
1794/// Why `load_from_disk` is running, which decides whether it may repair.
1795#[derive(Clone, Copy, Debug, PartialEq, Eq)]
1796enum LoadOrigin {
1797 /// A fresh open. Crash recovery is this handle's job: a torn WAL tail is
1798 /// the signature of a crash and truncating it is correct, and archives
1799 /// orphaned by an interrupted prune can be swept.
1800 Open,
1801 /// A reload driven by [`GraphDb::refresh`], because another process
1802 /// replaced the snapshot. Nothing here is crash recovery — the store is
1803 /// live and someone else is writing it — so this origin writes nothing.
1804 Reload,
1805}
1806
1807/// Authorization context carried by `write_batch_authz` / `query_write_authz`.
1808///
1809/// `None` at the call site = full authority (today's zero-cost behavior).
1810/// `Some(WriteAuthz)` = role-scoped: the decision table (plan §"authz decision
1811/// table") is evaluated per-op inside `commit_logged_batch` BEFORE any WAL
1812/// record is built. A denial returns an error with no WAL frame written.
1813///
1814/// The mask is ALWAYS `Omit`-mode: role-token paths must never acknowledge
1815/// hidden-node existence to callers.
1816#[derive(Clone, Debug)]
1817pub struct WriteAuthz {
1818 pub role: String,
1819 pub scope: WriteScope,
1820 /// Resolved by `mask_for_role` under the same write guard as the mutation.
1821 /// Always `Omit`-mode — never `Stub`.
1822 pub mask: crate::mask::NodeMask,
1823}
1824
1825/// The error every role surface gives when `roles.json` did not parse at open.
1826///
1827/// One text, so `mask_for_role` and [`GraphDb::roles_checked`] cannot drift
1828/// apart on the same cause.
1829fn roles_poisoned() -> GraphError {
1830 GraphError::Corrupt {
1831 detail: "roles.json was corrupt at open; fix the file and re-open to restore role access"
1832 .into(),
1833 }
1834}
1835
1836/// Write `bytes` to `snapshot.bin.bak` atomically with full fsync.
1837///
1838/// Uses [`RealFs::write_atomic`] which applies `F_FULLFSYNC` on macOS and
1839/// `sync_all` on other platforms, then renames the `.tmp` file into place and
1840/// syncs the directory entry. This is the only correct path for writing the
1841/// `.bak` — plain `std::fs::write + sync_all` misses both `F_FULLFSYNC` and
1842/// the directory sync.
1843pub fn write_snapshot_bak(dir: &std::path::Path, bytes: &[u8]) -> crate::Result<()> {
1844 use core_storage::fs::{FileId, Fs as _};
1845 RealFs::new(dir)
1846 .map_err(core_storage::GraphError::Io)?
1847 .write_atomic(FileId::SnapshotBak, bytes)
1848 .map_err(core_storage::GraphError::Io)
1849}
1850
1851/// Return the on-disk snapshot format version without decoding the full snapshot.
1852///
1853/// Reads only the 6-byte header (magic + version LE). Returns `None` when no
1854/// snapshot file exists (WAL-only store). Returns an error if the header is
1855/// malformed.
1856pub fn snapshot_version_at(dir: &std::path::Path) -> crate::Result<Option<u16>> {
1857 use std::io::Read as _;
1858 let path = dir.join("snapshot.bin");
1859 let mut header = [0u8; 6];
1860 let n = match std::fs::File::open(&path) {
1861 Ok(mut f) => f.read(&mut header).map_err(core_storage::GraphError::Io)?,
1862 Err(e) if e.kind() == std::io::ErrorKind::NotFound => return Ok(None),
1863 Err(e) => return Err(core_storage::GraphError::Io(e)),
1864 };
1865 core_storage::snapshot::peek_version(&header[..n])
1866}
1867
1868/// Options for [`GraphDb::snapshot_with`].
1869#[derive(Debug, Clone, Default)]
1870pub struct SnapshotOptions {
1871 /// When `true`, the WAL is preserved after the snapshot write.
1872 /// Pre-snapshot commits remain reachable via [`GraphDb::open_at`].
1873 /// When `false` (the default), the WAL is truncated to a minimal
1874 /// baseline so cold-start replay stays fast.
1875 pub keep_wal: bool,
1876 /// When `true`, the current WAL is renamed to `wal.<commit_seq>.archive`
1877 /// before a fresh WAL baseline is written (history-preserving snapshot).
1878 ///
1879 /// This is the feature opt-in: `false` (the default) leaves the existing
1880 /// truncation / keep-wal behaviour byte-identical. `archive_wal` takes
1881 /// precedence over `keep_wal` when both are set.
1882 ///
1883 /// Archives can be scanned by [`GraphDb::node_history`],
1884 /// [`GraphDb::edge_history`], [`GraphDb::was_linked`], and
1885 /// [`GraphDb::open_at`], extending the reachable history horizon across
1886 /// snapshot boundaries.
1887 pub archive_wal: bool,
1888}
1889
1890/// Derive the scan-label sym for the commit-skip fast-path.
1891///
1892/// Walks `ops` to find the plan's leading scan op (`ScanLabel`, `IndexScan`,
1893/// or `IndexIntersect`) with a concrete label string, then interns it.
1894///
1895/// Returns `None` in all cases where skipping is unsafe:
1896/// - Any `Expand` op is present (edge traversal; edges change results regardless
1897/// of node labels).
1898/// - The leading scan has no label (`ScanLabel { label: None }` — full scan).
1899/// - No recognizable leading scan op is found.
1900///
1901/// This is the conservative v0.4.3 boundary. The caller stores the result in
1902/// [`QuerySubEntry::scan_label`] at subscribe time; `None` means always execute.
1903fn extract_scan_label(ops: &[PlanOp], syms: &mut Interner) -> Option<u32> {
1904 // Any Expand → must always re-execute (edges can change join results).
1905 if ops.iter().any(|op| matches!(op, PlanOp::Expand { .. })) {
1906 return None;
1907 }
1908 for op in ops {
1909 match op {
1910 PlanOp::ScanLabel {
1911 label: Some(label), ..
1912 } => return Some(syms.intern(label)),
1913 PlanOp::IndexScan {
1914 label: Some(label), ..
1915 } => return Some(syms.intern(label)),
1916 PlanOp::IndexIntersect {
1917 label: Some(label), ..
1918 } => return Some(syms.intern(label)),
1919 _ => {}
1920 }
1921 }
1922 None
1923}
1924
1925/// How an as-of read is restricted — the argument to
1926/// [`GraphDb::query_at_scoped`].
1927///
1928/// Every variant is resolved against the graph **as it was at the requested
1929/// commit**, not against the current graph.
1930#[derive(Debug, Clone, Copy)]
1931pub enum AsOfScope<'a> {
1932 /// Everything the named role may see. The role *definition* is the current
1933 /// one — `roles.json` is a sidecar and has no past version — but its
1934 /// `keys` and `labels` are resolved against the as-of graph.
1935 Role(&'a str),
1936 /// An explicit node-key allow-list. Keys that did not exist at that commit
1937 /// resolve to nothing.
1938 Keys(&'a [String]),
1939 /// A role intersected with a client-supplied allow-list. The intersection
1940 /// is the never-widen rule: a client mask can only narrow a role.
1941 RoleAndKeys(&'a str, &'a [String]),
1942 /// Every live node in one namespace, as the graph was at that commit.
1943 ///
1944 /// A namespace cannot change — it is set at insert and immutable — so the
1945 /// answer is simply "the nodes that existed then and are in this
1946 /// namespace". A name no node uses resolves to nothing, never to
1947 /// everything.
1948 Namespace(&'a str),
1949}
1950
1951impl GraphDb<RealFs> {
1952 /// Open the database at `dir` with default options.
1953 ///
1954 /// Equivalent to `open_with_options(dir, OpenOptions::default())`.
1955 /// Old-format snapshots (V5, V6) are automatically migrated to the
1956 /// current version on a successful load (see [`OpenOptions::auto_migrate`]).
1957 pub fn open(dir: &std::path::Path) -> Result<Self> {
1958 Self::open_with_options(dir, OpenOptions::default())
1959 }
1960
1961 /// Open the database at `dir` with explicit options.
1962 ///
1963 /// When `opts.auto_migrate` is `true` (the default) and the on-disk
1964 /// snapshot is an older format version, this function:
1965 /// 1. Copies the current `snapshot.bin` to `snapshot.bin.bak` (atomic
1966 /// + fsynced) before any modification.
1967 /// 2. Rewrites `snapshot.bin` at the current format version via
1968 /// [`GraphDb::snapshot_with`] with `keep_wal: true` (WAL preserved).
1969 ///
1970 /// If migration fails the error is returned and the original files are
1971 /// intact (the `.bak` was written before the new snapshot was attempted).
1972 ///
1973 /// A clean open that finds the snapshot already at the current version
1974 /// deletes any leftover `.bak` file.
1975 ///
1976 /// WAL-only stores (no snapshot) are never auto-migrated on open.
1977 ///
1978 /// `opts.repair_wal` controls the other write this function can make; see
1979 /// [`OpenOptions::repair_wal`]. With both flags `false` the open touches
1980 /// no file on disk.
1981 pub fn open_with_options(dir: &std::path::Path, opts: OpenOptions) -> Result<Self> {
1982 Self::open_dir(dir, opts, true)
1983 }
1984
1985 /// Open without taking the cross-process write lock for the handle's
1986 /// lifetime.
1987 ///
1988 /// Only [`SharedDb`](crate::SharedDb) uses this: a long-lived server holds
1989 /// its handle open indefinitely, so it takes the lock per write instead of
1990 /// keeping every other process out of the store for as long as it runs.
1991 pub(crate) fn open_unlocked(dir: &std::path::Path) -> Result<Self> {
1992 Self::open_dir(dir, OpenOptions::default(), false)
1993 }
1994
1995 fn open_dir(dir: &std::path::Path, opts: OpenOptions, hold_lock: bool) -> Result<Self> {
1996 // Header-only peek — 6 bytes, no full decode.
1997 let snap_version = snapshot_version_at(dir)?;
1998
1999 // Full load: decode snapshot + replay WAL + rebuild indexes.
2000 let mut db = Self::open_generic(RealFs::new(dir)?, opts, hold_lock)?;
2001
2002 // A read-only handle writes nothing at open, so it never migrates —
2003 // the old-format snapshot is loaded and left exactly as it is.
2004 if opts.auto_migrate && !opts.read_only {
2005 match snap_version {
2006 Some(ver) if ver < core_storage::snapshot::VERSION => {
2007 let _tm = std::time::Instant::now();
2008 // Copy the original snapshot to .bak at OS level — no in-memory
2009 // buffer required for a 2+ GiB file.
2010 //
2011 // Crash-safety: snapshot.bin remains intact (write_atomic inside
2012 // snapshot_with uses a .tmp+rename) until the V8 write succeeds.
2013 // A torn .bak on crash is acceptable because the original
2014 // snapshot.bin is the authoritative source until after the rename.
2015 std::fs::copy(dir.join("snapshot.bin"), dir.join("snapshot.bin.bak"))
2016 .map_err(core_storage::GraphError::Io)?;
2017 trace_migrate!("bak copy done", _tm);
2018 // Rewrite snapshot at current version; keep WAL intact.
2019 db.snapshot_with(SnapshotOptions {
2020 keep_wal: true,
2021 ..SnapshotOptions::default()
2022 })?;
2023 trace_migrate!("snapshot_with done", _tm);
2024 }
2025 Some(_) => {
2026 // Already current version: remove any leftover .bak.
2027 let bak = dir.join("snapshot.bin.bak");
2028 if bak.exists() {
2029 std::fs::remove_file(&bak).map_err(core_storage::GraphError::Io)?;
2030 }
2031 }
2032 None => {
2033 // WAL-only store — nothing to migrate on open.
2034 }
2035 }
2036 }
2037
2038 Ok(db)
2039 }
2040
2041 /// Open a read-only view of the database as it existed after `commit`.
2042 ///
2043 /// Commit indices are 0-based over the current WAL: commit 0 is the state
2044 /// after the first WAL frame, commit N-1 is the state after the N-th (most
2045 /// recent) frame. Call [`GraphDb::open`] to read the full current state.
2046 ///
2047 /// **Replay base.** [`GraphDb::snapshot`] truncates the WAL when it runs,
2048 /// so as-of can only reach commits recorded in the current WAL (those
2049 /// written after the most recent snapshot, or all commits if no snapshot
2050 /// was ever taken). Commit 0 in `open_at` always refers to the first
2051 /// frame in the WAL that exists on disk, not the first ever write to the
2052 /// database. When the on-disk snapshot recorded that it truncated the
2053 /// WAL (V7, default `keep_wal: false`), it is loaded as the base state
2054 /// before frame replay, so the as-of view includes all pre-snapshot data.
2055 /// Snapshots written with `keep_wal: true` (and legacy V5/V6 snapshots)
2056 /// are ignored and replay is WAL-only, as before.
2057 ///
2058 /// **Read-only.** Every mutation method and `snapshot()` on the returned
2059 /// instance returns [`GraphError::ReadOnly`]. Queries, `explain()`, and
2060 /// `stats()` work normally.
2061 ///
2062 /// # Errors
2063 /// - [`GraphError::CommitOutOfRange`] if `commit >= wal_commit_count` (including
2064 /// when the WAL is empty after a snapshot).
2065 pub fn open_at(dir: &std::path::Path, commit: u64) -> Result<Self> {
2066 Self::open_at_with(RealFs::new(dir)?, commit)
2067 }
2068
2069 /// Run a **read-only** Cypher query against the graph as it existed at
2070 /// `commit` — the "time-travel" / agent-replay query. Opens a temporal view
2071 /// of this store's directory at that commit and executes the read there.
2072 ///
2073 /// The current instance is unaffected. Write statements are rejected (the
2074 /// temporal view is read-only). `commit` is a 0-based WAL commit index;
2075 /// `commit == wal_commit_count` (or `open_at`'s range) yields the newest
2076 /// state. Prefer this over holding many historical instances open.
2077 ///
2078 /// # Errors
2079 /// - [`GraphError::CommitOutOfRange`] if `commit` is past the WAL horizon.
2080 /// - A query error for a malformed or write query.
2081 pub fn query_at(
2082 &self,
2083 commit: u64,
2084 cypher: &str,
2085 params: &std::collections::BTreeMap<String, Value>,
2086 ) -> Result<ResultSet> {
2087 let temporal = self.open_at_for_read(commit, cypher)?;
2088 temporal.query(cypher, params)
2089 }
2090
2091 /// Run a **read-only** Cypher query at `commit`, restricted by `scope`.
2092 ///
2093 /// The **graph** is as of `commit`; the **role definition** is as it is
2094 /// now, because `roles.json` is a sidecar and is never a WAL record — it
2095 /// has no past version to read. A role's `keys` and `labels` are resolved
2096 /// against the commit-`commit` graph, so a role that may see a label sees
2097 /// exactly the nodes that carried it then, and an explicit key that did
2098 /// not exist yet resolves to nothing.
2099 ///
2100 /// [`AsOfScope::RoleAndKeys`] intersects the two: a client allow-list can
2101 /// only narrow what a role may see, never widen it.
2102 ///
2103 /// Write statements are rejected, exactly as [`GraphDb::query_at`] rejects
2104 /// them.
2105 ///
2106 /// # Errors
2107 /// - [`GraphError::CommitOutOfRange`] if `commit` is outside the retained
2108 /// range; the error carries that range.
2109 /// - [`GraphError::KeyNotFound`] with a `role:` prefix for an unknown role,
2110 /// or [`GraphError::Corrupt`] when `roles.json` was corrupt at open.
2111 /// - A query error for a malformed or write query.
2112 pub fn query_at_scoped(
2113 &self,
2114 commit: u64,
2115 cypher: &str,
2116 params: &std::collections::BTreeMap<String, Value>,
2117 scope: AsOfScope<'_>,
2118 ) -> Result<ResultSet> {
2119 let temporal = self.open_at_for_read(commit, cypher)?;
2120 let mask = temporal.mask_at_scope(scope)?;
2121 temporal.query_masked(cypher, params, &mask)
2122 }
2123
2124 /// As [`GraphDb::query_at_scoped`], with `namespace` intersected into
2125 /// whatever `scope` resolves to.
2126 ///
2127 /// This is what a surface needs when a caller passes `namespace` beside a
2128 /// `role` or a client mask on a time-travel read: [`AsOfScope`] names one
2129 /// restriction, and the namespace is a second one that composes with it
2130 /// rather than replacing it. The intersection is the never-widen rule — a
2131 /// namespace can only narrow what the scope already allows — and both legs
2132 /// are resolved against the graph as it was at `commit`.
2133 ///
2134 /// `AsOfScope::Namespace(ns)` is still the way to ask for a namespace alone.
2135 pub fn query_at_scoped_in_namespace(
2136 &self,
2137 commit: u64,
2138 cypher: &str,
2139 params: &std::collections::BTreeMap<String, Value>,
2140 scope: AsOfScope<'_>,
2141 namespace: &str,
2142 ) -> Result<ResultSet> {
2143 let temporal = self.open_at_for_read(commit, cypher)?;
2144 let mask = temporal
2145 .mask_at_scope(scope)?
2146 .intersect(&temporal.mask_for_namespace(namespace));
2147 temporal.query_masked(cypher, params, &mask)
2148 }
2149
2150 /// Run a **read-only** Cypher query at `commit`, restricted by a
2151 /// [`Scope`](crate::mask::Scope).
2152 ///
2153 /// [`AsOfScope`] names *one* restriction — a role, a key list, a namespace,
2154 /// or a role-and-keys pair. A `Scope` is the general shape a handle carries,
2155 /// and nesting can give it several role or namespace legs at once, so it
2156 /// cannot be spelled as an `AsOfScope`. This is the entry point a scoped
2157 /// handle uses for time travel; `query_at_scoped` stays the way to ask for
2158 /// one named restriction.
2159 ///
2160 /// Both the graph and the scope's key and namespace legs are resolved
2161 /// against `commit`; a role's *definition* is the current one, because
2162 /// `roles.json` is a sidecar with no past version — the same split
2163 /// [`GraphDb::query_at_scoped`] documents.
2164 ///
2165 /// The scope resolves **cold** here: a temporal handle is its own store, so
2166 /// its ids could never be served to a live read, but filling the scope's
2167 /// one-entry key memo from a handle thrown away at the end of this call
2168 /// would evict the live entry for nothing. See
2169 /// [`Scope::resolve_uncached`](crate::mask::Scope::resolve_uncached).
2170 ///
2171 /// # Errors
2172 /// - [`GraphError::CommitOutOfRange`] if `commit` is outside the retained
2173 /// range; the error carries that range.
2174 /// - [`GraphError::KeyNotFound`] with a `role:` prefix for an unknown role,
2175 /// or [`GraphError::Corrupt`] when `roles.json` was corrupt at open.
2176 /// - A query error for a malformed or write query.
2177 pub fn query_at_with_scope(
2178 &self,
2179 commit: u64,
2180 cypher: &str,
2181 params: &std::collections::BTreeMap<String, Value>,
2182 scope: &crate::mask::Scope,
2183 ) -> Result<ResultSet> {
2184 let temporal = self.open_at_for_read(commit, cypher)?;
2185 let mask = scope.resolve_uncached(&temporal)?;
2186 temporal.query_masked(cypher, params, &mask)
2187 }
2188
2189 /// Open the temporal view for a time-travel read and refuse write Cypher.
2190 ///
2191 /// Shared by [`GraphDb::query_at`] and [`GraphDb::query_at_scoped`] so both
2192 /// resolve the commit and reject writes identically.
2193 fn open_at_for_read(&self, commit: u64, cypher: &str) -> Result<Self> {
2194 let dir = self.fs.dir().to_path_buf();
2195 let temporal = Self::open_at(&dir, commit)?;
2196 if is_write_tokens(&lex(cypher).map_err(|e| GraphError::QueryError {
2197 detail: format!("lex: {e}"),
2198 })?) {
2199 return Err(GraphError::QueryError {
2200 detail: "query_at is read-only: write statements are not permitted in a \
2201 time-travel query"
2202 .into(),
2203 });
2204 }
2205 Ok(temporal)
2206 }
2207}
2208
2209impl<F: Fs> GraphDb<F> {
2210 /// Open over an arbitrary [`Fs`], repairing a torn WAL tail as usual.
2211 pub fn open_with(fs: F) -> Result<Self> {
2212 Self::open_with_repair(fs, true)
2213 }
2214
2215 /// As [`GraphDb::open_with`], but `repair_wal: false` decodes the valid WAL
2216 /// prefix without writing the truncation back. See
2217 /// [`OpenOptions::repair_wal`].
2218 pub fn open_with_repair(fs: F, repair_wal: bool) -> Result<Self> {
2219 Self::open_generic(
2220 fs,
2221 OpenOptions {
2222 repair_wal,
2223 ..OpenOptions::default()
2224 },
2225 true,
2226 )
2227 }
2228
2229 /// Shared open path.
2230 ///
2231 /// `hold_lock` requests the cross-process write lock for the whole handle
2232 /// lifetime — the right behaviour for a plain read-write `GraphDb`, whose
2233 /// owner writes through it directly. [`SharedDb`](crate::SharedDb) passes
2234 /// `false` and takes the lock per write instead, so that a long-lived
2235 /// server does not keep every other process out of the store.
2236 ///
2237 /// A read-only open never takes the lock regardless of `hold_lock`.
2238 fn open_generic(fs: F, opts: OpenOptions, hold_lock: bool) -> Result<Self> {
2239 let mut db = Self::new_empty(fs, opts);
2240 db.read_only = opts.read_only;
2241 if hold_lock && !opts.read_only {
2242 if !db.poll_lock(WRITE_LOCK_WAIT)? {
2243 return Err(GraphError::Busy { holder: None });
2244 }
2245 db.holds_lifetime_lock = true;
2246 }
2247 db.load_from_disk(LoadOrigin::Open)?;
2248 Ok(db)
2249 }
2250
2251 /// A handle with no state loaded: every field at its empty value, the
2252 /// filesystem and options in place. Only [`load_from_disk`] makes it
2253 /// usable.
2254 fn new_empty(fs: F, opts: OpenOptions) -> Self {
2255 Self {
2256 fs,
2257 ids: IdMap::new(),
2258 syms: Interner::new(),
2259 topo: Topology::new(),
2260 props: ColumnStore::new(),
2261 labels: Vec::new(),
2262 ns_names: vec![NS_DEFAULT.to_string()],
2263 node_ns: Vec::new(),
2264 edge_props: EdgeProps::new(),
2265 engine: RuleEngine::new(),
2266 view_store: ViewStore::new(),
2267 fulltext: FulltextIndex::new(),
2268 prop_index: PropertyIndex::new(),
2269 multiplicity: false,
2270 event_sink: None,
2271 fsync: FsyncPolicy::Strict,
2272 commit_seq: 0,
2273 roles: Some(vec![]),
2274 role_masks: Arc::new(crate::mask::RoleMaskCache::new()),
2275 store_id: crate::mask::StoreId::next(),
2276 subscriptions: Vec::new(),
2277 query_subscriptions: Vec::new(),
2278 sub_capacity: DEFAULT_SUB_CAPACITY,
2279 read_only: false,
2280 total_wal_commits: 0,
2281 base: None,
2282 fold_overlay: None,
2283 delta_tail: Vec::new(),
2284 commits_since_fold: 0,
2285 defer_events: false,
2286 deferred_events: Vec::new(),
2287 degraded: false,
2288 v8_sections_loaded: std::sync::atomic::AtomicBool::new(false),
2289 v8_sections_mutex: std::sync::Mutex::new(()),
2290 last_change: HashMap::new(),
2291 wal_archive_retention: None,
2292 wal_horizon_floor: 0,
2293 archive_genesis_chain: false,
2294 // Nothing is proven until `load_from_disk` has looked at the store.
2295 snapshot_preserved_history: false,
2296 pending_write_authz: None,
2297 slow_query_threshold_ms: std::env::var("MUSHROOMDB_SLOW_QUERY_MS")
2298 .ok()
2299 .and_then(|v| v.parse().ok())
2300 .unwrap_or(100),
2301 slow_queries: std::sync::Mutex::new(SlowQueryLog {
2302 entries: std::collections::VecDeque::new(),
2303 total: 0,
2304 }),
2305 warned_ambiguous_exactness: std::sync::Mutex::new(HashSet::new()),
2306 started_at: std::time::Instant::now(),
2307 wal_consumed: 0,
2308 snapshot_ident: None,
2309 open_opts: opts,
2310 holds_lifetime_lock: false,
2311 lock_denied: false,
2312 pinned: false,
2313 }
2314 }
2315
2316 /// Return every field describing stored graph state to its empty value,
2317 /// leaving this handle's own identity alone.
2318 ///
2319 /// Preserved on purpose: the filesystem, open options, lock ownership, the
2320 /// event sink and subscriptions, fsync policy, degraded flag, and the
2321 /// slow-query configuration and log. A caller that registered a sink or a
2322 /// subscription keeps it across a reload.
2323 fn reset_for_reload(&mut self) {
2324 self.ids = IdMap::new();
2325 self.syms = Interner::new();
2326 self.topo = Topology::new();
2327 self.props = ColumnStore::new();
2328 self.labels = Vec::new();
2329 self.ns_names = vec![NS_DEFAULT.to_string()];
2330 self.node_ns = Vec::new();
2331 self.edge_props = EdgeProps::new();
2332 self.engine = RuleEngine::new();
2333 self.view_store = ViewStore::new();
2334 self.fulltext = FulltextIndex::new();
2335 self.prop_index = PropertyIndex::new();
2336 // Cleared like every other declaration: a reload replays the store's own
2337 // WAL, and the opt-in comes back from it or not at all.
2338 self.multiplicity = false;
2339 self.commit_seq = 0;
2340 self.roles = Some(vec![]);
2341 // A fresh cache, not a cleared one: any reader snapshot still holding
2342 // the old `Arc` keeps it to itself, so nothing it memoised against the
2343 // pre-reload store can be read back through this handle.
2344 self.role_masks = Arc::new(crate::mask::RoleMaskCache::new());
2345 // The same move for memos this handle does not own. `commit_seq` is
2346 // zeroed just above and reseeded from `max(last_change)`, which a
2347 // delete-only commit leaves where it was — so a reload can land back on
2348 // a sequence a caller's `Scope` already cached a mask at. A new id is
2349 // what makes that entry stop matching.
2350 self.store_id = crate::mask::StoreId::next();
2351 self.total_wal_commits = 0;
2352 self.base = None;
2353 self.fold_overlay = None;
2354 self.delta_tail = Vec::new();
2355 self.commits_since_fold = 0;
2356 self.deferred_events = Vec::new();
2357 self.v8_sections_loaded
2358 .store(false, std::sync::atomic::Ordering::Release);
2359 self.last_change = HashMap::new();
2360 self.wal_horizon_floor = 0;
2361 self.archive_genesis_chain = false;
2362 // Re-derived by `load_from_disk` from the store it is about to read.
2363 self.snapshot_preserved_history = false;
2364 self.pending_write_authz = None;
2365 self.wal_consumed = 0;
2366 self.snapshot_ident = None;
2367 }
2368
2369 /// Load the snapshot base and replay the WAL into an empty handle — the
2370 /// whole of what opening a store does after the struct exists.
2371 ///
2372 /// Split out of the open path so that [`refresh`](GraphDb::refresh) can
2373 /// rebuild a handle in place, without ownership of `F`, when another
2374 /// process replaces the snapshot underneath it.
2375 ///
2376 /// `origin` decides whether the two repair writes this function can make
2377 /// are appropriate; see [`LoadOrigin`].
2378 fn load_from_disk(&mut self, origin: LoadOrigin) -> Result<usize> {
2379 // Both writes below are crash recovery, and only an open is entitled to
2380 // perform them. A read-only handle promises to touch nothing, and a
2381 // reload driven by `refresh` is looking at a store another process is
2382 // actively writing: what looks like a torn tail there is a peer
2383 // mid-append, and what looks like an orphaned archive may be one that
2384 // peer is about to reference.
2385 let may_repair = origin == LoadOrigin::Open && !self.open_opts.read_only;
2386 let repair_wal = self.open_opts.repair_wal && may_repair;
2387 let db = self;
2388 db.wal_horizon_floor = db.fs.read_horizon_floor()?;
2389 db.archive_genesis_chain = db.fs.has_genesis_marker();
2390 // Opening cleanup: remove orphaned archives — archives whose frames all
2391 // fall below the horizon floor. Orphans arise when a crash interrupted
2392 // the retention-prune sequence after the floor was written but before
2393 // all surplus archives were deleted. Safe to delete: floor already
2394 // accounts for their frames.
2395 if may_repair {
2396 db.cleanup_orphaned_archives()?;
2397 }
2398 let _t0 = std::time::Instant::now();
2399 // Peek 6 bytes to determine snapshot version without reading the full
2400 // file. For RealFs this is a true partial read (O(1)); for SimFs the
2401 // default impl reads all bytes and truncates (still correct).
2402 let snap_header = db.fs.read_prefix(FileId::Snapshot, 6)?;
2403 // V8, V9 and V10 share the mmap-able container; V9 only adds section 12
2404 // and V10 adds nothing but its version stamp. A version outside that set
2405 // falls through to the full-read path below, where `snapshot::decode`
2406 // either handles it (V5–V7) or refuses it by name — which is what stops
2407 // an older binary before it reaches the WAL.
2408 let is_v8 = snap_header.len() >= 6
2409 && &snap_header[0..4] == b"GDB1"
2410 && core_storage::snapshot::is_mmap_container(u16::from_le_bytes([
2411 snap_header[4],
2412 snap_header[5],
2413 ]));
2414 // The version this store is stamped with, or `None` when it has never
2415 // been snapshotted. Read from the same six bytes, with no second read.
2416 let snapshot_version = if snap_header.len() >= 6 && &snap_header[0..4] == b"GDB1" {
2417 Some(u16::from_le_bytes([snap_header[4], snap_header[5]]))
2418 } else {
2419 None
2420 };
2421 // No snapshot means no snapshot has ever truncated the WAL, so this
2422 // handle can prove the history is whole. Once a snapshot exists that
2423 // this handle did not take, it cannot: see `snapshot_preserved_history`.
2424 db.snapshot_preserved_history = snap_header.is_empty();
2425 if is_v8 {
2426 // V8: map the file zero-copy (RealFs) or read full bytes (SimFs).
2427 // No 2.4GB heap Vec is allocated on RealFs.
2428 let mapped = Arc::new(
2429 if let Some(snap_path) = db.fs.snapshot_path() {
2430 core_storage::v8::MappedBase::map(&snap_path)
2431 } else {
2432 let snap_bytes = db.fs.read(FileId::Snapshot)?;
2433 core_storage::v8::MappedBase::from_bytes(snap_bytes)
2434 }
2435 .map_err(|e| GraphError::Corrupt {
2436 detail: format!("v8: mmap open: {e:?}"),
2437 })?,
2438 );
2439 db.restore_v8_base(Arc::clone(&mapped))?;
2440 trace_open!("restore_v8_base", _t0);
2441 db.base = Some(mapped);
2442 trace_open!("base assigned", _t0);
2443 } else if !snap_header.is_empty() {
2444 // Legacy V5-V7: full read required for decode.
2445 let snap_bytes = db.fs.read(FileId::Snapshot)?;
2446 if let Some(state) = core_storage::snapshot::decode(&snap_bytes)? {
2447 db.restore_snapshot_state(state)?;
2448 }
2449 }
2450 // else: snap_header is empty = no snapshot file, fresh store.
2451 //
2452 // Seed commit_seq from the highest seq persisted in last_change so that
2453 // WAL-replay frames (which start at commit_seq+1) always exceed any seq
2454 // already stored in the snapshot. Without this, a db with one snapshot
2455 // commit would save last_change["a"]=1, then on reopen the first WAL
2456 // frame would replay at seq=1 again — colliding and making WAL-tail
2457 // mutations indistinguishable from the snapshot baseline.
2458 //
2459 // Safety invariant (seq-recycling):
2460 // Recycled seqs (those below the seeded baseline) were NEVER stored in
2461 // last_change because they belonged to a previous db lifetime — a new
2462 // db starts at commit_seq=0 with an empty last_change. Therefore no
2463 // CAS precondition can carry a recycled seq as its `expected` value
2464 // and accidentally match a live node's last_change entry.
2465 //
2466 // `expected:0` on a deleted-then-reinserted node:
2467 // After deletion, last_changed() returns None; callers that call
2468 // last_changed() and then use NodeUnchangedSince get None.unwrap_or(0)
2469 // = 0. The reinserted node gets seq > 0, so a subsequent CAS with
2470 // expected=0 correctly conflicts. The only way to observe actual=0 in
2471 // a CasConflict would be a caller that invented expected=0 without ever
2472 // calling last_changed() — unreachable via the documented API contract.
2473 if let Some(&max_seq) = db.last_change.values().max() {
2474 db.commit_seq = db.commit_seq.max(max_seq);
2475 }
2476 let bytes = db.fs.read(FileId::Wal)?;
2477 let (records, valid_len) = decode_all(&bytes);
2478 // The valid prefix is replayed either way; `repair_wal` only decides
2479 // whether the truncation is written back. A reader that races a live
2480 // appender must not persist a truncation the writer never asked for.
2481 if valid_len < bytes.len() && repair_wal {
2482 db.fs.write_atomic(FileId::Wal, &bytes[..valid_len])?;
2483 }
2484 // WAL-present path: build indexes eagerly BEFORE replay so that the
2485 // first replayed record does not trigger the lazy-init guard (which
2486 // would call reindex_all_load_state on an empty graph, defeating the
2487 // point of restoring IVF/HNSW blobs from the snapshot).
2488 if !records.is_empty() {
2489 db.ensure_v8_base_sections_loaded();
2490 trace_open!("lazy sections loaded (WAL path)", _t0);
2491 }
2492 let replayed = db.apply_frames(records)?;
2493 // ── The multiplicity declaration, recovered from the stamp ───────────
2494 //
2495 // The opt-in is re-emitted into every baseline WAL a snapshot writes, so
2496 // ordinarily the replay above has already found it. But
2497 // `snapshot_with(archive_wal)` renames the live WAL away and writes its
2498 // replacement afterwards, and between those two points the store holds
2499 // no live declaration at all. A crash there — or a single `Err` from any
2500 // call in between — used to opt the store back out on the next open
2501 // (defect #22): it would stop counting and write a **V9** snapshot while
2502 // the archives still carried discriminant 23, which is the exact state
2503 // the V10 stamp exists to prevent.
2504 //
2505 // The V10 stamp is what carries the conclusion. The archive clause is a
2506 // scope restriction, not a second proof — an earlier version of this
2507 // comment, and defect #22, claimed otherwise, and defect #33 corrects
2508 // it. Taking the two in order:
2509 //
2510 // **The stamp.** `snapshot_with` stamps the snapshot from
2511 // `self.multiplicity` *before* it touches the WAL, and nothing rewrites
2512 // a V10 snapshot at V9 while the store believes it is opted in. So a
2513 // V10 stamp says this store reached `enable_multiplicity` far enough to
2514 // write the snapshot — and, decisively, that every older binary already
2515 // refuses this store by name. Opting in here can cost such a reader
2516 // nothing it was not already being told.
2517 //
2518 // **What the archive clause does not prove.** It is *not* evidence that
2519 // the archive was taken while the store was opted in. A store can
2520 // archive at V9 and opt in afterwards, leaving a V10 snapshot standing
2521 // beside an archive whose WAL carries no declaration at all — see
2522 // `a_failed_opt_in_beside_an_archive_comes_back_opted_in`. The inference
2523 // held in the success case by coincidence, not by construction.
2524 //
2525 // **What it does buy: scope.** Without it the recovery would also fire
2526 // on a store that reached the V10 snapshot write and then failed with no
2527 // archive in sight. That store must stay opted out, and can: no WAL was
2528 // renamed away, nothing carries discriminant 23, and its next snapshot
2529 // rewrites at V9, which puts it back within reach of every older reader.
2530 // An archive is the marker for the one state that is not recoverable
2531 // that way — a WAL renamed away that may hold the only copy of the
2532 // declaration. `no_crash_leaves_discriminant_23_unguarded` pins that
2533 // line: it sweeps a workload with no archives at all and refuses a
2534 // V10-implies-enabled rule.
2535 //
2536 // **The invariant, whichever way the clause goes:** the recovery never
2537 // opts in a store whose snapshot is not V10. A V9 store has made no
2538 // promise to an older reader, so opting it in would start writing
2539 // discriminant 23 behind a stamp that does not guard it. Pinned by
2540 // `the_recovery_never_opts_in_a_store_whose_snapshot_is_not_v10` and
2541 // `the_recovery_does_not_opt_a_store_in_by_itself`.
2542 //
2543 // What this recovery cannot do is make the opt-in atomic; it is not,
2544 // and `enable_multiplicity` says so. See defects #32-#34.
2545 if !db.multiplicity
2546 && snapshot_version == Some(core_storage::snapshot::VERSION_10)
2547 && !db.fs.list_archives()?.is_empty()
2548 {
2549 db.multiplicity = true;
2550 }
2551 // The cursor sits at the end of the valid prefix, not the end of the
2552 // file: a torn or still-being-written tail is unconsumed by definition
2553 // and stays visible to `is_stale` until it decodes.
2554 db.wal_consumed = valid_len as u64;
2555 db.snapshot_ident = db.fs.snapshot_ident().map_err(GraphError::Io)?;
2556 trace_open!("wal replay done", _t0);
2557 // Rebuild view values after WAL replay only when there is no V8 base.
2558 // With a V8 base, view values are correct in the snapshot and are updated
2559 // incrementally during WAL replay (on_edge_changed / on_prop_changed).
2560 // A full rebuild would read overlay-only props (empty after restore_v8_base)
2561 // and overwrite correct base values with wrong results (e.g. NeighborAgg
2562 // Sum reads no "score" in overlay → writes 0.0, shadowing the correct
2563 // base value).
2564 if db.base.is_none() {
2565 let topo_view = TopologyView::owned(&db.topo);
2566 db.view_store
2567 .rebuild_all(&mut db.props, &topo_view, &db.ids, &db.syms, &db.labels);
2568 }
2569 // Rebuild full-text index after WAL replay. Corrects drift from
2570 // per-record incremental apply during replay.
2571 db.fulltext.rebuild_all(
2572 &db.ids,
2573 &db.labels,
2574 &db.syms,
2575 build_props_view(&db.props, &db.base),
2576 );
2577 db.prop_index.rebuild_all(
2578 &db.ids,
2579 &db.labels,
2580 &db.syms,
2581 build_props_view(&db.props, &db.base),
2582 );
2583 // Namespaces: one pass over the `ns` column, after the snapshot is
2584 // restored and the WAL replayed. Replay maintains `node_ns` record by
2585 // record as well; this pass is what makes a snapshot-only open right,
2586 // and it reads nothing on a store with no `ns` column.
2587 db.rebuild_node_ns();
2588 // A mid-build snapshot's HNSW blob carries `complete == false`.
2589 // Register it so `serve`'s ticker sees work without waiting for a write.
2590 db.register_outstanding_index_builds();
2591 // Load roles sidecar. Missing file = no roles (Some(vec![])).
2592 // Corrupt/unparseable = poisoned (None); mask_for_role will fail-loud.
2593 db.roles = Self::load_roles_from_fs(&db.fs)?;
2594 // Capture the initial MVCC fold so reader() is ready immediately.
2595 db.fold_now();
2596 trace_open!("open_with complete", _t0);
2597 Ok(replayed)
2598 }
2599
2600 /// Apply decoded WAL frames to in-memory state, exactly as the open-path
2601 /// replay does — same `apply` calls, same per-frame delta drain, same
2602 /// commit-seq and last-change bookkeeping. Rules therefore fire and derived
2603 /// edges appear identically whether a frame arrives at open, from a local
2604 /// commit, or from another process by way of [`refresh`](GraphDb::refresh).
2605 ///
2606 /// Returns the number of frames applied.
2607 ///
2608 /// Deltas are drained and discarded per frame: replayed frames are already
2609 /// reflected on disk, so they are not news to a subscriber, and draining
2610 /// inside the loop keeps `pending_deltas` O(1) over a large WAL (I-2).
2611 fn apply_frames(&mut self, records: Vec<WalRecord>) -> Result<usize> {
2612 if records.is_empty() {
2613 return Ok(0);
2614 }
2615 // Materialize any state retained in the mmap base before the first
2616 // frame lands, so a replayed record cannot trip the lazy-init guard and
2617 // rebuild indexes from an empty graph. Both calls are idempotent.
2618 self.ensure_v8_base_sections_loaded();
2619 self.engine.consume_retained_state_eager(
2620 &self.ids,
2621 &self.syms,
2622 &self.labels,
2623 build_props_view(&self.props, &self.base),
2624 );
2625 let applied = records.len();
2626 for rec in records {
2627 self.apply(&rec)?;
2628 let _ = self.engine.drain_deltas();
2629 // Track commit_seq during replay so last_change entries are
2630 // consistent with the seqs assigned by log_then_apply_with on
2631 // subsequent live commits. After N replayed frames, commit_seq=N;
2632 // live commits begin at N+1.
2633 self.commit_seq += 1;
2634 let replay_seq = self.commit_seq;
2635 self.update_last_change_from_rec(&rec, replay_seq);
2636 }
2637 // Enforce I-2: if the per-frame drain above is ever removed or skipped,
2638 // this assert catches the regression in debug builds immediately.
2639 debug_assert_eq!(
2640 self.engine.pending_delta_count(),
2641 0,
2642 "pending_deltas non-empty after replay — \
2643 per-frame drain must run inside the loop to keep memory O(1)"
2644 );
2645 // T2 note: the per-frame drain IS the suppression seam for replay.
2646 // Any future as-of replay path (Plan-15 T2) must drain here to feed
2647 // replaying subscribers; the mechanism is already in place.
2648 let _ = self.engine.drain_deltas(); // belt-and-braces no-op after loop drain
2649 Ok(applied)
2650 }
2651
2652 // ── Multi-process safety: cross-process write lock + WAL tailing ──────────
2653 //
2654 // mushroomdb is many-readers / one-writer across processes. Writers take an
2655 // advisory exclusive lock on the store's `LOCK` file; readers never do.
2656 // Every handle tracks how much of the WAL it has consumed, so it can pick
2657 // up another process's commits by decoding only the new tail rather than
2658 // reopening. See `docs/site/concurrency.md`.
2659
2660 /// Whether the store on disk has moved ahead of (or out from under) this
2661 /// handle's in-memory state.
2662 ///
2663 /// True when the WAL's length differs from this handle's cursor — another
2664 /// process committed, or is mid-append — or when the snapshot file's
2665 /// identity changed. Costs two metadata lookups and reads no file contents,
2666 /// so it is cheap enough for a read path to call.
2667 ///
2668 /// Always false for an as-of view from [`GraphDb::open_at`]: such a view is
2669 /// pinned to one commit and later commits are deliberately invisible to it.
2670 pub fn is_stale(&self) -> Result<bool> {
2671 if self.pinned {
2672 return Ok(false);
2673 }
2674 if self.fs.wal_len().map_err(GraphError::Io)? != self.wal_consumed {
2675 return Ok(true);
2676 }
2677 Ok(self.fs.snapshot_ident().map_err(GraphError::Io)? != self.snapshot_ident)
2678 }
2679
2680 /// Bring this handle up to date with every commit other processes have made,
2681 /// and return how many frames were applied.
2682 ///
2683 /// The WAL tail is decoded from this handle's cursor and applied through the
2684 /// same path the open replay uses, so rules fire and derived edges appear
2685 /// exactly as they would on a fresh open. Interners, id maps and indexes
2686 /// stay valid for the same reason.
2687 ///
2688 /// A frame another process is still writing is left alone: a trailing
2689 /// partial frame is a wait, not a corruption, and the handle stays stale
2690 /// until that frame is complete. Nothing is written to disk, so a read-only
2691 /// handle can refresh freely.
2692 ///
2693 /// When the snapshot file's identity changed, or the WAL is shorter than
2694 /// this handle's cursor, the WAL no longer continues our state — another
2695 /// process snapshotted or archived. The handle is then rebuilt from disk
2696 /// with the options it was opened with, and the return value is the number
2697 /// of frames in the new WAL.
2698 ///
2699 /// Returns 0 for an as-of view, which never follows later commits.
2700 ///
2701 /// # Errors
2702 ///
2703 /// An error here leaves the handle **degraded**: it got partway through
2704 /// applying the tail, or partway through a reload, so its in-memory state
2705 /// no longer matches any point on disk. Further mutations are refused and
2706 /// the handle must be reopened. Nothing on disk was damaged — the store
2707 /// itself is fine, and a fresh open recovers it.
2708 pub fn refresh(&mut self) -> Result<u64> {
2709 if self.pinned {
2710 return Ok(0);
2711 }
2712 let disk_ident = self.fs.snapshot_ident().map_err(GraphError::Io)?;
2713 let wal_len = self.fs.wal_len().map_err(GraphError::Io)?;
2714 if disk_ident != self.snapshot_ident || wal_len < self.wal_consumed {
2715 // The WAL no longer continues our state: rebuild from disk. State
2716 // is cleared first, so a failed load leaves an empty handle — mark
2717 // it degraded rather than let a caller read an empty graph as if
2718 // it were the store's contents.
2719 self.reset_for_reload();
2720 return match self.load_from_disk(LoadOrigin::Reload) {
2721 Ok(frames) => Ok(frames as u64),
2722 Err(e) => {
2723 self.degraded = true;
2724 Err(e)
2725 }
2726 };
2727 }
2728 if wal_len == self.wal_consumed {
2729 return Ok(0);
2730 }
2731 let tail = self
2732 .fs
2733 .read_range(FileId::Wal, self.wal_consumed)
2734 .map_err(GraphError::Io)?;
2735 let (records, valid_len) = decode_all(&tail);
2736 let applied = match self.apply_frames(records) {
2737 Ok(n) => n,
2738 Err(e) => {
2739 // Some frames landed and some did not, and the cursor cannot
2740 // say how many. Advancing it would skip the rest; leaving it
2741 // would replay what already applied. Neither is recoverable in
2742 // place, so refuse further writes and require a reopen.
2743 self.degraded = true;
2744 return Err(e);
2745 }
2746 };
2747 // Advance by the bytes actually decoded, never by the file length: an
2748 // incomplete trailing frame stays unconsumed for the next refresh.
2749 self.wal_consumed += valid_len as u64;
2750 if applied > 0 {
2751 // Peer commits must reach `reader()` snapshots taken from here on.
2752 // A full fold is what open does; refresh does not build per-commit
2753 // deltas, so there is nothing cheaper that stays correct.
2754 self.fold_now();
2755 }
2756 Ok(applied as u64)
2757 }
2758
2759 /// Byte offset of the WAL prefix this handle has applied.
2760 ///
2761 /// Exposed for tests that assert the cursor tracks appended bytes exactly.
2762 #[doc(hidden)]
2763 pub fn wal_consumed(&self) -> u64 {
2764 self.wal_consumed
2765 }
2766
2767 /// Rewind the WAL cursor after the group-commit drain thread truncated a
2768 /// failed group off the tail, so the cursor still describes the file.
2769 pub(crate) fn set_wal_consumed(&mut self, len: u64) {
2770 self.wal_consumed = len;
2771 }
2772
2773 /// One non-blocking attempt at the cross-process write lock.
2774 ///
2775 /// Takes `&self` so a caller can poll for the lock *before* it acquires the
2776 /// in-process write guard. That ordering is what keeps a busy peer in
2777 /// another process from stalling this process's readers.
2778 ///
2779 /// A handle that owns the lock for its lifetime always succeeds.
2780 pub(crate) fn try_cross_process_lock(&self) -> Result<bool> {
2781 if self.holds_lifetime_lock {
2782 return Ok(true);
2783 }
2784 self.fs.try_lock_exclusive().map_err(GraphError::Io)
2785 }
2786
2787 /// Poll for the cross-process write lock until `wait` elapses.
2788 ///
2789 /// One attempt is always made, so a zero wait is a single try. Returns
2790 /// `false` when the lock is still held elsewhere at the deadline; nothing
2791 /// has been written and retrying later is safe.
2792 ///
2793 /// Only the plain-`GraphDb` open path uses this, where the caller owns the
2794 /// handle outright. [`SharedDb`](crate::SharedDb) polls
2795 /// [`try_cross_process_lock`](GraphDb::try_cross_process_lock) itself so
2796 /// that it holds no in-process guard while it waits.
2797 fn poll_lock(&self, wait: std::time::Duration) -> Result<bool> {
2798 let deadline = std::time::Instant::now() + wait;
2799 loop {
2800 if self.try_cross_process_lock()? {
2801 return Ok(true);
2802 }
2803 let now = std::time::Instant::now();
2804 if now >= deadline {
2805 return Ok(false);
2806 }
2807 std::thread::sleep(LOCK_POLL_INTERVAL.min(deadline.saturating_duration_since(now)));
2808 }
2809 }
2810
2811 /// Open a cross-process write scope, given the outcome of an already-made
2812 /// lock attempt.
2813 ///
2814 /// The caller polls for the lock first — outside any in-process guard — and
2815 /// passes what it got. On success this refreshes, so the writes about to
2816 /// happen land on top of every other process's commits. On failure the
2817 /// handle refuses WAL-appending mutations and `snapshot()` with
2818 /// [`GraphError::Busy`] until [`end_write_lock`](GraphDb::end_write_lock)
2819 /// closes the scope, so a caller holding a guard cannot write behind
2820 /// another process's back.
2821 ///
2822 /// A handle that already owns the lock for its lifetime skips the refresh:
2823 /// no other process can have written, so there is nothing to pick up.
2824 pub(crate) fn enter_write_scope(&mut self, acquired: bool) -> Result<()> {
2825 self.lock_denied = !acquired;
2826 if !acquired || self.holds_lifetime_lock {
2827 return Ok(());
2828 }
2829 if let Err(e) = self.refresh() {
2830 // Do not hold a lock we cannot use: release it and let the caller
2831 // see the underlying failure.
2832 let _ = self.fs.unlock();
2833 self.lock_denied = true;
2834 return Err(e);
2835 }
2836 Ok(())
2837 }
2838
2839 /// Close a cross-process write scope opened by
2840 /// [`enter_write_scope`](GraphDb::enter_write_scope): release the lock and
2841 /// clear the Busy latch. Safe to call when the lock was never taken.
2842 pub(crate) fn end_write_lock(&mut self) {
2843 self.lock_denied = false;
2844 if !self.holds_lifetime_lock {
2845 // Releasing a lock we do not hold is a no-op; a failure to release
2846 // is reported by the OS closing the descriptor at handle drop.
2847 let _ = self.fs.unlock();
2848 }
2849 }
2850
2851 /// As-of replay for [`GraphDb::open_at`]: snapshot base (only when the
2852 /// snapshot truncated the WAL) plus the first `commit + 1` WAL frames;
2853 /// see [`GraphDb::open_at`] for the semantics. The per-frame drain
2854 /// mirrors `open_with` exactly so pending_delta_count is 0 on exit.
2855 /// Restore all persisted state from a decoded snapshot. Shared by
2856 /// `open_with` and (when the snapshot truncated the WAL) `open_at_with`.
2857 fn restore_snapshot_state(
2858 &mut self,
2859 state: core_storage::snapshot::SnapshotState,
2860 ) -> Result<()> {
2861 self.ids = state.ids;
2862 self.syms = state.syms;
2863 self.topo = state.topo;
2864 self.props = state.props;
2865 self.labels = state.labels;
2866 self.edge_props = state.edge_props;
2867 // Cross-section label integrity for V5/V7 snapshots: same invariants as
2868 // restore_v8_base. A crafted bincode snapshot with a short `labels` vec,
2869 // out-of-range sym ids, or a sentinel label on a live node would otherwise
2870 // open successfully and panic later in `NodeRef::label()` or
2871 // `neighborhood_masked()`. Catching it here turns those into typed
2872 // `GraphError::Corrupt` at open time.
2873 {
2874 let ids_len = self.ids.len();
2875 if self.labels.len() != ids_len {
2876 return Err(GraphError::Corrupt {
2877 detail: format!(
2878 "snapshot: labels vec has {} entries but id table has {} total slots",
2879 self.labels.len(),
2880 ids_len,
2881 ),
2882 });
2883 }
2884 let syms_len = self.syms.len() as u32;
2885 for (i, &sym) in self.labels.iter().enumerate() {
2886 let is_tombstoned = self.ids.is_tombstoned(i as u32);
2887 if sym == u32::MAX {
2888 if !is_tombstoned {
2889 return Err(GraphError::Corrupt {
2890 detail: format!(
2891 "snapshot: live node at id slot {i} has sentinel label (u32::MAX)"
2892 ),
2893 });
2894 }
2895 } else if sym >= syms_len {
2896 return Err(GraphError::Corrupt {
2897 detail: format!(
2898 "snapshot: label at id slot {i} references sym {sym} \
2899 which is out of interner range ({syms_len})"
2900 ),
2901 });
2902 }
2903 }
2904 }
2905 let defs: Vec<RuleDef> = state
2906 .rule_defs
2907 .iter()
2908 .map(|b| {
2909 decode_rule_def(b).map_err(|e| GraphError::Corrupt {
2910 detail: format!("snapshot rule_def deserialize: {e}"),
2911 })
2912 })
2913 .collect::<Result<Vec<_>>>()?;
2914 self.engine =
2915 RuleEngine::from_persist(defs, state.provenance, state.rule_tripped, state.rule_fires);
2916 // Candidate indexes are rebuilt lazily on the first mutation (see
2917 // RuleEngine::on_node_changed). HNSW blobs and IVF centroids from the
2918 // snapshot are retained without deserializing so that:
2919 // - clean-open (empty WAL): indexes stay empty; blobs load on first
2920 // ANN query via ensure_hnsw_loaded, or on first mutation via the
2921 // lazy-init guard which calls reindex_all_load_state (the scan
2922 // skips the HNSW build for every side the blob supplies).
2923 // - WAL-present: open_with calls consume_retained_state_eager before
2924 // replay so HNSW/IVF are live before any record fires the hooks.
2925 let ivf_bytes = if state.ivf_state.is_empty() {
2926 Vec::new()
2927 } else {
2928 bincode::serialize(&state.ivf_state).expect("IVF state serialize cannot fail")
2929 };
2930 // Store blobs without eagerly deserializing them.
2931 // `self.ids` is the snapshot's id table at this point — WAL replay has
2932 // not run — so its length is the line an interrupted build is detected
2933 // against.
2934 let snapshot_ids = self.ids.len() as u32;
2935 self.engine
2936 .store_snapshot_state(state.hnsw_state, ivf_bytes, snapshot_ids);
2937 // Restore view defs from snapshot (V5).
2938 // The ColumnStore already contains view values from the snapshot;
2939 // use restore_view (no collision check, no backfill) so the store
2940 // is aware of the definitions. rebuild_all runs after WAL replay.
2941 for def_bytes in &state.view_defs {
2942 let def: ViewDef =
2943 bincode::deserialize(def_bytes).map_err(|e| GraphError::Corrupt {
2944 detail: format!("snapshot view_def deserialize: {e}"),
2945 })?;
2946 self.view_store
2947 .restore_view(def)
2948 .map_err(|e| GraphError::Corrupt {
2949 detail: format!("snapshot view restore: {e}"),
2950 })?;
2951 }
2952 Ok(())
2953 }
2954
2955 /// Restore all persisted state from a V8 `MappedBase` snapshot, **except**
2956 /// topology (`self.topo` stays empty and serves as the WAL-replay overlay).
2957 ///
2958 /// `self.props` IS fully materialised from the base so that HNSW/IVF blob
2959 /// deserialization and view rebuild have access to all column data.
2960 fn restore_v8_base(&mut self, mapped: Arc<core_storage::v8::MappedBase>) -> Result<()> {
2961 self.ids = archived_to_idmap(mapped.ids().map_err(|e| GraphError::Corrupt {
2962 detail: format!("v8: ids section: {e:?}"),
2963 })?);
2964 self.syms = archived_to_interner(mapped.syms().map_err(|e| GraphError::Corrupt {
2965 detail: format!("v8: syms section: {e:?}"),
2966 })?);
2967
2968 // C1: self.props is left as an empty overlay. Column reads go through
2969 // props_view() (ColumnsView::with_base), which consults the archived base
2970 // section zero-copy. This avoids the O(columns) heap copy at every open.
2971
2972 // self.topo deliberately left as Topology::new() — overlay path.
2973
2974 let meta = decode_meta(mapped.meta_bytes().map_err(|e| GraphError::Corrupt {
2975 detail: format!("v8: meta section: {e:?}"),
2976 })?)
2977 .map_err(|e| GraphError::Corrupt {
2978 detail: format!("v8: meta decode: {e:?}"),
2979 })?;
2980 self.labels = meta.labels;
2981 // Cross-section label integrity: labels must cover every id slot (live
2982 // and tombstoned), every non-sentinel sym must be within the interner's
2983 // bound, and no live (non-tombstoned) node may carry the u32::MAX
2984 // sentinel label. Without this check, a crafted snapshot where the META
2985 // section (small, CRC-validated) holds a short `labels` vec, out-of-range
2986 // sym ids, or a sentinel label on a live node, would open successfully
2987 // and then panic in `NodeRef::label()`, `neighborhood_masked()`, and
2988 // related read paths. Catching the inconsistency here converts those
2989 // panics into typed `GraphError::Corrupt` at open time.
2990 {
2991 let ids_len = self.ids.len();
2992 if self.labels.len() != ids_len {
2993 return Err(GraphError::Corrupt {
2994 detail: format!(
2995 "v8: labels section has {} entries but id table has {} total slots",
2996 self.labels.len(),
2997 ids_len,
2998 ),
2999 });
3000 }
3001 let syms_len = self.syms.len() as u32;
3002 for (i, &sym) in self.labels.iter().enumerate() {
3003 let is_tombstoned = self.ids.is_tombstoned(i as u32);
3004 if sym == u32::MAX {
3005 // Sentinel is only valid for tombstoned slots.
3006 if !is_tombstoned {
3007 return Err(GraphError::Corrupt {
3008 detail: format!(
3009 "v8: live node at id slot {i} has sentinel label (u32::MAX)"
3010 ),
3011 });
3012 }
3013 } else if sym >= syms_len {
3014 return Err(GraphError::Corrupt {
3015 detail: format!(
3016 "v8: label at id slot {i} references sym {sym} \
3017 which is out of interner range ({syms_len})"
3018 ),
3019 });
3020 }
3021 }
3022 }
3023 // C3: self.edge_props stays as an empty overlay. Reads go through
3024 // edge_props_view() which consults the mmap'd base section zero-copy
3025 // via EdgePropsView::with_base. No heap decode at open time.
3026
3027 // Restore rule engine.
3028 let (rule_def_bytes, rule_tripped, rule_fires) =
3029 archived_rules_meta_to_owned(mapped.rules_meta_section().map_err(|e| {
3030 GraphError::Corrupt {
3031 detail: format!("v8: rules_meta section: {e:?}"),
3032 }
3033 })?);
3034 let defs: Vec<RuleDef> = rule_def_bytes
3035 .iter()
3036 .map(|b| {
3037 decode_rule_def(b).map_err(|e| GraphError::Corrupt {
3038 detail: format!("v8: rule_def deserialize: {e}"),
3039 })
3040 })
3041 .collect::<Result<Vec<_>>>()?;
3042 self.engine = RuleEngine::from_persist(defs, BTreeMap::new(), rule_tripped, rule_fires);
3043 // C4+C5: provenance, HNSW, and IVF sections are NOT read here.
3044 // `ensure_v8_base_sections_loaded` reads them on first use from
3045 // `self.base` (set by the caller immediately after this returns).
3046 // A clean open touches only: header + IDS + SYMS + META + RULES_META.
3047
3048 // Restore view definitions.
3049 let view_defs =
3050 archived_views_to_owned(mapped.views_section().map_err(|e| GraphError::Corrupt {
3051 detail: format!("v8: views section: {e:?}"),
3052 })?);
3053 for def_bytes in &view_defs {
3054 let def: ViewDef =
3055 bincode::deserialize(def_bytes).map_err(|e| GraphError::Corrupt {
3056 detail: format!("v8: view_def deserialize: {e}"),
3057 })?;
3058 self.view_store
3059 .restore_view(def)
3060 .map_err(|e| GraphError::Corrupt {
3061 detail: format!("v8: view restore: {e}"),
3062 })?;
3063 }
3064 // Load the last-change map from section 11 (small section; load eagerly).
3065 // Pre-Task-3 snapshots lack this section; `last_change_bytes` returns &[]
3066 // in that case and `decode_last_change_bytes` returns an empty map.
3067 let last_change_raw = mapped
3068 .last_change_bytes()
3069 .map_err(|e| GraphError::Corrupt {
3070 detail: format!("v8: last_change section: {e:?}"),
3071 })?;
3072 self.last_change = decode_last_change_bytes(last_change_raw);
3073
3074 // Validate that all deferred sections (provenance, HNSW, IVF) fit within
3075 // the file. Pure bounds check — no bytes read, no page faults triggered.
3076 // Catches truncated snapshots at open time before the lazy deferred reads.
3077 mapped.validate_section_bounds().map_err(|e| match e {
3078 GraphError::Corrupt { detail } => GraphError::Corrupt {
3079 detail: format!("v8: section bounds: {detail}"),
3080 },
3081 other => other,
3082 })?;
3083 Ok(())
3084 }
3085
3086 /// Read provenance, HNSW, and IVF sections from the mmap base into the
3087 /// engine's retained fields on first call. Subsequent calls are a no-op
3088 /// (AtomicBool fast-path).
3089 ///
3090 /// Must be called before any code path that reads or mutates engine
3091 /// provenance, HNSW, or IVF state:
3092 /// - WAL replay (before `consume_retained_state_eager`)
3093 /// - First mutation (`log_then_apply_with`)
3094 /// - Read-only paths (`stats`, `explain`, `node_edges`)
3095 /// - Snapshot (`snapshot_with`)
3096 ///
3097 /// No-op for fresh stores and V5-V7 opens (`self.base` is `None`).
3098 fn ensure_v8_base_sections_loaded(&self) {
3099 use std::sync::atomic::Ordering;
3100 if self.v8_sections_loaded.load(Ordering::Acquire) {
3101 return;
3102 }
3103 let _guard = self
3104 .v8_sections_mutex
3105 .lock()
3106 .expect("v8 sections mutex poisoned");
3107 if self.v8_sections_loaded.load(Ordering::Acquire) {
3108 return; // another caller populated while we waited
3109 }
3110 let _t = std::time::Instant::now();
3111 if let Some(base) = &self.base {
3112 // Provenance: raw rkyv bytes; CRC validated inside section_bytes.
3113 // Bounds are already validated at open time (restore_v8_base →
3114 // validate_section_bounds) — unreachable post-validate_section_bounds;
3115 // unwrap_or_default is a safety belt against impossible errors.
3116 let prov_bytes = base
3117 .provenance_raw_bytes()
3118 .map(|b| b.to_vec())
3119 .unwrap_or_default();
3120 self.engine.store_provenance_bytes(prov_bytes);
3121 // HNSW: decode rkyv blobs into owned map.
3122 let hnsw_state = base
3123 .hnsw_section()
3124 .map(archived_hnsw_to_owned)
3125 .unwrap_or_default();
3126 // IVF: raw bincode bytes; deserialized on first mutation/query.
3127 let ivf_bytes = base.ivf_bytes().map(|b| b.to_vec()).unwrap_or_default();
3128 // Called before WAL replay on a WAL-present open (`open_with`) and
3129 // before any write on a clean one, so this is the snapshot's count.
3130 let snapshot_ids = self.ids.len() as u32;
3131 self.engine
3132 .store_snapshot_state(hnsw_state, ivf_bytes, snapshot_ids);
3133 }
3134 self.v8_sections_loaded.store(true, Ordering::Release);
3135 if std::env::var("MUSHROOMDB_TRACE_OPEN").is_ok() {
3136 eprintln!(
3137 "[MUSHROOMDB_TRACE_OPEN] ensure_v8_base_sections_loaded: {:>9.3?}",
3138 _t.elapsed()
3139 );
3140 }
3141 }
3142
3143 /// Return a `TopologyView` that merges the mmap'd base (when present) with
3144 /// the in-memory WAL overlay. Used by all read paths in db.rs that need
3145 /// the full merged topology without going through `self.view()`.
3146 fn topo_view(&self) -> TopologyView<'_> {
3147 match self.base {
3148 None => TopologyView::owned(&self.topo),
3149 Some(ref base) => {
3150 // SAFETY: base lives as long as self; section bounds validated at open.
3151 // topology() uses access_unchecked; all field reads are bounds-checked in seam.rs.
3152 let archived = base
3153 .topology()
3154 .expect("base topology section bounds validated at open");
3155 TopologyView::with_base(&self.topo, archived)
3156 }
3157 }
3158 }
3159
3160 /// Return a `ColumnsView` that merges the mmap'd base columns (when a V8
3161 /// snapshot is open) with the in-memory WAL overlay. Reads consult the
3162 /// overlay first, then fall through to the archived base section zero-copy.
3163 fn props_view(&self) -> core_storage::v8::seam::ColumnsView<'_> {
3164 match self.base {
3165 None => core_storage::v8::seam::ColumnsView::owned(&self.props),
3166 Some(ref base) => {
3167 // columns() uses access_unchecked; field reads are bounds-checked in seam.rs.
3168 let archived = base
3169 .columns()
3170 .expect("base columns section bounds validated at open");
3171 core_storage::v8::seam::ColumnsView::with_base_cached(
3172 &self.props,
3173 archived,
3174 base.mixed_cache(),
3175 )
3176 .with_shared_strings(base_string_table(base))
3177 }
3178 }
3179 }
3180
3181 /// Return an `EdgePropsView` that merges the mmap'd base edge-props section
3182 /// (when a V8 snapshot is open) with the in-memory WAL overlay.
3183 ///
3184 /// Reads consult the overlay first (for post-snapshot mutations), then fall
3185 /// through to the archived base section zero-copy. Tombstones in the
3186 /// overlay mask deleted-from-base entries.
3187 fn edge_props_view(&self) -> EdgePropsView<'_> {
3188 match self.base {
3189 None => EdgePropsView::owned(&self.edge_props),
3190 Some(ref base) => {
3191 // edge_props_section() uses access_unchecked; field reads bounds-checked in seam.rs.
3192 let archived = base
3193 .edge_props_section()
3194 .expect("base edge_props section bounds validated at open");
3195 EdgePropsView::with_base(&self.edge_props, archived)
3196 }
3197 }
3198 }
3199
3200 fn open_at_with(fs: F, commit: u64) -> Result<Self> {
3201 // An as-of view never writes and is pinned to one commit: it takes no
3202 // cross-process lock and does not follow later commits.
3203 let mut db = Self::new_empty(
3204 fs,
3205 OpenOptions {
3206 repair_wal: false,
3207 auto_migrate: false,
3208 read_only: true,
3209 },
3210 );
3211 db.pinned = true; // read_only is set after replay, but pinning is immediate
3212 db.wal_horizon_floor = db.fs.read_horizon_floor()?;
3213 db.archive_genesis_chain = db.fs.has_genesis_marker();
3214 // Same orphaned-archive cleanup as open_with: floor was written first
3215 // during pruning, so a crash may have left stale archives below floor.
3216 db.cleanup_orphaned_archives()?;
3217 // Collect archive frames (oldest-first) and live WAL frames.
3218 // Archives represent pre-snapshot history; the snapshot captures the
3219 // cumulative state at the time of archiving. Crash-window guarantee:
3220 // A: crash before rename → WAL intact, no archive. Reopen: normal.
3221 // B: crash after rename, before new WAL → archive present, WAL
3222 // absent. Reopen: snapshot loaded (full state), no WAL replay.
3223 // C: crash after new baseline WAL written → normal post-archive.
3224 let archive_ns = db.fs.list_archives()?;
3225 let mut archive_frames_all: Vec<WalRecord> = Vec::new();
3226 for n in &archive_ns {
3227 let arc_bytes = db.fs.read_archive(*n)?;
3228 let (arc_frames, _) = decode_all(&arc_bytes);
3229 archive_frames_all.extend(arc_frames);
3230 }
3231 let total_archive_frames = archive_frames_all.len() as u64;
3232
3233 let live_bytes = db.fs.read(FileId::Wal)?;
3234 let (live_records, _valid_len) = decode_all(&live_bytes);
3235 let total_surviving = total_archive_frames + live_records.len() as u64;
3236 // Global total including any pruned history below the horizon floor.
3237 let total = db.wal_horizon_floor + total_surviving;
3238
3239 // Horizon and range check.
3240 if commit < db.wal_horizon_floor {
3241 return Err(GraphError::CommitOutOfRange {
3242 commit,
3243 total,
3244 floor: db.wal_horizon_floor,
3245 });
3246 }
3247 if commit >= total {
3248 return Err(GraphError::CommitOutOfRange {
3249 commit,
3250 total,
3251 floor: db.wal_horizon_floor,
3252 });
3253 }
3254
3255 // Local index into surviving frames (0 = first frame of oldest archive).
3256 let local = commit - db.wal_horizon_floor;
3257
3258 if local < total_archive_frames {
3259 // Target commit is in an archive. Correct replay from empty state
3260 // is only possible when the archive chain is an uninterrupted
3261 // genesis chain (first archive taken from a fresh store, no prior
3262 // WAL truncation) and no archives have been pruned (floor == 0).
3263 //
3264 // If either condition is violated the prefix needed to reconstruct
3265 // the requested state is gone; refuse rather than return wrong data.
3266 if db.wal_horizon_floor > 0 || !db.archive_genesis_chain {
3267 return Err(GraphError::CommitOutOfRange {
3268 commit,
3269 total,
3270 floor: db.wal_horizon_floor,
3271 });
3272 }
3273 // Replay all archive frames up to and including the target commit
3274 // from an empty database state. Archives must be replayed in order
3275 // so that dense-id intern tables are built up correctly.
3276 for rec in archive_frames_all.into_iter().take((local + 1) as usize) {
3277 db.apply(&rec)?;
3278 let _ = db.engine.drain_deltas();
3279 }
3280 } else {
3281 // Target commit is in the live WAL: load snapshot as base, then
3282 // replay the needed live WAL prefix.
3283 //
3284 // Base state: a truncating snapshot (wal_truncated=true) compacts
3285 // all pre-truncation / pre-archive commits. Dense-id records in
3286 // the live WAL reference ids/interns that the snapshot provides.
3287 // Peek 6 bytes (same pattern as open_with).
3288 let snap_header = db.fs.read_prefix(FileId::Snapshot, 6)?;
3289 let is_v8 = snap_header.len() >= 6
3290 && &snap_header[0..4] == b"GDB1"
3291 && core_storage::snapshot::is_mmap_container(u16::from_le_bytes([
3292 snap_header[4],
3293 snap_header[5],
3294 ]));
3295 if is_v8 {
3296 let state = if let Some(snap_path) = db.fs.snapshot_path() {
3297 let mapped = core_storage::v8::MappedBase::map(&snap_path).map_err(|e| {
3298 GraphError::Corrupt {
3299 detail: format!("v8: open_at mmap: {e:?}"),
3300 }
3301 })?;
3302 core_storage::snapshot::decode_v8_from_mapped(&mapped)?
3303 } else {
3304 let snap_bytes = db.fs.read(FileId::Snapshot)?;
3305 core_storage::snapshot::decode(&snap_bytes)?
3306 };
3307 if let Some(state) = state {
3308 if state.wal_truncated {
3309 db.restore_snapshot_state(state)?;
3310 }
3311 }
3312 } else if !snap_header.is_empty() {
3313 let snap_bytes = db.fs.read(FileId::Snapshot)?;
3314 if let Some(state) = core_storage::snapshot::decode(&snap_bytes)? {
3315 if state.wal_truncated {
3316 db.restore_snapshot_state(state)?;
3317 }
3318 }
3319 }
3320 // else: snap_header empty = no snapshot file.
3321 let live_local = local - total_archive_frames;
3322 for rec in live_records.into_iter().take((live_local + 1) as usize) {
3323 db.apply(&rec)?;
3324 let _ = db.engine.drain_deltas();
3325 }
3326 }
3327 // Pin: pending_delta_count must be 0 after as-of replay, mirroring T1's
3328 // post-loop assert in open_with.
3329 debug_assert_eq!(
3330 db.engine.pending_delta_count(),
3331 0,
3332 "pending_deltas non-empty after open_at replay — \
3333 per-frame drain must run inside the loop to keep memory O(1)"
3334 );
3335 let _ = db.engine.drain_deltas(); // belt-and-braces no-op
3336 // Rebuild view values after WAL replay so derived-edge-driven views
3337 // reflect the as-of state. open_at always uses the legacy path (no V8
3338 // base), so topo_view is always owned.
3339 {
3340 let topo_view = TopologyView::owned(&db.topo);
3341 db.view_store
3342 .rebuild_all(&mut db.props, &topo_view, &db.ids, &db.syms, &db.labels);
3343 }
3344 // Rebuild full-text index for as-of view (mirrors open_with pattern).
3345 db.fulltext.rebuild_all(
3346 &db.ids,
3347 &db.labels,
3348 &db.syms,
3349 build_props_view(&db.props, &db.base),
3350 );
3351 db.prop_index.rebuild_all(
3352 &db.ids,
3353 &db.labels,
3354 &db.syms,
3355 build_props_view(&db.props, &db.base),
3356 );
3357 // Namespaces on the temporal handle, built by the same pass the live
3358 // open uses, so an as-of mask narrows by the namespaces of that commit.
3359 db.rebuild_node_ns();
3360 // Load roles sidecar (current roles, not point-in-time).
3361 db.roles = Self::load_roles_from_fs(&db.fs)?;
3362 db.read_only = true;
3363 db.total_wal_commits = total;
3364 // Capture initial fold so reader() is immediately usable.
3365 db.fold_now();
3366 Ok(db)
3367 }
3368
3369 /// Whether this instance is a read-only as-of view.
3370 pub fn is_read_only(&self) -> bool {
3371 self.read_only
3372 }
3373
3374 // ── MVCC epoch reader ─────────────────────────────────────────────────────
3375
3376 /// Clone the current overlay state into a new `FrozenOverlay` and reset
3377 /// the delta tail. Called automatically every `FOLD_EVERY_K` commits and at
3378 /// the end of `open_with` / `open_at_with` to prime the reader.
3379 fn fold_now(&mut self) {
3380 let frozen = crate::reader::FrozenOverlay {
3381 ids: self.ids.clone(),
3382 syms: self.syms.clone(),
3383 topo: self.topo.clone(),
3384 props: self.props.clone(),
3385 labels: self.labels.clone(),
3386 edge_props: self.edge_props.clone(),
3387 roles: self.roles.clone(),
3388 fulltext: self.fulltext.clone(),
3389 };
3390 self.fold_overlay = Some(Arc::new(frozen));
3391 self.delta_tail.clear();
3392 self.commits_since_fold = 0;
3393 }
3394
3395 /// Capture a lock-free reader snapshot of the current db state.
3396 ///
3397 /// The read lock is held only for the duration of this call (to clone a
3398 /// handful of `Arc` handles). Subsequent query operations run without any
3399 /// lock.
3400 pub fn reader(&self) -> crate::reader::ReaderSnapshot {
3401 crate::reader::ReaderSnapshot::new(
3402 self.fold_overlay
3403 .clone()
3404 .expect("fold_overlay is always Some after open_with; call reader() after open"),
3405 self.base.clone(),
3406 self.delta_tail.clone(),
3407 // The snapshot's effective state is exactly this handle's state at
3408 // this commit, so it shares the memo and its version key.
3409 self.commit_seq,
3410 Arc::clone(&self.role_masks),
3411 )
3412 }
3413
3414 /// Append a delta the reader cannot apply, so that a corrupt overlay is
3415 /// reachable from a test.
3416 ///
3417 /// Compiled only under `test-hooks`, which the server's dev-dependency on
3418 /// this crate turns on. One call permanently corrupts every
3419 /// [`ReaderSnapshot`](crate::reader::ReaderSnapshot) taken from the handle,
3420 /// so it must not be in the published surface: `#[doc(hidden)]` hides it
3421 /// from rustdoc and from nothing else. The feature gate — not
3422 /// `#[cfg(test)]` — because its only callers are in `crates/server/tests`,
3423 /// a different crate, exactly as `core_rules`'s index counters are.
3424 ///
3425 /// [`ReaderSnapshot::effective`](crate::reader::ReaderSnapshot) folds the
3426 /// delta tail into a clone of the frozen overlay and answers
3427 /// [`GraphError::Corrupt`] when a record will not apply. Nothing a caller
3428 /// can do produces that state — `apply_one`'s failures are disagreements
3429 /// between the tail and the fold it is applied to, which the write path
3430 /// cannot create — so the `Corrupt` arm of every scoped reader method was
3431 /// reachable only by inspection until this hook existed. An `Intern` record
3432 /// claiming an id the frozen interner will not hand back is the smallest
3433 /// such disagreement.
3434 ///
3435 /// Only the tail is touched. This handle's own state is untouched and
3436 /// `commit_seq` does not move, so a role mask already memoised at this
3437 /// version stays memoised — which is exactly the state in which the HTTP
3438 /// role branches reach a scoped read with a corrupt overlay under them.
3439 #[cfg(any(test, feature = "test-hooks"))]
3440 #[doc(hidden)]
3441 pub fn push_unapplyable_delta_for_test(&mut self) {
3442 self.delta_tail.push(Arc::new(crate::reader::CommitDelta {
3443 records: vec![WalRecord::Intern {
3444 id: u32::MAX,
3445 text: "delta-tail-corruption".into(),
3446 }],
3447 derived_inserts: Vec::new(),
3448 derived_deletes: Vec::new(),
3449 }));
3450 }
3451
3452 /// Total number of WAL commits at the time [`open_at`] was called.
3453 /// Returns 0 for normal (non-as-of) instances.
3454 pub fn total_wal_commits(&self) -> u64 {
3455 self.total_wal_commits
3456 }
3457
3458 /// Apply a record to in-memory state. Used by both live writes and replay,
3459 /// so replay is definitionally identical to the original execution.
3460 fn apply(&mut self, rec: &WalRecord) -> Result<()> {
3461 // Before the record mutates anything: a store restored from a snapshot
3462 // defers building its candidate indexes until the first write, and that
3463 // build is a full node scan. Left where it used to fire — inside the
3464 // engine hook, after `props.set` and the label assignment — the scan
3465 // read the half-applied record and took the in-flight node's vector for
3466 // one the snapshot should have carried, which read as an interrupted
3467 // vector-index build and cost a full `RebuildRule` on the first
3468 // embedded write after every reopen. Hoisted here the scan sees exactly
3469 // the persisted state; the record's own hook then files its vector
3470 // through the ordinary insert path a line later.
3471 self.populate_indexes_before_write();
3472 match rec {
3473 WalRecord::InsertNode { label, key, props } => {
3474 let id = self.ids.try_insert(key)?;
3475 let sym = self.syms.intern(label);
3476 if self.labels.len() <= id as usize {
3477 // gap slots are sentinels, never valid label symbols
3478 self.labels.resize(id as usize + 1, u32::MAX);
3479 }
3480 self.labels[id as usize] = sym;
3481 let mut ns_name = NS_DEFAULT.to_string();
3482 for (field, value) in props {
3483 if field == NS_PROP {
3484 ns_name = namespace_of_value(Some(value)).to_string();
3485 }
3486 self.props.set(id, field, value.clone());
3487 }
3488 self.set_node_ns(id, &ns_name);
3489 // Initialize view values for the new node before the engine runs so
3490 // delta-based increments start from a known zero baseline.
3491 self.view_store
3492 .init_node_views(id, &mut self.props, &self.syms, &self.labels);
3493 // Fire rules for the newly inserted node.
3494 let cursor = self.engine.pending_delta_count();
3495 let mut eng = std::mem::take(&mut self.engine);
3496 {
3497 let mut gm = make_graph_mut(
3498 &self.ids,
3499 &mut self.syms,
3500 &self.labels,
3501 build_props_view(&self.props, &self.base),
3502 &mut self.topo,
3503 &self.base,
3504 &mut self.edge_props,
3505 );
3506 eng.on_node_changed(id, None, &mut gm);
3507 }
3508 self.engine = eng;
3509 // Process derived-edge deltas for view maintenance.
3510 // Fast path: skip the O(delta_count) allocation when no views exist.
3511 if !self.view_store.is_empty() {
3512 #[cfg(test)]
3513 DELTA_COPY_COUNT.with(|c| c.set(c.get() + 1));
3514 let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
3515 for d in &new_deltas {
3516 self.view_store.on_edge_changed(
3517 d.etype_sym,
3518 d.src_id,
3519 d.dst_id,
3520 d.fired,
3521 &mut self.props,
3522 &build_topo_view(&self.topo, &self.base),
3523 &self.ids,
3524 &self.syms,
3525 &self.labels,
3526 base_columns(&self.base),
3527 );
3528 }
3529 }
3530 // Full-text index maintenance: index enabled fields for this label.
3531 if self.fulltext.has_label(label) {
3532 for (field, value) in props {
3533 if self.fulltext.is_enabled(label, field) {
3534 self.fulltext.add_tokens(id, field, value);
3535 }
3536 }
3537 }
3538 // Property (equality) index maintenance.
3539 if self.prop_index.has_label(label) {
3540 for (field, value) in props {
3541 self.prop_index.set(label, field, id, value);
3542 }
3543 }
3544 }
3545 WalRecord::InsertEdge {
3546 edge_type,
3547 src_key,
3548 dst_key,
3549 } => {
3550 let src = self.ids.get(src_key).ok_or_else(|| GraphError::Corrupt {
3551 detail: format!("wal replay references unknown key {src_key}"),
3552 })?;
3553 let dst = self.ids.get(dst_key).ok_or_else(|| GraphError::Corrupt {
3554 detail: format!("wal replay references unknown key {dst_key}"),
3555 })?;
3556 let etype = self.syms.intern(edge_type);
3557 // Skip if the edge is already visible in the merged base+overlay
3558 // view. This keeps WAL replay idempotent when the WAL contains
3559 // pre-snapshot records that are already encoded in a V8 base
3560 // (keep_wal=true opens and crash-before-truncation scenarios).
3561 if self.base.is_some()
3562 && self
3563 .topo_view()
3564 .neighbors(etype, Direction::Out, src)
3565 .contains(&dst)
3566 {
3567 return Ok(());
3568 }
3569 self.topo.add_edge(etype, src, dst);
3570 // View maintenance for manual edge insert.
3571 self.view_store.on_edge_changed(
3572 etype,
3573 src,
3574 dst,
3575 true,
3576 &mut self.props,
3577 &build_topo_view(&self.topo, &self.base),
3578 &self.ids,
3579 &self.syms,
3580 &self.labels,
3581 base_columns(&self.base),
3582 );
3583 // Rule engine: via-hop rules must update when user edges change.
3584 let cursor = self.engine.pending_delta_count();
3585 let mut eng = std::mem::take(&mut self.engine);
3586 {
3587 let mut gm = make_graph_mut(
3588 &self.ids,
3589 &mut self.syms,
3590 &self.labels,
3591 build_props_view(&self.props, &self.base),
3592 &mut self.topo,
3593 &self.base,
3594 &mut self.edge_props,
3595 );
3596 eng.on_edge_changed(edge_type, src, dst, &mut gm);
3597 }
3598 self.engine = eng;
3599 if !self.view_store.is_empty() {
3600 let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
3601 for d in &new_deltas {
3602 self.view_store.on_edge_changed(
3603 d.etype_sym,
3604 d.src_id,
3605 d.dst_id,
3606 d.fired,
3607 &mut self.props,
3608 &build_topo_view(&self.topo, &self.base),
3609 &self.ids,
3610 &self.syms,
3611 &self.labels,
3612 base_columns(&self.base),
3613 );
3614 }
3615 }
3616 }
3617 WalRecord::SetProp { key, field, value } => {
3618 let id = self.ids.get(key).ok_or_else(|| GraphError::Corrupt {
3619 detail: format!("wal replay references unknown key {key}"),
3620 })?;
3621 let old_value = build_props_view(&self.props, &self.base)
3622 .get(id, field)
3623 .map(|vr| vr.into_value());
3624 self.props.set(id, field, value.clone());
3625 // Fire rules for the changed field.
3626 let cursor = self.engine.pending_delta_count();
3627 let mut eng = std::mem::take(&mut self.engine);
3628 {
3629 let mut gm = make_graph_mut(
3630 &self.ids,
3631 &mut self.syms,
3632 &self.labels,
3633 build_props_view(&self.props, &self.base),
3634 &mut self.topo,
3635 &self.base,
3636 &mut self.edge_props,
3637 );
3638 eng.on_node_changed(id, Some((field, old_value)), &mut gm);
3639 }
3640 self.engine = eng;
3641 // Derived-edge deltas → view updates.
3642 if !self.view_store.is_empty() {
3643 #[cfg(test)]
3644 DELTA_COPY_COUNT.with(|c| c.set(c.get() + 1));
3645 let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
3646 for d in &new_deltas {
3647 self.view_store.on_edge_changed(
3648 d.etype_sym,
3649 d.src_id,
3650 d.dst_id,
3651 d.fired,
3652 &mut self.props,
3653 &build_topo_view(&self.topo, &self.base),
3654 &self.ids,
3655 &self.syms,
3656 &self.labels,
3657 base_columns(&self.base),
3658 );
3659 }
3660 }
3661 // Neighbor-aggregate views that read `field` must also update.
3662 self.view_store.on_prop_changed(
3663 id,
3664 field,
3665 &mut self.props,
3666 &build_topo_view(&self.topo, &self.base),
3667 &self.ids,
3668 &self.syms,
3669 &self.labels,
3670 base_columns(&self.base),
3671 );
3672 // Full-text index maintenance: update tokens for this field if indexed.
3673 if self.fulltext.field_indexed(field) {
3674 let label_opt = self.labels.get(id as usize).and_then(|&sym| {
3675 if sym == u32::MAX {
3676 None
3677 } else {
3678 self.syms.resolve(sym)
3679 }
3680 });
3681 if let Some(label) = label_opt {
3682 if self.fulltext.is_enabled(label, field) {
3683 self.fulltext.remove_node_field(id, field);
3684 self.fulltext.add_tokens(id, field, value);
3685 }
3686 }
3687 }
3688 // Property (equality) index maintenance: re-key this node's value.
3689 if self.prop_index.field_indexed(field) {
3690 let label_opt = self.labels.get(id as usize).and_then(|&sym| {
3691 if sym == u32::MAX {
3692 None
3693 } else {
3694 self.syms.resolve(sym)
3695 }
3696 });
3697 if let Some(label) = label_opt {
3698 self.prop_index.set(label, field, id, value);
3699 }
3700 }
3701 }
3702 WalRecord::Intern { id, text } => {
3703 if let Some(existing) = self.syms.get(text) {
3704 if existing != *id {
3705 return Err(GraphError::Corrupt {
3706 detail: format!(
3707 "wal intern mismatch for {text:?}: have {existing}, record {id}"
3708 ),
3709 });
3710 }
3711 } else {
3712 let got = self.syms.intern(text);
3713 if got != *id {
3714 return Err(GraphError::Corrupt {
3715 detail: format!(
3716 "wal intern assigned {got} for {text:?}, record wanted {id}"
3717 ),
3718 });
3719 }
3720 }
3721 }
3722 WalRecord::InsertNodeId { label, key, props } => {
3723 let id = self.ids.try_insert(key)?;
3724 if self.labels.len() <= id as usize {
3725 self.labels.resize(id as usize + 1, u32::MAX);
3726 }
3727 self.labels[id as usize] = *label;
3728 let label_str = self
3729 .syms
3730 .resolve(*label)
3731 .ok_or_else(|| GraphError::Corrupt {
3732 detail: format!("wal InsertNodeId unknown label intern {label}"),
3733 })?
3734 .to_string();
3735 let mut ns_name = NS_DEFAULT.to_string();
3736 for (field_sym, value) in props {
3737 let field =
3738 self.syms
3739 .resolve(*field_sym)
3740 .ok_or_else(|| GraphError::Corrupt {
3741 detail: format!(
3742 "wal InsertNodeId unknown field intern {field_sym}"
3743 ),
3744 })?;
3745 if field == NS_PROP {
3746 ns_name = namespace_of_value(Some(value)).to_string();
3747 }
3748 self.props.set(id, field, value.clone());
3749 }
3750 self.set_node_ns(id, &ns_name);
3751 self.view_store
3752 .init_node_views(id, &mut self.props, &self.syms, &self.labels);
3753 let cursor = self.engine.pending_delta_count();
3754 let mut eng = std::mem::take(&mut self.engine);
3755 {
3756 let mut gm = make_graph_mut(
3757 &self.ids,
3758 &mut self.syms,
3759 &self.labels,
3760 build_props_view(&self.props, &self.base),
3761 &mut self.topo,
3762 &self.base,
3763 &mut self.edge_props,
3764 );
3765 eng.on_node_changed(id, None, &mut gm);
3766 }
3767 self.engine = eng;
3768 if !self.view_store.is_empty() {
3769 #[cfg(test)]
3770 DELTA_COPY_COUNT.with(|c| c.set(c.get() + 1));
3771 let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
3772 for d in &new_deltas {
3773 self.view_store.on_edge_changed(
3774 d.etype_sym,
3775 d.src_id,
3776 d.dst_id,
3777 d.fired,
3778 &mut self.props,
3779 &build_topo_view(&self.topo, &self.base),
3780 &self.ids,
3781 &self.syms,
3782 &self.labels,
3783 base_columns(&self.base),
3784 );
3785 }
3786 }
3787 if self.fulltext.has_label(&label_str) {
3788 for (field_sym, value) in props {
3789 let Some(field) = self.syms.resolve(*field_sym) else {
3790 continue;
3791 };
3792 if self.fulltext.is_enabled(&label_str, field) {
3793 self.fulltext.add_tokens(id, field, value);
3794 }
3795 }
3796 }
3797 if self.prop_index.has_label(&label_str) {
3798 for (field_sym, value) in props {
3799 let Some(field) = self.syms.resolve(*field_sym) else {
3800 continue;
3801 };
3802 self.prop_index.set(&label_str, field, id, value);
3803 }
3804 }
3805 }
3806 WalRecord::InsertEdgeId { etype, src, dst } => {
3807 // Replay-over-snapshot: dense ids in the pre-snapshot WAL may
3808 // already be tombstoned. Skip rather than attaching edges to
3809 // dead ids (DeleteNode keys the live re-insert, not the old id).
3810 if self.ids.is_tombstoned(*src)
3811 || self.ids.is_tombstoned(*dst)
3812 || self.ids.key_of(*src).is_none()
3813 || self.ids.key_of(*dst).is_none()
3814 {
3815 return Ok(());
3816 }
3817 // Skip if already visible in the merged view (same idempotency
3818 // guard as InsertEdge above: prevents double-counting when
3819 // pre-snapshot WAL records are replayed over a V8 base).
3820 if self.base.is_some()
3821 && self
3822 .topo_view()
3823 .neighbors(*etype, Direction::Out, *src)
3824 .contains(dst)
3825 {
3826 return Ok(());
3827 }
3828 self.topo.add_edge(*etype, *src, *dst);
3829 self.view_store.on_edge_changed(
3830 *etype,
3831 *src,
3832 *dst,
3833 true,
3834 &mut self.props,
3835 &build_topo_view(&self.topo, &self.base),
3836 &self.ids,
3837 &self.syms,
3838 &self.labels,
3839 base_columns(&self.base),
3840 );
3841 // Rule engine: via-hop rules fire when user via-edges are inserted.
3842 // Resolve etype back to string so on_edge_changed can match rules by name.
3843 if let Some(etype_str) = self.syms.resolve(*etype).map(|s| s.to_string()) {
3844 let cursor = self.engine.pending_delta_count();
3845 let mut eng = std::mem::take(&mut self.engine);
3846 {
3847 let mut gm = make_graph_mut(
3848 &self.ids,
3849 &mut self.syms,
3850 &self.labels,
3851 build_props_view(&self.props, &self.base),
3852 &mut self.topo,
3853 &self.base,
3854 &mut self.edge_props,
3855 );
3856 eng.on_edge_changed(&etype_str, *src, *dst, &mut gm);
3857 }
3858 self.engine = eng;
3859 if !self.view_store.is_empty() {
3860 let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
3861 for d in &new_deltas {
3862 self.view_store.on_edge_changed(
3863 d.etype_sym,
3864 d.src_id,
3865 d.dst_id,
3866 d.fired,
3867 &mut self.props,
3868 &build_topo_view(&self.topo, &self.base),
3869 &self.ids,
3870 &self.syms,
3871 &self.labels,
3872 base_columns(&self.base),
3873 );
3874 }
3875 }
3876 }
3877 }
3878 WalRecord::SetPropId { id, field, value } => {
3879 if self.ids.is_tombstoned(*id) || self.ids.key_of(*id).is_none() {
3880 return Ok(());
3881 }
3882 let field_str = self
3883 .syms
3884 .resolve(*field)
3885 .ok_or_else(|| GraphError::Corrupt {
3886 detail: format!("wal SetPropId unknown field intern {field}"),
3887 })?
3888 .to_string();
3889 let old_value = build_props_view(&self.props, &self.base)
3890 .get(*id, &field_str)
3891 .map(|vr| vr.into_value());
3892 self.props.set(*id, &field_str, value.clone());
3893 let cursor = self.engine.pending_delta_count();
3894 let mut eng = std::mem::take(&mut self.engine);
3895 {
3896 let mut gm = make_graph_mut(
3897 &self.ids,
3898 &mut self.syms,
3899 &self.labels,
3900 build_props_view(&self.props, &self.base),
3901 &mut self.topo,
3902 &self.base,
3903 &mut self.edge_props,
3904 );
3905 eng.on_node_changed(*id, Some((field_str.as_str(), old_value)), &mut gm);
3906 }
3907 self.engine = eng;
3908 if !self.view_store.is_empty() {
3909 #[cfg(test)]
3910 DELTA_COPY_COUNT.with(|c| c.set(c.get() + 1));
3911 let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
3912 for d in &new_deltas {
3913 self.view_store.on_edge_changed(
3914 d.etype_sym,
3915 d.src_id,
3916 d.dst_id,
3917 d.fired,
3918 &mut self.props,
3919 &build_topo_view(&self.topo, &self.base),
3920 &self.ids,
3921 &self.syms,
3922 &self.labels,
3923 base_columns(&self.base),
3924 );
3925 }
3926 }
3927 self.view_store.on_prop_changed(
3928 *id,
3929 &field_str,
3930 &mut self.props,
3931 &build_topo_view(&self.topo, &self.base),
3932 &self.ids,
3933 &self.syms,
3934 &self.labels,
3935 base_columns(&self.base),
3936 );
3937 if self.fulltext.field_indexed(&field_str) {
3938 let label_opt = self.labels.get(*id as usize).and_then(|&sym| {
3939 if sym == u32::MAX {
3940 None
3941 } else {
3942 self.syms.resolve(sym)
3943 }
3944 });
3945 if let Some(label) = label_opt {
3946 if self.fulltext.is_enabled(label, &field_str) {
3947 self.fulltext.remove_node_field(*id, &field_str);
3948 self.fulltext.add_tokens(*id, &field_str, value);
3949 }
3950 }
3951 }
3952 if self.prop_index.field_indexed(&field_str) {
3953 let label_opt = self.labels.get(*id as usize).and_then(|&sym| {
3954 if sym == u32::MAX {
3955 None
3956 } else {
3957 self.syms.resolve(sym)
3958 }
3959 });
3960 if let Some(label) = label_opt {
3961 self.prop_index.set(label, &field_str, *id, value);
3962 }
3963 }
3964 }
3965 WalRecord::CreateRule { def_bytes } => {
3966 let def: RuleDef = decode_rule_def(def_bytes).map_err(|e| GraphError::Corrupt {
3967 detail: format!("CreateRule def_bytes deserialize failed: {e}"),
3968 })?;
3969 // Replay-over-snapshot idempotency: the rule was captured in the snapshot
3970 // so the engine already has it; silently skip to avoid a spurious
3971 // RuleInvalid error in the crash window between snapshot write and WAL
3972 // truncation.
3973 if self.engine.rules().any(|r| r.name == def.name) {
3974 return Ok(());
3975 }
3976 let cursor = self.engine.pending_delta_count();
3977 let mut eng = std::mem::take(&mut self.engine);
3978 let result = {
3979 let mut gm = make_graph_mut(
3980 &self.ids,
3981 &mut self.syms,
3982 &self.labels,
3983 build_props_view(&self.props, &self.base),
3984 &mut self.topo,
3985 &self.base,
3986 &mut self.edge_props,
3987 );
3988 eng.create_rule(def, &mut gm)
3989 };
3990 self.engine = eng;
3991 result.map_err(|e| GraphError::RuleInvalid { detail: e })?;
3992 // Derived-edge fires from backfill → view updates.
3993 // Fast path: skip O(edge_count) allocation when no views exist.
3994 if !self.view_store.is_empty() {
3995 #[cfg(test)]
3996 DELTA_COPY_COUNT.with(|c| c.set(c.get() + 1));
3997 let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
3998 for d in &new_deltas {
3999 self.view_store.on_edge_changed(
4000 d.etype_sym,
4001 d.src_id,
4002 d.dst_id,
4003 d.fired,
4004 &mut self.props,
4005 &build_topo_view(&self.topo, &self.base),
4006 &self.ids,
4007 &self.syms,
4008 &self.labels,
4009 base_columns(&self.base),
4010 );
4011 }
4012 }
4013 }
4014 WalRecord::DeleteRule { name } => {
4015 // Replay-over-snapshot idempotency: the snapshot already captured the
4016 // post-delete state so the rule is absent; silently skip to avoid a
4017 // spurious RuleNotFound error in the crash window between snapshot write
4018 // and WAL truncation.
4019 if !self.engine.rules().any(|r| r.name == *name) {
4020 return Ok(());
4021 }
4022 let cursor = self.engine.pending_delta_count();
4023 let mut eng = std::mem::take(&mut self.engine);
4024 let result = {
4025 let mut gm = make_graph_mut(
4026 &self.ids,
4027 &mut self.syms,
4028 &self.labels,
4029 build_props_view(&self.props, &self.base),
4030 &mut self.topo,
4031 &self.base,
4032 &mut self.edge_props,
4033 );
4034 eng.delete_rule(name, &mut gm)
4035 };
4036 self.engine = eng;
4037 result.map_err(|_| GraphError::RuleNotFound { name: name.clone() })?;
4038 // Derived-edge retractions → view updates.
4039 if !self.view_store.is_empty() {
4040 #[cfg(test)]
4041 DELTA_COPY_COUNT.with(|c| c.set(c.get() + 1));
4042 let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
4043 for d in &new_deltas {
4044 self.view_store.on_edge_changed(
4045 d.etype_sym,
4046 d.src_id,
4047 d.dst_id,
4048 d.fired,
4049 &mut self.props,
4050 &build_topo_view(&self.topo, &self.base),
4051 &self.ids,
4052 &self.syms,
4053 &self.labels,
4054 base_columns(&self.base),
4055 );
4056 }
4057 }
4058 }
4059 WalRecord::RemoveProp { key, field } => {
4060 // Recovery-safe: unknown key or already-absent field is a
4061 // clean no-op. Crash-window replay over a snapshot that
4062 // already applied this record must not Err.
4063 let Some(id) = self.ids.get(key) else {
4064 return Ok(());
4065 };
4066 // Read old value through the seam for rule retraction.
4067 let old = build_props_view(&self.props, &self.base)
4068 .get(id, field)
4069 .map(|vr| vr.into_value());
4070 self.props.remove(id, field);
4071 // If the base still supplies the value after the overlay removal,
4072 // record a tombstone so ColumnsView::get does not resurrect it.
4073 // This covers both the base-only case AND the both-resident case:
4074 // base-only (in_overlay=false): old prop was only in base, remove
4075 // is a no-op on overlay, base still visible → tombstone needed.
4076 // both-resident (in_overlay=true): overlay had v2, base has v1;
4077 // removing overlay uncovers v1 → tombstone needed.
4078 // Idempotent on double-replay: second pass sees the tombstone →
4079 // get() returns None → condition is false → no duplicate tombstone.
4080 if build_props_view(&self.props, &self.base)
4081 .get(id, field)
4082 .is_some()
4083 {
4084 self.props.record_prop_tombstone(id, field);
4085 }
4086 let cursor = self.engine.pending_delta_count();
4087 let mut eng = std::mem::take(&mut self.engine);
4088 {
4089 let mut gm = make_graph_mut(
4090 &self.ids,
4091 &mut self.syms,
4092 &self.labels,
4093 build_props_view(&self.props, &self.base),
4094 &mut self.topo,
4095 &self.base,
4096 &mut self.edge_props,
4097 );
4098 eng.on_node_changed(id, Some((field, old)), &mut gm);
4099 }
4100 self.engine = eng;
4101 // Derived-edge deltas → view updates.
4102 if !self.view_store.is_empty() {
4103 #[cfg(test)]
4104 DELTA_COPY_COUNT.with(|c| c.set(c.get() + 1));
4105 let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
4106 for d in &new_deltas {
4107 self.view_store.on_edge_changed(
4108 d.etype_sym,
4109 d.src_id,
4110 d.dst_id,
4111 d.fired,
4112 &mut self.props,
4113 &build_topo_view(&self.topo, &self.base),
4114 &self.ids,
4115 &self.syms,
4116 &self.labels,
4117 base_columns(&self.base),
4118 );
4119 }
4120 }
4121 // Neighbor-aggregate views that read `field` must also update.
4122 self.view_store.on_prop_changed(
4123 id,
4124 field,
4125 &mut self.props,
4126 &build_topo_view(&self.topo, &self.base),
4127 &self.ids,
4128 &self.syms,
4129 &self.labels,
4130 base_columns(&self.base),
4131 );
4132 // Full-text index maintenance: remove tokens for this field.
4133 if self.fulltext.field_indexed(field) {
4134 self.fulltext.remove_node_field(id, field);
4135 }
4136 // Property (equality) index maintenance: drop this node's entry.
4137 if self.prop_index.field_indexed(field) {
4138 if let Some(label) = self.labels.get(id as usize).and_then(|&sym| {
4139 (sym != u32::MAX).then(|| self.syms.resolve(sym)).flatten()
4140 }) {
4141 self.prop_index.remove_node(label, field, id);
4142 }
4143 }
4144 }
4145 WalRecord::DeleteEdge {
4146 edge_type,
4147 src_key,
4148 dst_key,
4149 } => {
4150 // Recovery-safe: unknown keys, unknown etype, or already-
4151 // absent edge is a clean no-op (remove_edge returns false).
4152 let Some(src) = self.ids.get(src_key) else {
4153 return Ok(());
4154 };
4155 let Some(dst) = self.ids.get(dst_key) else {
4156 return Ok(());
4157 };
4158 let Some(etype) = self.syms.get(edge_type) else {
4159 return Ok(());
4160 };
4161 // I3: phantom-tombstone guard. When a V8 base is present, a
4162 // DeleteEdge WAL record for an edge that was already absorbed into
4163 // the new base (i.e. neither in overlay nor in base) must be skipped.
4164 // Without this guard, remove_edge records a tombstone for an edge
4165 // that no longer exists, incorrectly understating edge_count.
4166 if self.base.is_some()
4167 && !self
4168 .topo_view()
4169 .neighbors(etype, core_storage::topology::Direction::Out, src)
4170 .contains(&dst)
4171 {
4172 return Ok(());
4173 }
4174 self.topo.remove_edge(etype, src, dst);
4175 self.edge_props.remove_edge(etype, src, dst);
4176 // View maintenance for manual edge delete (topo already updated above).
4177 self.view_store.on_edge_changed(
4178 etype,
4179 src,
4180 dst,
4181 false,
4182 &mut self.props,
4183 &build_topo_view(&self.topo, &self.base),
4184 &self.ids,
4185 &self.syms,
4186 &self.labels,
4187 base_columns(&self.base),
4188 );
4189 // Rule engine: via-hop rules must retract when user via-edges are deleted.
4190 let cursor = self.engine.pending_delta_count();
4191 let mut eng = std::mem::take(&mut self.engine);
4192 {
4193 let mut gm = make_graph_mut(
4194 &self.ids,
4195 &mut self.syms,
4196 &self.labels,
4197 build_props_view(&self.props, &self.base),
4198 &mut self.topo,
4199 &self.base,
4200 &mut self.edge_props,
4201 );
4202 eng.on_edge_changed(edge_type, src, dst, &mut gm);
4203 }
4204 self.engine = eng;
4205 if !self.view_store.is_empty() {
4206 let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
4207 for d in &new_deltas {
4208 self.view_store.on_edge_changed(
4209 d.etype_sym,
4210 d.src_id,
4211 d.dst_id,
4212 d.fired,
4213 &mut self.props,
4214 &build_topo_view(&self.topo, &self.base),
4215 &self.ids,
4216 &self.syms,
4217 &self.labels,
4218 base_columns(&self.base),
4219 );
4220 }
4221 }
4222 }
4223 WalRecord::DeleteNode { key } => {
4224 // Recovery-safe: already-tombstoned / unknown key is a clean
4225 // no-op. Crash-window replay over a snapshot that already
4226 // applied this record cannot recover the retired id from the
4227 // key (`IdMap::get` is None), so every subsequent step is
4228 // skipped. Each step is independently idempotent if invoked
4229 // twice on a still-live id: retraction is a no-op on empty
4230 // provenance, `remove_edge` returns false, `remove_all` is a
4231 // no-op, `ids.delete` returns None, label sentinel is sticky.
4232 let Some(n) = self.ids.get(key) else {
4233 return Ok(());
4234 };
4235
4236 // (1) Retract derived edges + de-index while props/labels live.
4237 let cursor = self.engine.pending_delta_count();
4238 let mut eng = std::mem::take(&mut self.engine);
4239 {
4240 let mut gm = make_graph_mut(
4241 &self.ids,
4242 &mut self.syms,
4243 &self.labels,
4244 build_props_view(&self.props, &self.base),
4245 &mut self.topo,
4246 &self.base,
4247 &mut self.edge_props,
4248 );
4249 eng.on_node_removed(n, &mut gm);
4250 }
4251 self.engine = eng;
4252 // Derived-edge retractions → view updates for neighbors.
4253 if !self.view_store.is_empty() {
4254 #[cfg(test)]
4255 DELTA_COPY_COUNT.with(|c| c.set(c.get() + 1));
4256 let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
4257 for d in &new_deltas {
4258 self.view_store.on_edge_changed(
4259 d.etype_sym,
4260 d.src_id,
4261 d.dst_id,
4262 d.fired,
4263 &mut self.props,
4264 &build_topo_view(&self.topo, &self.base),
4265 &self.ids,
4266 &self.syms,
4267 &self.labels,
4268 base_columns(&self.base),
4269 );
4270 }
4271 }
4272
4273 // (2) Sweep ALL remaining edges incident to n, both directions,
4274 // every etype. This cascade is intentionally mask-independent:
4275 // topology integrity requires removing every edge touching the
4276 // deleted node regardless of the caller's visibility scope.
4277 // (The mask limits which nodes a role's read phase can return;
4278 // the WAL delete always executes with full storage authority.)
4279 // Collect then remove so neighbor slices stay valid during
4280 // iteration. Remove from topo first, then call view maintenance
4281 // so Avg/Min/Max recompute sees the correct (reduced) neighbor set.
4282 let etypes: Vec<u32> = self.topo.etypes().collect();
4283 let mut doomed = Vec::new();
4284 for et in &etypes {
4285 for &dst in self.topo.neighbors(*et, Direction::Out, n).as_ref() {
4286 doomed.push((*et, n, dst));
4287 }
4288 for &src in self.topo.neighbors(*et, Direction::In, n).as_ref() {
4289 doomed.push((*et, src, n));
4290 }
4291 }
4292 for (et, s, d) in doomed {
4293 self.topo.remove_edge(et, s, d);
4294 self.edge_props.remove_edge(et, s, d);
4295 // View maintenance: n's own view values will be cleared by
4296 // remove_all below; only update surviving neighbors.
4297 self.view_store.on_edge_changed(
4298 et,
4299 s,
4300 d,
4301 false,
4302 &mut self.props,
4303 &build_topo_view(&self.topo, &self.base),
4304 &self.ids,
4305 &self.syms,
4306 &self.labels,
4307 base_columns(&self.base),
4308 );
4309 }
4310
4311 // (3) Drop every remaining prop (`ColumnStore::remove_all`).
4312 self.props.remove_all(n);
4313 // Full-text index maintenance: remove all tokens for this node.
4314 self.fulltext.remove_node(n);
4315 // Property (equality) index maintenance: drop all entries for n.
4316 self.prop_index.remove_node_all(n);
4317
4318 // (4) Retire the dense id and stamp the label sentinel.
4319 self.ids.delete(key);
4320 if let Some(slot) = self.labels.get_mut(n as usize) {
4321 *slot = u32::MAX;
4322 }
4323 }
4324 WalRecord::Batch(inner) => {
4325 // Apply each inner record in order through the same apply path.
4326 // Inner records are validated free of nested Batch by encode_record.
4327 for rec in inner {
4328 self.apply(rec)?;
4329 }
4330 }
4331 WalRecord::RebuildRule { name } => {
4332 // Replay-over-snapshot idempotency: the snapshot may already
4333 // reflect a later delete_rule, so the rule is absent; skip.
4334 if !self.engine.rules().any(|r| r.name == *name) {
4335 return Ok(());
4336 }
4337 let cursor = self.engine.pending_delta_count();
4338 let mut eng = std::mem::take(&mut self.engine);
4339 let result = {
4340 let mut gm = make_graph_mut(
4341 &self.ids,
4342 &mut self.syms,
4343 &self.labels,
4344 build_props_view(&self.props, &self.base),
4345 &mut self.topo,
4346 &self.base,
4347 &mut self.edge_props,
4348 );
4349 eng.rebuild(name, &mut gm)
4350 };
4351 self.engine = eng;
4352 result.map_err(|_| GraphError::RuleNotFound { name: name.clone() })?;
4353 // Derived-edge delta changes → view updates.
4354 if !self.view_store.is_empty() {
4355 #[cfg(test)]
4356 DELTA_COPY_COUNT.with(|c| c.set(c.get() + 1));
4357 let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
4358 for d in &new_deltas {
4359 self.view_store.on_edge_changed(
4360 d.etype_sym,
4361 d.src_id,
4362 d.dst_id,
4363 d.fired,
4364 &mut self.props,
4365 &build_topo_view(&self.topo, &self.base),
4366 &self.ids,
4367 &self.syms,
4368 &self.labels,
4369 base_columns(&self.base),
4370 );
4371 }
4372 }
4373 }
4374 WalRecord::CreateView { def_bytes } => {
4375 let def: ViewDef =
4376 bincode::deserialize(def_bytes).map_err(|e| GraphError::Corrupt {
4377 detail: format!("CreateView def_bytes deserialize failed: {e}"),
4378 })?;
4379 // Replay-over-snapshot idempotency: view already present → skip.
4380 if self.view_store.has_view(&def.name) {
4381 return Ok(());
4382 }
4383 self.view_store
4384 .create_view(
4385 def,
4386 &mut self.props,
4387 &build_topo_view(&self.topo, &self.base),
4388 &self.ids,
4389 &self.syms,
4390 &self.labels,
4391 )
4392 .map_err(|e| GraphError::RuleInvalid { detail: e })?;
4393 }
4394 WalRecord::DeleteView { name } => {
4395 // Replay-over-snapshot idempotency: view already absent → skip.
4396 if !self.view_store.has_view(name) {
4397 return Ok(());
4398 }
4399 self.view_store
4400 .delete_view(name, &mut self.props, &self.ids, &self.labels, &self.syms)
4401 .map_err(|_| GraphError::RuleNotFound { name: name.clone() })?;
4402 }
4403 WalRecord::EnableFulltext { label, field } => {
4404 // Replay-over-snapshot idempotency: already enabled → skip.
4405 if self.fulltext.is_enabled(label, field) {
4406 return Ok(());
4407 }
4408 self.fulltext.enable(label, field);
4409 // Backfill: index all live nodes of this label that have the field.
4410 let n = self.ids.len() as u32;
4411 for id in 0..n {
4412 let Some(&sym) = self.labels.get(id as usize) else {
4413 continue;
4414 };
4415 if sym == u32::MAX {
4416 continue; // tombstoned
4417 }
4418 let Some(lbl) = self.syms.resolve(sym) else {
4419 continue;
4420 };
4421 if lbl != label {
4422 continue;
4423 }
4424 if let Some(value) = build_props_view(&self.props, &self.base)
4425 .get(id, field)
4426 .map(|vr| vr.into_value())
4427 {
4428 self.fulltext.add_tokens(id, field, &value);
4429 }
4430 }
4431 }
4432 WalRecord::DisableFulltext { label, field } => {
4433 // Replay-over-snapshot idempotency: already disabled → skip.
4434 if !self.fulltext.is_enabled(label, field) {
4435 return Ok(());
4436 }
4437 // If another label still indexes this field, the postings column
4438 // is kept — but it must not contain node_ids from the now-disabled
4439 // label. Remove them before calling disable() so the field_indexed
4440 // guard inside disable() sees the correct post-removal state.
4441 if self.fulltext.field_indexed_by_other(label, field) {
4442 if let Some(label_sym) = self.syms.get(label) {
4443 for (node_id, &lsym) in self.labels.iter().enumerate() {
4444 if lsym == label_sym {
4445 self.fulltext.remove_node_field(node_id as u32, field);
4446 }
4447 }
4448 }
4449 }
4450 self.fulltext.disable(label, field);
4451 }
4452 WalRecord::EnableIndex { label, field } => {
4453 // Replay-over-snapshot idempotency: already enabled → skip.
4454 if self.prop_index.is_enabled(label, field) {
4455 return Ok(());
4456 }
4457 self.prop_index.enable(label, field);
4458 // Backfill: index all live nodes of this label that have the field.
4459 let n = self.ids.len() as u32;
4460 for id in 0..n {
4461 let Some(&sym) = self.labels.get(id as usize) else {
4462 continue;
4463 };
4464 if sym == u32::MAX {
4465 continue; // tombstoned
4466 }
4467 let Some(lbl) = self.syms.resolve(sym) else {
4468 continue;
4469 };
4470 if lbl != label {
4471 continue;
4472 }
4473 if let Some(value) = build_props_view(&self.props, &self.base)
4474 .get(id, field)
4475 .map(|vr| vr.into_value())
4476 {
4477 self.prop_index.set(label, field, id, &value);
4478 }
4479 }
4480 }
4481 WalRecord::DisableIndex { label, field } => {
4482 self.prop_index.disable(label, field);
4483 }
4484 // ── insert-count multiplicity (§5.13) ────────────────────────────
4485 //
4486 // Two shapes, told apart by `count`: the opt-in declaration, and an
4487 // absolute count for one triple. Absolute is what makes this
4488 // idempotent over a snapshot base — a pre-snapshot frame replayed
4489 // over a base that already folded it in lands on the same number
4490 // rather than adding to it, which is the failure a delta (or a count
4491 // derived from `InsertEdgeId` records) would have.
4492 WalRecord::SetEdgeCount {
4493 etype,
4494 src,
4495 dst,
4496 count,
4497 } => {
4498 if rec.is_multiplicity_decl() {
4499 self.multiplicity = true;
4500 } else {
4501 self.edge_props.set(
4502 *etype,
4503 *src,
4504 *dst,
4505 EDGE_COUNT_PROP,
4506 Value::Int(*count as i64),
4507 );
4508 }
4509 }
4510 // History markers carry no replay state — rules re-derive edges
4511 // deterministically on open/replay. Skip unconditionally.
4512 WalRecord::DerivedEdgeAdded { .. } | WalRecord::DerivedEdgeRetracted { .. } => {}
4513 // ── rename_node ──────────────────────────────────────────────────
4514 WalRecord::RenameNode { old_key, new_key } => {
4515 // Recovery-safe: if old_key is already gone (key was renamed
4516 // by a snapshot or a prior replay frame), skip cleanly.
4517 if self.ids.get(old_key).is_none() {
4518 return Ok(());
4519 }
4520 // The rename only updates the key-table; the dense id, all
4521 // topo edges, props, labels, and rule state are id-indexed and
4522 // require no change.
4523 self.ids
4524 .rename(old_key, new_key)
4525 .map_err(|e| GraphError::Corrupt {
4526 detail: format!("wal replay RenameNode {old_key}→{new_key}: {e}"),
4527 })?;
4528 }
4529 }
4530 Ok(())
4531 }
4532
4533 /// Intern `s` in `syms` and emit a WAL `Intern` record so `*Id` records
4534 /// replay on WAL-only `open_at` (no snapshot intern table). Apply is
4535 /// idempotent when the string is already bound. Always emit: after
4536 /// `snapshot()` the WAL is truncated and live intern is not on disk.
4537 fn intern_wal(&mut self, s: &str) -> (u32, WalRecord) {
4538 let id = if let Some(id) = self.syms.get(s) {
4539 id
4540 } else {
4541 self.syms.intern(s)
4542 };
4543 (
4544 id,
4545 WalRecord::Intern {
4546 id,
4547 text: s.to_string(),
4548 },
4549 )
4550 }
4551
4552 /// Rewrite user-facing records into dense-id records. On `Err`, no live
4553 /// state is left mutated: speculative interns made while building the
4554 /// output are rolled back, so a later successful mutation cannot log an
4555 /// `Intern` record whose id replay would never reproduce.
4556 fn rewrite_wal_dense(&mut self, recs: Vec<WalRecord>) -> Result<Vec<WalRecord>> {
4557 self.rewrite_wal_dense_planned(recs.into_iter().map(PlannedRec::Rec).collect())
4558 }
4559
4560 /// [`rewrite_wal_dense`](Self::rewrite_wal_dense) for a frame that still
4561 /// carries [`PlannedRec::DuplicateCount`] entries — the shape a batch
4562 /// produces, where a duplicate's count can only be named once this pass has
4563 /// assigned the frame's own ids.
4564 fn rewrite_wal_dense_planned(&mut self, recs: Vec<PlannedRec>) -> Result<Vec<WalRecord>> {
4565 let syms_checkpoint = self.syms.len();
4566 let result = self.rewrite_wal_dense_inner(recs);
4567 if result.is_err() {
4568 self.syms.truncate(syms_checkpoint);
4569 }
4570 result
4571 }
4572
4573 fn rewrite_wal_dense_inner(&mut self, recs: Vec<PlannedRec>) -> Result<Vec<WalRecord>> {
4574 let mut out = Vec::with_capacity(recs.len());
4575 // Node ids allocated by later apply(InsertNodeId) in this same batch.
4576 let mut pending: std::collections::HashMap<String, u32> = std::collections::HashMap::new();
4577 // Namespace of each node inserted earlier in this same frame, so a SET
4578 // on a node this frame created is measured against the namespace it was
4579 // created in rather than against the store, where it does not exist yet.
4580 let mut pending_ns: std::collections::HashMap<String, String> =
4581 std::collections::HashMap::new();
4582 let mut interned = std::collections::HashSet::<u32>::new();
4583 let mut next = u32::try_from(self.ids.len()).map_err(|_| GraphError::Corrupt {
4584 detail: "id space exhausted".into(),
4585 })?;
4586 // Insert counts this frame has already raised. `edge_insert_count`
4587 // reads committed state, which cannot see a count queued earlier in
4588 // this same frame, so N duplicates of one pair would otherwise all
4589 // compute `committed + 1` and the last would win.
4590 let mut pending_counts: HashMap<(u32, u32, u32), u64> = HashMap::new();
4591 let lookup = |ids: &IdMap,
4592 pending: &std::collections::HashMap<String, u32>,
4593 key: &str|
4594 -> Option<u32> { ids.get(key).or_else(|| pending.get(key).copied()) };
4595 for rec in recs {
4596 // A duplicate insert's count, resolved here and nowhere else.
4597 //
4598 // This is the only pass that knows the frame's own ids: a node
4599 // created earlier in the same frame has no dense id until the
4600 // `InsertNodeId` above allocates one, and an edge type first used in
4601 // this frame is not in `syms` until `intern_wal` puts it there.
4602 // Resolving the count in the batch's validate pass instead — where
4603 // it used to live — meant that a duplicate whose endpoints or type
4604 // were created in the same frame silently produced no count at all,
4605 // which is exactly the shape a mirror rebuild writes (defect #24).
4606 let rec = match rec {
4607 PlannedRec::Rec(rec) => rec,
4608 PlannedRec::DuplicateCount {
4609 edge_type,
4610 src_key,
4611 dst_key,
4612 } => {
4613 let (etype, intern) = self.intern_wal(&edge_type);
4614 if interned.insert(etype) {
4615 out.push(intern);
4616 }
4617 let src = lookup(&self.ids, &pending, &src_key).ok_or_else(|| {
4618 GraphError::Corrupt {
4619 detail: format!("dense WAL rewrite missing src {src_key}"),
4620 }
4621 })?;
4622 let dst = lookup(&self.ids, &pending, &dst_key).ok_or_else(|| {
4623 GraphError::Corrupt {
4624 detail: format!("dense WAL rewrite missing dst {dst_key}"),
4625 }
4626 })?;
4627 let count = pending_counts
4628 .get(&(etype, src, dst))
4629 .copied()
4630 .unwrap_or_else(|| self.edge_insert_count(etype, src, dst))
4631 .saturating_add(1);
4632 pending_counts.insert((etype, src, dst), count);
4633 out.push(WalRecord::SetEdgeCount {
4634 etype,
4635 src,
4636 dst,
4637 count,
4638 });
4639 continue;
4640 }
4641 };
4642 match rec {
4643 WalRecord::InsertNode { label, key, props } => {
4644 // Namespace validation and normalisation, on the one seam
4645 // every user-visible node insert passes through: insert_node,
4646 // a batch, ingest, Cypher CREATE and MERGE all arrive here
4647 // before the WAL append, and replay never does.
4648 let (props, ns_name) = Self::normalise_insert_ns(&key, props)?;
4649 pending_ns.insert(key.clone(), ns_name);
4650 let (label_id, intern) = self.intern_wal(&label);
4651 if interned.insert(label_id) {
4652 out.push(intern);
4653 }
4654 let mut props_id = Vec::with_capacity(props.len());
4655 for (field, value) in props {
4656 let (field_id, intern) = self.intern_wal(&field);
4657 if interned.insert(field_id) {
4658 out.push(intern);
4659 }
4660 props_id.push((field_id, value));
4661 }
4662 if lookup(&self.ids, &pending, &key).is_none() {
4663 pending.insert(key.clone(), next);
4664 next = next.checked_add(1).ok_or_else(|| GraphError::Corrupt {
4665 detail: "id space exhausted".into(),
4666 })?;
4667 }
4668 out.push(WalRecord::InsertNodeId {
4669 label: label_id,
4670 key,
4671 props: props_id,
4672 });
4673 }
4674 WalRecord::SetProp { key, field, value } => {
4675 // A namespace is set at insert and fixed after: the write is
4676 // refused when it would move the node, and dropped when it
4677 // names the namespace the node is already in. Checked here
4678 // so set_prop, a batch, Cypher SET/MERGE and every upsert
4679 // that merges props get the same answer.
4680 if field == NS_PROP {
4681 let Value::Str(ref to) = value else {
4682 return Err(GraphError::RuleInvalid {
4683 detail: format!(
4684 "node {key}: {NS_PROP} must be a string naming a namespace, \
4685 got {value:?}"
4686 ),
4687 });
4688 };
4689 let from = pending_ns
4690 .get(&key)
4691 .cloned()
4692 .or_else(|| self.namespace_of(&key))
4693 .unwrap_or_else(|| NS_DEFAULT.to_string());
4694 let to = to.clone();
4695 if to != from {
4696 return Err(GraphError::NamespaceImmutable {
4697 key: key.clone(),
4698 from,
4699 to,
4700 });
4701 }
4702 continue;
4703 }
4704 let id =
4705 lookup(&self.ids, &pending, &key).ok_or_else(|| GraphError::Corrupt {
4706 detail: format!("dense WAL rewrite missing key {key}"),
4707 })?;
4708 let (field_id, intern) = self.intern_wal(&field);
4709 if interned.insert(field_id) {
4710 out.push(intern);
4711 }
4712 out.push(WalRecord::SetPropId {
4713 id,
4714 field: field_id,
4715 value,
4716 });
4717 }
4718 WalRecord::InsertEdge {
4719 edge_type,
4720 src_key,
4721 dst_key,
4722 } => {
4723 let (etype, intern) = self.intern_wal(&edge_type);
4724 if interned.insert(etype) {
4725 out.push(intern);
4726 }
4727 let src = lookup(&self.ids, &pending, &src_key).ok_or_else(|| {
4728 GraphError::Corrupt {
4729 detail: format!("dense WAL rewrite missing src {src_key}"),
4730 }
4731 })?;
4732 let dst = lookup(&self.ids, &pending, &dst_key).ok_or_else(|| {
4733 GraphError::Corrupt {
4734 detail: format!("dense WAL rewrite missing dst {dst_key}"),
4735 }
4736 })?;
4737 out.push(WalRecord::InsertEdgeId { etype, src, dst });
4738 }
4739 WalRecord::RenameNode {
4740 ref old_key,
4741 ref new_key,
4742 } => {
4743 // Track the rename in `pending` so subsequent InsertEdge /
4744 // SetProp records in this batch can resolve the new key.
4745 let id = lookup(&self.ids, &pending, old_key).ok_or_else(|| {
4746 GraphError::Corrupt {
4747 detail: format!(
4748 "dense WAL rewrite: RenameNode old key {old_key} not found"
4749 ),
4750 }
4751 })?;
4752 pending.remove(old_key.as_str());
4753 pending.insert(new_key.clone(), id);
4754 out.push(rec);
4755 }
4756 // # Symbol-order invariant (load-bearing)
4757 //
4758 // Write-time and replay-time symbol assignment must agree: every
4759 // symbol in a `Batch` frame has to receive the same dense id when
4760 // the frame's records are replayed in order as it received when
4761 // the frame was written.
4762 //
4763 // A rule's backfill interns its `edge_type` lazily
4764 // (`core_rules::engine`, every `g.syms.intern(&def.edge_type)`
4765 // site), and that backfill runs from `apply` — during the
4766 // `CreateRule` record itself, and again from any later
4767 // `InsertNodeId` in the same frame that makes the rule fire. At
4768 // write time the whole batch is rewritten before any of it is
4769 // applied, so a later `InsertEdge` in the same batch would win the
4770 // lower id for its edge type; on replay the rule's lazy intern
4771 // gets there first and steals it, and the `Intern` record fails at
4772 // the `wal intern assigned …` check in `apply`.
4773 //
4774 // Pre-interning the rule's `edge_type` here, and emitting its
4775 // `Intern` record ahead of the `CreateRule` record, makes both
4776 // orders identical. `weight_prop` needs no pre-intern:
4777 // `EdgeProps::set` keys props by `String`, never through the
4778 // interner. `via_edge` needs none either: via-hop rules resolve it
4779 // with `syms.get` and skip when it is absent.
4780 //
4781 // `RebuildRule` and `DeleteRule` need no such handling here:
4782 // `RebuildRule` has no `BatchOp` variant, so it never appears
4783 // inside a `Batch` today — it is only ever issued as its own
4784 // standalone commit (`rebuild_rule`, or the auto-rebuild path
4785 // that logs it as a second commit after the triggering op).
4786 // `DeleteRule` does have a `BatchOp` variant and can appear
4787 // inside a `Batch`, but it carries only a rule `name` — no
4788 // `edge_type` or other symbol that needs pre-interning — so
4789 // only `CreateRule` needs this arm.
4790 WalRecord::CreateRule { ref def_bytes } => {
4791 let def = decode_rule_def(def_bytes).map_err(|e| GraphError::Corrupt {
4792 detail: format!("CreateRule def_bytes deserialize failed: {e}"),
4793 })?;
4794 let (etype, intern) = self.intern_wal(&def.edge_type);
4795 if interned.insert(etype) {
4796 out.push(intern);
4797 }
4798 out.push(rec);
4799 }
4800 other => out.push(other),
4801 }
4802 }
4803 Ok(out)
4804 }
4805
4806 fn log_dense(&mut self, recs: Vec<WalRecord>) -> Result<()> {
4807 let recs = self.rewrite_wal_dense(recs)?;
4808 match recs.len() {
4809 0 => Ok(()),
4810 1 => self.log_then_apply(recs.into_iter().next().unwrap()),
4811 _ => self.log_then_apply(WalRecord::Batch(recs)),
4812 }
4813 }
4814
4815 /// Durable write, then notify the event sink. Replay (`apply` during
4816 /// `open`) never enters this function, so it is the replay-silent seam.
4817 fn log_then_apply(&mut self, rec: WalRecord) -> Result<()> {
4818 self.log_then_apply_with(rec, None, self.fsync)
4819 }
4820
4821 /// Whether this frame must fsync under `policy`.
4822 ///
4823 /// Batched contract: user-visible batches (>1 mutation) fsync; single
4824 /// mutations do not. The dense rewrite wraps a single mutation in a
4825 /// `Batch([Intern.., <one *Id record>])`, so `Intern` records are excluded
4826 /// from the count — removing that filter would make every single-op write
4827 /// fsync under Batched (or, if the threshold were raised instead, skip a
4828 /// needed fsync for real two-op batches).
4829 fn wal_needs_sync(policy: FsyncPolicy, rec: &WalRecord) -> bool {
4830 match policy {
4831 FsyncPolicy::Relaxed => false,
4832 FsyncPolicy::Strict => true,
4833 FsyncPolicy::Batched => match rec {
4834 // Intern + one mutation is the single-op rewrite, not a user batch.
4835 WalRecord::Batch(inner) => {
4836 inner
4837 .iter()
4838 .filter(|r| !matches!(r, WalRecord::Intern { .. }))
4839 .count()
4840 > 1
4841 }
4842 _ => false,
4843 },
4844 }
4845 }
4846
4847 /// # Apply-infallibility invariant (load-bearing)
4848 ///
4849 /// The ordering is: WAL append → fsync → apply. If `apply` returned `Err`
4850 /// for a `Batch` frame after a successful WAL write, the WAL would contain
4851 /// the full frame while in-memory state would reflect only the ops before
4852 /// the failure. On reopen, WAL replay would then apply the entire batch —
4853 /// diverging permanently from what the pre-crash process had in memory.
4854 ///
4855 /// For `Batch` frames this situation cannot arise because:
4856 /// - All validation runs via `commit_logged_batch`/`MutPreview` **before**
4857 /// the WAL write. `MutPreview` uses the same `&mut self` that apply will
4858 /// use, with no concurrent mutation between validation exit and apply entry.
4859 /// - Every `apply` arm for a validated op is either infallible by construction
4860 /// (`InsertNode`, `RemoveProp`, `DeleteEdge`, `DeleteNode`), has idempotency
4861 /// guards that return `Ok(())` (`CreateRule`, `DeleteRule`), or is
4862 /// guaranteed-present by validation (`InsertEdge`/`SetProp` key lookups).
4863 /// - `on_node_changed` and `on_node_removed` return `()` — never `Err`.
4864 ///
4865 /// A `debug_assert!` below fires in debug builds if `apply` ever returns
4866 /// `Err` for a `Batch` frame, making any future regression immediately visible
4867 /// in tests rather than silently diverging crash-recovery behaviour.
4868 fn log_then_apply_with(
4869 &mut self,
4870 rec: WalRecord,
4871 ingest: Option<(String, usize)>,
4872 policy: FsyncPolicy,
4873 ) -> Result<()> {
4874 // Read-only guard: as-of instances must never write the WAL.
4875 if self.read_only {
4876 return Err(GraphError::ReadOnly);
4877 }
4878 // Degraded guard: fsync failure left WAL truncated, or a refresh failed
4879 // partway; in-memory state is ahead of (or out of step with) the
4880 // on-disk WAL, so further mutations would deepen the divergence.
4881 // Reopen the database to recover. Checked before the lock guard: this
4882 // is the more serious condition and the more useful error.
4883 if self.degraded {
4884 return Err(GraphError::Io(std::io::Error::other(
4885 "database degraded after group-commit fsync failure; reopen required",
4886 )));
4887 }
4888 // Cross-process guard: this write scope asked for the store's write
4889 // lock and did not get it. Writing anyway would append frames on top of
4890 // a WAL another process is extending, so refuse instead.
4891 if self.lock_denied {
4892 return Err(GraphError::Busy { holder: None });
4893 }
4894 // Ensure retained provenance bytes are decoded into the live mutable
4895 // fields before any mutation touches self.engine.provenance. This is a
4896 // no-op if provenance was never stored (fresh store) or has already been
4897 // consumed (subsequent mutations). WAL replay calls apply() directly
4898 // and is covered by consume_retained_state_eager before replay.
4899 self.ensure_v8_base_sections_loaded();
4900 self.engine.ensure_provenance_loaded_mut();
4901 // Invariant (I-1): no stale deltas may enter from a previous apply.
4902 // If any engine method ever accumulates deltas before erroring, they would
4903 // contaminate the *next* commit's event stream. This assert fires in debug
4904 // builds, making any future regression visible at the earliest point.
4905 debug_assert_eq!(
4906 self.engine.pending_delta_count(),
4907 0,
4908 "stale engine deltas at log_then_apply_with entry — \
4909 a previous apply arm may have accumulated deltas before erroring; \
4910 the caller must drain_deltas() on any error path before returning"
4911 );
4912 let frame = encode_record(&rec);
4913 self.fs.append(FileId::Wal, &frame)?;
4914 // The cursor advances by exactly the bytes appended: these frames are
4915 // ours and already applied, so a later refresh must not replay them.
4916 self.wal_consumed += frame.len() as u64;
4917 if Self::wal_needs_sync(policy, &rec) {
4918 self.fs.sync(FileId::Wal)?;
4919 }
4920 // Marker writing always needs the engine deltas, but the engine only
4921 // accumulates them when emit_deltas is true (normally gated on subscribers
4922 // or views being present). Enable emission for this apply if it is
4923 // currently off, then restore the original state unconditionally via an
4924 // RAII guard — this prevents a panic in apply() from leaking the flag.
4925 // The same guard resets the engine's transient chaining state. A panic
4926 // unwinding out of a rule hook would otherwise leave `chain_depth`
4927 // non-zero, which makes every later `begin_chain` decide chaining is
4928 // already running and silently switch it off for good.
4929 struct RestoreEmitDeltas(*mut RuleEngine, bool);
4930 impl Drop for RestoreEmitDeltas {
4931 fn drop(&mut self) {
4932 // SAFETY: pointer into self (GraphDb); guard is dropped within
4933 // this frame before log_then_apply_with returns.
4934 unsafe {
4935 (*self.0).set_emit_deltas(self.1);
4936 (*self.0).reset_chain_state();
4937 }
4938 }
4939 }
4940 let original_emit = self.engine.emit_deltas();
4941 if !original_emit {
4942 self.engine.set_emit_deltas(true);
4943 }
4944 // SAFETY: raw pointer into self; guard dropped within this frame.
4945 let _emit_guard = RestoreEmitDeltas(&mut self.engine as *mut _, original_emit);
4946
4947 let apply_result = self.apply(&rec);
4948 // For Batch frames, post-validation apply must be infallible (see above).
4949 // A debug_assert here catches any future change that makes apply fallible
4950 // before the caller notices via silent WAL/memory divergence.
4951 if matches!(&rec, WalRecord::Batch(_)) {
4952 debug_assert!(
4953 apply_result.is_ok(),
4954 "Batch apply returned Err after successful WAL write — \
4955 the validate-then-apply invariant has been violated; \
4956 see log_then_apply_with invariant doc"
4957 );
4958 }
4959 if apply_result.is_err() {
4960 // Discard any partial deltas accumulated by the failed apply.
4961 // They must not ride the next commit's event stream (I-1).
4962 // _emit_guard restores emit_deltas on drop automatically.
4963 let _ = self.engine.drain_deltas();
4964 let _ = self.engine.take_rebuild_needed();
4965 apply_result?;
4966 }
4967 self.commit_seq += 1;
4968 let seq = self.commit_seq;
4969 // Update per-node last-change map for the committed record.
4970 // Must happen after commit_seq is incremented so the seq is correct.
4971 self.update_last_change_from_rec(&rec, seq);
4972 // Drain engine deltas and distribute to subscribers before the existing
4973 // MutationEvent sink fires — both happen post-fsync, post-apply.
4974 // _emit_guard restores emit_deltas after this line when it drops.
4975 let engine_deltas = self.engine.drain_deltas();
4976
4977 // Append history-marker WAL records for any derived-edge changes so
4978 // that `edge_history` and `was_linked` can surface rule-attributed
4979 // events. Markers are STATE NO-OPS during replay; they are written
4980 // without an additional fsync (the triggering commit's sync already
4981 // happened; the next commit's sync covers these lazily).
4982 if !engine_deltas.is_empty() {
4983 let markers: Vec<WalRecord> = engine_deltas
4984 .iter()
4985 .map(|d| {
4986 if d.fired {
4987 WalRecord::DerivedEdgeAdded {
4988 rule: d.rule.clone(),
4989 edge_type: d.edge_type.clone(),
4990 src_key: d.src_key.clone(),
4991 dst_key: d.dst_key.clone(),
4992 }
4993 } else {
4994 WalRecord::DerivedEdgeRetracted {
4995 rule: d.rule.clone(),
4996 edge_type: d.edge_type.clone(),
4997 src_key: d.src_key.clone(),
4998 dst_key: d.dst_key.clone(),
4999 }
5000 }
5001 })
5002 .collect();
5003 let marker_frame = if markers.len() == 1 {
5004 markers.into_iter().next().unwrap()
5005 } else {
5006 WalRecord::Batch(markers)
5007 };
5008 // Ignore append errors: markers are best-effort history
5009 // annotations. Losing them does not affect state correctness.
5010 // The cursor only advances when the bytes actually landed.
5011 let marker_bytes = encode_record(&marker_frame);
5012 if self.fs.append(FileId::Wal, &marker_bytes).is_ok() {
5013 self.wal_consumed += marker_bytes.len() as u64;
5014 }
5015 }
5016
5017 // Record MVCC CommitDelta for the epoch reader. The WAL record is
5018 // stored as-is (including any nested Batch / Intern records); the
5019 // ReaderSnapshot's apply_one function handles all variants.
5020 {
5021 let derived_inserts = engine_deltas
5022 .iter()
5023 .filter(|d| d.fired)
5024 .map(|d| (d.etype_sym, d.src_id, d.dst_id))
5025 .collect();
5026 let derived_deletes = engine_deltas
5027 .iter()
5028 .filter(|d| !d.fired)
5029 .map(|d| (d.etype_sym, d.src_id, d.dst_id))
5030 .collect();
5031 let delta = Arc::new(crate::reader::CommitDelta {
5032 records: vec![rec.clone()],
5033 derived_inserts,
5034 derived_deletes,
5035 });
5036 self.delta_tail.push(delta);
5037 self.commits_since_fold += 1;
5038 if self.commits_since_fold >= crate::reader::FOLD_EVERY_K {
5039 self.fold_now();
5040 }
5041 }
5042
5043 if self.defer_events {
5044 // Group-commit drain thread: hold events until after the group
5045 // fsync so subscribers only observe durable data (R2).
5046 self.deferred_events.push(DeferredEvent {
5047 rec: rec.clone(),
5048 engine_deltas,
5049 seq,
5050 ingest,
5051 });
5052 } else {
5053 self.distribute_events(&rec, &engine_deltas, seq);
5054 self.emit_committed(&rec, ingest);
5055 }
5056 // Drift is only known after apply, so auto-rebuild cannot join the
5057 // triggering op's WAL frame. Issue RebuildRule as a second commit.
5058 // Skip when `rec` is itself RebuildRule: rebuild resets drift, so a
5059 // retrigger loop is impossible if the fit succeeded, but we still
5060 // drain the flag so a leftover cannot re-enter.
5061 // One slice of any outstanding vector-index build rides here too, so a
5062 // store that is being written to finishes its build without anyone
5063 // calling `pump_index_build`. A rule that becomes whole joins the same
5064 // RebuildRule loop below.
5065 let mut rebuilds = self.engine.take_rebuild_needed();
5066 if !matches!(&rec, WalRecord::RebuildRule { .. }) {
5067 // Not after `CreateRule`: that record's own apply already did the
5068 // rule's first slice, and pumping again here would make one
5069 // `create_rule` call do two slices' work under one lock.
5070 // Nothing pending is the overwhelmingly common case and must cost
5071 // a map lookup, not an engine swap: a store being written to has
5072 // long since populated its indexes, so the `pump_index_build`
5073 // entry point owns the not-yet-populated case on its own.
5074 if !matches!(&rec, WalRecord::CreateRule { .. })
5075 && !self.engine.builds_in_progress().is_empty()
5076 {
5077 rebuilds.extend(self.pump_one_slice().into_iter().map(|b| b.rule));
5078 }
5079 let mut failed = Vec::new();
5080 for name in rebuilds {
5081 if self.engine.rules().any(|r| r.name == name) {
5082 // User op is already durable. A failed second commit must
5083 // not surface as the caller's error.
5084 if let Err(e) =
5085 self.log_then_apply(WalRecord::RebuildRule { name: name.clone() })
5086 {
5087 eprintln!(
5088 "auto-rebuild of rule {name:?} failed after durable user commit: {e}"
5089 );
5090 failed.push(name);
5091 }
5092 }
5093 }
5094 for name in failed {
5095 self.engine.queue_rebuild_needed(name);
5096 }
5097 }
5098 Ok(())
5099 }
5100
5101 /// Install a post-commit hook. Replaces any previous sink.
5102 ///
5103 /// The sink runs inside `log_then_apply` after a successful
5104 /// durable commit, while the caller still holds `&mut self`. When this
5105 /// database is behind a [`crate::SharedDb`], that means the **write
5106 /// guard is held**. The sink must never call `read` / `write` (or any
5107 /// other method) on the same `SharedDb` — the `RwLock` is not
5108 /// re-entrant and doing so deadlocks. The sink is `Send + Sync`;
5109 /// `std::sync::mpsc::Sender` is not `Sync` and will not type-check.
5110 /// Intended examples: `std::sync::mpsc::SyncSender`,
5111 /// `tokio::sync::mpsc::Sender`, `tokio::sync::broadcast::Sender`
5112 /// (non-blocking `send`), or `Arc<Mutex<Vec<MutationEvent>>>`.
5113 pub fn set_event_sink(&mut self, sink: Box<dyn Fn(MutationEvent) + Send + Sync>) {
5114 self.event_sink = Some(sink);
5115 }
5116
5117 /// Whether a post-commit event sink is currently installed.
5118 pub fn has_event_sink(&self) -> bool {
5119 self.event_sink.is_some()
5120 }
5121
5122 /// Set WAL fsync cadence. Default [`FsyncPolicy::Strict`].
5123 pub fn set_fsync_policy(&mut self, p: FsyncPolicy) {
5124 self.fsync = p;
5125 }
5126
5127 /// Return the current WAL fsync cadence.
5128 pub fn fsync_policy(&self) -> FsyncPolicy {
5129 self.fsync
5130 }
5131
5132 // ── Group-commit event deferral ───────────────────────────────────────────
5133
5134 /// Enable or disable deferred event mode.
5135 ///
5136 /// When `true`, event notifications (subscription `DbEvent`s and legacy
5137 /// `MutationEvent` sink calls) are buffered rather than fired immediately.
5138 /// Call [`flush_deferred_events`] after the group fsync to deliver them,
5139 /// or [`discard_deferred_events`] if the fsync failed and the group must
5140 /// be treated as lost.
5141 pub fn set_deferred_events_mode(&mut self, defer: bool) {
5142 self.defer_events = defer;
5143 }
5144
5145 /// Fire all buffered events accumulated since [`set_deferred_events_mode`]
5146 /// was set to true. Clears the buffer.
5147 ///
5148 /// Called by the drain thread AFTER a successful group fsync, so
5149 /// subscribers observe only data that is durably on disk.
5150 pub fn flush_deferred_events(&mut self) {
5151 let events = std::mem::take(&mut self.deferred_events);
5152 for de in events {
5153 self.distribute_events(&de.rec, &de.engine_deltas, de.seq);
5154 self.emit_committed(&de.rec, de.ingest);
5155 }
5156 }
5157
5158 /// Discard all buffered events without firing them.
5159 ///
5160 /// Called by the drain thread when a group fsync fails: the WAL has been
5161 /// truncated back to the pre-group offset, so the committed-but-unsynced
5162 /// ops must not be observable to subscribers.
5163 pub fn discard_deferred_events(&mut self) {
5164 self.deferred_events.clear();
5165 }
5166
5167 // ── Degraded state ────────────────────────────────────────────────────────
5168
5169 /// Mark this database as degraded.
5170 ///
5171 /// Called by the group-commit drain thread after a group fsync failure and
5172 /// WAL truncation: the in-memory state is now ahead of the on-disk WAL, so
5173 /// further mutations would deepen the divergence. All subsequent calls to
5174 /// [`log_then_apply_with`] return `Err` until the database is reopened.
5175 pub fn set_degraded(&mut self) {
5176 self.degraded = true;
5177 }
5178
5179 fn emit(&self, ev: MutationEvent) {
5180 if let Some(sink) = &self.event_sink {
5181 sink(ev);
5182 }
5183 }
5184
5185 fn emit_committed(&self, rec: &WalRecord, ingest: Option<(String, usize)>) {
5186 match rec {
5187 WalRecord::Batch(inner) => {
5188 for r in inner {
5189 if let Some(ev) = event_from_record(r, &self.syms, &self.ids) {
5190 self.emit(ev);
5191 }
5192 }
5193 match ingest {
5194 Some((label, inserted)) => {
5195 self.emit(MutationEvent::Ingested { label, inserted })
5196 }
5197 None => {
5198 let ops = inner
5199 .iter()
5200 .filter(|r| !matches!(r, WalRecord::Intern { .. }))
5201 .count();
5202 if ops > 1 {
5203 self.emit(MutationEvent::BatchApplied { ops });
5204 }
5205 }
5206 }
5207 }
5208 other => {
5209 if let Some(ev) = event_from_record(other, &self.syms, &self.ids) {
5210 self.emit(ev);
5211 }
5212 }
5213 }
5214 }
5215
5216 // -----------------------------------------------------------------------
5217 // Subscription API
5218 // -----------------------------------------------------------------------
5219
5220 /// Distribute post-commit events to all live subscribers.
5221 ///
5222 /// Build a row-key → row-data map from a [`ResultSet`].
5223 ///
5224 /// Each row is serialized to JSON to form its key; a debug fallback is used
5225 /// if serialization fails. Used by both the initial-seed path in
5226 /// [`Self::subscribe_query`] and the per-commit diff path in
5227 /// [`Self::distribute_events`] to keep the two in sync.
5228 fn result_to_row_map(
5229 result: &core_query::ResultSet,
5230 ) -> std::collections::HashMap<String, Vec<Option<Value>>> {
5231 (0..result.len())
5232 .map(|i| {
5233 let row = result.row(i).to_vec();
5234 let key = serde_json::to_string(&row).unwrap_or_else(|_| format!("{row:?}"));
5235 (key, row)
5236 })
5237 .collect()
5238 }
5239
5240 /// Collect the set of label syms touched by a WAL record.
5241 ///
5242 /// Returns `Some(set)` when every record in this commit can be attributed to
5243 /// a known label sym. Returns `None` when the commit must not be skipped:
5244 /// edge records, unresolvable key→label lookups, or any record type not in
5245 /// the explicit handled set.
5246 ///
5247 /// Handled record types and their actions:
5248 /// - `InsertNode` → look up label in interner (fails → None)
5249 /// - `InsertNodeId` → label sym is carried directly
5250 /// - `SetProp` → resolve key→id→label (fails → None)
5251 /// - `DeleteNode` → resolve key→id→label (fails → None)
5252 /// - `Batch` → recurse into every inner record
5253 /// - `InsertEdge`, `DeleteEdge`, `InsertEdgeId` → always None (edge records)
5254 /// - everything else → None (conservative)
5255 fn commit_touched_labels(
5256 rec: &WalRecord,
5257 syms: &Interner,
5258 ids: &IdMap,
5259 labels: &[u32],
5260 ) -> Option<BTreeSet<u32>> {
5261 let mut out = BTreeSet::new();
5262 if Self::collect_touched_labels(rec, syms, ids, labels, &mut out) {
5263 Some(out)
5264 } else {
5265 None
5266 }
5267 }
5268
5269 fn collect_touched_labels(
5270 rec: &WalRecord,
5271 syms: &Interner,
5272 ids: &IdMap,
5273 labels: &[u32],
5274 out: &mut BTreeSet<u32>,
5275 ) -> bool {
5276 match rec {
5277 // String-key insert: the dense rewrite converts this to
5278 // [Intern, InsertNodeId], so this arm fires only for legacy WAL
5279 // records written before the dense path was added.
5280 WalRecord::InsertNode { label, .. } => {
5281 if let Some(sym) = syms.get(label) {
5282 out.insert(sym);
5283 true
5284 } else {
5285 false
5286 }
5287 }
5288 // Dense-id insert (produced by rewrite_wal_dense for every
5289 // insert_node call in the current codebase).
5290 WalRecord::InsertNodeId { label, .. } => {
5291 out.insert(*label);
5292 true
5293 }
5294 // String-key prop set: dense path converts to [Intern, SetPropId].
5295 WalRecord::SetProp { key, .. } => {
5296 if let Some(sym) = Self::resolve_key_label_sym(key, ids, labels) {
5297 out.insert(sym);
5298 true
5299 } else {
5300 false
5301 }
5302 }
5303 // Dense-id prop set (produced by rewrite_wal_dense for set_prop).
5304 WalRecord::SetPropId { id, .. } => {
5305 if let Some(sym) = labels.get(*id as usize).copied().filter(|&s| s != u32::MAX) {
5306 out.insert(sym);
5307 true
5308 } else {
5309 false
5310 }
5311 }
5312 WalRecord::DeleteNode { key } => {
5313 if let Some(sym) = Self::resolve_key_label_sym(key, ids, labels) {
5314 out.insert(sym);
5315 true
5316 } else {
5317 false
5318 }
5319 }
5320 WalRecord::Batch(inner) => inner
5321 .iter()
5322 .all(|r| Self::collect_touched_labels(r, syms, ids, labels, out)),
5323 // Intern is a pure metadata record — it does not touch any node's
5324 // label and is safe to skip for the label-skip predicate.
5325 WalRecord::Intern { .. } => true,
5326 // Edge records: always re-execute (edges can change join results).
5327 WalRecord::InsertEdge { .. }
5328 | WalRecord::DeleteEdge { .. }
5329 | WalRecord::InsertEdgeId { .. } => false,
5330 _ => false,
5331 }
5332 }
5333
5334 /// Resolve a node key to its label sym via the dense id table.
5335 /// Returns `None` if the key is unknown or the label is a tombstone sentinel.
5336 fn resolve_key_label_sym(key: &str, ids: &IdMap, labels: &[u32]) -> Option<u32> {
5337 let id = ids.get(key)?;
5338 let sym = labels.get(id as usize).copied()?;
5339 (sym != u32::MAX).then_some(sym)
5340 }
5341
5342 /// Distribute post-commit events to all live subscribers.
5343 ///
5344 /// Called from `log_then_apply_with` after apply + fsync, before the
5345 /// legacy MutationEvent sink. Prunes dead `Weak` entries in-place.
5346 ///
5347 /// Query subscriptions (subscribe_query) re-execute their plan on every
5348 /// call and diff the result against the previous run. Zero overhead when
5349 /// no query subscriptions are active.
5350 fn distribute_events(&mut self, rec: &WalRecord, engine_deltas: &[EngineEdgeDelta], seq: u64) {
5351 if self.subscriptions.is_empty() && self.query_subscriptions.is_empty() {
5352 return;
5353 }
5354
5355 if !self.subscriptions.is_empty() {
5356 // Build write events from the WAL record.
5357 let write_events: Vec<DbEvent> =
5358 Self::write_events_from_record(rec, seq, &self.syms, &self.ids);
5359
5360 // Build edge events from engine deltas. Weight is looked up from
5361 // edge_props at distribution time (after apply), so it's always fresh.
5362 let edge_events: Vec<DbEvent> = engine_deltas
5363 .iter()
5364 .map(|d| {
5365 if d.fired {
5366 // The score lives under the rule's declared weight_prop,
5367 // which is not always the literal "weight".
5368 let prop = self
5369 .engine
5370 .rules()
5371 .find(|r| r.name == d.rule)
5372 .and_then(|r| r.weight_prop.as_deref());
5373 let weight = prop.and_then(|p| {
5374 self.edge_props
5375 .get(d.etype_sym, d.src_id, d.dst_id, p)
5376 .and_then(|v| {
5377 if let core_storage::Value::Float(f) = v {
5378 Some(*f)
5379 } else {
5380 None
5381 }
5382 })
5383 });
5384 DbEvent::EdgeFired {
5385 rule: d.rule.clone(),
5386 src_key: d.src_key.clone(),
5387 dst_key: d.dst_key.clone(),
5388 edge_type: d.edge_type.clone(),
5389 weight,
5390 commit_seq: seq,
5391 }
5392 } else {
5393 DbEvent::EdgeRetracted {
5394 rule: d.rule.clone(),
5395 src_key: d.src_key.clone(),
5396 dst_key: d.dst_key.clone(),
5397 edge_type: d.edge_type.clone(),
5398 commit_seq: seq,
5399 }
5400 }
5401 })
5402 .collect();
5403
5404 // Prune dead entries; push matching events to live ones.
5405 self.subscriptions.retain(|entry| {
5406 let Some(inner) = entry.inner.upgrade() else {
5407 return false;
5408 };
5409 for ev in &write_events {
5410 if event_matches(ev, &entry.filter) {
5411 inner.push(ev.clone());
5412 }
5413 }
5414 for ev in &edge_events {
5415 if event_matches(ev, &entry.filter) {
5416 inner.push(ev.clone());
5417 }
5418 }
5419 true
5420 });
5421
5422 // Turn off delta accumulation if all subscribers dropped and no views remain.
5423 if self.subscriptions.is_empty() && self.view_store.is_empty() {
5424 self.engine.set_emit_deltas(false);
5425 }
5426 }
5427
5428 // Query subscriptions: full re-run per commit, then diff rows.
5429 // IMPORTANT: full re-execution on every commit — use LIMIT to bound cost.
5430 // Differential evaluation is roadmap / Phase 5.
5431 if !self.query_subscriptions.is_empty() {
5432 // Take the list out so we can call self.view() without borrow conflict.
5433 let mut query_subs = std::mem::take(&mut self.query_subscriptions);
5434 let empty_params = BTreeMap::new();
5435 query_subs.retain_mut(|entry| {
5436 let Some(inner) = entry.inner.upgrade() else {
5437 return false; // subscriber dropped — prune
5438 };
5439 // Label-skip: if the plan has a known scan label and this commit
5440 // can be proven to touch only different labels (and no rule-derived
5441 // edge deltas fired), the result set cannot have changed — skip.
5442 if let Some(scan_sym) = entry.scan_label {
5443 if engine_deltas.is_empty() {
5444 let touched =
5445 Self::commit_touched_labels(rec, &self.syms, &self.ids, &self.labels);
5446 if touched.map(|t| !t.contains(&scan_sym)).unwrap_or(false) {
5447 return true; // safe to skip — result set unchanged
5448 }
5449 }
5450 }
5451 QUERY_SUB_EXECS_TL.with(|c| c.set(c.get() + 1));
5452 let result = match execute(&self.view(), &entry.ops, &Params(&empty_params)) {
5453 Ok(r) => r,
5454 Err(e) => {
5455 // Keep the subscription alive; skip the diff for this commit.
5456 // Re-run errors are transient (e.g., planner change) and
5457 // self-heal when the next commit succeeds.
5458 eprintln!("[mushroomdb] subscribe_query re-run failed: {e}");
5459 return true;
5460 }
5461 };
5462 // Build new row map: serialized-key → row data.
5463 let new_row_map = Self::result_to_row_map(&result);
5464 // Removed rows: in prev but not in new.
5465 for (key, row) in &entry.prev_row_map {
5466 if !new_row_map.contains_key(key) {
5467 inner.push(DbEvent::QueryRowRemoved {
5468 columns: entry.columns.clone(),
5469 row: row.clone(),
5470 });
5471 }
5472 }
5473 // Added rows: in new but not in prev.
5474 for (key, row) in &new_row_map {
5475 if !entry.prev_row_map.contains_key(key) {
5476 inner.push(DbEvent::QueryRowAdded {
5477 columns: entry.columns.clone(),
5478 row: row.clone(),
5479 });
5480 }
5481 }
5482 entry.prev_row_map = new_row_map;
5483 true
5484 });
5485 self.query_subscriptions = query_subs;
5486 }
5487 }
5488
5489 /// Returns `true` if any live subscriber or view definition requires delta
5490 /// accumulation. Used to set `engine.emit_deltas` on subscribe/view DDL.
5491 fn needs_emit_deltas(&self) -> bool {
5492 !self.view_store.is_empty()
5493 || self
5494 .subscriptions
5495 .iter()
5496 .any(|e| e.inner.upgrade().is_some())
5497 }
5498
5499 /// Convert a WAL record into `DbEvent` write events with the given seq.
5500 fn write_events_from_record(
5501 rec: &WalRecord,
5502 seq: u64,
5503 intern: &Interner,
5504 ids: &IdMap,
5505 ) -> Vec<DbEvent> {
5506 match rec {
5507 WalRecord::InsertNode { label, key, .. } => vec![DbEvent::NodeInserted {
5508 label: label.clone(),
5509 key: key.clone(),
5510 commit_seq: seq,
5511 }],
5512 // *Id arms run after a successful apply, so resolution can only
5513 // fail on a programming error. Skip the event rather than emit a
5514 // fabricated "" that clients can't tell from a real empty value
5515 // (mirrors event_from_record returning None).
5516 WalRecord::InsertNodeId { label, key, .. } => intern
5517 .resolve(*label)
5518 .map(|label| DbEvent::NodeInserted {
5519 label: label.to_string(),
5520 key: key.clone(),
5521 commit_seq: seq,
5522 })
5523 .into_iter()
5524 .collect(),
5525 WalRecord::SetProp { key, field, .. } => vec![DbEvent::PropSet {
5526 key: key.clone(),
5527 field: field.clone(),
5528 commit_seq: seq,
5529 }],
5530 WalRecord::SetPropId { id, field, .. } => ids
5531 .key_of(*id)
5532 .zip(intern.resolve(*field))
5533 .map(|(key, field)| DbEvent::PropSet {
5534 key: key.to_string(),
5535 field: field.to_string(),
5536 commit_seq: seq,
5537 })
5538 .into_iter()
5539 .collect(),
5540 WalRecord::RemoveProp { key, field } => vec![DbEvent::PropRemoved {
5541 key: key.clone(),
5542 field: field.clone(),
5543 commit_seq: seq,
5544 }],
5545 WalRecord::InsertEdge {
5546 edge_type,
5547 src_key,
5548 dst_key,
5549 } => vec![DbEvent::EdgeInserted {
5550 edge_type: edge_type.clone(),
5551 src: src_key.clone(),
5552 dst: dst_key.clone(),
5553 commit_seq: seq,
5554 }],
5555 WalRecord::InsertEdgeId { etype, src, dst } => (|| {
5556 Some(DbEvent::EdgeInserted {
5557 edge_type: intern.resolve(*etype)?.to_string(),
5558 src: ids.key_of(*src)?.to_string(),
5559 dst: ids.key_of(*dst)?.to_string(),
5560 commit_seq: seq,
5561 })
5562 })()
5563 .into_iter()
5564 .collect(),
5565 WalRecord::DeleteEdge {
5566 edge_type,
5567 src_key,
5568 dst_key,
5569 } => vec![DbEvent::EdgeDeleted {
5570 edge_type: edge_type.clone(),
5571 src: src_key.clone(),
5572 dst: dst_key.clone(),
5573 commit_seq: seq,
5574 }],
5575 WalRecord::DeleteNode { key } => vec![DbEvent::NodeDeleted {
5576 key: key.clone(),
5577 commit_seq: seq,
5578 }],
5579 WalRecord::Batch(inner) => inner
5580 .iter()
5581 .flat_map(|r| Self::write_events_from_record(r, seq, intern, ids))
5582 .collect(),
5583 WalRecord::CreateRule { .. }
5584 | WalRecord::DeleteRule { .. }
5585 | WalRecord::RebuildRule { .. }
5586 | WalRecord::CreateView { .. }
5587 | WalRecord::DeleteView { .. }
5588 | WalRecord::EnableFulltext { .. }
5589 | WalRecord::DisableFulltext { .. }
5590 | WalRecord::EnableIndex { .. }
5591 | WalRecord::DisableIndex { .. }
5592 | WalRecord::Intern { .. }
5593 // History markers produce no DbEvent — the engine delta already
5594 // fired the EdgeFired/EdgeRetracted subscription events.
5595 | WalRecord::DerivedEdgeAdded { .. }
5596 | WalRecord::DerivedEdgeRetracted { .. }
5597 // A count is not an edge event: the pair it counts already fired one
5598 // when it was first inserted.
5599 | WalRecord::SetEdgeCount { .. }
5600 | WalRecord::RenameNode { .. } => vec![],
5601 }
5602 }
5603
5604 /// Subscribe to edge-fire and edge-retract events for one named rule.
5605 ///
5606 /// Returns `Err(GraphError::RuleNotFound)` if `rule_name` is not
5607 /// currently registered. Dropping the returned [`Subscription`] handle
5608 /// unregisters the subscriber — no further events are queued, no
5609 /// resources leak.
5610 pub fn subscribe_rule(&mut self, rule_name: &str) -> core_storage::Result<Subscription> {
5611 if self.read_only {
5612 return Err(core_storage::GraphError::ReadOnly);
5613 }
5614 if !self.engine.rules().any(|r| r.name == rule_name) {
5615 return Err(core_storage::GraphError::RuleNotFound {
5616 name: rule_name.to_string(),
5617 });
5618 }
5619 let inner = SubInner::new(self.sub_capacity());
5620 self.subscriptions.push(SubEntry {
5621 filter: SubFilter::Rule(rule_name.to_string()),
5622 inner: std::sync::Arc::downgrade(&inner),
5623 });
5624 self.engine.set_emit_deltas(true);
5625 Ok(Subscription(inner))
5626 }
5627
5628 /// Subscribe to edge-fire and edge-retract events for **all** rules.
5629 ///
5630 /// Returns `Err(GraphError::ReadOnly)` if called on an as-of instance —
5631 /// as-of instances never commit, so `distribute_events` never runs and the
5632 /// subscription would never deliver events.
5633 pub fn subscribe_all_rules(&mut self) -> core_storage::Result<Subscription> {
5634 if self.read_only {
5635 return Err(core_storage::GraphError::ReadOnly);
5636 }
5637 let inner = SubInner::new(self.sub_capacity());
5638 self.subscriptions.push(SubEntry {
5639 filter: SubFilter::AllRules,
5640 inner: std::sync::Arc::downgrade(&inner),
5641 });
5642 self.engine.set_emit_deltas(true);
5643 Ok(Subscription(inner))
5644 }
5645
5646 /// Subscribe to write events: node insert/delete, prop set/remove.
5647 ///
5648 /// Does not include edge-fire / edge-retract (rule-derived edge events).
5649 ///
5650 /// Returns `Err(GraphError::ReadOnly)` if called on an as-of instance —
5651 /// as-of instances never commit, so `distribute_events` never runs and the
5652 /// subscription would never deliver events.
5653 pub fn subscribe_writes(&mut self) -> core_storage::Result<Subscription> {
5654 if self.read_only {
5655 return Err(core_storage::GraphError::ReadOnly);
5656 }
5657 let inner = SubInner::new(self.sub_capacity());
5658 self.subscriptions.push(SubEntry {
5659 filter: SubFilter::Writes,
5660 inner: std::sync::Arc::downgrade(&inner),
5661 });
5662 self.engine.set_emit_deltas(true);
5663 Ok(Subscription(inner))
5664 }
5665
5666 /// Subscribe to incremental Cypher query results.
5667 ///
5668 /// Parses and plans `cypher`; rejects the query if the plan is not in the
5669 /// allowlisted subset (see [`core_query::cypher::is_subscribable`]):
5670 /// - `MATCH (n:Label) WHERE … RETURN … [LIMIT n]`
5671 /// - `MATCH (a)-[r:TYPE]->(b) RETURN … [LIMIT n]` (exactly one hop)
5672 ///
5673 /// SKIP is not supported — it shifts the result window on every commit,
5674 /// causing spurious Added/Removed churn for rows whose data never changed.
5675 /// Multi-hop Expand chains are not supported; each additional MATCH clause
5676 /// widens scope beyond the documented single-scan / single-hop subset.
5677 ///
5678 /// After each successful commit, the plan is **fully re-executed** and the
5679 /// result is diffed against the previous run. Added rows produce
5680 /// [`DbEvent::QueryRowAdded`]; removed rows produce
5681 /// [`DbEvent::QueryRowRemoved`].
5682 ///
5683 /// **Full re-run per commit; use LIMIT to bound execution cost.**
5684 /// The existing 1 M intermediate-row cap applies. Differential evaluation
5685 /// is roadmap / Phase 5.
5686 ///
5687 /// Returns `Err(GraphError::ReadOnly)` if called on an as-of instance —
5688 /// as-of instances never commit, so `distribute_events` never runs and the
5689 /// subscription would never deliver events.
5690 ///
5691 /// Returns `Err(GraphError::QueryError)` if the query fails to parse, plan,
5692 /// or if the plan shape is not in the allowlist.
5693 pub fn subscribe_query(&mut self, cypher: &str) -> Result<Subscription> {
5694 if self.read_only {
5695 return Err(GraphError::ReadOnly);
5696 }
5697 let tokens = lex(cypher).map_err(|e| GraphError::QueryError {
5698 detail: format!("lex: {e}"),
5699 })?;
5700 let ast = parse(&tokens).map_err(|e| GraphError::QueryError {
5701 detail: format!("parse: {e}"),
5702 })?;
5703 let ops = plan(&ast).map_err(|e| GraphError::QueryError {
5704 detail: format!("plan: {e}"),
5705 })?;
5706 if !is_subscribable(&ops) {
5707 return Err(GraphError::QueryError {
5708 detail: "subscribe_query only supports allowlisted plan shapes: \
5709 MATCH (n:Label) WHERE … RETURN … [LIMIT n] or \
5710 MATCH (a)-[r:TYPE]->(b) RETURN … [LIMIT n] (exactly one hop). \
5711 Not supported: multi-hop Expand chains, SKIP (creates \
5712 unstable offset windows), ORDER BY, DISTINCT, aggregates, \
5713 variable-length paths, OPTIONAL MATCH, WITH, UNWIND. \
5714 Use LIMIT to bound re-execution cost."
5715 .to_string(),
5716 });
5717 }
5718 // Execute once to capture initial state (initial rows are not emitted as
5719 // events — the subscriber learns the baseline via the first query call).
5720 let empty_params = BTreeMap::new();
5721 let initial = execute(&self.view(), &ops, &Params(&empty_params)).map_err(|e| {
5722 GraphError::QueryError {
5723 detail: format!("execute: {e}"),
5724 }
5725 })?;
5726 let columns = initial.columns().to_vec();
5727 let prev_row_map = Self::result_to_row_map(&initial);
5728 let inner = SubInner::new(self.sub_capacity());
5729 // Derive the scan-label sym for the commit-skip fast-path. Any Expand op
5730 // or unrecognized leading scan → None (always re-execute).
5731 let scan_label = extract_scan_label(&ops, &mut self.syms);
5732 self.query_subscriptions.push(QuerySubEntry {
5733 ops,
5734 columns,
5735 prev_row_map,
5736 inner: std::sync::Arc::downgrade(&inner),
5737 scan_label,
5738 });
5739 Ok(Subscription(inner))
5740 }
5741
5742 /// Queue capacity used for new subscriptions.
5743 fn sub_capacity(&self) -> usize {
5744 self.sub_capacity
5745 }
5746
5747 /// Override per-subscriber queue capacity for subsequently created
5748 /// subscriptions on this db instance.
5749 ///
5750 /// Default is [`DEFAULT_SUB_CAPACITY`] (65,536 events). Use a smaller
5751 /// value in tests to exercise the [`DbEvent::Lagged`] path without
5752 /// generating tens of thousands of events.
5753 ///
5754 /// This is a test-support escape hatch. Calling it in production reduces
5755 /// subscriber reliability (more Lagged events). It is hidden from rustdoc
5756 /// to discourage accidental production use.
5757 #[doc(hidden)]
5758 pub fn set_sub_capacity(&mut self, capacity: usize) {
5759 self.sub_capacity = capacity;
5760 }
5761
5762 // -----------------------------------------------------------------------
5763
5764 /// Start an atomic batch.
5765 ///
5766 /// The returned [`BatchBuilder`] borrows `self` mutably until
5767 /// [`BatchBuilder::commit`]. Builder methods queue ops only — no
5768 /// validation, no WAL I/O. `commit` validates every queued op against
5769 /// live state plus preceding ops in this batch (duplicate key inside
5770 /// the batch is `Err`; an edge between two nodes created earlier in
5771 /// the batch is valid; `delete_node` then insert of the same key is a
5772 /// fresh identity). Validation never mutates the database. Any failure
5773 /// leaves WAL bytes and in-memory state identical to before `commit`.
5774 /// On success, one `WalRecord::Batch` frame is appended (one fsync)
5775 /// and each inner record is applied in order so rules fire per record.
5776 /// An empty batch, or a batch of only no-ops, writes zero WAL bytes.
5777 ///
5778 /// **Rule-window limitation:** batch validation cannot see edges that a
5779 /// rule created earlier in the *same* batch will derive at apply time, so
5780 /// a `delete_edge` / `insert_edge` in that window is silently no-oped
5781 /// where sequential calls would return `Err(RuleOwned)`. State integrity
5782 /// is unaffected (idempotent apply, provenance intact). Create rules in
5783 /// their own batch, or sequentially, when later ops may touch derived
5784 /// edges.
5785 pub fn batch(&mut self) -> BatchBuilder<'_, F> {
5786 BatchBuilder {
5787 db: self,
5788 ops: Vec::new(),
5789 }
5790 }
5791
5792 /// Closure-style atomic write batch.
5793 ///
5794 /// Equivalent to calling [`GraphDb::batch`], invoking `build` to queue ops,
5795 /// then committing. All ops queued inside `build` are validated in order and
5796 /// committed as a single `WalRecord::Batch` frame (one fsync). Rules fire
5797 /// once per inner record, in order, after commit — semantically identical to
5798 /// sequential single-op writes.
5799 ///
5800 /// **Error semantics — validate-then-apply.** `build` queues ops without
5801 /// touching the database. [`BatchBuilder::commit`] validates every op against
5802 /// live state plus earlier ops in this batch before writing anything. If op N
5803 /// fails validation (duplicate key, unknown key, rule-owned edge, …) the
5804 /// entire batch is rejected: no WAL bytes are written and no in-memory state
5805 /// changes. The database is identical to its state before `write_batch` was
5806 /// called.
5807 ///
5808 /// **Atomicity is crash-level, NOT isolation-level.** On replay after a crash,
5809 /// a partial (torn) `Batch` frame applies NONE of its ops — the frame is
5810 /// either fully applied or not at all. However, while applying a committed
5811 /// batch, concurrent readers may observe intermediate states as ops are applied
5812 /// sequentially in memory. There is no interactive transaction isolation in v1.
5813 /// This is documented as "crash-atomic write batches; no interactive
5814 /// transactions or read isolation."
5815 ///
5816 /// **Returns** `(nodes_inserted, edges_inserted)`. An empty or all-noop batch
5817 /// writes zero WAL bytes and returns `(0, 0)`.
5818 ///
5819 /// # Example
5820 ///
5821 /// ```rust,ignore
5822 /// let (nodes, edges) = db.write_batch(|b| {
5823 /// b.insert_node("Person", "alice", vec![("age".into(), Value::Int(30))]);
5824 /// b.insert_node("Person", "bob", vec![]);
5825 /// b.insert_edge("KNOWS", "alice", "bob");
5826 /// b.set_prop("alice", "role", Value::Str("admin".into()));
5827 /// b.delete_node("old_key");
5828 /// })?;
5829 /// // One fsync; on crash replay: all five ops land or none do.
5830 /// ```
5831 pub fn write_batch<C>(&mut self, build: C) -> Result<(usize, usize)>
5832 where
5833 C: FnOnce(&mut BatchBuilder<'_, F>),
5834 {
5835 let mut b = self.batch();
5836 build(&mut b);
5837 b.commit()
5838 }
5839
5840 /// Insert `rows` as nodes of `label`. One call is one atomic batch:
5841 /// auto-declared KeyMatch rules (if any) first, then the accepted node
5842 /// inserts, so incremental fire sees the new rules. Per-row key problems
5843 /// are collected in [`IngestReport::row_errors`] and skipped; a commit
5844 /// `Err` means nothing was applied.
5845 ///
5846 /// Auto-FK rule names are `auto_fk_<src_label_lowercase>_<field>` so
5847 /// distinct source labels sharing an FK field each get their own rule.
5848 pub fn ingest(
5849 &mut self,
5850 label: &str,
5851 rows: Vec<BTreeMap<String, Value>>,
5852 opts: &IngestOptions,
5853 ) -> Result<IngestReport> {
5854 self.ingest_with_edges(label, rows, opts, &[])
5855 }
5856
5857 /// [`ingest`] plus user edges in the **same** previewed WAL batch.
5858 /// A failing edge rejects the whole request; nothing is applied.
5859 pub fn ingest_with_edges(
5860 &mut self,
5861 label: &str,
5862 rows: Vec<BTreeMap<String, Value>>,
5863 opts: &IngestOptions,
5864 edges: &[(String, String, String)],
5865 ) -> Result<IngestReport> {
5866 crate::ingest::run(self, label, rows, opts, edges)
5867 }
5868
5869 /// Parse `json` as an array of objects and ingest via [`GraphDb::ingest`].
5870 ///
5871 /// JSON `null` fields are silently omitted (not stored, not a row error).
5872 /// Nested objects and arrays-of-objects are a per-row error (row skipped).
5873 /// Parse failures and a top-level value that is not an array of objects
5874 /// return [`GraphError::IngestError`].
5875 pub fn ingest_json(
5876 &mut self,
5877 label: &str,
5878 json: &str,
5879 opts: &IngestOptions,
5880 ) -> Result<IngestReport> {
5881 crate::ingest::run_json(self, label, json, opts)
5882 }
5883
5884 fn commit_logged_batch(
5885 &mut self,
5886 ops: Vec<BatchOp>,
5887 ingest: Option<(String, usize)>,
5888 // Two-source rule: write_batch_authz threads authz here directly (never
5889 // touches pending_write_authz); query_write_authz sets the field instead
5890 // and passes None. Only one source is non-None per call.
5891 param_authz: Option<WriteAuthz>,
5892 ) -> Result<BatchOutcome> {
5893 // Read-only guard: catches empty-batch calls before the early-return
5894 // that skips log_then_apply_with, ensuring all mutation entry points fail.
5895 if self.read_only {
5896 return Err(GraphError::ReadOnly);
5897 }
5898 // Ensure provenance is decoded before MutPreview accesses it
5899 // (note_delete_rule / is_rule_owned may call engine.provenance()).
5900 self.engine.ensure_provenance_loaded_mut();
5901
5902 // ── Authz pre-check ──────────────────────────────────────────────────
5903 // Evaluate the decision table per-op BEFORE MutPreview so that a denial
5904 // produces no WAL frame (all-or-nothing at the authz boundary extends
5905 // the existing validate-then-apply contract to role-scope checks).
5906 //
5907 // `batch_created` tracks key→label for nodes created by earlier ops in
5908 // THIS batch, so InsertEdgeUpsert can count same-batch placeholder nodes
5909 // as visible without needing to call `self.ids.get` on not-yet-committed
5910 // keys (they won't be there yet).
5911 //
5912 // Two-source rule: param_authz (write_batch_authz path) takes precedence;
5913 // fall back to self.pending_write_authz (query_write_authz/Cypher path).
5914 // Cloning the field copy avoids a simultaneous borrow of self.ids below.
5915 let authz_opt = param_authz.or_else(|| self.pending_write_authz.clone());
5916 if let Some(ref authz) = authz_opt {
5917 let mut batch_created: BTreeMap<String, String> = BTreeMap::new();
5918 for op in &ops {
5919 self.check_single_op_authz(authz, op, &batch_created)?;
5920 // Update batch_created after a passing authz check so that
5921 // subsequent ops in this batch see the nodes as "about to exist".
5922 match op {
5923 BatchOp::InsertNode { label, key, .. } => {
5924 // Only track genuinely new nodes (absent from the
5925 // snapshot at authz-check time). A pre-existing visible
5926 // key would be a DuplicateKey — not a real creation —
5927 // so MutPreview handles it. Letting it into batch_created
5928 // would allow a later SetProp to bypass update_labels
5929 // via the "batch-created → always updatable" ruling
5930 // (delete+recreate exploit, fix for I1 review round 2).
5931 //
5932 // Accepted edge: for a delete+recreate-with-different-
5933 // label batch, node_status resolves the pre-delete
5934 // (store) label for any subsequent update checks. This
5935 // grants no net-new capability — a role that can delete+
5936 // create can already place arbitrary props via
5937 // InsertNode's own props field.
5938 if self.ids.get(key.as_str()).is_none() {
5939 batch_created.insert(key.clone(), label.clone());
5940 }
5941 }
5942 BatchOp::InsertEdgeUpsert {
5943 placeholder_label,
5944 src_key,
5945 dst_key,
5946 ..
5947 } => {
5948 // Both endpoints will be created if not already in store.
5949 for ep_key in [src_key, dst_key] {
5950 if self.ids.get(ep_key.as_str()).is_none()
5951 && !batch_created.contains_key(ep_key.as_str())
5952 {
5953 batch_created.insert(ep_key.clone(), placeholder_label.clone());
5954 }
5955 }
5956 }
5957 _ => {}
5958 }
5959 }
5960 }
5961
5962 let mut outcome = BatchOutcome::default();
5963 let recs = {
5964 let mut preview = MutPreview::new(self);
5965 let mut recs = Vec::with_capacity(ops.len());
5966 // Which node row we are on, counted over the node-insert ops only.
5967 // A caller that queues its rows in order reads this straight back
5968 // as the index into its own list.
5969 let mut node_row = 0usize;
5970 // Every field name the store knows, which a `Replace` needs to work
5971 // out what it removes. Resolved on the first `Replace` in the frame
5972 // and reused, so N replaces read the field list once, not N times.
5973 let mut store_fields: Option<Vec<String>> = None;
5974 // Duplicate inserts this frame has to count, each paired with the
5975 // position in `recs` it belongs at. The count itself is named in the
5976 // dense rewrite and not here: a duplicate's endpoints and edge type
5977 // may all be created by earlier ops in this same frame, and nothing
5978 // in the frame has a dense id yet. See [`PlannedRec`].
5979 let mut deferred_counts: Vec<(usize, String, String, String)> = Vec::new();
5980 for op in ops {
5981 match op {
5982 BatchOp::InsertNode { label, key, props } => {
5983 node_row += 1;
5984 preview.check_insert_node(&key, &props)?;
5985 preview.note_insert_node(&label, &key, &props);
5986 recs.push(WalRecord::InsertNode { label, key, props });
5987 }
5988 BatchOp::InsertNodeOnConflict {
5989 label,
5990 key,
5991 props,
5992 on_conflict,
5993 } => {
5994 let row = node_row;
5995 node_row += 1;
5996 if !preview.has_key(&key) {
5997 // No conflict: an ordinary insert on any policy —
5998 // except that a supplied view-owned field is the
5999 // same mistake here as on a taken key, and gets the
6000 // same row error rather than a frame error. Without
6001 // this, one op answered one request two ways
6002 // depending on whether the store already had the
6003 // key (defect #19).
6004 if let Some(why) = preview.supplied_view_owned_prop(&key, &props) {
6005 outcome.row_errors.push((row, why));
6006 continue;
6007 }
6008 preview.note_insert_node(&label, &key, &props);
6009 recs.push(WalRecord::InsertNode { label, key, props });
6010 continue;
6011 }
6012 match on_conflict {
6013 OnConflict::Error => {
6014 return Err(GraphError::DuplicateKey { key });
6015 }
6016 OnConflict::Skip => outcome.skipped += 1,
6017 OnConflict::Replace => {
6018 if store_fields.is_none() {
6019 store_fields = Some(preview.db.props_view().field_names());
6020 }
6021 let fields = store_fields.as_deref().unwrap_or_default();
6022 match preview.plan_replace(&label, &key, &props, fields) {
6023 Ok((writes, kept_view_owned)) => {
6024 outcome.kept_view_owned += kept_view_owned;
6025 for (field, value) in writes {
6026 match value {
6027 Some(value) => {
6028 preview.note_set_prop(&key, &field, &value);
6029 recs.push(WalRecord::SetProp {
6030 key: key.clone(),
6031 field,
6032 value,
6033 });
6034 }
6035 None => {
6036 preview.note_remove_prop(&key, &field);
6037 recs.push(WalRecord::RemoveProp {
6038 key: key.clone(),
6039 field,
6040 });
6041 }
6042 }
6043 }
6044 outcome.replaced += 1;
6045 }
6046 Err(why) => outcome.row_errors.push((row, why)),
6047 }
6048 }
6049 }
6050 }
6051 BatchOp::InsertEdge {
6052 edge_type,
6053 src_key,
6054 dst_key,
6055 } => {
6056 if preview.prepare_insert_edge(&edge_type, &src_key, &dst_key)? {
6057 preview.note_insert_edge(&edge_type, &src_key, &dst_key);
6058 recs.push(WalRecord::InsertEdge {
6059 edge_type,
6060 src_key,
6061 dst_key,
6062 });
6063 } else if preview.db.multiplicity {
6064 // A duplicate inside a batch counts the way a
6065 // duplicate through `insert_edge` does: `ingest` and
6066 // Cypher `CREATE` reach this choke-point and not
6067 // that one, and a count only one entry point keeps
6068 // would be worse than no count at all.
6069 //
6070 // This is the one gate on discriminant 23 from the
6071 // batch path: a store that never opted in queues
6072 // nothing here and so writes no such record.
6073 deferred_counts.push((recs.len(), edge_type, src_key, dst_key));
6074 }
6075 }
6076 BatchOp::SetProp { key, field, value } => {
6077 if let Some(view_name) = preview.db.view_store.view_for_prop(&field) {
6078 return Err(GraphError::ViewPropReadOnly {
6079 view_name: view_name.to_string(),
6080 });
6081 }
6082 preview.check_live_key(&key)?;
6083 preview.note_set_prop(&key, &field, &value);
6084 recs.push(WalRecord::SetProp { key, field, value });
6085 }
6086 BatchOp::RemoveProp { key, field } => {
6087 if preview.prepare_remove_prop(&key, &field)? {
6088 preview.note_remove_prop(&key, &field);
6089 recs.push(WalRecord::RemoveProp { key, field });
6090 }
6091 }
6092 BatchOp::DeleteEdge {
6093 edge_type,
6094 src_key,
6095 dst_key,
6096 } => {
6097 if preview.prepare_delete_edge(&edge_type, &src_key, &dst_key)? {
6098 preview.note_delete_edge(&edge_type, &src_key, &dst_key);
6099 recs.push(WalRecord::DeleteEdge {
6100 edge_type,
6101 src_key,
6102 dst_key,
6103 });
6104 }
6105 }
6106 BatchOp::DeleteNode { key } => {
6107 preview.check_live_key(&key)?;
6108 preview.note_delete_node(&key);
6109 recs.push(WalRecord::DeleteNode { key });
6110 }
6111 BatchOp::CreateRule(def) => {
6112 preview.check_create_rule(&def)?;
6113 let def_bytes =
6114 bincode::serialize(&def).map_err(|e| GraphError::Corrupt {
6115 detail: format!("serialize rule: {e}"),
6116 })?;
6117 preview.note_create_rule(&def);
6118 recs.push(WalRecord::CreateRule { def_bytes });
6119 }
6120 BatchOp::DeleteRule { name } => {
6121 preview.check_delete_rule(&name)?;
6122 preview.note_delete_rule(&name);
6123 recs.push(WalRecord::DeleteRule { name });
6124 }
6125 BatchOp::RenameNode { old_key, new_key } => {
6126 preview.check_rename_node(&old_key, &new_key)?;
6127 preview.note_rename_node(&old_key, &new_key);
6128 recs.push(WalRecord::RenameNode { old_key, new_key });
6129 }
6130 BatchOp::InsertEdgeUpsert {
6131 edge_type,
6132 src_key,
6133 dst_key,
6134 placeholder_label,
6135 } => {
6136 // Auto-create any missing endpoints as plain InsertNode ops.
6137 // Rules fire and last-change is updated for each created node.
6138 for key in [&src_key, &dst_key] {
6139 if !preview.has_key(key) {
6140 // A placeholder endpoint carries no props, so
6141 // the view-owned check has nothing to refuse.
6142 preview.check_insert_node(key, &[])?;
6143 preview.note_insert_node(&placeholder_label, key, &[]);
6144 recs.push(WalRecord::InsertNode {
6145 label: placeholder_label.clone(),
6146 key: key.clone(),
6147 props: vec![],
6148 });
6149 }
6150 }
6151 if preview.prepare_insert_edge(&edge_type, &src_key, &dst_key)? {
6152 preview.note_insert_edge(&edge_type, &src_key, &dst_key);
6153 recs.push(WalRecord::InsertEdge {
6154 edge_type,
6155 src_key,
6156 dst_key,
6157 });
6158 } else if preview.db.multiplicity {
6159 // Same choke-point, same gate as `BatchOp::InsertEdge`
6160 // above: an upsert that finds the pair already there
6161 // is a duplicate insert and counts as one.
6162 deferred_counts.push((recs.len(), edge_type, src_key, dst_key));
6163 }
6164 }
6165 }
6166 }
6167 // Splice the deferred counts back into the positions they were
6168 // raised at, so a count still sits exactly where the duplicate did
6169 // — before any later op in the frame that deletes the pair.
6170 let mut planned: Vec<PlannedRec> =
6171 Vec::with_capacity(recs.len() + deferred_counts.len());
6172 let mut deferred = deferred_counts.into_iter().peekable();
6173 for (i, rec) in recs.into_iter().enumerate() {
6174 while deferred.peek().is_some_and(|(at, ..)| *at == i) {
6175 let (_, edge_type, src_key, dst_key) = deferred.next().expect("just peeked");
6176 planned.push(PlannedRec::DuplicateCount {
6177 edge_type,
6178 src_key,
6179 dst_key,
6180 });
6181 }
6182 planned.push(PlannedRec::Rec(rec));
6183 }
6184 for (_, edge_type, src_key, dst_key) in deferred {
6185 planned.push(PlannedRec::DuplicateCount {
6186 edge_type,
6187 src_key,
6188 dst_key,
6189 });
6190 }
6191 planned
6192 };
6193 // A frame that is nothing but skips or refused rows writes no WAL, but
6194 // it still has counts to report, so the early returns carry `outcome`
6195 // rather than zeros.
6196 if recs.is_empty() {
6197 return Ok(outcome);
6198 }
6199 // rewrite_wal_dense converts every InsertNode/InsertEdge into its
6200 // *Id form, so only the dense variants can appear in `recs` here.
6201 let recs = self.rewrite_wal_dense_planned(recs)?;
6202 // The rewrite can empty a non-empty batch: a `SET n.ns` naming the
6203 // namespace the node is already in is a no-op and is dropped there. An
6204 // empty `Batch` frame would still take a commit sequence and a WAL
6205 // record, so a batch that turns out to be nothing writes nothing.
6206 if recs.is_empty() {
6207 return Ok(outcome);
6208 }
6209 outcome.nodes_inserted = recs
6210 .iter()
6211 .filter(|r| matches!(r, WalRecord::InsertNodeId { .. }))
6212 .count();
6213 outcome.edges_inserted = recs
6214 .iter()
6215 .filter(|r| matches!(r, WalRecord::InsertEdgeId { .. }))
6216 .count();
6217 // Ingest / write_batch / query_write: one Batch frame, one fsync per call
6218 // under Strict. Pass self.fsync directly so Strict stays Strict —
6219 // wal_needs_sync(Strict, _) always returns true regardless of op count.
6220 // Mapping Strict → Batched (the prior bug) caused wal_needs_sync to
6221 // short-circuit on single-op batches and silently skip the fsync.
6222 // Batched fsyncs only for multi-op batches; Relaxed always skips.
6223 self.log_then_apply_with(WalRecord::Batch(recs), ingest, self.fsync)?;
6224 Ok(outcome)
6225 }
6226
6227 fn commit_batch(&mut self, ops: Vec<BatchOp>) -> Result<(usize, usize)> {
6228 self.commit_logged_batch(ops, None, None).map(inserted_pair)
6229 }
6230
6231 /// Commit one submission WITHOUT an fsync — for use inside `commit_group`
6232 /// and the group-commit drain thread, which do a single group fsync later.
6233 fn commit_batch_nosync(&mut self, ops: Vec<BatchOp>) -> Result<(usize, usize)> {
6234 // Restore fsync policy even on panic via a raw-pointer drop guard.
6235 // A panic here would poison the RwLock anyway, but the correct policy
6236 // must be in place if the guard is ever unwrapped.
6237 struct RestoreFsync(*mut FsyncPolicy, FsyncPolicy);
6238 impl Drop for RestoreFsync {
6239 fn drop(&mut self) {
6240 // SAFETY: the pointer is valid for the full duration of
6241 // commit_batch_nosync; the guard is dropped before the frame
6242 // returns, and GraphDb outlives this frame.
6243 unsafe {
6244 *self.0 = self.1;
6245 }
6246 }
6247 }
6248 let saved = self.fsync;
6249 // SAFETY: raw pointer into self; guard dropped within this frame.
6250 let _g = RestoreFsync(&mut self.fsync as *mut FsyncPolicy, saved);
6251 self.fsync = FsyncPolicy::Relaxed;
6252 self.commit_logged_batch(ops, None, None).map(inserted_pair)
6253 }
6254
6255 /// Commit multiple op-batches as a **group**: each submission gets its own
6256 /// WAL `Batch` frame, but there is exactly **one** `Fs::sync` for the whole
6257 /// group (under `Strict` / `Batched` policy; `Relaxed` skips all syncs).
6258 ///
6259 /// # Durability semantics
6260 ///
6261 /// A crash before the group fsync may lose **all** submissions in the group.
6262 /// A crash after the group fsync preserves all of them. No submission is
6263 /// ever torn: each WAL frame is either fully applied on replay or dropped
6264 /// in its entirety (CRC-protected frame boundaries).
6265 ///
6266 /// Events and subscription notifications fire per-submission immediately
6267 /// after apply, which may be before the group fsync. From a subscriber's
6268 /// perspective this is equivalent to the `Relaxed` durability window.
6269 /// Submitters using [`SharedDb::submit_batch`] only unblock after the group
6270 /// fsync, so from their perspective durability is fully guaranteed.
6271 ///
6272 /// # MVCC interplay
6273 ///
6274 /// Each submission records its own `CommitDelta`; the fold-every-K counter
6275 /// increments per submission (not per group), preserving existing reader
6276 /// snapshot semantics.
6277 ///
6278 /// # Returns
6279 ///
6280 /// One `Result<(nodes_inserted, edges_inserted)>` per input group element,
6281 /// in order. Failures are per-submission (validation errors); the group
6282 /// fsync error (if any) is returned as the second tuple element.
6283 pub fn commit_group(
6284 &mut self,
6285 groups: Vec<Vec<BatchOp>>,
6286 ) -> (Vec<Result<(usize, usize)>>, Option<GraphError>) {
6287 let mut results = Vec::with_capacity(groups.len());
6288 for ops in groups {
6289 results.push(self.commit_batch_nosync(ops));
6290 }
6291 let any_ok = results.iter().any(|r| r.is_ok());
6292 let sync_err = if self.fsync != FsyncPolicy::Relaxed && any_ok {
6293 self.fs
6294 .sync(core_storage::fs::FileId::Wal)
6295 .map_err(GraphError::Io)
6296 .err()
6297 } else {
6298 None
6299 };
6300 (results, sync_err)
6301 }
6302
6303 /// Like [`commit_group`] but skips the group fsync entirely.
6304 ///
6305 /// Used by the drain thread to apply submissions under the write lock and
6306 /// then perform the single fsync OUTSIDE the lock (via
6307 /// `core_storage::sync_wal_at`), reducing the write-lock hold time visible
6308 /// to concurrent readers.
6309 pub fn commit_group_nosync(
6310 &mut self,
6311 groups: Vec<Vec<BatchOp>>,
6312 ) -> Vec<Result<(usize, usize)>> {
6313 let mut results = Vec::with_capacity(groups.len());
6314 for ops in groups {
6315 results.push(self.commit_batch_nosync(ops));
6316 }
6317 results
6318 }
6319
6320 pub fn insert_node(
6321 &mut self,
6322 label: &str,
6323 key: &str,
6324 props: Vec<(String, Value)>,
6325 ) -> Result<()> {
6326 if self.read_only {
6327 return Err(GraphError::ReadOnly);
6328 }
6329 MutPreview::new(self).check_insert_node(key, &props)?;
6330 self.log_dense(vec![WalRecord::InsertNode {
6331 label: label.into(),
6332 key: key.into(),
6333 props,
6334 }])
6335 }
6336
6337 /// Insert a user edge. `Ok(true)` when the pair was new, `Ok(false)` when it
6338 /// was already there — the question is "was this pair new", and a duplicate
6339 /// does not make it so.
6340 ///
6341 /// On a store that called [`enable_multiplicity`](Self::enable_multiplicity)
6342 /// a duplicate is no longer a total no-op: it raises the pair's insert count
6343 /// (§5.13). Adjacency is still a set, so [`degree`](Self::degree) is
6344 /// unchanged and the return value is still `Ok(false)`; the count is visible
6345 /// only through [`degree_multiplicity`](Self::degree_multiplicity) and the
6346 /// reserved [`EDGE_COUNT_PROP`]. On every other store a duplicate writes
6347 /// nothing at all, as it always has.
6348 pub fn insert_edge(&mut self, edge_type: &str, src_key: &str, dst_key: &str) -> Result<bool> {
6349 if self.read_only {
6350 return Err(GraphError::ReadOnly);
6351 }
6352 if !MutPreview::new(self).prepare_insert_edge(edge_type, src_key, dst_key)? {
6353 // The pair exists. The only thing left to record is that it was
6354 // asked for again, and only where the store asked to be told.
6355 if let Some(rec) = self.edge_count_record(edge_type, src_key, dst_key) {
6356 self.log_then_apply(rec)?;
6357 }
6358 return Ok(false);
6359 }
6360 self.log_dense(vec![WalRecord::InsertEdge {
6361 edge_type: edge_type.into(),
6362 src_key: src_key.into(),
6363 dst_key: dst_key.into(),
6364 }])?;
6365 Ok(true)
6366 }
6367
6368 pub fn set_prop(&mut self, key: &str, field: &str, value: Value) -> Result<()> {
6369 if self.read_only {
6370 return Err(GraphError::ReadOnly);
6371 }
6372 if let Some(view_name) = self.view_store.view_for_prop(field) {
6373 return Err(GraphError::ViewPropReadOnly {
6374 view_name: view_name.to_string(),
6375 });
6376 }
6377 MutPreview::new(self).check_live_key(key)?;
6378 self.log_dense(vec![WalRecord::SetProp {
6379 key: key.into(),
6380 field: field.into(),
6381 value,
6382 }])
6383 }
6384
6385 /// Set several properties on one live node in a single WAL commit.
6386 ///
6387 /// Every per-property check [`set_prop`](Self::set_prop) runs — view-owned
6388 /// names, live key, the `ns` immutability rule and its type — is evaluated
6389 /// for the whole list before any record is logged. The first refusal
6390 /// returns and the node is unchanged. An empty list writes nothing.
6391 pub fn set_props(&mut self, key: &str, props: Vec<(String, Value)>) -> Result<()> {
6392 if self.read_only {
6393 return Err(GraphError::ReadOnly);
6394 }
6395 MutPreview::new(self).check_live_key(key)?;
6396 for (field, _) in &props {
6397 if let Some(view_name) = self.view_store.view_for_prop(field) {
6398 return Err(GraphError::ViewPropReadOnly {
6399 view_name: view_name.to_string(),
6400 });
6401 }
6402 }
6403 if props.is_empty() {
6404 return Ok(());
6405 }
6406 self.write_batch(|b| {
6407 for (field, value) in props {
6408 b.set_prop(key, &field, value);
6409 }
6410 })
6411 .map(|_| ())
6412 }
6413
6414 /// Remove a property. Returns `Ok(false)` (and does not log) if the field
6415 /// is already absent. Unknown or tombstoned keys are `Err(KeyNotFound)`.
6416 /// A field a view owns is `Err(ViewPropReadOnly)` — stated once, in
6417 /// [`MutPreview::prepare_remove_prop`], so that the batch ops reaching that
6418 /// same choke-point cannot miss it.
6419 pub fn remove_prop(&mut self, key: &str, field: &str) -> Result<bool> {
6420 if self.read_only {
6421 return Err(GraphError::ReadOnly);
6422 }
6423 if !MutPreview::new(self).prepare_remove_prop(key, field)? {
6424 return Ok(false);
6425 }
6426 self.log_then_apply(WalRecord::RemoveProp {
6427 key: key.into(),
6428 field: field.into(),
6429 })?;
6430 Ok(true)
6431 }
6432
6433 /// Delete a user edge. Returns `Ok(false)` (and does not log) if the edge
6434 /// is absent. Unknown keys are `Err(KeyNotFound)`. Rule-owned edges — in
6435 /// provenance, or a pair a live rule would derive — are `Err(RuleOwned)`
6436 /// (the rule would just put the edge back; delete or change the rule).
6437 pub fn delete_edge(&mut self, edge_type: &str, src_key: &str, dst_key: &str) -> Result<bool> {
6438 if self.read_only {
6439 return Err(GraphError::ReadOnly);
6440 }
6441 if !MutPreview::new(self).prepare_delete_edge(edge_type, src_key, dst_key)? {
6442 return Ok(false);
6443 }
6444 self.log_then_apply(WalRecord::DeleteEdge {
6445 edge_type: edge_type.into(),
6446 src_key: src_key.into(),
6447 dst_key: dst_key.into(),
6448 })?;
6449 Ok(true)
6450 }
6451
6452 /// Delete a live node. Unknown or already-tombstoned keys are
6453 /// `Err(KeyNotFound)` and are not logged. Validation runs before the WAL
6454 /// write; `apply` of a logged `DeleteNode` for an already-tombstoned key
6455 /// (crash window) is a clean no-op.
6456 ///
6457 /// Returns a [`DeleteReport`] with counts of manual and derived edges
6458 /// removed (computed from live state before the deletion is applied).
6459 pub fn delete_node(&mut self, key: &str) -> Result<DeleteReport> {
6460 if self.read_only {
6461 return Err(GraphError::ReadOnly);
6462 }
6463 // Provenance must be loaded before we query provenance_touching.
6464 self.engine.ensure_provenance_loaded_mut();
6465 let id = self
6466 .ids
6467 .get(key)
6468 .ok_or_else(|| GraphError::KeyNotFound { key: key.into() })?;
6469
6470 // Count edges before the delete is applied so we can report counts.
6471 let derived_set: BTreeSet<(u32, u32, u32)> = self
6472 .engine
6473 .provenance_touching(id)
6474 .map(|(_, etype, src, dst)| (etype, src, dst))
6475 .collect();
6476 let derived_edges = derived_set.len() as u64;
6477
6478 let mut total_topo = 0u64;
6479 let tv = self.topo_view();
6480 for et in tv.etypes() {
6481 total_topo += tv.neighbors(et, Direction::Out, id).len() as u64
6482 + tv.neighbors(et, Direction::In, id).len() as u64;
6483 }
6484 // For symmetric rules (e.g. Overlap), a→b and b→a are two separate directed
6485 // triples in both the topo scan (Out and In from id) and in provenance_touching.
6486 // The subtraction remains correct because both counts include both directions.
6487 let manual_edges = total_topo.saturating_sub(derived_edges);
6488
6489 self.log_then_apply(WalRecord::DeleteNode { key: key.into() })?;
6490 Ok(DeleteReport {
6491 manual_edges,
6492 derived_edges,
6493 })
6494 }
6495
6496 /// Rename a live node's key. The dense id (and therefore all edges,
6497 /// props, history, and last-change tracking) is unaffected.
6498 ///
6499 /// Returns `Err(KeyNotFound)` if `old` is not a live key.
6500 /// Returns `Err(DuplicateKey)` if `new` is already live.
6501 pub fn rename_node(&mut self, old: &str, new: &str) -> Result<()> {
6502 if self.read_only {
6503 return Err(GraphError::ReadOnly);
6504 }
6505 MutPreview::new(self).check_rename_node(old, new)?;
6506 self.log_then_apply(WalRecord::RenameNode {
6507 old_key: old.into(),
6508 new_key: new.into(),
6509 })
6510 }
6511
6512 /// Return the IVF drift counter for the dst-side candidate index of `rule`.
6513 /// `None` if the rule does not exist or is not approximate.
6514 ///
6515 /// The drift counter increments on IVF insert/remove after the last fit.
6516 /// When dst-side drift exceeds [`core_rules::IVF_DRIFT_REBUILD`], apply
6517 /// WAL-logs `RebuildRule` as a second commit (rebuild resets the counter).
6518 pub fn ivf_dst_drift(&self, rule: &str) -> Option<u64> {
6519 // SideIvfExport = (centroids, node→cluster, drift)
6520 self.engine
6521 .export_ivf_state()
6522 .remove(rule)
6523 .map(|(_src, dst)| dst.2)
6524 }
6525
6526 /// Validate and WAL-log a new rule, then backfill derived edges inside apply.
6527 /// Validation and duplicate-name check run before logging so invalid rules
6528 /// never enter the WAL.
6529 pub fn create_rule(&mut self, def: RuleDef) -> Result<()> {
6530 if self.read_only {
6531 return Err(GraphError::ReadOnly);
6532 }
6533 MutPreview::new(self).check_create_rule(&def)?;
6534 let def_bytes = bincode::serialize(&def).map_err(|e| GraphError::Corrupt {
6535 detail: format!("serialize rule: {e}"),
6536 })?;
6537 self.log_then_apply(WalRecord::CreateRule { def_bytes })
6538 }
6539
6540 /// Override this handle's HNSW build-slice size, or `None` to restore
6541 /// [`core_rules::HNSW_BUILD_BATCH`].
6542 ///
6543 /// Exposed for tests that need a small slice without a large corpus; not
6544 /// part of the stable surface.
6545 #[doc(hidden)]
6546 pub fn set_hnsw_build_batch(&mut self, batch: Option<usize>) {
6547 self.engine.set_hnsw_build_batch(batch);
6548 }
6549
6550 /// Rules whose vector index is still being built, in name order.
6551 ///
6552 /// The same list [`GraphDb::stats`] reports per rule in `building`.
6553 /// After a clean open this includes a build a snapshot cut short, so
6554 /// `serve`'s ticker can pump it without a write.
6555 pub fn builds_in_progress(&self) -> Vec<BuildProgress> {
6556 self.engine.builds_in_progress()
6557 }
6558
6559 /// Advance any vector index still building and backfill each rule that
6560 /// finishes. Returns what is still outstanding.
6561 ///
6562 /// A map lookup when nothing is pending, so it is cheap to call on a timer.
6563 /// One write lock and at most [`core_rules::HNSW_BUILD_BATCH`] vector
6564 /// inserts per pending rule per call, so a caller can drive a large build
6565 /// to completion without ever holding the lock for more than a slice.
6566 ///
6567 /// A rule that finishes here is backfilled through the same
6568 /// `WalRecord::RebuildRule` second commit that IVF drift already uses, so
6569 /// its derived edges are produced by [`GraphDb::rebuild_rule`]'s code path
6570 /// and appear all at once.
6571 ///
6572 /// Every ordinary write pumps one slice on its own (see the post-commit
6573 /// hook in `log_then_apply_with`), so this is for quiescent stores and for
6574 /// operators who want the build finished before traffic arrives.
6575 pub fn pump_index_build(&mut self) -> Result<Vec<BuildProgress>> {
6576 Ok(self.pump_index_build_reporting()?.1)
6577 }
6578
6579 /// [`GraphDb::pump_index_build`], also reporting the builds that **this**
6580 /// call finished, so a progress display can say so.
6581 ///
6582 /// A build can be registered and completed inside a single call — that is
6583 /// what a mid-build snapshot looks like on reopen, where the index scan
6584 /// finishes the graph and only the backfill is outstanding — and the
6585 /// outstanding list alone cannot show that anything happened.
6586 pub fn pump_index_build_reporting(
6587 &mut self,
6588 ) -> Result<(Vec<BuildProgress>, Vec<BuildProgress>)> {
6589 // A read-only handle cannot issue the `RebuildRule` a finished build
6590 // needs, so it would advance the index and then silently fail to
6591 // produce the edges. Refusing is the honest answer.
6592 if self.read_only {
6593 return Err(GraphError::ReadOnly);
6594 }
6595 let finished = self.pump_one_slice();
6596 for done in &finished {
6597 // The index is whole but the rule still owns no edges. A failed
6598 // second commit must leave the rule re-pumpable rather than
6599 // silently edge-less, so the error is surfaced here — unlike the
6600 // post-commit hook, this call is not riding someone else's commit.
6601 self.log_then_apply(WalRecord::RebuildRule {
6602 name: done.rule.clone(),
6603 })?;
6604 }
6605 Ok((finished, self.engine.builds_in_progress()))
6606 }
6607
6608 /// Run the deferred candidate-index build, if it is still owed, against the
6609 /// graph as it stands *now* — before the caller applies anything.
6610 ///
6611 /// A no-op bool test once the indexes are populated, which is after the
6612 /// first write of the handle's life, and for a store with no rules at all.
6613 fn populate_indexes_before_write(&mut self) {
6614 if !self.engine.needs_index_population() {
6615 return;
6616 }
6617 // The retained snapshot blobs arrive with the V8 base sections; without
6618 // them the scan would rebuild every graph the snapshot already holds.
6619 self.ensure_v8_base_sections_loaded();
6620 if !self.engine.needs_index_population() {
6621 return;
6622 }
6623 let mut eng = std::mem::take(&mut self.engine);
6624 {
6625 let gm = make_graph_mut(
6626 &self.ids,
6627 &mut self.syms,
6628 &self.labels,
6629 build_props_view(&self.props, &self.base),
6630 &mut self.topo,
6631 &self.base,
6632 &mut self.edge_props,
6633 );
6634 eng.populate_indexes(&gm);
6635 }
6636 self.engine = eng;
6637 }
6638
6639 /// One slice of build work for every pending rule. Returns the rules whose
6640 /// index just became whole, which the caller must `RebuildRule`.
6641 ///
6642 /// Goes through the engine even with nothing pending when the indexes have
6643 /// not been populated yet: that call adopts the persisted graphs and, for
6644 /// an incomplete blob already registered at open, leaves the remainder to
6645 /// this slice rather than inserting it inline.
6646 fn pump_one_slice(&mut self) -> Vec<BuildProgress> {
6647 // The retained snapshot blobs — and the id count an interrupted build
6648 // is recognised against — arrive with the V8 base sections, which a
6649 // clean open reads lazily. Without this a freshly opened handle pumps
6650 // against empty retained state and concludes there is nothing to do,
6651 // which is precisely the store `build-index` exists for.
6652 self.ensure_v8_base_sections_loaded();
6653 let mut eng = std::mem::take(&mut self.engine);
6654 let finished = {
6655 let mut gm = make_graph_mut(
6656 &self.ids,
6657 &mut self.syms,
6658 &self.labels,
6659 build_props_view(&self.props, &self.base),
6660 &mut self.topo,
6661 &self.base,
6662 &mut self.edge_props,
6663 );
6664 eng.pump_index_build(&mut gm)
6665 };
6666 self.engine = eng;
6667 finished
6668 }
6669
6670 /// Register a sliced build a snapshot cut short, from blobs with
6671 /// `complete == false`.
6672 ///
6673 /// Peeks the V8 mmap for incomplete entries without copying complete
6674 /// graphs. V5–V7 already hold the blobs in the engine from restore.
6675 fn register_outstanding_index_builds(&mut self) {
6676 if self.engine.indexes_populated() {
6677 return;
6678 }
6679 let extra = self.collect_incomplete_hnsw_blobs();
6680 let mut eng = std::mem::take(&mut self.engine);
6681 {
6682 let gm = make_graph_mut(
6683 &self.ids,
6684 &mut self.syms,
6685 &self.labels,
6686 build_props_view(&self.props, &self.base),
6687 &mut self.topo,
6688 &self.base,
6689 &mut self.edge_props,
6690 );
6691 eng.register_incomplete_hnsw_builds(&extra, &gm);
6692 }
6693 self.engine = eng;
6694 }
6695
6696 /// Incomplete `(src, dst)` HNSW blobs from the V8 mmap, copied only when
6697 /// `complete` is false. Empty when there is no mmap base (V5–V7 uses the
6698 /// engine's retained map instead).
6699 fn collect_incomplete_hnsw_blobs(&self) -> BTreeMap<String, (Vec<u8>, Vec<u8>)> {
6700 let Some(base) = &self.base else {
6701 return BTreeMap::new();
6702 };
6703 let Ok(archived) = base.hnsw_section() else {
6704 return BTreeMap::new();
6705 };
6706 archived
6707 .rules
6708 .iter()
6709 .filter_map(|e| {
6710 let src = e.src_blob.as_slice();
6711 let dst = e.dst_blob.as_slice();
6712 if core_rules::hnsw::hnsw_blob_complete(src) == Some(false)
6713 || core_rules::hnsw::hnsw_blob_complete(dst) == Some(false)
6714 {
6715 Some((e.name.as_str().to_string(), (src.to_vec(), dst.to_vec())))
6716 } else {
6717 None
6718 }
6719 })
6720 .collect()
6721 }
6722
6723 /// WAL-log rule deletion. Returns RuleNotFound if the rule does not exist.
6724 pub fn delete_rule(&mut self, name: &str) -> Result<()> {
6725 if self.read_only {
6726 return Err(GraphError::ReadOnly);
6727 }
6728 MutPreview::new(self).check_delete_rule(name)?;
6729 self.log_then_apply(WalRecord::DeleteRule { name: name.into() })
6730 }
6731
6732 /// Return a snapshot of all registered rules.
6733 pub fn rules(&self) -> Vec<RuleDef> {
6734 self.engine.rules().cloned().collect()
6735 }
6736
6737 // -----------------------------------------------------------------------
6738 // Rule suggestion API
6739 // -----------------------------------------------------------------------
6740
6741 /// Profile the database and suggest linking rules with previewed edge counts.
6742 ///
6743 /// Uses the default seed ([`core_rules::SUGGEST_DEFAULT_SEED`]) for deterministic
6744 /// sampling. Suggestions are sorted by estimated edge count (descending).
6745 /// **NO auto-accept** — call [`GraphDb::create_rule`] explicitly to apply.
6746 pub fn suggest_rules(&self) -> Vec<core_rules::RuleSuggestion> {
6747 self.suggest_rules_seeded(core_rules::SUGGEST_DEFAULT_SEED)
6748 }
6749
6750 /// Like [`suggest_rules`] but with a caller-supplied RNG seed for
6751 /// reproducibility. Same seed + same data = identical output.
6752 pub fn suggest_rules_seeded(&self, seed: u64) -> Vec<core_rules::RuleSuggestion> {
6753 self.suggest_rules_with_config(&core_rules::suggest::SuggestConfig::default(), seed)
6754 .suggestions
6755 }
6756
6757 /// [`suggest_rules_seeded`] with a fully custom [`SuggestConfig`].
6758 ///
6759 /// Returns a [`core_rules::SuggestReport`] that includes both the candidate list
6760 /// and a `truncated` flag indicating whether the global budget fired before all
6761 /// candidates were evaluated.
6762 pub fn suggest_rules_with_config(
6763 &self,
6764 config: &core_rules::suggest::SuggestConfig,
6765 seed: u64,
6766 ) -> core_rules::SuggestReport {
6767 use std::collections::BTreeMap;
6768
6769 // Collect (node_id, key) pairs per label, skipping tombstoned nodes.
6770 let mut label_nodes: BTreeMap<String, Vec<(u32, String)>> = BTreeMap::new();
6771 for id in 0..self.ids.len() as u32 {
6772 let Some(key) = self.ids.key_of(id) else {
6773 continue;
6774 };
6775 let Some(&sym) = self.labels.get(id as usize) else {
6776 continue;
6777 };
6778 if sym == u32::MAX {
6779 continue; // tombstoned
6780 }
6781 let Some(label) = self.syms.resolve(sym) else {
6782 continue;
6783 };
6784 label_nodes
6785 .entry(label.to_string())
6786 .or_default()
6787 .push((id, key.to_string()));
6788 }
6789
6790 let existing = self.rules();
6791 let pv = build_props_view(&self.props, &self.base);
6792 let all_fields: Vec<String> = pv.field_names();
6793
6794 core_rules::suggest::suggest_rules(
6795 &label_nodes,
6796 &|id, field| pv.get(id, field).map(|vr| vr.into_value()),
6797 &all_fields,
6798 &existing,
6799 config,
6800 seed,
6801 )
6802 }
6803
6804 /// Recompute a rule's derived edges from scratch. WAL-logged so un-trip
6805 /// plus later mutations replay identically (rebuild is a pure function
6806 /// of state).
6807 ///
6808 /// Only exit from the tripped latch: if the full desired set fits the
6809 /// budget, it is applied completely and `tripped` clears; if it still
6810 /// exceeds the budget, provenance is left untouched and `tripped` stays
6811 /// true. Counts as a fire evaluation (see [`RuleStats::fires`]).
6812 /// Unknown rule → `RuleNotFound`, nothing logged.
6813 pub fn rebuild_rule(&mut self, name: &str) -> Result<()> {
6814 if self.read_only {
6815 return Err(GraphError::ReadOnly);
6816 }
6817 if !self.engine.rules().any(|r| r.name == name) {
6818 return Err(GraphError::RuleNotFound { name: name.into() });
6819 }
6820 self.log_then_apply(WalRecord::RebuildRule { name: name.into() })
6821 }
6822
6823 // -----------------------------------------------------------------------
6824 // Materialized view API
6825 // -----------------------------------------------------------------------
6826
6827 /// Register a new materialized property view, backfill its values for all
6828 /// existing nodes, and WAL-log the definition.
6829 ///
6830 /// # Errors
6831 /// - `ReadOnly`: called on an as-of instance.
6832 /// - `RuleInvalid`: name collision, view_prop collision, or invalid def.
6833 pub fn create_view(&mut self, def: ViewDef) -> Result<()> {
6834 if self.read_only {
6835 return Err(GraphError::ReadOnly);
6836 }
6837 // Pre-validate before WAL write.
6838 def.validate()
6839 .map_err(|e| GraphError::RuleInvalid { detail: e })?;
6840 if self.view_store.has_view(&def.name) {
6841 return Err(GraphError::RuleInvalid {
6842 detail: format!("view {:?} already exists", def.name),
6843 });
6844 }
6845 if let Some(existing) = self.view_store.view_for_prop(&def.view_prop) {
6846 return Err(GraphError::RuleInvalid {
6847 detail: format!(
6848 "view_prop {:?} is already used by view {:?}",
6849 def.view_prop, existing
6850 ),
6851 });
6852 }
6853 let def_bytes = bincode::serialize(&def).map_err(|e| GraphError::Corrupt {
6854 detail: format!("serialize view: {e}"),
6855 })?;
6856 // Enable delta accumulation before the view is registered so subsequent
6857 // incremental edge events reach view maintenance from this point onward.
6858 // (The backfill inside create_view reads topo directly; it does not rely
6859 // on pending deltas.)
6860 self.engine.set_emit_deltas(true);
6861 self.log_then_apply(WalRecord::CreateView { def_bytes })
6862 }
6863
6864 /// Remove a named view and delete its values from every node.
6865 ///
6866 /// # Errors
6867 /// - `ReadOnly`: called on an as-of instance.
6868 /// - `RuleNotFound`: view does not exist.
6869 pub fn delete_view(&mut self, name: &str) -> Result<()> {
6870 if self.read_only {
6871 return Err(GraphError::ReadOnly);
6872 }
6873 if !self.view_store.has_view(name) {
6874 return Err(GraphError::RuleNotFound { name: name.into() });
6875 }
6876 let result = self.log_then_apply(WalRecord::DeleteView { name: name.into() });
6877 // After deletion, disable accumulation if no listeners remain.
6878 if !self.needs_emit_deltas() {
6879 self.engine.set_emit_deltas(false);
6880 }
6881 result
6882 }
6883
6884 /// Snapshot of all registered view definitions.
6885 pub fn views(&self) -> Vec<ViewDef> {
6886 self.view_store.views().cloned().collect()
6887 }
6888
6889 // -----------------------------------------------------------------------
6890 // Full-text-lite API
6891 // -----------------------------------------------------------------------
6892
6893 /// Enable full-text indexing for all nodes of `label` on property `field`.
6894 ///
6895 /// After this call, every subsequent write to `(label, field)` is reflected
6896 /// in the index incrementally. Existing nodes are backfilled immediately.
6897 /// The declaration is persisted as a WAL record; the index itself is rebuilt
6898 /// from scratch on re-open (no snapshot format changes).
6899 ///
6900 /// # Errors
6901 /// - [`GraphError::ReadOnly`]: called on an as-of instance.
6902 /// - [`GraphError::RuleInvalid`]: `(label, field)` is already indexed.
6903 pub fn enable_fulltext(&mut self, label: &str, field: &str) -> Result<()> {
6904 if self.read_only {
6905 return Err(GraphError::ReadOnly);
6906 }
6907 if self.fulltext.is_enabled(label, field) {
6908 return Err(GraphError::RuleInvalid {
6909 detail: format!("full-text index for ({label:?}, {field:?}) already enabled"),
6910 });
6911 }
6912 self.log_then_apply(WalRecord::EnableFulltext {
6913 label: label.into(),
6914 field: field.into(),
6915 })
6916 }
6917
6918 /// Disable full-text indexing for `(label, field)` and drop its postings.
6919 ///
6920 /// # Errors
6921 /// - [`GraphError::ReadOnly`]: called on an as-of instance.
6922 /// - [`GraphError::RuleNotFound`]: `(label, field)` is not currently indexed.
6923 pub fn disable_fulltext(&mut self, label: &str, field: &str) -> Result<()> {
6924 if self.read_only {
6925 return Err(GraphError::ReadOnly);
6926 }
6927 if !self.fulltext.is_enabled(label, field) {
6928 return Err(GraphError::RuleNotFound {
6929 name: format!("fulltext({label},{field})"),
6930 });
6931 }
6932 self.log_then_apply(WalRecord::DisableFulltext {
6933 label: label.into(),
6934 field: field.into(),
6935 })
6936 }
6937
6938 /// Whether `(label, field)` is currently indexed for full-text search.
6939 pub fn is_fulltext_enabled(&self, label: &str, field: &str) -> bool {
6940 self.fulltext.is_enabled(label, field)
6941 }
6942
6943 /// Every `(label, field)` pair with a live full-text index, sorted.
6944 ///
6945 /// Note that [`GraphDb::search`] is keyed by field alone — a pair only
6946 /// declares which nodes are *indexed*, so callers that want to search
6947 /// everything indexed should query each distinct field once.
6948 pub fn fulltext_pairs(&self) -> Vec<(String, String)> {
6949 let mut v: Vec<(String, String)> = self.fulltext.enabled_pairs().cloned().collect();
6950 v.sort();
6951 v
6952 }
6953
6954 /// Enable an equality index for all nodes of `label` on scalar property
6955 /// `field`. Subsequent `WHERE n.field = value` lookups become O(matches)
6956 /// instead of an O(N_label) scan. Existing nodes are backfilled; the
6957 /// declaration persists via WAL and the postings rebuild on re-open.
6958 ///
6959 /// # Errors
6960 /// - [`GraphError::ReadOnly`]: called on an as-of instance.
6961 /// - [`GraphError::RuleInvalid`]: `(label, field)` is already indexed.
6962 pub fn enable_index(&mut self, label: &str, field: &str) -> Result<()> {
6963 if self.read_only {
6964 return Err(GraphError::ReadOnly);
6965 }
6966 if self.prop_index.is_enabled(label, field) {
6967 return Err(GraphError::RuleInvalid {
6968 detail: format!("property index for ({label:?}, {field:?}) already enabled"),
6969 });
6970 }
6971 self.log_then_apply(WalRecord::EnableIndex {
6972 label: label.into(),
6973 field: field.into(),
6974 })
6975 }
6976
6977 /// Disable the equality index for `(label, field)` and drop its postings.
6978 ///
6979 /// # Errors
6980 /// - [`GraphError::ReadOnly`]: called on an as-of instance.
6981 /// - [`GraphError::RuleNotFound`]: `(label, field)` is not currently indexed.
6982 pub fn disable_index(&mut self, label: &str, field: &str) -> Result<()> {
6983 if self.read_only {
6984 return Err(GraphError::ReadOnly);
6985 }
6986 if !self.prop_index.is_enabled(label, field) {
6987 return Err(GraphError::RuleNotFound {
6988 name: format!("index({label},{field})"),
6989 });
6990 }
6991 self.log_then_apply(WalRecord::DisableIndex {
6992 label: label.into(),
6993 field: field.into(),
6994 })
6995 }
6996
6997 /// Whether `(label, field)` currently has an equality index.
6998 pub fn is_index_enabled(&self, label: &str, field: &str) -> bool {
6999 self.prop_index.is_enabled(label, field)
7000 }
7001
7002 /// Start recording insert-count multiplicity on this store (§5.13).
7003 ///
7004 /// Adjacency stays a set and nothing about an existing read changes: a
7005 /// duplicate [`insert_edge`](Self::insert_edge) still returns `Ok(false)`
7006 /// and still leaves [`degree`](Self::degree) alone. What it gains is that
7007 /// the duplicate is *counted*, as the reserved edge property
7008 /// [`EDGE_COUNT_PROP`], readable through
7009 /// [`degree_multiplicity`](Self::degree_multiplicity).
7010 ///
7011 /// # This is a one-way step, and that is why it is a call
7012 ///
7013 /// The count is durable, so it is written to the WAL — as discriminant 23,
7014 /// which no release before v0.6.10 knows. A reader meeting an unknown WAL
7015 /// discriminant cannot know what the record would have changed, so it
7016 /// cannot degrade the way an unreadable index blob can. **After this call
7017 /// the store can no longer be read by an older binary, and there is no call
7018 /// that undoes it.** Gating the record behind this method is what keeps
7019 /// that step a decision an operator makes when they want the feature,
7020 /// rather than one everybody takes by upgrading.
7021 ///
7022 /// # It fails loudly, and that costs a snapshot
7023 ///
7024 /// An older binary does not refuse discriminant 23 — it truncates the WAL
7025 /// at it and, with `repair_wal`, persists the truncation. So this call also
7026 /// writes a **V10 snapshot**, a version no earlier release knows, and it
7027 /// writes it *first*: the snapshot is read before the WAL, so an older
7028 /// binary stops at `snapshot: unsupported version 10` with the WAL
7029 /// untouched. Taking the snapshot before appending the record is what makes
7030 /// the guard unconditional — the store is never, at any interruption point,
7031 /// carrying the record without the stamp that announces it.
7032 ///
7033 /// The snapshot keeps the WAL (`keep_wal: true`): opting in is not a
7034 /// compaction, and history reachable by [`open_at`](Self::open_at) stays
7035 /// reachable. On a large store the call therefore costs one full snapshot
7036 /// write.
7037 ///
7038 /// # What it costs a store that archives
7039 ///
7040 /// Writing `snapshot.bin` is also how the archive path decides whether the
7041 /// store may have a *genesis chain* — whether `open_at` can replay
7042 /// archive-resident commits from empty state. The rule is conservative: a
7043 /// snapshot that was already on disk might have been a truncating one, and
7044 /// once the handle that took it is gone this binary cannot tell. A
7045 /// `keep_wal` snapshot taken by **this** handle is the case where it can, so
7046 /// opting in and then archiving **in the same session** keeps the chain.
7047 ///
7048 /// Opting in, closing the store, and archiving in a *later* session does
7049 /// not — but that is the answer any store with a prior snapshot gets, not
7050 /// something this call causes. A store that wants the chain should take its
7051 /// first archive in the session that opted in.
7052 ///
7053 /// Calling it on a store that has already opted in writes nothing and
7054 /// returns `Ok(())`: an operator should not have to ask first.
7055 ///
7056 /// # This call is not atomic, and an `Err` does not undo it
7057 ///
7058 /// There is no rollback here, and there never was one. An `Err` means this
7059 /// handle stopped believing the store is opted in — `self.multiplicity` is
7060 /// reset, so this handle reports `false` from then on — and nothing more. It
7061 /// says nothing about what reached disk. Two reachable failures leave the
7062 /// opt-in standing:
7063 ///
7064 /// * **The declaration landed and only its fsync failed.** `log_then_apply`
7065 /// appends, then syncs; a failed barrier leaves `MULTIPLICITY_ENABLED`
7066 /// already in `wal.bin`. The next open replays it and the store is opted
7067 /// in. No archive is involved — this one predates the recovery below.
7068 /// * **The declaration never landed, but the V10 snapshot did, on a store
7069 /// that already had an archive.** The open-time recovery in
7070 /// `load_from_disk` reads V10-beside-an-archive as an interrupted archive
7071 /// sequence and opts the store in.
7072 ///
7073 /// So a failed call may leave the opt-in on disk immediately (the first
7074 /// case) or conjure it at the next open (the second), and nothing puts the
7075 /// store back out. Treat `Err` as "the outcome is unknown", not as "nothing
7076 /// happened".
7077 ///
7078 /// **This is safe, and the ordering is the reason.** The V10 stamp is
7079 /// written *before* the declaration, so every one of these intermediate
7080 /// states is one an older binary refuses by name rather than truncates at.
7081 /// The failure direction costs a refusal, never a commit. That ordering is
7082 /// the property worth protecting, not the atomicity this call never had.
7083 ///
7084 /// **To know where the store stands, ask the store.** Reopen it and call
7085 /// [`is_multiplicity_enabled`](Self::is_multiplicity_enabled); that is the
7086 /// only answer that accounts for what reached disk.
7087 ///
7088 /// The one case that really does leave the store opted out is a failure with
7089 /// no archive present and no record written: a stray V10 snapshot remains,
7090 /// costing an older reader a refusal it did not strictly need, and *that*
7091 /// store's next snapshot rewrites at V9.
7092 ///
7093 /// # Errors
7094 /// - [`GraphError::ReadOnly`]: called on an as-of instance.
7095 /// - Anything [`snapshot_with`](Self::snapshot_with) can return, and
7096 /// anything the WAL append or its fsync can return. See the atomicity
7097 /// section above for what the store is left holding.
7098 pub fn enable_multiplicity(&mut self) -> Result<()> {
7099 if self.read_only {
7100 return Err(GraphError::ReadOnly);
7101 }
7102 if self.multiplicity {
7103 return Ok(());
7104 }
7105 // The snapshot goes first, and the order is the guard.
7106 //
7107 // An older reader refuses a V10 snapshot by name and stops; it does not
7108 // refuse discriminant 23, it truncates the WAL at it. So the store must
7109 // never hold the record without the snapshot that announces it — not
7110 // even for the width of one fsync. Writing the snapshot before the
7111 // record makes the only reachable intermediate state "V10 snapshot, no
7112 // record", which is merely conservative: this binary reads it as a
7113 // store that has not opted in, and an older one refuses it.
7114 //
7115 // `keep_wal: true` because opting in is not a compaction: an operator
7116 // asking for multiplicity has not asked to lose the history `open_at`
7117 // can reach.
7118 self.multiplicity = true;
7119 let forced = self
7120 .snapshot_with(SnapshotOptions {
7121 keep_wal: true,
7122 ..SnapshotOptions::default()
7123 })
7124 .and_then(|()| self.log_then_apply(core_storage::wal::MULTIPLICITY_ENABLED));
7125 if forced.is_err() {
7126 // This handle stops believing it is opted in. That is all this line
7127 // does — it is not a rollback, and cannot be one: the declaration
7128 // may already be in `wal.bin` (the append succeeded and only the
7129 // fsync failed), and even when it is not, the V10 snapshot beside an
7130 // existing archive is enough for the open-time recovery to opt the
7131 // store in. See the "not atomic" section on this method.
7132 //
7133 // It fails in the safe direction either way: the V10 stamp reached
7134 // disk before anything a v0.6.9 reader would truncate at, so the
7135 // worst an interruption costs that reader is a refusal by name.
7136 self.multiplicity = false;
7137 }
7138 forced
7139 }
7140
7141 /// Whether this store records insert-count multiplicity.
7142 ///
7143 /// `false` on every store that has not called
7144 /// [`enable_multiplicity`](Self::enable_multiplicity) — which is every
7145 /// store that did not ask for it, including one upgraded from an earlier
7146 /// release.
7147 pub fn is_multiplicity_enabled(&self) -> bool {
7148 self.multiplicity
7149 }
7150
7151 /// How many times `(etype, src, dst)` has been inserted: the reserved
7152 /// `count` edge property, or 1 when it is absent.
7153 ///
7154 /// Answers 1 for a pair on a store that never opted in, which is the truth
7155 /// available there — the pair was inserted at least once, and the store
7156 /// kept no record of any second insert.
7157 fn edge_insert_count(&self, etype: u32, src: u32, dst: u32) -> u64 {
7158 match self.edge_props_view().get(etype, src, dst, EDGE_COUNT_PROP) {
7159 Some(Value::Int(n)) if n > 0 => n as u64,
7160 _ => 1,
7161 }
7162 }
7163
7164 /// The `SetEdgeCount` record a duplicate insert of `(edge_type, src_key,
7165 /// dst_key)` should log, or `None` when nothing should be written.
7166 ///
7167 /// `None` when the store has not opted in, so **no discriminant-23 record
7168 /// is written at all** — the gate the whole feature rests on.
7169 ///
7170 /// The other two `None`s are unreachable from the one caller. This is the
7171 /// single-mutation path, where `prepare_insert_edge` has already refused a
7172 /// missing endpoint and an existing pair's edge type is necessarily
7173 /// interned. A batch is the case where a pair's endpoints and type can all
7174 /// be created by the same frame, and it does not come through here: it
7175 /// queues a [`PlannedRec::DuplicateCount`] and names the count in the dense
7176 /// rewrite, which is the only pass that knows the frame's own ids.
7177 fn edge_count_record(
7178 &self,
7179 edge_type: &str,
7180 src_key: &str,
7181 dst_key: &str,
7182 ) -> Option<WalRecord> {
7183 if !self.multiplicity {
7184 return None;
7185 }
7186 let etype = self.syms.get(edge_type)?;
7187 let src = self.ids.get(src_key)?;
7188 let dst = self.ids.get(dst_key)?;
7189 Some(WalRecord::SetEdgeCount {
7190 etype,
7191 src,
7192 dst,
7193 count: self.edge_insert_count(etype, src, dst).saturating_add(1),
7194 })
7195 }
7196
7197 /// Search a full-text-indexed field.
7198 ///
7199 /// Returns `(node_key, match_count)` pairs sorted by match_count descending,
7200 /// ties broken by key (lexicographic). Tombstoned nodes are excluded.
7201 ///
7202 /// **Query syntax:**
7203 /// - Space-separated terms are AND'd: `"foo bar"` requires both.
7204 /// - `OR` between terms forms disjunction: `"foo OR bar"` matches either.
7205 /// - Trailing `*` on a term is a prefix match: `"rust*"` matches `rustlang`, `rusty`.
7206 /// - `AND` keyword is accepted explicitly and is the default.
7207 /// - Tokenization is unicode-alphanumeric (same as index time); case-insensitive.
7208 ///
7209 /// **Unindexed field:** returns `Ok(vec![])` if `field` is not indexed.
7210 /// Pin: this is the documented, tested, stable behavior for v1.
7211 ///
7212 /// **Memory / performance:** O(postings) lookup; no scan. The index is
7213 /// in-memory and proportional to total indexed text across all enabled fields.
7214 ///
7215 /// **v2 grammar:** supports `"phrase"`, `-negation`, `prefix*`, `OR`, `AND`.
7216 /// Results are BM25-scored (k1=1.2, b=0.75) and sorted by score descending,
7217 /// key ascending for deterministic tiebreaking.
7218 pub fn search(&self, field: &str, query: &str) -> Vec<(String, f64)> {
7219 // Resolve node_ids to keys (excluding tombstones) then re-sort by
7220 // (score DESC, key ASC) to give a deterministic, key-lexicographic
7221 // tiebreak. FulltextIndex::search sorts by (score DESC, node_id ASC)
7222 // which diverges from key order when nodes were not inserted in key-lex order.
7223 let mut results: Vec<(String, f64)> = self
7224 .fulltext
7225 .search(field, query, 0)
7226 .into_iter()
7227 .filter_map(|(id, score)| self.ids.key_of(id).map(|key| (key.to_string(), score)))
7228 .collect();
7229 results.sort_by(|a, b| {
7230 b.1.partial_cmp(&a.1)
7231 .unwrap_or(std::cmp::Ordering::Equal)
7232 .then(a.0.cmp(&b.0))
7233 });
7234 results
7235 }
7236
7237 /// [`search`](Self::search), stopping at the `k` best hits.
7238 ///
7239 /// Same ranking and the same deterministic tiebreak, but the index drops
7240 /// everything past `k` before any key is resolved, so a caller that wants
7241 /// the top few out of a field that matched thousands does not pay to
7242 /// materialise and re-sort the tail. `k == 0` means no limit, exactly as
7243 /// [`search`](Self::search) behaves.
7244 ///
7245 /// The BM25 scoring itself is not bounded by `k` — every candidate is
7246 /// scored either way — so this trims the resolve and the sort, not the
7247 /// search.
7248 pub fn search_top(&self, field: &str, query: &str, k: usize) -> Vec<(String, f64)> {
7249 // A tombstoned id resolves to nothing, so asking the index for exactly
7250 // `k` could return fewer. Over-fetching a little and truncating after
7251 // the filter keeps the count right without unbounding the call.
7252 let want = if k == 0 { 0 } else { k.saturating_mul(2) };
7253 let mut results: Vec<(String, f64)> = self
7254 .fulltext
7255 .search(field, query, want)
7256 .into_iter()
7257 .filter_map(|(id, score)| self.ids.key_of(id).map(|key| (key.to_string(), score)))
7258 .collect();
7259 results.sort_by(|a, b| {
7260 b.1.partial_cmp(&a.1)
7261 .unwrap_or(std::cmp::Ordering::Equal)
7262 .then(a.0.cmp(&b.0))
7263 });
7264 if k > 0 {
7265 results.truncate(k);
7266 }
7267 results
7268 }
7269
7270 /// Hybrid search: Reciprocal Rank Fusion (RRF) over fulltext + vector results.
7271 ///
7272 /// Takes up to `4*k` fulltext hits for `(text_field, query_text)` and up to
7273 /// `4*k` vector hits for `(vector_field, query_vec, min=0.0)`, then fuses
7274 /// them with RRF using a fixed constant of 60.
7275 ///
7276 /// ```text
7277 /// score(d) = Σ 1 / (60 + rank_i(d)) (rank 1-based per list)
7278 /// ```
7279 ///
7280 /// Returns the top `k` nodes by fused score, ties broken by node key
7281 /// ascending (deterministic).
7282 ///
7283 /// # Vector leg fallback
7284 ///
7285 /// When `query_vec` is empty the vector leg is skipped entirely and
7286 /// results are ranked by the text list alone through the same RRF path
7287 /// (each text result scores `1/(60 + rank)` from that single list).
7288 ///
7289 /// When `label` is `None`, the vector leg **always** returns empty results.
7290 /// Internally `label` is mapped to `""`, which does not match any rule-created
7291 /// HNSW index (all such indexes are keyed to a specific non-empty label), and
7292 /// the brute-force fallback finds no nodes with an empty label. The fused
7293 /// ranking is therefore text-only in this case.
7294 pub fn search_hybrid(
7295 &self,
7296 text_field: &str,
7297 query_text: &str,
7298 vector_field: &str,
7299 query_vec: &[f64],
7300 label: Option<&str>,
7301 k: usize,
7302 ) -> Vec<(String, f64)> {
7303 self.search_hybrid_inner(
7304 text_field,
7305 query_text,
7306 vector_field,
7307 query_vec,
7308 label,
7309 k,
7310 None,
7311 )
7312 }
7313
7314 /// [`search_hybrid`](Self::search_hybrid) with **each leg** filtered to the
7315 /// mask before the fusion.
7316 ///
7317 /// Filtering the fused list afterwards would quietly return fewer than `k`.
7318 /// Each leg over-fetches `4*k` candidates, so when the visible nodes rank
7319 /// below `4*k` hidden ones neither leg carries them into the fusion at all
7320 /// and the post-filter has nothing left to keep. Filtering first spends the
7321 /// `4*k` on **visible** hits, so a scoped call is as long as the corpus it
7322 /// can see allows.
7323 ///
7324 /// The ranks that enter RRF are therefore the ranks of the visible corpus,
7325 /// not the visible entries of the store-wide ranking. The constant stays 60
7326 /// and the tiebreak stays key-ascending.
7327 #[allow(clippy::too_many_arguments)]
7328 pub fn search_hybrid_scoped(
7329 &self,
7330 text_field: &str,
7331 query_text: &str,
7332 vector_field: &str,
7333 query_vec: &[f64],
7334 label: Option<&str>,
7335 k: usize,
7336 mask: &crate::mask::NodeMask,
7337 ) -> Vec<(String, f64)> {
7338 self.search_hybrid_inner(
7339 text_field,
7340 query_text,
7341 vector_field,
7342 query_vec,
7343 label,
7344 k,
7345 Some(mask),
7346 )
7347 }
7348
7349 /// The body shared by [`search_hybrid`](Self::search_hybrid) and
7350 /// [`search_hybrid_scoped`](Self::search_hybrid_scoped). `mask = None` is
7351 /// the unscoped contract unchanged: the filter below is then a no-op and
7352 /// the vector leg is the same unmasked call it has always been.
7353 #[allow(clippy::too_many_arguments)]
7354 fn search_hybrid_inner(
7355 &self,
7356 text_field: &str,
7357 query_text: &str,
7358 vector_field: &str,
7359 query_vec: &[f64],
7360 label: Option<&str>,
7361 k: usize,
7362 mask: Option<&crate::mask::NodeMask>,
7363 ) -> Vec<(String, f64)> {
7364 use std::collections::HashMap;
7365
7366 const RRF_K: f64 = 60.0;
7367 let pool = 4 * k;
7368
7369 // Accumulate per-node RRF scores.
7370 let mut scores: HashMap<String, f64> = HashMap::new();
7371
7372 // Text leg. The mask bites on the candidates, before `take(pool)`, so
7373 // the over-fetch is a budget of visible hits rather than one a hidden
7374 // prefix can exhaust.
7375 let text_hits = self.search(text_field, query_text);
7376 let visible_text = text_hits
7377 .into_iter()
7378 .filter(|(key, _count)| mask.is_none_or(|m| m.contains_node(self, key)));
7379 for (rank0, (key, _count)) in visible_text.take(pool).enumerate() {
7380 let rank = (rank0 + 1) as f64;
7381 *scores.entry(key).or_insert(0.0) += 1.0 / (RRF_K + rank);
7382 }
7383
7384 // Vector leg (skipped when query_vec is empty). The masked variant
7385 // applies the mask before its own k-truncation, for the same reason.
7386 if !query_vec.is_empty() {
7387 // `ExactnessCaller::Hybrid`: the leg is the same one
7388 // `find_similar_vector_masked` runs, but the advice its warning
7389 // gives has to fit *this* signature, which has no `exact`.
7390 let vec_hits = self
7391 .find_similar_vector_as(
7392 vector_field,
7393 label,
7394 query_vec,
7395 pool,
7396 0.0,
7397 mask,
7398 None,
7399 false,
7400 ExactnessCaller::Hybrid,
7401 )
7402 .expect("find_similar_vector_as is infallible without where_");
7403 for (rank0, (key, _sim)) in vec_hits.into_iter().enumerate() {
7404 let rank = (rank0 + 1) as f64;
7405 *scores.entry(key).or_insert(0.0) += 1.0 / (RRF_K + rank);
7406 }
7407 }
7408
7409 // Sort: score DESC, then key ASC for deterministic tie-breaking.
7410 let mut ranked: Vec<(String, f64)> = scores.into_iter().collect();
7411 ranked.sort_by(|a, b| {
7412 b.1.partial_cmp(&a.1)
7413 .unwrap_or(std::cmp::Ordering::Equal)
7414 .then(a.0.cmp(&b.0))
7415 });
7416 ranked.truncate(k);
7417 ranked
7418 }
7419
7420 /// For DST/testing: scratch BM25 search over live nodes without the index.
7421 /// Walks every live node, re-stems field tokens, computes corpus stats, and
7422 /// returns BM25-ranked results.
7423 ///
7424 /// The oracle: the ordered key list of `search(field, q)` must equal that of
7425 /// `scratch_search(field, q)` at every quiescent state.
7426 #[doc(hidden)]
7427 pub fn scratch_search(&self, field: &str, query: &str) -> Vec<(String, f64)> {
7428 use core_storage::fulltext::{parse_query, value_tokens_stemmed_with_positions};
7429 use std::collections::BTreeMap;
7430
7431 let groups = parse_query(query);
7432 if groups.is_empty() {
7433 return vec![];
7434 }
7435
7436 // --- Pass 1: collect all live indexed nodes with stemmed token data ---
7437 struct NodeData {
7438 key: String,
7439 /// stemmed_token → positions (sorted)
7440 tokens: BTreeMap<String, Vec<u32>>,
7441 dl: u32,
7442 }
7443
7444 let mut nodes: Vec<NodeData> = Vec::new();
7445 for id in 0..self.ids.len() as u32 {
7446 let Some(key) = self.ids.key_of(id) else {
7447 continue;
7448 };
7449 let Some(&sym) = self.labels.get(id as usize) else {
7450 continue;
7451 };
7452 if sym == u32::MAX {
7453 continue;
7454 }
7455 let label = match self.syms.resolve(sym) {
7456 Some(l) => l,
7457 None => continue,
7458 };
7459 if !self.fulltext.is_enabled(label, field) {
7460 continue;
7461 }
7462 let Some(value) = self.props_view().get(id, field).map(|vr| vr.into_value()) else {
7463 continue;
7464 };
7465 // Use value_tokens_stemmed_with_positions so list elements are
7466 // separated by POSITION_GAP — identical to the index path, which
7467 // prevents phrase queries from matching across element boundaries.
7468 let stemmed_with_pos = match &value {
7469 Value::Str(_) | Value::List(_) => value_tokens_stemmed_with_positions(&value),
7470 _ => continue,
7471 };
7472 let dl = stemmed_with_pos.len() as u32;
7473 let mut tok_map: BTreeMap<String, Vec<u32>> = BTreeMap::new();
7474 for (tok, pos) in stemmed_with_pos {
7475 tok_map.entry(tok).or_default().push(pos);
7476 }
7477 nodes.push(NodeData {
7478 key: key.to_string(),
7479 tokens: tok_map,
7480 dl,
7481 });
7482 }
7483
7484 if nodes.is_empty() {
7485 return vec![];
7486 }
7487
7488 // --- BM25 corpus stats ---
7489 let n = nodes.len() as f64;
7490 let avg_dl: f64 = nodes.iter().map(|nd| nd.dl as f64).sum::<f64>() / n;
7491 // df per stemmed token across all live indexed nodes.
7492 let mut df_map: BTreeMap<&str, f64> = BTreeMap::new();
7493 for nd in &nodes {
7494 for tok in nd.tokens.keys() {
7495 *df_map.entry(tok.as_str()).or_insert(0.0) += 1.0;
7496 }
7497 }
7498
7499 const K1: f64 = 1.2;
7500 const B: f64 = 0.75;
7501
7502 // --- Pass 2: score each node against each OR-group ---
7503 let mut results: Vec<(String, f64)> = Vec::new();
7504 for nd in &nodes {
7505 let dl = nd.dl as f64;
7506 let mut total_score = 0.0f64;
7507
7508 'group: for group in &groups {
7509 let mut group_score = 0.0f64;
7510
7511 for term in group {
7512 if term.negated {
7513 // Negated: if doc has this stemmed token → group fails.
7514 let present = if term.prefix {
7515 nd.tokens.keys().any(|t| t.starts_with(term.token.as_str()))
7516 } else {
7517 nd.tokens.contains_key(term.token.as_str())
7518 };
7519 if present {
7520 continue 'group;
7521 }
7522 continue;
7523 }
7524 if term.prefix {
7525 // Prefix: sum BM25 for all matching stemmed tokens.
7526 let mut prefix_matched = false;
7527 for (tok, positions) in &nd.tokens {
7528 if tok.starts_with(term.token.as_str()) {
7529 let tf = positions.len() as f64;
7530 let df = df_map.get(tok.as_str()).copied().unwrap_or(1.0);
7531 let idf = ((n - df + 0.5) / (df + 0.5) + 1.0).ln();
7532 let tf_norm =
7533 tf * (K1 + 1.0) / (tf + K1 * (1.0 - B + B * dl / avg_dl));
7534 group_score += idf * tf_norm;
7535 prefix_matched = true;
7536 }
7537 }
7538 if !prefix_matched {
7539 continue 'group;
7540 }
7541 } else {
7542 // term.token is already stemmed by parse_query; use directly.
7543 match nd.tokens.get(term.token.as_str()) {
7544 None => continue 'group,
7545 Some(positions) => {
7546 let tf = positions.len() as f64;
7547 let df = df_map.get(term.token.as_str()).copied().unwrap_or(1.0);
7548 let idf = ((n - df + 0.5) / (df + 0.5) + 1.0).ln();
7549 let tf_norm =
7550 tf * (K1 + 1.0) / (tf + K1 * (1.0 - B + B * dl / avg_dl));
7551 group_score += idf * tf_norm;
7552 }
7553 }
7554 }
7555 }
7556
7557 if group_score > 0.0 {
7558 total_score += group_score;
7559 }
7560 }
7561
7562 if total_score > 0.0 {
7563 results.push((nd.key.clone(), total_score));
7564 }
7565 }
7566
7567 results.sort_by(|a, b| {
7568 b.1.partial_cmp(&a.1)
7569 .unwrap_or(std::cmp::Ordering::Equal)
7570 .then(a.0.cmp(&b.0))
7571 });
7572 results
7573 }
7574
7575 /// Return the current view-maintained value of `view_prop` for node `key`.
7576 /// Equivalent to `get_prop` but documents that it reads a view-managed column.
7577 pub fn get_view_prop(&self, key: &str, view_prop: &str) -> Option<Value> {
7578 let id = self.ids.get(key)?;
7579 self.props_view()
7580 .get(id, view_prop)
7581 .map(|vr| vr.into_value())
7582 }
7583
7584 /// For testing / DST oracle: scratch recompute of a view value for one node.
7585 ///
7586 /// Returns `None` if the node does not exist, the view does not exist, or
7587 /// the view has no result for the node (e.g. Avg with no qualifying neighbors).
7588 #[doc(hidden)]
7589 pub fn scratch_view_value(&self, key: &str, view_name: &str) -> Option<Value> {
7590 let node = self.ids.get(key)?;
7591 let def = self.view_store.views().find(|v| v.name == view_name)?;
7592 // Use TopologyView so that NeighborAgg sees base + overlay edges
7593 // without materialising a temporary Topology (I1).
7594 let topo_view = self.topo_view();
7595 core_rules::views::compute_view_value(
7596 def,
7597 node,
7598 self.props_view(),
7599 &topo_view,
7600 &self.ids,
7601 &self.syms,
7602 &self.labels,
7603 )
7604 }
7605
7606 // -----------------------------------------------------------------------
7607 // Graph algorithm API
7608 // -----------------------------------------------------------------------
7609
7610 /// Run PageRank over the unified topology (manual + derived edges).
7611 ///
7612 /// Returns a [`PageRankReport`] with scores sorted descending (ties: key
7613 /// ascending). Set `config.edge_type` to restrict to one edge type.
7614 /// `config.converged` is `true` only when the power iteration converged
7615 /// within `config.max_iters` and within any time budget.
7616 pub fn pagerank(&self, config: &crate::algo::PageRankConfig) -> crate::algo::PageRankReport {
7617 let topo = build_topo_view(&self.topo, &self.base);
7618 let edge_props = self.edge_props_view();
7619 crate::algo::pagerank(
7620 &topo,
7621 &self.ids,
7622 &self.syms,
7623 &self.labels,
7624 &edge_props,
7625 config,
7626 )
7627 }
7628
7629 /// Weakly-connected components over the unified topology (treated as
7630 /// undirected regardless of how edges were inserted).
7631 ///
7632 /// Component IDs are the key of the smallest member in the component
7633 /// (deterministic). Result sorted by (component_id, key).
7634 pub fn connected_components(&self, config: &crate::algo::WccConfig) -> crate::algo::WccReport {
7635 let topo = build_topo_view(&self.topo, &self.base);
7636 let edge_props = self.edge_props_view();
7637 crate::algo::wcc(
7638 &topo,
7639 &self.ids,
7640 &self.syms,
7641 &self.labels,
7642 &edge_props,
7643 config,
7644 )
7645 }
7646
7647 /// Degree centrality for every live node.
7648 ///
7649 /// `direction`: `AlgoDir::Out` = out-degree, `AlgoDir::In` = in-degree,
7650 /// `AlgoDir::Both` = out + in (total directed degree).
7651 ///
7652 /// For one-shot ranking use this; for a live property updated on every
7653 /// write, create a Degree materialized view instead (see `docs/site/algorithms.md`).
7654 pub fn degree_centrality(
7655 &self,
7656 config: &crate::algo::DegreeConfig,
7657 ) -> crate::algo::DegreeReport {
7658 let topo = build_topo_view(&self.topo, &self.base);
7659 let edge_props = self.edge_props_view();
7660 crate::algo::degree_centrality(
7661 &topo,
7662 &self.ids,
7663 &self.syms,
7664 &self.labels,
7665 &edge_props,
7666 config,
7667 )
7668 }
7669
7670 /// Louvain community detection over the unified topology (undirected).
7671 ///
7672 /// See [`crate::algo::LouvainConfig`] for edge-type/weight/label
7673 /// restriction and [`crate::algo::CommunityReport`] for the shape of the
7674 /// result (communities sorted size-desc, then smallest member key asc).
7675 pub fn communities(&self, config: &crate::algo::LouvainConfig) -> crate::algo::CommunityReport {
7676 let topo = build_topo_view(&self.topo, &self.base);
7677 let edge_props = self.edge_props_view();
7678 crate::algo::louvain(
7679 &topo,
7680 &self.ids,
7681 &self.syms,
7682 &self.labels,
7683 &edge_props,
7684 config,
7685 )
7686 }
7687
7688 /// Write a vector of `(node_key, score)` pairs as `prop_name` on each node,
7689 /// atomically via a single write-batch (one WAL frame, one fsync).
7690 ///
7691 /// # Errors
7692 /// - [`GraphError::ReadOnly`]: called on an as-of instance.
7693 /// - [`GraphError::RuleInvalid`]: `prop_name` is managed by an existing view
7694 /// (collision check mirrors `create_view`).
7695 /// - [`GraphError::KeyNotFound`]: a key in `scores` does not exist as a live node.
7696 pub fn write_scores(&mut self, prop_name: &str, scores: &[(String, f64)]) -> Result<()> {
7697 if self.read_only {
7698 return Err(GraphError::ReadOnly);
7699 }
7700 // Collision check: refuse if prop_name is view-managed.
7701 if let Some(view_name) = self.view_store.view_for_prop(prop_name) {
7702 return Err(GraphError::RuleInvalid {
7703 detail: format!(
7704 "prop {:?} is managed by view {:?} and cannot be written as scores",
7705 prop_name, view_name
7706 ),
7707 });
7708 }
7709 // Refuse if prop_name is a view name itself (confusing namespace collision).
7710 if self.view_store.has_view(prop_name) {
7711 return Err(GraphError::RuleInvalid {
7712 detail: format!(
7713 "prop_name {:?} collides with an existing view name",
7714 prop_name
7715 ),
7716 });
7717 }
7718 // Write all scores in a single crash-atomic batch.
7719 self.write_batch(|b| {
7720 for (key, score) in scores {
7721 b.set_prop(key, prop_name, Value::Float(*score));
7722 }
7723 })?;
7724 Ok(())
7725 }
7726
7727 /// Return the value of `field` for the node with key `key`, or `None` if
7728 /// the node or field is absent. Reads through the overlay-over-base
7729 /// `ColumnsView`, materialising base values on demand (zero heap cost for
7730 /// overlay hits; one clone per base hit).
7731 pub fn get_prop(&self, key: &str, field: &str) -> Option<Value> {
7732 let id = self.ids.get(key)?;
7733 self.props_view().get(id, field).map(|vr| vr.into_value())
7734 }
7735
7736 pub fn has_node(&self, key: &str) -> bool {
7737 self.ids.get(key).is_some()
7738 }
7739
7740 /// Borrow the raw id map. Used by `NodeMask::from_keys` to resolve keys.
7741 pub(crate) fn ids(&self) -> &IdMap {
7742 &self.ids
7743 }
7744
7745 // -----------------------------------------------------------------------
7746 // Namespaces
7747 // -----------------------------------------------------------------------
7748
7749 /// The index `name` already has in `ns_names`, if any.
7750 fn ns_index_of(&self, name: &str) -> Option<u32> {
7751 self.ns_names
7752 .iter()
7753 .position(|n| n == name)
7754 .map(|i| i as u32)
7755 }
7756
7757 /// The index for `name`, appending it to `ns_names` when it is new.
7758 ///
7759 /// The table holds one entry per distinct namespace in the store — a
7760 /// tenant count, not a node count — so the linear scan is cheaper than a
7761 /// map and keeps `namespaces()` allocation-free of a second index.
7762 fn ns_index_for(&mut self, name: &str) -> u32 {
7763 match self.ns_index_of(name) {
7764 Some(i) => i,
7765 None => {
7766 self.ns_names.push(name.to_string());
7767 (self.ns_names.len() - 1) as u32
7768 }
7769 }
7770 }
7771
7772 /// The namespace name at `idx`, or [`NS_DEFAULT`] for an index this handle
7773 /// does not know (unreachable; the default is the narrowing answer).
7774 fn ns_name(&self, idx: u32) -> &str {
7775 self.ns_names
7776 .get(idx as usize)
7777 .map(String::as_str)
7778 .unwrap_or(NS_DEFAULT)
7779 }
7780
7781 /// The namespace index of dense node `id`, defaulting for an id with no
7782 /// entry (a node inserted before this handle rebuilt the array cannot
7783 /// exist: every insert path maintains it).
7784 fn node_ns_idx(&self, id: u32) -> u32 {
7785 self.node_ns
7786 .get(id as usize)
7787 .copied()
7788 .unwrap_or(NS_DEFAULT_IDX)
7789 }
7790
7791 /// File node `id` under namespace `name`, growing `node_ns` as `labels`
7792 /// grows. Called from `apply` for every node insert, live and replayed.
7793 fn set_node_ns(&mut self, id: u32, name: &str) {
7794 let idx = if name == NS_DEFAULT {
7795 NS_DEFAULT_IDX
7796 } else {
7797 self.ns_index_for(name)
7798 };
7799 if self.node_ns.len() <= id as usize {
7800 self.node_ns.resize(id as usize + 1, NS_DEFAULT_IDX);
7801 }
7802 self.node_ns[id as usize] = idx;
7803 }
7804
7805 /// Rebuild `node_ns` from the `ns` column — one pass, at the end of an
7806 /// open or a reload, after the snapshot is restored and the WAL replayed.
7807 ///
7808 /// A store with no `ns` column reads nothing: the column-name check fails
7809 /// and the vector is filled with one constant.
7810 fn rebuild_node_ns(&mut self) {
7811 let total = self.ids.len();
7812 self.ns_names.truncate(1);
7813 self.node_ns.clear();
7814 self.node_ns.resize(total, NS_DEFAULT_IDX);
7815 let has_ns_column = {
7816 let cv = self.props_view();
7817 cv.field_names().iter().any(|f| f == NS_PROP)
7818 };
7819 if !has_ns_column {
7820 return;
7821 }
7822 // Collected first so the props view is released before `ns_index_for`
7823 // takes `&mut self`.
7824 let named: Vec<(u32, String)> = {
7825 let cv = self.props_view();
7826 (0..total as u32)
7827 .filter_map(|id| match cv.get(id, NS_PROP).map(|vr| vr.into_value()) {
7828 Some(Value::Str(s)) if s != NS_DEFAULT => Some((id, s)),
7829 _ => None,
7830 })
7831 .collect()
7832 };
7833 for (id, name) in named {
7834 let idx = self.ns_index_for(&name);
7835 self.node_ns[id as usize] = idx;
7836 }
7837 }
7838
7839 /// Every namespace with at least one live node, in name order.
7840 ///
7841 /// `["default"]` on any store that has never named a namespace, including
7842 /// an empty one: a store is always at least its default namespace.
7843 pub fn namespaces(&self) -> Vec<String> {
7844 let mut out: BTreeSet<&str> = BTreeSet::new();
7845 out.insert(NS_DEFAULT);
7846 for (id, &idx) in self.node_ns.iter().enumerate() {
7847 if idx == NS_DEFAULT_IDX || !self.is_live_node(id as u32) {
7848 continue;
7849 }
7850 out.insert(self.ns_name(idx));
7851 }
7852 out.into_iter().map(str::to_string).collect()
7853 }
7854
7855 /// The namespace of `key`, or `None` when the key names no live node.
7856 pub fn namespace_of(&self, key: &str) -> Option<String> {
7857 let id = self.ids.get(key)?;
7858 if !self.is_live_node(id) {
7859 return None;
7860 }
7861 Some(self.ns_name(self.node_ns_idx(id)).to_string())
7862 }
7863
7864 /// Every live node in `namespace`, as a visibility mask.
7865 ///
7866 /// Built off `node_ns` on whichever handle this is, so on a temporal handle
7867 /// it is the namespace's membership at that commit. A name no node uses
7868 /// gives an empty mask — a namespace scope never widens.
7869 pub fn mask_for_namespace(&self, namespace: &str) -> crate::mask::NodeMask {
7870 let Some(idx) = self.ns_index_of(namespace) else {
7871 return crate::mask::NodeMask::from_ids(std::collections::HashSet::new());
7872 };
7873 let visible: std::collections::HashSet<u32> = (0..self.ids.len() as u32)
7874 .filter(|&id| self.node_ns_idx(id) == idx && self.is_live_node(id))
7875 .collect();
7876 crate::mask::NodeMask::from_ids(visible)
7877 }
7878
7879 /// Live-node test used by the namespace accessors: a deleted node keeps its
7880 /// dense id and its `node_ns` slot, and the label sentinel is what marks it
7881 /// gone — the same test `mask_for_role`'s label leg applies implicitly.
7882 fn is_live_node(&self, id: u32) -> bool {
7883 self.labels
7884 .get(id as usize)
7885 .is_some_and(|&sym| sym != u32::MAX)
7886 && self.ids.key_of(id).is_some()
7887 }
7888
7889 /// Per-namespace live node counts for [`Stats`], in name order.
7890 fn namespace_stats(&self) -> Vec<NamespaceStats> {
7891 let mut counts: BTreeMap<&str, usize> = BTreeMap::new();
7892 counts.insert(NS_DEFAULT, 0);
7893 for id in 0..self.ids.len() as u32 {
7894 if !self.is_live_node(id) {
7895 continue;
7896 }
7897 *counts
7898 .entry(self.ns_name(self.node_ns_idx(id)))
7899 .or_insert(0) += 1;
7900 }
7901 counts
7902 .into_iter()
7903 .filter(|&(name, n)| n > 0 || name == NS_DEFAULT)
7904 .map(|(name, nodes_live)| NamespaceStats {
7905 name: name.to_string(),
7906 nodes_live,
7907 })
7908 .collect()
7909 }
7910
7911 /// The namespace a create-class op would put its node in: the `ns` entry of
7912 /// the props it carries, normalised, with absent meaning [`NS_DEFAULT`].
7913 fn created_namespace<'a>(key: &str, props: &'a [(String, Value)]) -> Result<&'a str> {
7914 Ok(namespace_of_value(Self::sole_ns_entry(key, props)?))
7915 }
7916
7917 /// The one `ns` entry in a node's props, or `None` when it carries none.
7918 ///
7919 /// A props list naming `ns` twice is refused. Without that refusal the
7920 /// write path and the authorisation path can read the same list
7921 /// differently — one taking the first entry, the other the last — and
7922 /// `CREATE (n:L {ns: 'mine', ns: 'theirs'})` lands a node in a namespace
7923 /// the role was checked against the other of. One entry is the only shape
7924 /// where "the node's namespace" is a single fact, so it is the only shape
7925 /// accepted, and every reader of it agrees by construction.
7926 fn sole_ns_entry<'a>(key: &str, props: &'a [(String, Value)]) -> Result<Option<&'a Value>> {
7927 let mut found: Option<&'a Value> = None;
7928 for (field, value) in props {
7929 if field != NS_PROP {
7930 continue;
7931 }
7932 if found.is_some() {
7933 return Err(GraphError::RuleInvalid {
7934 detail: format!(
7935 "node {key}: {NS_PROP} is given more than once; a node has exactly \
7936 one namespace"
7937 ),
7938 });
7939 }
7940 found = Some(value);
7941 }
7942 Ok(found)
7943 }
7944
7945 /// The definition of the role a write authorisation names.
7946 ///
7947 /// `None` when `roles.json` was corrupt at open or the role has since been
7948 /// removed — neither can reach a write, because the authorisation carries a
7949 /// mask `mask_for_role` already resolved for that name.
7950 fn role_def_for(&self, role: &str) -> Option<&RoleDef> {
7951 self.roles.as_ref()?.iter().find(|r| r.name == role)
7952 }
7953
7954 /// Validate the `ns` entry of a node's props and drop an explicit default.
7955 ///
7956 /// Runs on the write path only (see `rewrite_wal_dense`), never on replay:
7957 /// a record that reached the WAL was already accepted here.
7958 fn normalise_insert_ns(
7959 key: &str,
7960 props: Vec<(String, Value)>,
7961 ) -> Result<(Vec<(String, Value)>, String)> {
7962 // One `ns` or none: this is where that is enforced, so every later
7963 // reader of the list — the authorisation gate, the two `apply` arms,
7964 // `node_ns` — is looking at a single entry and cannot disagree about
7965 // which one counts.
7966 Self::sole_ns_entry(key, &props)?;
7967 let mut name = NS_DEFAULT.to_string();
7968 let mut out = Vec::with_capacity(props.len());
7969 for (field, value) in props {
7970 if field != NS_PROP {
7971 out.push((field, value));
7972 continue;
7973 }
7974 let Value::Str(ref s) = value else {
7975 return Err(GraphError::RuleInvalid {
7976 detail: format!(
7977 "node {key}: {NS_PROP} must be a string naming a namespace, \
7978 got {value:?}"
7979 ),
7980 });
7981 };
7982 if !valid_namespace(s) {
7983 return Err(GraphError::RuleInvalid {
7984 detail: format!(
7985 "node {key}: {s:?} is not a valid namespace name — 1 to {NS_MAX_LEN} \
7986 characters of [A-Za-z0-9_.-]"
7987 ),
7988 });
7989 }
7990 name = s.clone();
7991 // An explicit default stores nothing, so a single-tenant store
7992 // never grows an `ns` column.
7993 if name != NS_DEFAULT {
7994 out.push((field, value));
7995 }
7996 }
7997 Ok((out, name))
7998 }
7999
8000 // -----------------------------------------------------------------------
8001 // RBAC role resolution
8002 // -----------------------------------------------------------------------
8003
8004 /// Parse `roles.json` bytes from `fs`.
8005 ///
8006 /// Return values:
8007 /// `Ok(Some(roles))` — file absent (returns `vec![]`) **or** file present
8008 /// and valid; in both cases `mask_for_role` uses the
8009 /// list normally (an absent file means no roles defined).
8010 /// `Ok(None)` — file present but corrupt or unrecognised version
8011 /// → poisoned state; `mask_for_role` returns `Err` for
8012 /// any role name until the file is fixed and the DB
8013 /// re-opened (or `apply_schema` is called to repair it).
8014 ///
8015 /// Note: `None` signals corruption, not absence — the opposite of what an
8016 /// optional "file missing" convention would suggest. The open path stores
8017 /// this result on `db.roles` directly.
8018 fn load_roles_from_fs(fs: &F) -> Result<Option<Vec<RoleDef>>> {
8019 let bytes = fs.read(FileId::Roles).map_err(GraphError::Io)?;
8020 if bytes.is_empty() {
8021 // Empty bytes means either the file is absent or zero-byte — both
8022 // are treated identically as "no roles defined". A zero-byte
8023 // roles.json does NOT widen access: an absent file and a zero-byte
8024 // file both resolve to an empty role list (sees nothing by default).
8025 return Ok(Some(vec![]));
8026 }
8027 match serde_json::from_slice::<RolesFile>(&bytes) {
8028 Ok(f) if matches!(f.version, 1..=4) => Ok(Some(f.roles)),
8029 // Corrupt or unrecognised version (>4): poison the roles state.
8030 // Never widen: a version this binary does not know may carry a
8031 // narrowing this binary would not apply.
8032 _ => Ok(None),
8033 }
8034 }
8035
8036 /// Resolve a role to a node-visibility mask against the current graph state.
8037 ///
8038 /// Returns `Err` when:
8039 /// - `roles.json` was present but corrupt at open (poisoned state), or
8040 /// - `role` does not match any defined role name.
8041 ///
8042 /// The mask union is: explicit `keys` (unknown keys silently ignored) plus
8043 /// all live nodes carrying any label in `labels` that also pass the role's
8044 /// [`visible_where`](crate::roles::RoleDef::visible_where) predicate, if it
8045 /// has one. Label resolution is live — new nodes of an allowed label are
8046 /// visible without re-applying the schema, and a property edited out of the
8047 /// predicate takes its node out of the mask on the next read. An empty
8048 /// union = empty mask = sees nothing.
8049 ///
8050 /// This is the one resolver every read path calls, live and as-of alike, so
8051 /// the predicate applies everywhere at once. On an as-of handle the role
8052 /// *definition* is the current one and the graph is the historical one: the
8053 /// predicate is evaluated against the property values at the commit being
8054 /// read.
8055 ///
8056 /// The result is memoised per `(role, commit_seq)`, so a scoped reader
8057 /// between two writes resolves the role once. See
8058 /// [`RoleMaskCache`](crate::mask::RoleMaskCache) for why that cannot go
8059 /// stale.
8060 pub fn mask_for_role(&self, role: &str) -> Result<crate::mask::NodeMask> {
8061 self.role_masks
8062 .get_or_build(role, self.commit_seq, || self.build_mask_for_role(role))
8063 .map(|m| (*m).clone())
8064 }
8065
8066 /// The mask an [`AsOfScope`] names, resolved against this handle.
8067 ///
8068 /// Shared by [`GraphDb::query_at_scoped`] and
8069 /// [`GraphDb::query_at_scoped_in_namespace`] so one scope resolves one way
8070 /// however the namespace leg is added.
8071 fn mask_at_scope(&self, scope: AsOfScope<'_>) -> Result<crate::mask::NodeMask> {
8072 // One resolver answers "what may this role see" — `mask_for_role` — and
8073 // it runs against this handle, so on a temporal one the answer is the
8074 // as-of one.
8075 Ok(match scope {
8076 AsOfScope::Role(role) => self.mask_for_role(role)?,
8077 AsOfScope::Keys(keys) => {
8078 crate::mask::NodeMask::from_keys(self, keys.iter().map(String::as_str))
8079 }
8080 AsOfScope::RoleAndKeys(role, keys) => {
8081 self.mask_for_role(role)?
8082 .intersect(&crate::mask::NodeMask::from_keys(
8083 self,
8084 keys.iter().map(String::as_str),
8085 ))
8086 }
8087 AsOfScope::Namespace(namespace) => self.mask_for_namespace(namespace),
8088 })
8089 }
8090
8091 /// Resolve `role` against the current graph, ignoring the memo.
8092 fn build_mask_for_role(&self, role: &str) -> Result<crate::mask::NodeMask> {
8093 let roles = self.roles.as_ref().ok_or_else(roles_poisoned)?;
8094 let def = roles
8095 .iter()
8096 .find(|r| r.name == role)
8097 .ok_or_else(|| GraphError::KeyNotFound {
8098 key: format!("role:{role}"),
8099 })?;
8100
8101 let mut visible = std::collections::HashSet::new();
8102
8103 // Key leg: resolve explicit keys to dense ids (unknown keys ignored).
8104 // An administrative grant, never narrowed by the predicate.
8105 for key in &def.keys {
8106 if let Some(id) = self.ids.get(key) {
8107 visible.insert(id);
8108 }
8109 }
8110
8111 // Label leg: live scan — iterate labels vec for matching symbol, and
8112 // when the role carries a predicate, test the property as well. The
8113 // property comes from the store's own merged view (overlay over the
8114 // mmap'd base), so an as-of handle reads the values of its own commit.
8115 let props = def.visible_where.as_ref().map(|_| self.props_view());
8116 for label_name in &def.labels {
8117 if let Some(sym) = self.syms.get(label_name) {
8118 for (i, &s) in self.labels.iter().enumerate() {
8119 if s != sym {
8120 continue;
8121 }
8122 let id = i as u32;
8123 match (&def.visible_where, &props) {
8124 (Some(pred), Some(view)) => {
8125 let value = view.get(id, &pred.field).map(|vr| vr.into_value());
8126 if pred.holds(value.as_ref()) {
8127 visible.insert(id);
8128 }
8129 }
8130 _ => {
8131 visible.insert(id);
8132 }
8133 }
8134 }
8135 }
8136 }
8137
8138 // Namespace leg: an intersection over the whole union, the key leg
8139 // included. A namespace is a tenancy boundary, so a key naming a node in
8140 // another tenant's namespace is not an administrative grant — and
8141 // `apply_schema` has already refused that role, so this only has to be
8142 // right about the node that moved into existence afterwards.
8143 if def.namespaces.is_some() {
8144 visible.retain(|&id| def.sees_namespace(self.ns_name(self.node_ns_idx(id))));
8145 }
8146
8147 Ok(crate::mask::NodeMask::from_ids(visible))
8148 }
8149
8150 /// Return the current list of role definitions.
8151 ///
8152 /// Returns an empty list when no roles are defined or when `roles.json`
8153 /// was corrupt at open (check [`mask_for_role`](Self::mask_for_role) for
8154 /// the fail-loud error in that case, or call
8155 /// [`roles_checked`](Self::roles_checked), which is this readout with that
8156 /// error in it).
8157 pub fn roles(&self) -> Vec<RoleDef> {
8158 self.roles.as_deref().unwrap_or(&[]).to_vec()
8159 }
8160
8161 /// The role definitions, or the poison error when `roles.json` was corrupt
8162 /// at open.
8163 ///
8164 /// [`roles`](Self::roles) answers `[]` both for a store that defines no
8165 /// roles and for one whose sidecar did not parse, and a caller validating a
8166 /// role name at boot cannot tell those apart. The wrong reading of the pair
8167 /// is the dangerous one: a store with no roles at all is an unrestricted
8168 /// store, so a poisoned file would read as "nothing is restricted here".
8169 ///
8170 /// This is the same answer, for the same cause, that
8171 /// [`mask_for_role`](Self::mask_for_role) gives on the first read.
8172 pub fn roles_checked(&self) -> Result<Vec<RoleDef>> {
8173 match self.roles.as_deref() {
8174 Some(roles) => Ok(roles.to_vec()),
8175 None => Err(roles_poisoned()),
8176 }
8177 }
8178
8179 // ── Role-scoped write authz ───────────────────────────────────────────────
8180
8181 /// Execute `ops` with optional role-scoped write authorization.
8182 ///
8183 /// - `None` → full authority, identical to [`write_batch`](Self::write_batch)
8184 /// (zero-cost bypass of all authz checks).
8185 /// - `Some(authz)` → the decision table is evaluated per-op BEFORE any WAL
8186 /// record is built. A denial returns an error with no WAL frame written
8187 /// (all-or-nothing at the authz boundary, then at the MutPreview boundary).
8188 ///
8189 /// See the plan's "authz decision table" section for the full semantics.
8190 pub fn write_batch_authz(
8191 &mut self,
8192 authz: Option<&WriteAuthz>,
8193 ops: Vec<BatchOp>,
8194 ) -> Result<(usize, usize)> {
8195 // Thread authz as a direct parameter — never touches pending_write_authz.
8196 self.commit_logged_batch(ops, None, authz.cloned())
8197 .map(inserted_pair)
8198 }
8199
8200 /// Execute a Cypher write statement with role-scoped write authorization.
8201 ///
8202 /// Resolves scope + mask from `self.roles` inside the call (same write-guard
8203 /// lifetime as execution, satisfying §5 lock discipline). The resolved
8204 /// `WriteAuthz` is stored as `pending_write_authz` for the duration of the
8205 /// call so that all inner `batch.commit()` calls are authz-checked.
8206 ///
8207 /// MERGE is handled specially: the MERGE scope precondition (§3.3) is
8208 /// checked in `exec_merge` BEFORE `has_node` to close the §6.2
8209 /// timing-oracle item (hidden ≡ absent for unscoped roles).
8210 ///
8211 /// Roles with `write: None` (v1 behavior) → `RoleWriteDenied` with
8212 /// "this endpoint is not permitted".
8213 pub fn query_write_authz(
8214 &mut self,
8215 role: &str,
8216 cypher: &str,
8217 params: &BTreeMap<String, Value>,
8218 ) -> Result<ResultSet> {
8219 // Resolve scope (fails fast if role has no write scope).
8220 // write:None → byte-identical v1 blanket-403 body (plan §v1-sidecar mandate).
8221 let scope =
8222 {
8223 let roles = self.roles.as_deref().ok_or_else(|| GraphError::Corrupt {
8224 detail: "roles.json was corrupt at open; re-open to restore role access".into(),
8225 })?;
8226 let def = roles.iter().find(|r| r.name == role).ok_or_else(|| {
8227 GraphError::KeyNotFound {
8228 key: format!("role:{role}"),
8229 }
8230 })?;
8231 def.write
8232 .clone()
8233 .ok_or_else(|| GraphError::RoleWriteDenied {
8234 reason: "role-bound token: writes are not permitted".into(),
8235 })?
8236 };
8237 // Resolve mask inside the call (same guard, §5 coherence).
8238 let mask = self.mask_for_role(role)?;
8239 self.pending_write_authz = Some(WriteAuthz {
8240 role: role.into(),
8241 scope,
8242 mask,
8243 });
8244 // RAII guard: always clears pending_write_authz on scope exit, including
8245 // on panic or early-return, mirroring the RestoreEmitDeltas precedent.
8246 struct ClearPendingAuthzOnDrop(*mut Option<WriteAuthz>);
8247 impl Drop for ClearPendingAuthzOnDrop {
8248 fn drop(&mut self) {
8249 // SAFETY: pointer into the owning GraphDb; guard is dropped
8250 // within this function's frame before it returns.
8251 unsafe { *self.0 = None };
8252 }
8253 }
8254 // SAFETY: raw pointer into self; guard dropped before this fn returns.
8255 let _authz_guard = ClearPendingAuthzOnDrop(&mut self.pending_write_authz as *mut _);
8256 let tokens = lex(cypher).map_err(|e| GraphError::QueryError {
8257 detail: format!("lex: {e}"),
8258 })?;
8259 let stmt = parse_write(&tokens).map_err(|e| GraphError::QueryError {
8260 detail: format!("parse: {e}"),
8261 })?;
8262 self.exec_write_stmt(stmt, params)
8263 }
8264
8265 /// Execute `ops` with optional role-scoped write authorization, suppressing
8266 /// fsync (for use inside the group-commit drain thread, which performs one
8267 /// group fsync after releasing the write lock).
8268 ///
8269 /// Identical to [`write_batch_authz`] except the fsync policy is temporarily
8270 /// forced to `Relaxed` for the duration of the call, matching the drain-thread
8271 /// contract established by [`commit_batch_nosync`].
8272 pub(crate) fn write_batch_authz_nosync(
8273 &mut self,
8274 authz: Option<&WriteAuthz>,
8275 ops: Vec<BatchOp>,
8276 ) -> Result<(usize, usize)> {
8277 let saved = self.fsync;
8278 struct RestoreFsync(*mut FsyncPolicy, FsyncPolicy);
8279 impl Drop for RestoreFsync {
8280 fn drop(&mut self) {
8281 // SAFETY: pointer into the owning GraphDb; guard is dropped
8282 // within the enclosing function's frame before it returns.
8283 unsafe { *self.0 = self.1 };
8284 }
8285 }
8286 // SAFETY: raw pointer into self; guard dropped before this fn returns.
8287 let _g = RestoreFsync(&mut self.fsync as *mut FsyncPolicy, saved);
8288 self.fsync = FsyncPolicy::Relaxed;
8289 self.commit_logged_batch(ops, None, authz.cloned())
8290 .map(inserted_pair)
8291 }
8292
8293 /// Execute a `/ingest` request with role-scoped write authorization.
8294 ///
8295 /// Resolves the role's `WriteScope` and `NodeMask` inside this call (same
8296 /// write-guard lifetime as the mutation, satisfying §5 lock discipline).
8297 /// Sets `pending_write_authz` for the duration of the call so that the
8298 /// `commit_ingest` → `commit_logged_batch` path picks up the authz context
8299 /// and evaluates the decision table per-op before any WAL write.
8300 ///
8301 /// §7.3: roles with empty `create_labels` will see every `InsertNode` op
8302 /// denied by the decision table with the appropriate §4.3 scope reason;
8303 /// no special HTTP-layer check is needed.
8304 ///
8305 /// Roles with `write: None` return `RoleWriteDenied` with
8306 /// "writes are not permitted" (byte-identical to v1 blanket 403).
8307 pub fn ingest_with_edges_authz(
8308 &mut self,
8309 role: &str,
8310 label: &str,
8311 rows: Vec<std::collections::BTreeMap<String, Value>>,
8312 opts: &crate::ingest::IngestOptions,
8313 edges: &[(String, String, String)],
8314 ) -> Result<crate::ingest::IngestReport> {
8315 // Resolve scope (fails fast if role has no write scope).
8316 // write:None → byte-identical v1 blanket-403 body (plan §v1-sidecar mandate).
8317 let scope =
8318 {
8319 let roles = self.roles.as_deref().ok_or_else(|| GraphError::Corrupt {
8320 detail: "roles.json was corrupt at open; re-open to restore role access".into(),
8321 })?;
8322 let def = roles.iter().find(|r| r.name == role).ok_or_else(|| {
8323 GraphError::KeyNotFound {
8324 key: format!("role:{role}"),
8325 }
8326 })?;
8327 def.write
8328 .clone()
8329 .ok_or_else(|| GraphError::RoleWriteDenied {
8330 reason: "role-bound token: writes are not permitted".into(),
8331 })?
8332 };
8333 let mask = self.mask_for_role(role)?;
8334 self.pending_write_authz = Some(WriteAuthz {
8335 role: role.into(),
8336 scope,
8337 mask,
8338 });
8339 // RAII guard: always clears pending_write_authz on scope exit, including
8340 // on panic or early-return, mirroring the RestoreEmitDeltas precedent.
8341 struct ClearPendingAuthzOnDrop(*mut Option<WriteAuthz>);
8342 impl Drop for ClearPendingAuthzOnDrop {
8343 fn drop(&mut self) {
8344 // SAFETY: pointer into the owning GraphDb; guard is dropped
8345 // within this function's frame before it returns.
8346 unsafe { *self.0 = None };
8347 }
8348 }
8349 // SAFETY: raw pointer into self; guard dropped before this fn returns.
8350 let _authz_guard = ClearPendingAuthzOnDrop(&mut self.pending_write_authz as *mut _);
8351 self.ingest_with_edges(label, rows, opts, edges)
8352 }
8353
8354 /// Evaluate the write-authz decision table for one `BatchOp`.
8355 ///
8356 /// Called by `commit_logged_batch` for each op when `pending_write_authz`
8357 /// is `Some`, BEFORE MutPreview. A denial returns an error immediately;
8358 /// the remaining ops are not evaluated and no WAL frame is written.
8359 ///
8360 /// `batch_created` carries the key→label pairs of nodes that earlier ops in
8361 /// THIS batch will create. Used by `InsertEdgeUpsert` to count same-batch
8362 /// placeholder nodes as visible (spec: "a placeholder endpoint the SAME
8363 /// batch creates counts as visible if its label passed the create-class gate").
8364 fn check_single_op_authz(
8365 &self,
8366 authz: &WriteAuthz,
8367 op: &BatchOp,
8368 batch_created: &BTreeMap<String, String>,
8369 ) -> Result<()> {
8370 // Helper: 3-way node status under the authz mask.
8371 //
8372 // Batch-created nodes (from earlier InsertNode in THIS batch) are treated
8373 // as Visible with their recorded label — their create gate already passed
8374 // and they are not yet in self.ids (not committed). This fixes the
8375 // MERGE+ON CREATE SET case where InsertNode + SetProp arrive together:
8376 // the SetProp must not see the node as Absent.
8377 let node_status = |key: &str| -> NodeAuthzStatus {
8378 if let Some(label) = batch_created.get(key) {
8379 return NodeAuthzStatus::Visible(label.clone());
8380 }
8381 match self.ids.get(key) {
8382 None => NodeAuthzStatus::Absent,
8383 Some(id) if !authz.mask.contains_id(id) => NodeAuthzStatus::Hidden,
8384 Some(id) => {
8385 let label = self
8386 .labels
8387 .get(id as usize)
8388 .and_then(|&sym| {
8389 if sym == u32::MAX {
8390 None
8391 } else {
8392 self.syms.resolve(sym).map(str::to_string)
8393 }
8394 })
8395 .unwrap_or_default();
8396 NodeAuthzStatus::Visible(label)
8397 }
8398 }
8399 };
8400
8401 // Helper: is an InsertEdgeUpsert endpoint visible?
8402 // A same-batch placeholder counts as visible if its label passed
8403 // the create-class gate (spec "upsert placeholder-counts-as-visible").
8404 let upsert_ep_visible = |ep_key: &str, placeholder_label: &str| -> bool {
8405 // In store and visible?
8406 if let Some(id) = self.ids.get(ep_key) {
8407 return authz.mask.contains_id(id);
8408 }
8409 // Created by an earlier op in this batch?
8410 if let Some(created_label) = batch_created.get(ep_key) {
8411 return authz.scope.create_labels.contains(created_label);
8412 }
8413 // Will be created by THIS InsertEdgeUpsert: placeholder_label
8414 // must pass the create-class gate.
8415 authz
8416 .scope
8417 .create_labels
8418 .contains(&placeholder_label.to_string())
8419 };
8420
8421 match op {
8422 // RenameNode / CreateRule / DeleteRule: defense-in-depth gate.
8423 // These ops are never routed to role-scoped paths by the HTTP layer,
8424 // but we 403 them here to close any future bypass route.
8425 //
8426 // InsertNodeOnConflict joins them: it is reachable only from the
8427 // embedded Python binding, which has no role token, and `Replace`
8428 // is a create and an update at once. Rather than split the decision
8429 // table for an op no role-scoped path constructs, refuse it — a
8430 // role-scoped caller writes through the ops that are already in the
8431 // table.
8432 BatchOp::RenameNode { .. }
8433 | BatchOp::CreateRule(_)
8434 | BatchOp::DeleteRule { .. }
8435 | BatchOp::InsertNodeOnConflict { .. } => {
8436 return Err(GraphError::RoleWriteDenied {
8437 reason: "role-bound token: this endpoint is not permitted".into(),
8438 });
8439 }
8440
8441 // ── CREATE-class: InsertNode ─────────────────────────────────────
8442 //
8443 // Decision table row 1 (scope-before-lookup): check label in
8444 // create_labels BEFORE any key lookup. This is the structural
8445 // closure of the §6.2 timing-oracle item — the denial fires even
8446 // when the store is EMPTY (see test_create_scope_denied_empty_store).
8447 BatchOp::InsertNode { label, key, props } => {
8448 if !authz.scope.create_labels.contains(label) {
8449 return Err(GraphError::RoleWriteDenied {
8450 reason: format!(
8451 "role-bound token: label '{}' not in write scope (create_labels)",
8452 label
8453 ),
8454 });
8455 }
8456 // A role bound to namespaces may only create inside them. The
8457 // never-widen rule is about what a write makes visible to *any*
8458 // party, not only to the writer: a node this role could never
8459 // read back is a write into somebody else's tenancy. Also a
8460 // scope check, so it runs before the key lookup — it discloses
8461 // nothing about the store. Covers Cypher `CREATE` and the node
8462 // `MERGE` creates, both of which arrive as this op.
8463 // Resolved before the role lookup so a props list naming `ns`
8464 // twice is refused for every role, scoped or not: it is the same
8465 // malformed write the seam refuses, and leaving it to the seam
8466 // would mean the gate had already read one of the two.
8467 let target = Self::created_namespace(key, props)?;
8468 if let Some(def) = self.role_def_for(&authz.role) {
8469 if !def.sees_namespace(target) {
8470 return Err(GraphError::RoleWriteDenied {
8471 reason: format!(
8472 "role-bound token: namespace '{target}' not in the role's \
8473 namespaces"
8474 ),
8475 });
8476 }
8477 }
8478 // Row 2/3: key lookup.
8479 match self.ids.get(key.as_str()) {
8480 Some(id) if authz.mask.contains_id(id) => {
8481 // Visible: DuplicateKey — let MutPreview handle this.
8482 }
8483 Some(_) => {
8484 // Hidden: indistinguishable from absent to the role.
8485 return Err(GraphError::RoleWriteDenied {
8486 reason: "role-bound token: target node not visible".into(),
8487 });
8488 }
8489 None => {
8490 // Absent: proceed (create).
8491 }
8492 }
8493 }
8494
8495 // ── UPDATE-class: SetProp, RemoveProp ────────────────────────────
8496 BatchOp::SetProp { key, .. } | BatchOp::RemoveProp { key, .. } => {
8497 if batch_created.contains_key(key.as_str()) {
8498 // Batch-created node: create gate already passed this batch.
8499 // Updating it in the same batch is always allowed, regardless
8500 // of update_labels (ruling §3.5: "writer just created it").
8501 } else {
8502 let label = match node_status(key) {
8503 NodeAuthzStatus::Visible(lbl) => lbl,
8504 _ => {
8505 return Err(GraphError::RoleWriteDenied {
8506 reason: "role-bound token: target node not visible".into(),
8507 });
8508 }
8509 };
8510 if !authz.scope.update_labels.contains(&label) {
8511 return Err(GraphError::RoleWriteDenied {
8512 reason: format!(
8513 "role-bound token: label '{}' not in write scope (update_labels)",
8514 label
8515 ),
8516 });
8517 }
8518 }
8519 }
8520
8521 // ── DELETE-class: DeleteNode ─────────────────────────────────────
8522 BatchOp::DeleteNode { key } => {
8523 let label = match node_status(key) {
8524 NodeAuthzStatus::Visible(lbl) => lbl,
8525 _ => {
8526 return Err(GraphError::RoleWriteDenied {
8527 reason: "role-bound token: target node not visible".into(),
8528 });
8529 }
8530 };
8531 if !authz.scope.delete_labels.contains(&label) {
8532 return Err(GraphError::RoleWriteDenied {
8533 reason: format!(
8534 "role-bound token: label '{}' not in write scope (delete_labels)",
8535 label
8536 ),
8537 });
8538 }
8539 }
8540
8541 // ── DELETE-class: DeleteEdge ─────────────────────────────────────
8542 //
8543 // Derived-edge rejection runs BEFORE the delete_edge_types scope
8544 // check (spec §3.5: "existing derived-edge rejection precedes
8545 // delete_edge_types check").
8546 BatchOp::DeleteEdge {
8547 edge_type,
8548 src_key,
8549 dst_key,
8550 } => {
8551 // Check provenance ownership BEFORE scope (spec §3.5 ordering).
8552 if let (Some(src_id), Some(dst_id), Some(et_sym)) = (
8553 self.ids.get(src_key.as_str()),
8554 self.ids.get(dst_key.as_str()),
8555 self.syms.get(edge_type.as_str()),
8556 ) {
8557 if self.engine.is_owned(et_sym, src_id, dst_id) {
8558 return Err(GraphError::RuleOwned {
8559 detail: format!(
8560 "edge {edge_type} {src_key}→{dst_key} is rule-owned; \
8561 delete or change the owning rule"
8562 ),
8563 });
8564 }
8565 // Also check would_derive via MutPreview (empty overlay, pre-batch).
8566 let preview = MutPreview::new(self);
8567 if preview.would_derive(edge_type, src_key, dst_key) {
8568 return Err(GraphError::RuleOwned {
8569 detail: format!(
8570 "edge {edge_type} {src_key}→{dst_key} is rule-owned; \
8571 delete or change the owning rule, or a live rule would \
8572 re-derive it"
8573 ),
8574 });
8575 }
8576 }
8577 // Scope check (AFTER derived-edge check, BEFORE endpoint visibility).
8578 if !authz.scope.delete_edge_types.contains(edge_type) {
8579 return Err(GraphError::RoleWriteDenied {
8580 reason: format!(
8581 "role-bound token: edge type '{}' not in write scope (delete_edge_types)",
8582 edge_type
8583 ),
8584 });
8585 }
8586 // Both endpoints must be visible.
8587 for ep_key in [src_key.as_str(), dst_key.as_str()] {
8588 match self.ids.get(ep_key) {
8589 None => {
8590 return Err(GraphError::RoleWriteDenied {
8591 reason: "role-bound token: edge endpoint not visible".into(),
8592 });
8593 }
8594 Some(id) if !authz.mask.contains_id(id) => {
8595 return Err(GraphError::RoleWriteDenied {
8596 reason: "role-bound token: edge endpoint not visible".into(),
8597 });
8598 }
8599 _ => {}
8600 }
8601 }
8602 }
8603
8604 // ── EDGE-CREATE: InsertEdge ──────────────────────────────────────
8605 //
8606 // Scope check BEFORE endpoint lookup (preserves timing symmetry).
8607 BatchOp::InsertEdge {
8608 edge_type,
8609 src_key,
8610 dst_key,
8611 } => {
8612 if !authz.scope.create_edge_types.contains(edge_type) {
8613 return Err(GraphError::RoleWriteDenied {
8614 reason: format!(
8615 "role-bound token: edge type '{}' not in write scope (create_edge_types)",
8616 edge_type
8617 ),
8618 });
8619 }
8620 // Both endpoints must be visible. A node created by an earlier
8621 // InsertNode in the same batch (tracked in batch_created) counts
8622 // as visible if its label passed the create-class gate.
8623 for ep_key in [src_key.as_str(), dst_key.as_str()] {
8624 if batch_created.contains_key(ep_key) {
8625 // Created earlier this batch — already scope-checked.
8626 continue;
8627 }
8628 match self.ids.get(ep_key) {
8629 None => {
8630 return Err(GraphError::RoleWriteDenied {
8631 reason: "role-bound token: edge endpoint not visible".into(),
8632 });
8633 }
8634 Some(id) if !authz.mask.contains_id(id) => {
8635 return Err(GraphError::RoleWriteDenied {
8636 reason: "role-bound token: edge endpoint not visible".into(),
8637 });
8638 }
8639 _ => {}
8640 }
8641 }
8642 }
8643
8644 // ── EDGE-CREATE: InsertEdgeUpsert ────────────────────────────────
8645 //
8646 // Scope check first; then endpoint visibility using same-batch
8647 // placeholder awareness (spec: "a placeholder endpoint the SAME
8648 // batch creates counts as visible if its label passed the
8649 // create-class gate").
8650 BatchOp::InsertEdgeUpsert {
8651 edge_type,
8652 src_key,
8653 dst_key,
8654 placeholder_label,
8655 } => {
8656 if !authz.scope.create_edge_types.contains(edge_type) {
8657 return Err(GraphError::RoleWriteDenied {
8658 reason: format!(
8659 "role-bound token: edge type '{}' not in write scope (create_edge_types)",
8660 edge_type
8661 ),
8662 });
8663 }
8664 // Check placeholder label against create_labels (create-class gate).
8665 // This ensures the auto-created endpoints are scope-allowed.
8666 for ep_key in [src_key.as_str(), dst_key.as_str()] {
8667 if !upsert_ep_visible(ep_key, placeholder_label) {
8668 return Err(GraphError::RoleWriteDenied {
8669 reason: "role-bound token: edge endpoint not visible".into(),
8670 });
8671 }
8672 }
8673 // A placeholder is created with no props, so it lands in the
8674 // default namespace. A role that cannot read `default` must not
8675 // create one there, for the same reason it may not create a node
8676 // there outright.
8677 //
8678 // The refusal is byte-identical to the hidden-endpoint one above,
8679 // and deliberately so: this arm fires only for an endpoint that
8680 // does **not** exist, and the one above only for an endpoint that
8681 // does. Two different strings would make the pair an existence
8682 // oracle — ask for an upsert and read off whether the key is
8683 // taken. Hidden ≡ absent is the rule everywhere else in this
8684 // table and it holds here too.
8685 if let Some(def) = self.role_def_for(&authz.role) {
8686 if !def.sees_namespace(NS_DEFAULT) {
8687 for ep_key in [src_key.as_str(), dst_key.as_str()] {
8688 if self.ids.get(ep_key).is_none() && !batch_created.contains_key(ep_key)
8689 {
8690 return Err(GraphError::RoleWriteDenied {
8691 reason: "role-bound token: edge endpoint not visible".into(),
8692 });
8693 }
8694 }
8695 }
8696 }
8697 }
8698 }
8699 Ok(())
8700 }
8701
8702 /// Write `roles` to `roles.json` atomically and update the in-memory list.
8703 ///
8704 /// Called by `apply_schema` when roles change. Never called on unchanged
8705 /// re-apply — this preserves byte-identical idempotency.
8706 pub(crate) fn commit_roles(&mut self, roles: Vec<RoleDef>) -> Result<()> {
8707 let file = RolesFile::new_versioned(roles.clone());
8708 let bytes = serde_json::to_vec(&file).map_err(|e| GraphError::Corrupt {
8709 detail: format!("roles serialization: {e}"),
8710 })?;
8711 self.fs
8712 .write_atomic(FileId::Roles, &bytes)
8713 .map_err(GraphError::Io)?;
8714 self.roles = Some(roles);
8715 // Rewriting the sidecar is not a commit, so `commit_seq` does not move
8716 // and a memoised mask would still match its version. Install a fresh
8717 // cache instead of clearing the shared one: a reader snapshot frozen
8718 // against the old definitions keeps the old `Arc` to itself and can
8719 // never publish an answer this handle would read back.
8720 self.role_masks = Arc::new(crate::mask::RoleMaskCache::new());
8721 // Refresh the MVCC frozen overlay so that reader() immediately sees the
8722 // updated role definitions without waiting for the next K-commit fold.
8723 self.fold_now();
8724 Ok(())
8725 }
8726
8727 fn view(&self) -> GraphView<'_> {
8728 GraphView {
8729 ids: &self.ids,
8730 syms: &self.syms,
8731 labels: &self.labels,
8732 props: self.props_view(),
8733 topo: self.topo_view(),
8734 edge_props: self.edge_props_view(),
8735 mask: None,
8736 prop_index: Some(&self.prop_index),
8737 }
8738 }
8739
8740 fn view_masked<'a>(&'a self, mask: &'a crate::mask::NodeMask) -> GraphView<'a> {
8741 GraphView {
8742 ids: &self.ids,
8743 syms: &self.syms,
8744 labels: &self.labels,
8745 props: self.props_view(),
8746 topo: self.topo_view(),
8747 edge_props: self.edge_props_view(),
8748 mask: Some(&mask.visible),
8749 prop_index: Some(&self.prop_index),
8750 }
8751 }
8752
8753 /// Execute a read-only Cypher query with a node visibility mask.
8754 ///
8755 /// Only nodes whose key is in `mask` are accessible: label scans, key
8756 /// lookups, and neighbor expansions all respect the mask. Edges where
8757 /// either endpoint is hidden are silently dropped.
8758 ///
8759 /// Returns `Err` with a "masked queries are read-only" message when
8760 /// `cypher` is a write statement (CREATE / MERGE / MATCH…SET / DELETE).
8761 pub fn query_masked(
8762 &self,
8763 cypher: &str,
8764 params: &std::collections::BTreeMap<String, Value>,
8765 mask: &crate::mask::NodeMask,
8766 ) -> Result<ResultSet> {
8767 // Reject write statements up front.
8768 let tokens = lex(cypher).map_err(|e| GraphError::QueryError {
8769 detail: format!("lex: {e}"),
8770 })?;
8771 if is_write_tokens(&tokens) {
8772 return Err(GraphError::MaskedReadOnly);
8773 }
8774 let union = parse_read(&tokens).map_err(|e| GraphError::QueryError {
8775 detail: format!("parse: {e}"),
8776 })?;
8777 // Each UNION part executes against the same masked view, so the mask
8778 // applies uniformly across the chain.
8779 execute_union(&self.view_masked(mask), &union, &Params(params)).map_err(|e| {
8780 GraphError::QueryError {
8781 detail: format!("execute: {e}"),
8782 }
8783 })
8784 }
8785
8786 pub fn node_ref(&self, key: &str) -> Option<NodeRef<'_, F>> {
8787 let id = self.ids.get(key)?;
8788 Some(NodeRef { db: self, id })
8789 }
8790
8791 /// BFS neighborhood expansion restricted to visible nodes in `mask`.
8792 ///
8793 /// Hidden nodes are never used as traversal intermediaries in either
8794 /// [`MaskMode::Omit`] or [`MaskMode::Stub`] — a visible node reachable
8795 /// only through a hidden node will not appear in results.
8796 ///
8797 /// In [`MaskMode::Stub`] mode, hidden nodes that are direct neighbours of
8798 /// a visited visible node are appended to the result as stub rows
8799 /// (`label` column is `null`, same key+depth columns as visible rows).
8800 /// They are NOT added to the BFS frontier.
8801 ///
8802 /// Returns `None` when `key` does not exist (caller should 404).
8803 ///
8804 /// **SECURITY**: role-token callers always pass an Omit-mode mask, so
8805 /// stub rows are never produced on the role path.
8806 pub fn neighborhood_masked(
8807 &self,
8808 key: &str,
8809 depth: u32,
8810 edge_types: Option<&[&str]>,
8811 dir: Dir,
8812 mask: &crate::mask::NodeMask,
8813 ) -> Option<ResultSet> {
8814 let start_id = self.ids.get(key)?;
8815 let view = self.view_masked(mask);
8816 let resolved: Option<Vec<u32>> = edge_types.map(|names| {
8817 names
8818 .iter()
8819 .filter_map(|name| view.syms.get(name))
8820 .collect()
8821 });
8822 let nb = neighborhood(&view, start_id, depth, resolved.as_deref(), dir);
8823 let mut rs = ResultSet::new(vec!["key".into(), "label".into(), "depth".into()]);
8824 // Collect visible BFS results (start_id at depth 0, BFS nodes after).
8825 let mut visited: Vec<(u32, u32)> = Vec::with_capacity(nb.nodes.len() + 1);
8826 visited.push((start_id, 0));
8827 for (nid, d) in &nb.nodes {
8828 let k = view.key_of(*nid);
8829 let label = view
8830 .label_of(*nid)
8831 .expect("real nodes always have a label; u32::MAX sentinel cannot occur");
8832 rs.push_row(vec![
8833 Some(Value::Str(k.to_string())),
8834 Some(Value::Str(label.to_string())),
8835 Some(Value::Int(*d as i64)),
8836 ]);
8837 visited.push((*nid, *d));
8838 }
8839 // Stub mode: add hidden direct neighbours of each visited node as stubs.
8840 // Hidden nodes are edge-endpoints only — they are not added to the BFS
8841 // frontier, so the BFS never expands through them.
8842 if mask.mode() == crate::mask::MaskMode::Stub {
8843 let raw_view = self.view();
8844 let mut seen: std::collections::HashSet<u32> =
8845 visited.iter().map(|(id, _)| *id).collect();
8846 for (node_id, node_depth) in &visited {
8847 if *node_depth >= depth {
8848 continue;
8849 }
8850 for e in expand(&raw_view, *node_id, resolved.as_deref(), dir) {
8851 let nbr = if e.src == *node_id { e.dst } else { e.src };
8852 if !mask.contains_id(nbr) && seen.insert(nbr) {
8853 if let Some(k) = self.ids.key_of(nbr) {
8854 rs.push_row(vec![
8855 Some(Value::Str(k.to_string())),
8856 None,
8857 Some(Value::Int((*node_depth + 1) as i64)),
8858 ]);
8859 }
8860 }
8861 }
8862 }
8863 }
8864 Some(rs)
8865 }
8866
8867 /// [`neighborhood_masked`](Self::neighborhood_masked) with the **subject
8868 /// check** a scoped caller needs: a start key the mask hides answers exactly
8869 /// as an absent one does.
8870 ///
8871 /// `neighborhood_masked` expands from any existing key, hidden or not,
8872 /// because a full-token caller supplying a client mask already knows which
8873 /// keys exist. A scoped caller does not, so telling it apart a hidden key
8874 /// from an absent one would be an existence oracle.
8875 ///
8876 /// Expansion itself is unchanged: hidden nodes are neither returned nor used
8877 /// as traversal intermediaries, so a visible node reachable only through a
8878 /// hidden one stays out of the result.
8879 ///
8880 /// Hidden or unknown `key` → [`GraphError::KeyNotFound`].
8881 pub fn neighborhood_scoped(
8882 &self,
8883 key: &str,
8884 depth: u32,
8885 edge_types: Option<&[&str]>,
8886 dir: Dir,
8887 mask: &crate::mask::NodeMask,
8888 ) -> Result<ResultSet> {
8889 if !mask.contains_node(self, key) {
8890 return Err(GraphError::KeyNotFound { key: key.into() });
8891 }
8892 self.neighborhood_masked(key, depth, edge_types, dir, mask)
8893 .ok_or_else(|| GraphError::KeyNotFound { key: key.into() })
8894 }
8895
8896 /// Live node's key, label, and columnar props. Unknown or tombstoned → `None`.
8897 pub fn node_info(&self, key: &str) -> Option<NodeInfo> {
8898 let n = self.node_ref(key)?;
8899 Some(NodeInfo {
8900 key: n.key().to_string(),
8901 label: n.label().to_string(),
8902 props: n.props(),
8903 })
8904 }
8905
8906 /// Look up a node with mask awareness.
8907 ///
8908 /// | Key state | Omit mode | Stub mode |
8909 /// |-------------------|-----------------|------------------------|
8910 /// | does not exist | `None` (→ 404) | `None` (→ 404) |
8911 /// | exists, visible | `Some(Visible)` | `Some(Visible)` |
8912 /// | exists, hidden | `None` (→ 404) | `Some(Restricted)` |
8913 ///
8914 /// **SECURITY**: only call from client-mask (full-token) paths.
8915 /// Role-token paths must use [`node_info`] after an explicit visibility check.
8916 pub fn node_info_masked(
8917 &self,
8918 key: &str,
8919 mask: &crate::mask::NodeMask,
8920 ) -> Option<MaskedNodeResult> {
8921 let id = self.ids.get(key)?;
8922 if mask.contains_id(id) {
8923 Some(MaskedNodeResult::Visible(self.node_info(key)?))
8924 } else {
8925 match mask.mode() {
8926 crate::mask::MaskMode::Stub => Some(MaskedNodeResult::Restricted),
8927 crate::mask::MaskMode::Omit => None,
8928 }
8929 }
8930 }
8931
8932 /// Get edges for `key` with mask-aware hidden-endpoint handling.
8933 ///
8934 /// - Omit mode: edges to hidden endpoints are excluded (same as role-path filtering).
8935 /// - Stub mode: edges to hidden endpoints are included; `src_restricted`/`dst_restricted`
8936 /// is `true` for each hidden endpoint.
8937 ///
8938 /// Unknown key → [`GraphError::KeyNotFound`].
8939 ///
8940 /// **SECURITY**: only call from client-mask (full-token) paths.
8941 pub fn node_edges_masked(
8942 &self,
8943 key: &str,
8944 mask: &crate::mask::NodeMask,
8945 ) -> Result<Vec<MaskedEdge>> {
8946 self.ensure_v8_base_sections_loaded();
8947 let id = self
8948 .ids
8949 .get(key)
8950 .ok_or_else(|| GraphError::KeyNotFound { key: key.into() })?;
8951 let derived: BTreeSet<(u32, u32, u32)> = self
8952 .engine
8953 .provenance_touching(id)
8954 .map(|(_rule, etype, src, dst)| (etype, src, dst))
8955 .collect();
8956 let mut edges = Vec::new();
8957 let tv = self.topo_view();
8958 for etype in tv.etypes() {
8959 // etype comes from the archived CSR (access_unchecked, no eager CRC).
8960 // A bit-flip in the large TOPOLOGY section can produce an etype id
8961 // that is not in the interner. Return Corrupt rather than panic.
8962 let edge_type = self
8963 .syms
8964 .resolve(etype)
8965 .ok_or_else(|| GraphError::Corrupt {
8966 detail: format!("v8: topology etype {etype} not in interner"),
8967 })?
8968 .to_string();
8969 for dir in [Direction::Out, Direction::In] {
8970 for &nbr in tv.neighbors(etype, dir, id).as_ref() {
8971 let nbr_restricted = !mask.contains_id(nbr);
8972 if nbr_restricted && mask.mode() == crate::mask::MaskMode::Omit {
8973 continue;
8974 }
8975 let nbr_key = self
8976 .ids
8977 .key_of(nbr)
8978 .ok_or_else(|| GraphError::Corrupt {
8979 detail: format!("topology id {nbr} has no key"),
8980 })?
8981 .to_string();
8982 let (src_id, dst_id, src_key, dst_key, src_restricted, dst_restricted) =
8983 match dir {
8984 Direction::Out => {
8985 (id, nbr, key.to_string(), nbr_key, false, nbr_restricted)
8986 }
8987 Direction::In => {
8988 (nbr, id, nbr_key, key.to_string(), nbr_restricted, false)
8989 }
8990 };
8991 edges.push(MaskedEdge {
8992 edge_type: edge_type.clone(),
8993 src_key,
8994 src_restricted,
8995 dst_key,
8996 dst_restricted,
8997 derived: derived.contains(&(etype, src_id, dst_id)),
8998 });
8999 }
9000 }
9001 }
9002 edges.sort_by(|a, b| {
9003 a.edge_type
9004 .cmp(&b.edge_type)
9005 .then(a.src_key.cmp(&b.src_key))
9006 .then(a.dst_key.cmp(&b.dst_key))
9007 });
9008 edges.dedup_by(|a, b| {
9009 a.edge_type == b.edge_type && a.src_key == b.src_key && a.dst_key == b.dst_key
9010 });
9011 Ok(edges)
9012 }
9013
9014 /// [`node_edges_masked`](Self::node_edges_masked) with the **subject check**
9015 /// a scoped caller needs, and a plain [`EdgeInfo`] list.
9016 ///
9017 /// `node_edges_masked` raises [`GraphError::KeyNotFound`] only when `key` is
9018 /// unknown; a key that exists but is hidden still yields its (filtered) edge
9019 /// list, which is correct for a full-token client mask and an existence
9020 /// oracle for a scoped one. Here a hidden subject answers exactly as an
9021 /// absent one does.
9022 ///
9023 /// Every edge naming a hidden endpoint is dropped, whatever the mask's
9024 /// [`MaskMode`](crate::mask::MaskMode): a scoped caller never sees a
9025 /// restricted stub, so there is nothing for it to render.
9026 ///
9027 /// Hidden or unknown `key` → [`GraphError::KeyNotFound`].
9028 pub fn node_edges_scoped(
9029 &self,
9030 key: &str,
9031 mask: &crate::mask::NodeMask,
9032 ) -> Result<Vec<EdgeInfo>> {
9033 if !mask.contains_node(self, key) {
9034 return Err(GraphError::KeyNotFound { key: key.into() });
9035 }
9036 Ok(self
9037 .node_edges_masked(key, mask)?
9038 .into_iter()
9039 .filter(|e| !e.src_restricted && !e.dst_restricted)
9040 .map(|e| EdgeInfo {
9041 edge_type: e.edge_type,
9042 src_key: e.src_key,
9043 dst_key: e.dst_key,
9044 derived: e.derived,
9045 })
9046 .collect())
9047 }
9048
9049 /// Every directed edge incident on `key`, both directions, every etype.
9050 ///
9051 /// Walk is `topology.etypes()` × `{Out, In}` × `neighbors()`. `derived` is
9052 /// membership in [`RuleEngine::provenance_touching`] (O(degree) via the
9053 /// Plan-8 `by_node` index). Sorted by `(edge_type, src_key, dst_key)`.
9054 /// Unknown key → [`GraphError::KeyNotFound`].
9055 pub fn node_edges(&self, key: &str) -> Result<Vec<EdgeInfo>> {
9056 self.ensure_v8_base_sections_loaded();
9057 let id = self
9058 .ids
9059 .get(key)
9060 .ok_or_else(|| GraphError::KeyNotFound { key: key.into() })?;
9061 let derived: BTreeSet<(u32, u32, u32)> = self
9062 .engine
9063 .provenance_touching(id)
9064 .map(|(_rule, etype, src, dst)| (etype, src, dst))
9065 .collect();
9066 let mut edges = Vec::new();
9067 let tv = self.topo_view();
9068 for etype in tv.etypes() {
9069 // Same guard as node_edges_masked: etype from unchecked-CRC CSR.
9070 let edge_type = self
9071 .syms
9072 .resolve(etype)
9073 .ok_or_else(|| GraphError::Corrupt {
9074 detail: format!("v8: topology etype {etype} not in interner"),
9075 })?
9076 .to_string();
9077 for dir in [Direction::Out, Direction::In] {
9078 for &nbr in tv.neighbors(etype, dir, id).as_ref() {
9079 let (src, dst, src_key, dst_key) = match dir {
9080 Direction::Out => (
9081 id,
9082 nbr,
9083 key.to_string(),
9084 self.ids
9085 .key_of(nbr)
9086 .ok_or_else(|| GraphError::Corrupt {
9087 detail: format!("topology id {nbr} has no key"),
9088 })?
9089 .to_string(),
9090 ),
9091 Direction::In => (
9092 nbr,
9093 id,
9094 self.ids
9095 .key_of(nbr)
9096 .ok_or_else(|| GraphError::Corrupt {
9097 detail: format!("topology id {nbr} has no key"),
9098 })?
9099 .to_string(),
9100 key.to_string(),
9101 ),
9102 };
9103 edges.push(EdgeInfo {
9104 edge_type: edge_type.clone(),
9105 src_key,
9106 dst_key,
9107 derived: derived.contains(&(etype, src, dst)),
9108 });
9109 }
9110 }
9111 }
9112 edges.sort_by(|a, b| {
9113 a.edge_type
9114 .cmp(&b.edge_type)
9115 .then(a.src_key.cmp(&b.src_key))
9116 .then(a.dst_key.cmp(&b.dst_key))
9117 });
9118 // Self-loops appear in both Out and In; sort makes the pair adjacent
9119 // (sort key matches PartialEq for this case) so one pass drops the dup.
9120 edges.dedup();
9121 Ok(edges)
9122 }
9123
9124 // ── Backup ────────────────────────────────────────────────────────────────
9125
9126 /// Copy this store to `dest` as a consistent, verified snapshot.
9127 ///
9128 /// Copies every durable file in the database directory — `snapshot.bin`,
9129 /// `wal.bin`, all `wal.<N>.archive` files, `wal.floor`, `wal.genesis`, and
9130 /// `roles.json` — into a freshly created `dest` directory using OS-level
9131 /// `copy` calls (no large in-process buffers).
9132 ///
9133 /// # Consistency guarantee
9134 ///
9135 /// The guarantee is **process-local**: the caller holds `&self`, which
9136 /// prevents any concurrent writer in the **same process** from modifying
9137 /// the files during the copy. Running `mushroomdb backup` against a
9138 /// directory that is **concurrently being written by another process** (e.g.
9139 /// `mushroomdb serve`) is **unsafe** — the copy can be torn. The post-copy
9140 /// `verified: true` result reduces but does not eliminate the risk of a
9141 /// silent corrupt backup (CRC catches many bit-flips; it cannot catch a
9142 /// consistent mid-write snapshot).
9143 ///
9144 /// **The safe path for a live-served store is `POST /backup` on the HTTP
9145 /// server.** That handler acquires the read lock on the shared database
9146 /// before calling this method, which is the correct cross-process
9147 /// synchronisation point because the server is the single process writing
9148 /// the files.
9149 ///
9150 /// After copying, opens the destination read-only and runs the CRC section
9151 /// verifier (`verify_snapshot`) to confirm byte-for-byte integrity.
9152 /// `BackupReport::verified` reflects whether both checks passed.
9153 ///
9154 /// Returns `Err` when `self` is not backed by a `RealFs` (e.g. `SimFs`).
9155 pub fn backup_to(&self, dest: &std::path::Path) -> Result<BackupReport> {
9156 // Derive source directory from snapshot_path (RealFs only).
9157 let src_dir = match self.fs.snapshot_path() {
9158 Some(p) => p.parent().map(|d| d.to_path_buf()).ok_or_else(|| {
9159 GraphError::Io(std::io::Error::other("snapshot has no parent dir"))
9160 })?,
9161 None => {
9162 return Err(GraphError::Io(std::io::Error::other(
9163 "backup_to requires a real filesystem (RealFs)",
9164 )))
9165 }
9166 };
9167
9168 std::fs::create_dir_all(dest)?;
9169
9170 let mut files: Vec<String> = Vec::new();
9171 let mut bytes: u64 = 0;
9172
9173 // Helper: copy src_dir/name → dest/name if the file exists.
9174 let mut try_copy = |name: &str| -> std::io::Result<()> {
9175 let src_path = src_dir.join(name);
9176 if src_path.exists() {
9177 let n = std::fs::copy(&src_path, dest.join(name))?;
9178 bytes += n;
9179 files.push(name.to_string());
9180 }
9181 Ok(())
9182 };
9183
9184 try_copy("snapshot.bin")?;
9185 try_copy("snapshot.bin.bak")?;
9186 try_copy("wal.bin")?;
9187 try_copy("wal.floor")?;
9188 try_copy("wal.genesis")?;
9189 try_copy("roles.json")?;
9190
9191 // Copy WAL archives.
9192 let archives = self.fs.list_archives()?;
9193 for n in &archives {
9194 let name = format!("wal.{n}.archive");
9195 let n_bytes = std::fs::copy(src_dir.join(&name), dest.join(&name))?;
9196 bytes += n_bytes;
9197 files.push(name);
9198 }
9199
9200 files.sort();
9201
9202 // Post-copy verification: open dest and run CRC checks.
9203 let snap_in_dest = dest.join("snapshot.bin").exists();
9204 let crc_ok = if snap_in_dest {
9205 crate::verify_snapshot(dest)
9206 .map(|results| results.iter().all(|(_, _, _, r)| r.is_ok()))
9207 .unwrap_or(false)
9208 } else {
9209 true // WAL-only store: nothing to CRC-check in snapshot
9210 };
9211 let opens_ok = GraphDb::<core_storage::fs::RealFs>::open(dest).is_ok();
9212 let verified = crc_ok && opens_ok;
9213
9214 Ok(BackupReport {
9215 files,
9216 bytes,
9217 verified,
9218 })
9219 }
9220
9221 // ── Export helpers ────────────────────────────────────────────────────────
9222
9223 /// All live nodes, sorted by key (deterministic).
9224 ///
9225 /// Reads base + WAL overlay. Tombstoned nodes are excluded.
9226 pub fn all_nodes_for_export(&self) -> Vec<NodeInfo> {
9227 self.ensure_v8_base_sections_loaded();
9228 let pv = self.props_view();
9229 let mut nodes = Vec::new();
9230 for id in 0..self.ids.len() as u32 {
9231 let Some(key) = self.ids.key_of(id) else {
9232 continue;
9233 };
9234 let Some(&sym) = self.labels.get(id as usize) else {
9235 continue;
9236 };
9237 if sym == u32::MAX {
9238 continue; // tombstoned
9239 }
9240 let Some(label) = self.syms.resolve(sym) else {
9241 continue;
9242 };
9243 let mut props = BTreeMap::new();
9244 for field in pv.field_names() {
9245 if let Some(vr) = pv.get(id, &field) {
9246 props.insert(field, vr.into_value());
9247 }
9248 }
9249 nodes.push(NodeInfo {
9250 key: key.to_string(),
9251 label: label.to_string(),
9252 props,
9253 });
9254 }
9255 nodes.sort_by(|a, b| a.key.cmp(&b.key));
9256 nodes
9257 }
9258
9259 /// All directed edges, sorted by `(edge_type, src, dst)`. Each edge appears once.
9260 ///
9261 /// Derived edges carry `derived: true` and the creating rule's name in `rule`.
9262 /// Manual edges carry `derived: false` and `rule: None`.
9263 /// `weight` is the creating rule's `weight_prop` value read off the edge
9264 /// (numeric only), mirroring the convention used by [`GraphDb::explain`]
9265 /// and [`GraphDb::weighted_edges`]. Deterministic across runs on the same
9266 /// store state.
9267 pub fn all_edges_for_export(&self) -> Vec<ExportEdge> {
9268 self.ensure_v8_base_sections_loaded();
9269
9270 // Build (etype_sym, src_id, dst_id) → rule_name for O(1) derivation lookup.
9271 let mut prov: HashMap<(u32, u32, u32), String> = HashMap::new();
9272 for (rule_name, triples) in self.engine.provenance() {
9273 for &(etype, src, dst) in triples {
9274 prov.insert((etype, src, dst), rule_name.clone());
9275 }
9276 }
9277
9278 // rule_name → weight_prop, for O(1) lookup per derived edge.
9279 let weight_props: HashMap<&str, Option<&str>> = self
9280 .engine
9281 .rules()
9282 .map(|r| (r.name.as_str(), r.weight_prop.as_deref()))
9283 .collect();
9284
9285 let tv = self.topo_view();
9286 let ep = self.edge_props_view();
9287 let mut edges = Vec::new();
9288
9289 for id in 0..self.ids.len() as u32 {
9290 let Some(key) = self.ids.key_of(id) else {
9291 continue;
9292 };
9293 let Some(&lsym) = self.labels.get(id as usize) else {
9294 continue;
9295 };
9296 if lsym == u32::MAX {
9297 continue; // tombstoned
9298 }
9299
9300 for etype_sym in tv.etypes() {
9301 // etype from archived CSR (access_unchecked, no eager CRC).
9302 // Skip edges whose etype is not in the interner; this can only
9303 // occur with a corrupt large TOPOLOGY section (bit-flip on an
9304 // etype field in the archived data). The function returns Vec,
9305 // not Result, so we continue rather than propagate.
9306 let Some(edge_type) = self.syms.resolve(etype_sym) else {
9307 continue;
9308 };
9309 let edge_type = edge_type.to_string();
9310 for &nbr in tv.neighbors(etype_sym, Direction::Out, id).as_ref() {
9311 let Some(dst_key) = self.ids.key_of(nbr) else {
9312 continue; // skip corrupt entries
9313 };
9314 let prov_key = (etype_sym, id, nbr);
9315 let rule = prov.get(&prov_key).cloned();
9316 let derived = rule.is_some();
9317 let weight = rule
9318 .as_deref()
9319 .and_then(|rn| weight_props.get(rn).copied().flatten())
9320 .and_then(|prop| match ep.get(etype_sym, id, nbr, prop) {
9321 Some(Value::Float(f)) => Some(f),
9322 Some(Value::Int(i)) => Some(i as f64),
9323 _ => None,
9324 });
9325 edges.push(ExportEdge {
9326 edge_type: edge_type.clone(),
9327 src: key.to_string(),
9328 dst: dst_key.to_string(),
9329 derived,
9330 rule,
9331 weight,
9332 });
9333 }
9334 }
9335 }
9336
9337 edges.sort_by(|a, b| {
9338 a.edge_type
9339 .cmp(&b.edge_type)
9340 .then(a.src.cmp(&b.src))
9341 .then(a.dst.cmp(&b.dst))
9342 });
9343 edges
9344 }
9345
9346 /// What each edge type *is*, without building one record per edge.
9347 ///
9348 /// [`all_edges_for_export`](Self::all_edges_for_export) answers the same
9349 /// question by materialising every edge — three `String`s apiece, a
9350 /// provenance `HashMap` over every derived edge, and a final sort. That is
9351 /// the right shape for an export, and the wrong one for a summary: on a
9352 /// store with 1.3 M derived edges it allocates hundreds of megabytes to
9353 /// produce nine lines. This walks the topology instead, summing neighbour
9354 /// slice lengths and collecting *label symbols* rather than label strings,
9355 /// so the per-edge cost is an integer add and a set insert on a set with
9356 /// as many members as the store has labels.
9357 ///
9358 /// The rule names come off the rule *definitions*, which each declare the
9359 /// `edge_type` they derive, so naming them costs one pass over the rules
9360 /// rather than one provenance lookup per edge. That is also why `rules`
9361 /// is a list: two rules may derive the same type — the association store
9362 /// derives `INDUSTRY_ALIGNMENT` from both a talent→company and a
9363 /// talent→job rule — and naming only one of them would be a half-truth.
9364 /// A type with no rules is one written by hand.
9365 ///
9366 /// `sample` is the first edge of the type in the store's own id order,
9367 /// which is insertion order: deterministic for a given store, and not the
9368 /// same as key order, which cannot be had without resolving a key per
9369 /// edge. Sorted by `edge_type`.
9370 pub fn edge_type_census(&self) -> Vec<EdgeTypeCensus> {
9371 self.ensure_v8_base_sections_loaded();
9372
9373 let mut rules_by_type: BTreeMap<&str, BTreeSet<&str>> = BTreeMap::new();
9374 for r in self.engine.rules() {
9375 rules_by_type
9376 .entry(r.edge_type.as_str())
9377 .or_default()
9378 .insert(r.name.as_str());
9379 }
9380
9381 let tv = self.topo_view();
9382 let node_count = self.ids.len() as u32;
9383 let mut out = Vec::new();
9384 for etype_sym in tv.etypes() {
9385 // An etype the interner cannot resolve means a corrupt TOPOLOGY
9386 // section; skip it rather than name it, as `all_edges_for_export`
9387 // does for the same reason.
9388 let Some(edge_type) = self.syms.resolve(etype_sym) else {
9389 continue;
9390 };
9391 let mut edges: u64 = 0;
9392 let mut src_syms: BTreeSet<u32> = BTreeSet::new();
9393 let mut dst_syms: BTreeSet<u32> = BTreeSet::new();
9394 let mut sample: Option<(u32, u32)> = None;
9395 for id in 0..node_count {
9396 let Some(&lsym) = self.labels.get(id as usize) else {
9397 continue;
9398 };
9399 if lsym == u32::MAX {
9400 continue; // tombstoned
9401 }
9402 let nbrs = tv.neighbors(etype_sym, Direction::Out, id);
9403 let nbrs = nbrs.as_ref();
9404 if nbrs.is_empty() {
9405 continue;
9406 }
9407 edges += nbrs.len() as u64;
9408 src_syms.insert(lsym);
9409 for &nbr in nbrs {
9410 if let Some(&dsym) = self.labels.get(nbr as usize) {
9411 if dsym != u32::MAX {
9412 dst_syms.insert(dsym);
9413 }
9414 }
9415 }
9416 if sample.is_none() {
9417 sample = Some((id, nbrs[0]));
9418 }
9419 }
9420 let resolve = |syms: &BTreeSet<u32>| -> Vec<String> {
9421 syms.iter()
9422 .filter_map(|&s| self.syms.resolve(s))
9423 .map(ToString::to_string)
9424 .collect()
9425 };
9426 out.push(EdgeTypeCensus {
9427 edge_type: edge_type.to_string(),
9428 edges,
9429 src_labels: resolve(&src_syms),
9430 dst_labels: resolve(&dst_syms),
9431 rules: rules_by_type
9432 .get(edge_type)
9433 .map(|rs| rs.iter().map(ToString::to_string).collect())
9434 .unwrap_or_default(),
9435 sample: sample.and_then(|(s, d)| {
9436 Some((
9437 self.ids.key_of(s)?.to_string(),
9438 self.ids.key_of(d)?.to_string(),
9439 ))
9440 }),
9441 });
9442 }
9443 out.sort_by(|a, b| a.edge_type.cmp(&b.edge_type));
9444 out
9445 }
9446
9447 /// All directed edges of `edge_type`, with the raw value of `weight_prop`
9448 /// on each edge when given.
9449 ///
9450 /// `weight` is `Some(f)` only when `weight_prop` is set and the edge
9451 /// carries that property with a numeric (`Int`/`Float`) value; otherwise
9452 /// `None` — callers that want a default weight (e.g. `1.0` for missing
9453 /// props) apply it themselves, matching the convention used internally
9454 /// by [`GraphDb::pagerank`], [`GraphDb::connected_components`],
9455 /// [`GraphDb::degree_centrality`], and [`GraphDb::communities`].
9456 ///
9457 /// Sorted by `(src, dst)` for determinism. Reads the unified topology
9458 /// (manual + rule-derived edges). An unknown `edge_type` returns an
9459 /// empty vec.
9460 pub fn weighted_edges(
9461 &self,
9462 edge_type: &str,
9463 weight_prop: Option<&str>,
9464 ) -> Vec<(String, String, Option<f64>)> {
9465 let Some(etype_sym) = self.syms.get(edge_type) else {
9466 return Vec::new();
9467 };
9468 let tv = self.topo_view();
9469 let ep = self.edge_props_view();
9470 let mut out = Vec::new();
9471 for id in 0..self.ids.len() as u32 {
9472 let Some(key) = self.ids.key_of(id) else {
9473 continue;
9474 };
9475 let Some(&sym) = self.labels.get(id as usize) else {
9476 continue;
9477 };
9478 if sym == u32::MAX {
9479 continue; // tombstoned
9480 }
9481 for &nbr in tv.neighbors(etype_sym, Direction::Out, id).as_ref() {
9482 let Some(dst_key) = self.ids.key_of(nbr) else {
9483 continue;
9484 };
9485 let weight = weight_prop.and_then(|prop| match ep.get(etype_sym, id, nbr, prop) {
9486 Some(Value::Float(f)) => Some(f),
9487 Some(Value::Int(i)) => Some(i as f64),
9488 _ => None,
9489 });
9490 out.push((key.to_string(), dst_key.to_string(), weight));
9491 }
9492 }
9493 out.sort_by(|a, b| a.0.cmp(&b.0).then(a.1.cmp(&b.1)));
9494 out
9495 }
9496
9497 pub fn nodes_with_label(&self, label: &str) -> Vec<NodeRef<'_, F>> {
9498 self.view()
9499 .nodes_with_label(label)
9500 .into_iter()
9501 .map(|id| NodeRef { db: self, id })
9502 .collect()
9503 }
9504
9505 pub fn find_nodes(&self, label: &str, filter: &Filter) -> Vec<NodeRef<'_, F>> {
9506 let view = self.view();
9507 view.nodes_with_label(label)
9508 .into_iter()
9509 .filter(|&id| {
9510 eval_filter(filter, &|field| {
9511 view.prop(id, field).map(|vr| vr.into_value())
9512 })
9513 })
9514 .map(|id| NodeRef { db: self, id })
9515 .collect()
9516 }
9517
9518 /// Returns `true` if any approximate (HNSW) VectorSimilar rule covers
9519 /// `field`. Use as a capability probe: when `true`, `find_similar_vector`
9520 /// with `label = None` will use the native ANN path rather than the O(n)
9521 /// brute-force scan.
9522 pub fn has_vector_rule(&self, field: &str) -> bool {
9523 self.engine.hnsw_has_rule(field)
9524 }
9525
9526 /// How many HNSW graphs this handle has built from scratch since it was
9527 /// opened (one per side of an approximate rule).
9528 ///
9529 /// An open that restored every graph from the snapshot reports `0`.
9530 /// Exposed for tests that assert the open path reuses the persisted index
9531 /// rather than rebuilding it; not part of the stable surface.
9532 #[doc(hidden)]
9533 pub fn hnsw_build_count(&self) -> u64 {
9534 self.engine.hnsw_build_count()
9535 }
9536
9537 /// How many rules this handle still holds a lazily-decoded HNSW graph for.
9538 ///
9539 /// Zero before the first ANN query on a clean open, and again once the
9540 /// live indexes own the graphs. See [`core_rules::RuleEngine::lazy_hnsw_len`].
9541 /// Exposed for tests that assert the lazy copies are released; not part of
9542 /// the stable surface.
9543 #[doc(hidden)]
9544 pub fn lazy_hnsw_len(&self) -> usize {
9545 self.engine.lazy_hnsw_len()
9546 }
9547
9548 /// Find nodes whose `field` vector is most similar to `q` (cosine
9549 /// similarity), returning up to `k` results with similarity ≥ `min`,
9550 /// sorted descending.
9551 ///
9552 /// When `label` is `None` the search spans all labels (via
9553 /// `hnsw_search_any_dst` or a full brute-force scan); when `label` is
9554 /// `Some(lbl)` it restricts to nodes with that label.
9555 ///
9556 /// Uses the HNSW index when one is available (fast path); otherwise falls
9557 /// back to an O(n) brute-force scan.
9558 ///
9559 /// **The index supplies candidates, never scores.** Its own distances are
9560 /// `f32` (accurate to ~1e-6, so an exact duplicate scores 0.9999999), so
9561 /// every candidate is re-scored from the `f64` property vectors by
9562 /// [`exact_vector_similarity`] before `min`, the ordering and the reported
9563 /// score are decided. `k + VECTOR_RESCORE_MARGIN` candidates are fetched so
9564 /// the re-ordering cannot drop a true top-`k` member; see that constant for
9565 /// the rule. The score a caller receives is therefore the same number the
9566 /// brute-force path would have produced, to `f64` precision, and `min = 1.0`
9567 /// finds an exact duplicate.
9568 pub fn find_similar_vector(
9569 &self,
9570 field: &str,
9571 label: Option<&str>,
9572 q: &[f64],
9573 k: usize,
9574 min: f64,
9575 ) -> Vec<(String, f64)> {
9576 self.find_similar_vector_filtered(field, label, q, k, min, None, None, false)
9577 .expect("find_similar_vector_filtered is infallible without where_")
9578 }
9579
9580 /// Like [`find_similar_vector`] but restricts results to nodes visible in
9581 /// `mask`. Hidden nodes never appear in results; the mask is applied
9582 /// **before** k-truncation so a caller still receives up to `k` visible
9583 /// hits.
9584 ///
9585 /// # HNSW path (widening beam)
9586 ///
9587 /// When an HNSW index covers the request, the beam starts at an over-fetch
9588 /// of `k × n / |visible|` (plus the rescore margin) when the mask's
9589 /// selectivity is known from the index length, otherwise at `k` plus that
9590 /// margin. If fewer than `k` visible candidates remain after the mask and
9591 /// `min` filter, the beam doubles — the same ×2 loop exact `VectorSimilar`
9592 /// rules use, capped at `ef_max()` (`EF_MAX` = 4,096). Reaching the cap,
9593 /// or a beam that comes back short of its own width, falls through to the
9594 /// exhaustive masked scan rather than returning a short result.
9595 ///
9596 /// Every surviving candidate is re-scored from the `f64` property vectors,
9597 /// exactly as [`find_similar_vector`] does and for the same reason.
9598 ///
9599 /// # Brute-force path
9600 ///
9601 /// When no HNSW index covers the request, or the beam cannot admit `k`
9602 /// hits, the function builds a masked [`GraphView`] so that `nodes_all` /
9603 /// `nodes_with_label` return only visible nodes, guaranteeing exact `k`
9604 /// results (or all visible nodes if fewer than `k` exist).
9605 pub fn find_similar_vector_masked(
9606 &self,
9607 field: &str,
9608 label: Option<&str>,
9609 q: &[f64],
9610 k: usize,
9611 min: f64,
9612 mask: &crate::mask::NodeMask,
9613 ) -> Vec<(String, f64)> {
9614 self.find_similar_vector_filtered(field, label, q, k, min, Some(mask), None, false)
9615 .expect("find_similar_vector_filtered is infallible without where_")
9616 }
9617
9618 /// Exact or ANN kNN with optional key-list `mask` and property `where_`.
9619 ///
9620 /// `where_` present and failing [`PropPredicate::validate_named`] `"where"`
9621 /// → `QueryError`. `exact=true` or `where_=Some` skip HNSW and GEMM-brute
9622 /// the candidate set (`label ∩ mask ∩ holds(where)`). `mask` alone still
9623 /// uses HNSW when an index covers the field.
9624 #[allow(clippy::too_many_arguments)]
9625 pub fn find_similar_vector_filtered(
9626 &self,
9627 field: &str,
9628 label: Option<&str>,
9629 q: &[f64],
9630 k: usize,
9631 min: f64,
9632 mask: Option<&crate::mask::NodeMask>,
9633 where_: Option<&PropPredicate>,
9634 exact: bool,
9635 ) -> Result<Vec<(String, f64)>> {
9636 self.find_similar_vector_as(
9637 field,
9638 label,
9639 q,
9640 k,
9641 min,
9642 mask,
9643 where_,
9644 exact,
9645 ExactnessCaller::Vector,
9646 )
9647 }
9648
9649 /// [`find_similar_vector_filtered`](Self::find_similar_vector_filtered)
9650 /// with the caller shape named, so the exactness warning can advise the
9651 /// signature that actually reached it. Everything else is identical.
9652 #[allow(clippy::too_many_arguments)]
9653 fn find_similar_vector_as(
9654 &self,
9655 field: &str,
9656 label: Option<&str>,
9657 q: &[f64],
9658 k: usize,
9659 min: f64,
9660 mask: Option<&crate::mask::NodeMask>,
9661 where_: Option<&PropPredicate>,
9662 exact: bool,
9663 caller: ExactnessCaller,
9664 ) -> Result<Vec<(String, f64)>> {
9665 if let Some(pred) = where_ {
9666 pred.validate_named("where")
9667 .map_err(|detail| GraphError::QueryError { detail })?;
9668 }
9669
9670 // Ensure any HNSW blobs retained from the snapshot are deserialized
9671 // before the first ANN query on a clean-open (no-WAL) path. The
9672 // section read has to come first: on a clean open nothing else has
9673 // called it, so without it `retained_hnsw_blobs` is empty,
9674 // `ensure_hnsw_loaded` caches an empty map in its `OnceLock`, and every
9675 // approximate query on the handle runs brute force — correct results,
9676 // silently off the index. Both calls are idempotent and cheap once hot.
9677 self.ensure_v8_base_sections_loaded();
9678 self.engine.ensure_hnsw_loaded();
9679 let norm: f64 = q.iter().map(|x| x * x).sum::<f64>().sqrt();
9680 if norm == 0.0 {
9681 return Ok(vec![]);
9682 }
9683 if let Some(m) = mask {
9684 if k == 0 || m.is_empty() {
9685 return Ok(vec![]);
9686 }
9687 }
9688 let q_unit: Vec<f64> = q.iter().map(|x| x / norm).collect();
9689
9690 // `where` implies exact: a predicate must not ride a silent ANN.
9691 let skip_hnsw = exact || where_.is_some();
9692 if !skip_hnsw {
9693 if let Some(mask) = mask {
9694 if let Some(out) =
9695 self.find_similar_hnsw_masked(field, label, &q_unit, k, min, mask, caller)
9696 {
9697 return Ok(out);
9698 }
9699 } else if let Some(out) = self.find_similar_hnsw(field, label, &q_unit, k, min) {
9700 return Ok(out);
9701 }
9702 }
9703
9704 let view = match mask {
9705 Some(m) => self.view_masked(m),
9706 None => self.view(),
9707 };
9708 let candidate_ids = Self::vector_candidates(&view, label, where_);
9709 Ok(self.brute_vector_hits(&view, candidate_ids, field, &q_unit, k, min))
9710 }
9711
9712 /// Unmasked HNSW path. `None` when no populated index covers the request.
9713 fn find_similar_hnsw(
9714 &self,
9715 field: &str,
9716 label: Option<&str>,
9717 q_unit: &[f64],
9718 k: usize,
9719 min: f64,
9720 ) -> Option<Vec<(String, f64)>> {
9721 // Try HNSW fast path.
9722 // `None` label searches across all VectorSimilar rules covering `field`
9723 // (merging their results); `Some(lbl)` restricts to rules whose
9724 // dst_label matches. Returns `None` when no populated HNSW index
9725 // covers the request — the O(n) brute-force fallback handles that case.
9726 let over_k = k.saturating_add(VECTOR_RESCORE_MARGIN);
9727 let hits = match label {
9728 Some(lbl) => self.engine.hnsw_search_dst(field, lbl, q_unit, over_k)?,
9729 None => self.engine.hnsw_search_any_dst(field, q_unit, over_k)?,
9730 };
9731 // Candidates only: the index's `f32` similarity is discarded and
9732 // each hit is re-scored against the `f64` vectors.
9733 let view = self.view();
9734 let mut out: Vec<(String, f64)> = hits
9735 .into_iter()
9736 .filter_map(|(id, _)| {
9737 let sim = exact_vector_similarity(&view, id, field, q_unit)?;
9738 if sim < min {
9739 return None;
9740 }
9741 Some((self.ids.key_of(id)?.to_string(), sim))
9742 })
9743 .collect();
9744 out.sort_by(|a, b| {
9745 b.1.partial_cmp(&a.1)
9746 .unwrap_or(std::cmp::Ordering::Equal)
9747 .then_with(|| a.0.cmp(&b.0))
9748 });
9749 out.truncate(k);
9750 Some(out)
9751 }
9752
9753 /// Masked HNSW widening beam. `None` when no index covers the request or
9754 /// the beam cannot admit `k` visible hits (caller falls through to brute).
9755 #[allow(clippy::too_many_arguments)]
9756 fn find_similar_hnsw_masked(
9757 &self,
9758 field: &str,
9759 label: Option<&str>,
9760 q_unit: &[f64],
9761 k: usize,
9762 min: f64,
9763 mask: &crate::mask::NodeMask,
9764 caller: ExactnessCaller,
9765 ) -> Option<Vec<(String, f64)>> {
9766 let index_len = match label {
9767 Some(lbl) => self.engine.hnsw_dst_len(field, lbl, q_unit.len()),
9768 None => self.engine.hnsw_any_dst_len(field, q_unit.len()),
9769 };
9770 let n = index_len?;
9771 // The `?` above is the coverage test: past it, an index exists and this
9772 // masked, non-exact call is about to ride it.
9773 self.note_ambiguous_exactness(field, label, caller);
9774 // Same ceiling the exact-rule widening loop in `hnsw_candidates`
9775 // consults — including the `with_ef_max` test hook.
9776 let cap = ef_max();
9777 let visible = mask.len();
9778 let mut ef = k.saturating_add(VECTOR_RESCORE_MARGIN);
9779 if visible > 0 && n > 0 {
9780 let over = k
9781 .saturating_mul(n)
9782 .div_ceil(visible)
9783 .saturating_add(VECTOR_RESCORE_MARGIN);
9784 ef = ef.max(over);
9785 }
9786 loop {
9787 let hits = match label {
9788 Some(lbl) => self
9789 .engine
9790 .hnsw_search_dst_with_ef(field, lbl, q_unit, ef, ef),
9791 None => self
9792 .engine
9793 .hnsw_search_any_dst_with_ef(field, q_unit, ef, ef),
9794 };
9795 let hits = hits?;
9796 let full = hits.len() == ef;
9797 let mut out = self.score_masked_hnsw_hits(&hits, field, q_unit, min, mask);
9798 if out.len() >= k {
9799 out.truncate(k);
9800 return Some(out);
9801 }
9802 // Short of its width (frontier exhausted) or at the ceiling:
9803 // a wider beam reaches nothing new, so the scan answers.
9804 if !full || ef >= cap {
9805 return None;
9806 }
9807 ef = ef.saturating_mul(2);
9808 }
9809 }
9810
9811 /// Say once, per `(field, label)` index and caller shape, that a masked
9812 /// search is answering approximately.
9813 ///
9814 /// A mask narrows *which nodes may be returned*. It does not choose a
9815 /// kernel — `exact=true` and a `where=` predicate do, and nothing else
9816 /// does. A caller who needed exact answers, passed `mask=` alone, and read
9817 /// the mask as a promise of exhaustiveness gets a correct-looking
9818 /// approximate answer and no signal at all; that is a silent wrong answer,
9819 /// and it has cost an integration team real time.
9820 ///
9821 /// The fix is a question, not a behaviour change. Making a mask imply
9822 /// `exact` would turn every existing masked caller's ANN into an O(n) GEMM
9823 /// without asking them, which is a worse trade than the ambiguity.
9824 ///
9825 /// Printed once per index for the reason the dimension-mismatch skip in
9826 /// `core_rules::hnsw` is: a line on every call is a line callers learn to
9827 /// scroll past.
9828 ///
9829 /// `caller` decides the advice. The same leg is reached from two signatures
9830 /// and only one of them has an `exact` argument to pass; see
9831 /// [`ExactnessCaller`].
9832 fn note_ambiguous_exactness(&self, field: &str, label: Option<&str>, caller: ExactnessCaller) {
9833 let entry = (field.to_string(), label.unwrap_or("").to_string(), caller);
9834 let first = match self.warned_ambiguous_exactness.lock() {
9835 Ok(mut seen) => seen.insert(entry),
9836 Err(poisoned) => poisoned.into_inner().insert(entry),
9837 };
9838 if !first {
9839 return;
9840 }
9841 AMBIGUOUS_EXACTNESS_WARNS.with(|c| c.set(c.get().saturating_add(1)));
9842 let which = match label {
9843 Some(lbl) => format!(" (label `{lbl}`)"),
9844 None => String::new(),
9845 };
9846 let subject = caller.subject();
9847 let advice = caller.advice();
9848 let line = format!(
9849 "mushroomdb: {subject} on field `{field}`{which} is answering \
9850 approximately. A mask narrows which nodes may be returned; it does not \
9851 change which kernel runs, and an index covers this field. For an exact \
9852 answer over the same visible candidate set, {advice} Further masked \
9853 searches of this shape on this index are silent."
9854 );
9855 AMBIGUOUS_EXACTNESS_LAST.with(|c| *c.borrow_mut() = Some(line.clone()));
9856 eprintln!("{line}");
9857 }
9858
9859 /// `label ∩ mask ∩ holds(where)`. Index fast path when `label` is `Some`
9860 /// and `(label, where.field)` is enabled; otherwise scan with `visible()`.
9861 fn vector_candidates(
9862 view: &GraphView<'_>,
9863 label: Option<&str>,
9864 where_: Option<&PropPredicate>,
9865 ) -> Vec<u32> {
9866 if let (Some(lbl), Some(pred)) = (label, where_) {
9867 let indexed = view
9868 .prop_index
9869 .is_some_and(|idx| idx.is_enabled(lbl, &pred.field));
9870 if indexed {
9871 match (&pred.eq, &pred.in_) {
9872 (Some(eq), None) => {
9873 if let Some(ids) = view.nodes_with_prop(lbl, &pred.field, eq) {
9874 return ids;
9875 }
9876 }
9877 (None, Some(allowed)) => {
9878 let mut seen = HashSet::new();
9879 let mut out = Vec::new();
9880 for v in allowed {
9881 if let Some(ids) = view.nodes_with_prop(lbl, &pred.field, v) {
9882 for id in ids {
9883 if seen.insert(id) {
9884 out.push(id);
9885 }
9886 }
9887 }
9888 }
9889 return out;
9890 }
9891 _ => {}
9892 }
9893 }
9894 }
9895
9896 let mut ids: Vec<u32> = match label {
9897 Some(lbl) => view
9898 .nodes_with_label(lbl)
9899 .into_iter()
9900 .filter(|&id| view.visible(id))
9901 .collect(),
9902 None => view.nodes_all(),
9903 };
9904 if let Some(pred) = where_ {
9905 ids.retain(|&id| match view.prop(id, &pred.field) {
9906 None => pred.holds(None),
9907 Some(vr) => pred.holds(Some(vr.as_value())),
9908 });
9909 }
9910 ids
9911 }
9912
9913 /// Exact brute kNN: pack candidates at `q_unit`'s dim, GEMV, keep
9914 /// `score >= min`, sort `(sim desc, key asc)`, truncate to `k`.
9915 fn brute_vector_hits(
9916 &self,
9917 view: &GraphView<'_>,
9918 candidate_ids: impl IntoIterator<Item = u32>,
9919 field: &str,
9920 q_unit: &[f64],
9921 k: usize,
9922 min: f64,
9923 ) -> Vec<(String, f64)> {
9924 let rows: Vec<(u32, std::borrow::Cow<'_, [f64]>)> = candidate_ids
9925 .into_iter()
9926 .filter_map(|id| crate::exact_knn::vector_f64(view, id, field).map(|v| (id, v)))
9927 .collect();
9928 let packed =
9929 crate::exact_knn::pack(rows.iter().map(|(id, v)| (*id, v.as_ref())), q_unit.len());
9930 let scores = crate::exact_knn::gemv(&packed, q_unit);
9931 let mut scored: Vec<(String, f64)> = packed
9932 .ids
9933 .iter()
9934 .zip(scores.iter())
9935 .filter_map(|(&id, &sim)| {
9936 if sim < min {
9937 return None;
9938 }
9939 let key = self.ids.key_of(id)?.to_string();
9940 Some((key, sim))
9941 })
9942 .collect();
9943 scored.sort_by(|a, b| {
9944 b.1.partial_cmp(&a.1)
9945 .unwrap_or(std::cmp::Ordering::Equal)
9946 .then_with(|| a.0.cmp(&b.0))
9947 });
9948 scored.truncate(k);
9949 scored
9950 }
9951
9952 /// Exact cosine top-k for each key in `keys`, scored only against `keys`.
9953 ///
9954 /// `min` is cosine similarity in [-1, 1], inclusive (`score >= min`), the
9955 /// same unit and inequality as `find_similar_vector`. Self-matches are
9956 /// excluded. Unknown keys, keys with no `field`, zero-norm or wrong-dim
9957 /// embeddings are omitted as both query and candidate. Duplicate keys are
9958 /// collapsed, first-seen order. Empty `keys` → empty `Ok(vec![])`. Never
9959 /// uses HNSW. `n > PAIRWISE_MAX_N` → `QueryError`.
9960 #[allow(clippy::type_complexity)]
9961 pub fn pairwise_similar(
9962 &self,
9963 keys: &[&str],
9964 field: &str,
9965 k: usize,
9966 min: f64,
9967 ) -> Result<Vec<(String, Vec<(String, f64)>)>> {
9968 let mut seen = HashSet::new();
9969 let mut unique_ids = Vec::new();
9970 for key in keys {
9971 let Some(id) = self.ids.get(key) else {
9972 continue;
9973 };
9974 if seen.insert(id) {
9975 unique_ids.push(id);
9976 }
9977 }
9978 let max_n = crate::exact_knn::pairwise_max_n();
9979 if unique_ids.len() > max_n {
9980 return Err(GraphError::QueryError {
9981 detail: format!(
9982 "pairwise_similar: n={} exceeds PAIRWISE_MAX_N ({max_n})",
9983 unique_ids.len()
9984 ),
9985 });
9986 }
9987 if unique_ids.is_empty() {
9988 return Ok(Vec::new());
9989 }
9990
9991 let view = self.view();
9992 let mut rows: Vec<(u32, std::borrow::Cow<'_, [f64]>)> = Vec::new();
9993 let mut counts: HashMap<usize, usize> = HashMap::new();
9994 for id in unique_ids {
9995 let Some(v) = crate::exact_knn::vector_f64(&view, id, field) else {
9996 continue;
9997 };
9998 let norm: f64 = v.iter().map(|x| x * x).sum::<f64>().sqrt();
9999 if norm == 0.0 {
10000 continue;
10001 }
10002 *counts.entry(v.len()).or_default() += 1;
10003 rows.push((id, v));
10004 }
10005 if rows.is_empty() {
10006 return Ok(Vec::new());
10007 }
10008 let dim = counts
10009 .into_iter()
10010 .max_by_key(|&(d, c)| (c, d))
10011 .map(|(d, _)| d)
10012 .expect("rows non-empty");
10013 let packed = crate::exact_knn::pack(rows.iter().map(|(id, v)| (*id, v.as_ref())), dim);
10014 let n = packed.ids.len();
10015 if n == 0 {
10016 return Ok(Vec::new());
10017 }
10018 let src_keys: Vec<String> = packed
10019 .ids
10020 .iter()
10021 .map(|&id| self.ids.key_of(id).unwrap_or("").to_string())
10022 .collect();
10023
10024 let mut out = Vec::with_capacity(n);
10025 if n <= crate::exact_knn::pairwise_gram_max() {
10026 let sims = crate::exact_knn::gram(&packed);
10027 for i in 0..n {
10028 out.push(Self::topk_from_row(
10029 &src_keys,
10030 i,
10031 &sims[i * n..(i + 1) * n],
10032 k,
10033 min,
10034 ));
10035 }
10036 } else {
10037 for i in 0..n {
10038 let row = &packed.data[i * packed.dim..(i + 1) * packed.dim];
10039 let scores = crate::exact_knn::gemv(&packed, row);
10040 out.push(Self::topk_from_row(&src_keys, i, &scores, k, min));
10041 }
10042 }
10043 Ok(out)
10044 }
10045
10046 /// [`pairwise_similar`](Self::pairwise_similar) over the keys the mask
10047 /// admits — intersected **before** the matmul, never filtered after it.
10048 ///
10049 /// A hidden vector packed into the Gram is a row every visible key is
10050 /// scored against. It can take a visible neighbour's place in the top-`k`,
10051 /// and because the packed dimension is a majority vote over the candidate
10052 /// rows it can decide whether a visible pair is scored at all. Dropping
10053 /// hidden names from the finished answer leaves both effects standing, so
10054 /// the intersection happens first and the answer is byte-for-byte the one
10055 /// `pairwise_similar` gives for the visible keys alone.
10056 ///
10057 /// The caps therefore measure the **post-filter** count: a key set over
10058 /// [`PAIRWISE_MAX_N`](crate::PAIRWISE_MAX_N) unscoped can come under it
10059 /// scoped and succeed, because the work the cap refuses is work this call
10060 /// no longer does. A filtered count still over the cap is still refused.
10061 #[allow(clippy::type_complexity)]
10062 pub fn pairwise_similar_scoped(
10063 &self,
10064 keys: &[&str],
10065 field: &str,
10066 k: usize,
10067 min: f64,
10068 mask: &crate::mask::NodeMask,
10069 ) -> Result<Vec<(String, Vec<(String, f64)>)>> {
10070 let visible: Vec<&str> = keys
10071 .iter()
10072 .copied()
10073 .filter(|key| mask.contains_node(self, key))
10074 .collect();
10075 self.pairwise_similar(&visible, field, k, min)
10076 }
10077
10078 /// Neighbours of packed row `i`: drop self, keep `score >= min`, sort
10079 /// `(sim desc, key asc)`, truncate to `k`. Packed srcs with no survivors
10080 /// still appear as `(src, [])`.
10081 fn topk_from_row(
10082 src_keys: &[String],
10083 i: usize,
10084 scores: &[f64],
10085 k: usize,
10086 min: f64,
10087 ) -> (String, Vec<(String, f64)>) {
10088 let mut neigh: Vec<(String, f64)> = scores
10089 .iter()
10090 .enumerate()
10091 .filter_map(|(j, &sim)| {
10092 if i == j || sim < min {
10093 return None;
10094 }
10095 Some((src_keys[j].clone(), sim))
10096 })
10097 .collect();
10098 neigh.sort_by(|a, b| {
10099 b.1.partial_cmp(&a.1)
10100 .unwrap_or(std::cmp::Ordering::Equal)
10101 .then_with(|| a.0.cmp(&b.0))
10102 });
10103 neigh.truncate(k);
10104 (src_keys[i].clone(), neigh)
10105 }
10106
10107 /// Re-score HNSW candidates from the `f64` vectors, drop hidden / below-`min`
10108 /// hits, order by score then key. The index's own `f32` similarity is discarded.
10109 fn score_masked_hnsw_hits(
10110 &self,
10111 hits: &[(u32, f64)],
10112 field: &str,
10113 q_unit: &[f64],
10114 min: f64,
10115 mask: &crate::mask::NodeMask,
10116 ) -> Vec<(String, f64)> {
10117 let view = self.view_masked(mask);
10118 let mut out: Vec<(String, f64)> = hits
10119 .iter()
10120 .copied()
10121 .filter(|&(id, _)| mask.contains_id(id))
10122 .filter_map(|(id, _)| {
10123 let sim = exact_vector_similarity(&view, id, field, q_unit)?;
10124 if sim < min {
10125 return None;
10126 }
10127 Some((self.ids.key_of(id)?.to_string(), sim))
10128 })
10129 .collect();
10130 out.sort_by(|a, b| {
10131 b.1.partial_cmp(&a.1)
10132 .unwrap_or(std::cmp::Ordering::Equal)
10133 .then_with(|| a.0.cmp(&b.0))
10134 });
10135 out
10136 }
10137
10138 /// Read a single property from an edge.
10139 ///
10140 /// Returns `None` when the edge does not exist, the field is absent, or any
10141 /// of the string keys cannot be resolved to interned ids. Only edge props
10142 /// written by rules (weight fields) are accessible without a `set_edge_prop`
10143 /// binding; topology-only edges (no props set) return `None` for every field.
10144 pub fn get_edge_prop(
10145 &self,
10146 edge_type: &str,
10147 src_key: &str,
10148 dst_key: &str,
10149 field: &str,
10150 ) -> Option<Value> {
10151 let etype = self.syms.get(edge_type)?;
10152 let src = self.ids.get(src_key)?;
10153 let dst = self.ids.get(dst_key)?;
10154 self.edge_props_view().get(etype, src, dst, field)
10155 }
10156
10157 /// Lex → parse → plan → execute `cypher` over a read-only view.
10158 /// Every pipeline `Err(String)` becomes `GraphError::QueryError` with a
10159 /// stage prefix (`lex:` / `parse:` / `plan:` / `execute:`).
10160 pub fn query(&self, cypher: &str, params: &BTreeMap<String, Value>) -> Result<ResultSet> {
10161 let tokens = lex(cypher).map_err(|e| GraphError::QueryError {
10162 detail: format!("lex: {e}"),
10163 })?;
10164 let union = parse_read(&tokens).map_err(|e| GraphError::QueryError {
10165 detail: format!("parse: {e}"),
10166 })?;
10167 let t0 = std::time::Instant::now();
10168 let result = execute_union(&self.view(), &union, &Params(params)).map_err(|e| {
10169 GraphError::QueryError {
10170 detail: format!("execute: {e}"),
10171 }
10172 });
10173 let elapsed_ms = t0.elapsed().as_millis() as u64;
10174 let threshold = self.slow_query_threshold_ms;
10175 if threshold > 0 && elapsed_ms >= threshold {
10176 eprintln!("[mushroomdb] slow query ({elapsed_ms}ms): {cypher}");
10177 let entry = SlowQueryEntry {
10178 ms: elapsed_ms,
10179 query: cypher.to_string(),
10180 at_commit: self.commit_seq,
10181 };
10182 if let Ok(mut log) = self.slow_queries.lock() {
10183 if log.entries.len() == SLOW_QUERY_RING_CAP {
10184 log.entries.pop_front();
10185 }
10186 log.entries.push_back(entry);
10187 log.total += 1;
10188 }
10189 }
10190 result
10191 }
10192
10193 /// Convenience entry-point that accepts a slice of `(name, value)` pairs
10194 /// instead of a pre-built `BTreeMap`. Equivalent to building the map and
10195 /// calling [`GraphDb::query`].
10196 pub fn query_with_params(&self, cypher: &str, params: &[(&str, Value)]) -> Result<ResultSet> {
10197 let map: BTreeMap<String, Value> = params
10198 .iter()
10199 .map(|(k, v)| (k.to_string(), v.clone()))
10200 .collect();
10201 self.query(cypher, &map)
10202 }
10203
10204 /// Execute a Cypher write statement (CREATE / MATCH…SET / MATCH…DELETE / MERGE).
10205 ///
10206 /// All mutations flow through the same `insert_node` / `set_prop` /
10207 /// `delete_edge` / `insert_edge` path as the Rust API so the rule engine
10208 /// fires and the WAL captures everything with one fsync per statement.
10209 ///
10210 /// Returns a one-row [`ResultSet`] with columns `created`, `properties_set`,
10211 /// and `deleted` matching the write-result contract.
10212 ///
10213 /// **Mutation routing**: mutations are collected into a single
10214 /// [`BatchBuilder`] and committed atomically (one WAL `Batch` frame, one
10215 /// fsync). The MATCH phase for SET/DELETE uses a read-only `execute` call
10216 /// over `self.view()` — the borrow is dropped before the batch is opened.
10217 ///
10218 /// **Limitations (v1)**:
10219 /// - SET RHS must be a literal, `$param`, or arithmetic; bare property copy → named error.
10220 /// - `DETACH DELETE n` → calls `delete_node` for each matched node (removes all edges).
10221 /// - Bare `DELETE n` → error if n has any incident edges; succeeds for isolated nodes.
10222 /// - MERGE supports `ON CREATE SET` / `ON MATCH SET` in the same write batch.
10223 /// - Deleting a derived edge → named error "cannot delete derived edge".
10224 pub fn query_write(
10225 &mut self,
10226 cypher: &str,
10227 params: &BTreeMap<String, Value>,
10228 ) -> Result<ResultSet> {
10229 let tokens = lex(cypher).map_err(|e| GraphError::QueryError {
10230 detail: format!("lex: {e}"),
10231 })?;
10232 let stmt = parse_write(&tokens).map_err(|e| GraphError::QueryError {
10233 detail: format!("parse: {e}"),
10234 })?;
10235 self.exec_write_stmt(stmt, params)
10236 }
10237
10238 fn exec_write_stmt(
10239 &mut self,
10240 stmt: WriteStatement,
10241 params: &BTreeMap<String, Value>,
10242 ) -> Result<ResultSet> {
10243 match stmt {
10244 WriteStatement::Create(s) => self.exec_create(s, params),
10245 WriteStatement::MatchSet(s) => self.exec_match_set(s, params),
10246 WriteStatement::MatchDelete(s) => self.exec_match_delete(s, params),
10247 WriteStatement::MatchDeleteNode(s) => self.exec_match_delete_node(s, params),
10248 WriteStatement::Merge(s) => self.exec_merge(s, params),
10249 }
10250 }
10251
10252 fn exec_create(
10253 &mut self,
10254 stmt: core_query::cypher::CreateStmt,
10255 params: &BTreeMap<String, Value>,
10256 ) -> Result<ResultSet> {
10257 // Extract the node key from props: require a string-valued `id` field.
10258 let mut var_to_key: BTreeMap<String, String> = BTreeMap::new();
10259 for node in &stmt.nodes {
10260 let var = node.var.as_deref().unwrap_or("_cn0");
10261 let key = node
10262 .props
10263 .iter()
10264 .find(|(f, _)| f == "id")
10265 .and_then(|(_, v)| {
10266 if let Value::Str(s) = v {
10267 Some(s.clone())
10268 } else {
10269 None
10270 }
10271 })
10272 .ok_or_else(|| GraphError::QueryError {
10273 detail: format!(
10274 "CREATE node ({}:{}) requires a string 'id' property",
10275 var, node.label
10276 ),
10277 })?;
10278 var_to_key.insert(var.to_string(), key);
10279 }
10280
10281 let mut batch = self.batch();
10282 let mut created: usize = 0;
10283 for node in &stmt.nodes {
10284 let var = node.var.as_deref().unwrap_or("_cn0");
10285 let key = &var_to_key[var];
10286 batch.insert_node(&node.label, key, node.props.clone());
10287 created += 1;
10288 }
10289 for edge in &stmt.edges {
10290 let src_key = var_to_key
10291 .get(&edge.src_var)
10292 .ok_or_else(|| GraphError::QueryError {
10293 detail: format!("CREATE edge src variable '{}' is not bound", edge.src_var),
10294 })?;
10295 let dst_key = var_to_key
10296 .get(&edge.dst_var)
10297 .ok_or_else(|| GraphError::QueryError {
10298 detail: format!("CREATE edge dst variable '{}' is not bound", edge.dst_var),
10299 })?;
10300 batch.insert_edge(&edge.etype, src_key, dst_key);
10301 }
10302 batch.commit()?;
10303
10304 // Optional RETURN clause: project created bindings as a read result.
10305 if let Some(returns) = stmt.returns {
10306 // Each created node is looked up by its key via a separate MATCH pattern.
10307 // Multiple single-node patterns cross-join to produce 1 output row with
10308 // all variables bound (each pattern returns exactly 1 row).
10309 let patterns: Vec<Pattern> = stmt
10310 .nodes
10311 .iter()
10312 .map(|node| {
10313 let var = node.var.as_deref().unwrap_or("_cn0");
10314 let key = var_to_key[var].clone();
10315 Pattern {
10316 start: NodePat {
10317 var: Some(var.to_string()),
10318 label: Some(node.label.clone()),
10319 props: vec![("id".to_string(), Operand::Lit(Value::Str(key)))],
10320 },
10321 chain: vec![],
10322 shortest: false,
10323 }
10324 })
10325 .collect();
10326 let q = Query {
10327 matches: patterns,
10328 optional_clauses: vec![],
10329 where_expr: None,
10330 unwinds: vec![],
10331 post_unwind_where: None,
10332 stages: vec![],
10333 returns,
10334 distinct: false,
10335 order_by: vec![],
10336 skip: None,
10337 limit: None,
10338 };
10339 let ops = plan(&q).map_err(|e| GraphError::QueryError {
10340 detail: format!("plan: {e}"),
10341 })?;
10342 return execute(&self.view(), &ops, &Params(params)).map_err(|e| {
10343 GraphError::QueryError {
10344 detail: format!("execute: {e}"),
10345 }
10346 });
10347 }
10348
10349 let mut rs = write_result_set();
10350 rs.push_row(vec![
10351 Some(Value::Int(created as i64)),
10352 Some(Value::Int(0)),
10353 Some(Value::Int(0)),
10354 ]);
10355 Ok(rs)
10356 }
10357
10358 fn exec_match_set(
10359 &mut self,
10360 stmt: core_query::cypher::MatchSetStmt,
10361 params: &BTreeMap<String, Value>,
10362 ) -> Result<ResultSet> {
10363 let project_returns = stmt.returns.clone();
10364 // Collect unique node vars targeted by SET clauses, plus RETURN bindings
10365 // so the post-write projection can look them up by key.
10366 let mut set_vars: Vec<String> = Vec::new();
10367 for s in &stmt.sets {
10368 if !set_vars.contains(&s.var) {
10369 set_vars.push(s.var.clone());
10370 }
10371 }
10372 let rel_vars = pattern_rel_vars(&stmt.matches);
10373 // `count` is the engine's, on an edge: it is the insert-count §5.13
10374 // maintains, and a `SET` that overwrote it would make the number mean
10375 // whatever the last writer said rather than how many times the pair was
10376 // inserted. Refused by name here, before the match runs, so the caller
10377 // is told what is actually wrong instead of meeting the executor's
10378 // generic "did not resolve to a node key" — and so the answer does not
10379 // depend on whether the pattern happened to match a row. The same name
10380 // on a *node* is an ordinary property and is untouched.
10381 for s in &stmt.sets {
10382 if s.field == EDGE_COUNT_PROP && rel_vars.iter().any(|r| r == &s.var) {
10383 return Err(GraphError::QueryError {
10384 detail: format!(
10385 "cannot SET {}.{EDGE_COUNT_PROP}: `{EDGE_COUNT_PROP}` is a reserved edge \
10386 property holding the pair's insert count",
10387 s.var
10388 ),
10389 });
10390 }
10391 }
10392 let mut lookup_vars = set_vars.clone();
10393 for v in pattern_node_vars(&stmt.matches) {
10394 add_var(&mut lookup_vars, &v);
10395 }
10396 if let Some(ref returns) = project_returns {
10397 for v in ret_node_vars(returns) {
10398 if !rel_vars.iter().any(|r| r == &v) {
10399 add_var(&mut lookup_vars, &v);
10400 }
10401 }
10402 }
10403
10404 // Synthesize a read query: MATCH … WHERE … RETURN <lookup_vars>, <set_values…>
10405 // SET values are projected as ScalarExpr items so that arithmetic expressions
10406 // (e.g. `SET n.score = n.score * 1.5`) are evaluated in the matched-row context.
10407 let mut set_returns: Vec<RetItem> = lookup_vars
10408 .iter()
10409 .map(|v| RetItem {
10410 value: RetVal::Var(v.clone()),
10411 alias: None,
10412 })
10413 .collect();
10414 // One computed column per SET clause; alias is `__sv_<i>`.
10415 let set_val_cols: Vec<String> = stmt
10416 .sets
10417 .iter()
10418 .enumerate()
10419 .map(|(i, _)| format!("__sv_{i}"))
10420 .collect();
10421 for (sc, col) in stmt.sets.iter().zip(&set_val_cols) {
10422 set_returns.push(RetItem {
10423 value: RetVal::ScalarExpr(sc.value.clone()),
10424 alias: Some(col.clone()),
10425 });
10426 }
10427 // Capture relationship types while r is bound; SET does not change them.
10428 for r in &rel_vars {
10429 set_returns.push(RetItem {
10430 value: RetVal::FuncCall {
10431 name: "type".into(),
10432 args: vec![Operand::Var(r.clone())],
10433 },
10434 alias: Some(rel_type_alias(r)),
10435 });
10436 }
10437
10438 let read_q = Query {
10439 matches: stmt.matches.clone(),
10440 optional_clauses: vec![],
10441 where_expr: stmt.where_expr.clone(),
10442 unwinds: vec![],
10443 post_unwind_where: None,
10444 stages: vec![],
10445 returns: set_returns,
10446 distinct: false,
10447 order_by: vec![],
10448 skip: None,
10449 limit: None,
10450 };
10451 let ops = plan(&read_q).map_err(|e| GraphError::QueryError {
10452 detail: format!("plan: {e}"),
10453 })?;
10454 // MATCH phase is read-only; borrow ends before batch opens.
10455 //
10456 // When a role-scoped write is in flight, run the MATCH read through
10457 // view_masked so hidden nodes are invisible → hidden ≡ absent ≡
10458 // zero-rows (no SetProp ops generated, no existence-oracle 403).
10459 // Full-authority writes (pending_write_authz=None) keep view().
10460 let match_rs = {
10461 let mask_opt = self.pending_write_authz.as_ref().map(|a| a.mask.clone());
10462 if let Some(ref mask) = mask_opt {
10463 execute(&self.view_masked(mask), &ops, &Params(params))
10464 } else {
10465 execute(&self.view(), &ops, &Params(params))
10466 }
10467 }
10468 .map_err(|e| GraphError::QueryError {
10469 detail: format!("execute: {e}"),
10470 })?;
10471
10472 // Collect (key, field, value) for each matched row × each SET clause.
10473 let mut set_ops: Vec<(String, String, Value)> = Vec::new();
10474 for row_i in 0..match_rs.len() {
10475 for (sc, col) in stmt.sets.iter().zip(&set_val_cols) {
10476 let key = match match_rs.get(row_i, &sc.var) {
10477 Some(Value::Str(k)) => k.clone(),
10478 _ => {
10479 return Err(GraphError::QueryError {
10480 detail: format!(
10481 "SET variable '{}' did not resolve to a node key",
10482 sc.var
10483 ),
10484 })
10485 }
10486 };
10487 // The SET value was already evaluated by the executor.
10488 let value = match match_rs.get(row_i, col) {
10489 Some(v) => v.clone(),
10490 None => {
10491 return Err(GraphError::QueryError {
10492 detail: format!(
10493 "SET value for {}.{} evaluated to null",
10494 sc.var, sc.field
10495 ),
10496 })
10497 }
10498 };
10499 set_ops.push((key, sc.field.clone(), value));
10500 }
10501 }
10502
10503 // Apply as one atomic batch.
10504 let props_set = set_ops.len();
10505 let mut batch = self.batch();
10506 for (key, field, value) in set_ops {
10507 batch.set_prop(&key, &field, value);
10508 }
10509 batch.commit()?;
10510
10511 if let Some(returns) = project_returns {
10512 return project_set_return_rows(self, &rel_vars, &match_rs, &returns, params);
10513 }
10514
10515 let mut rs = write_result_set();
10516 rs.push_row(vec![
10517 Some(Value::Int(0)),
10518 Some(Value::Int(props_set as i64)),
10519 Some(Value::Int(0)),
10520 ]);
10521 Ok(rs)
10522 }
10523
10524 fn exec_match_delete(
10525 &mut self,
10526 stmt: core_query::cypher::MatchDeleteStmt,
10527 params: &BTreeMap<String, Value>,
10528 ) -> Result<ResultSet> {
10529 // Collect unique node vars needed to identify edge endpoints.
10530 let mut node_vars: Vec<String> = Vec::new();
10531 for ed in &stmt.deletes {
10532 if !node_vars.contains(&ed.src_var) {
10533 node_vars.push(ed.src_var.clone());
10534 }
10535 if !node_vars.contains(&ed.dst_var) {
10536 node_vars.push(ed.dst_var.clone());
10537 }
10538 }
10539
10540 // Synthesize read query.
10541 let returns: Vec<RetItem> = node_vars
10542 .iter()
10543 .map(|v| RetItem {
10544 value: RetVal::Var(v.clone()),
10545 alias: None,
10546 })
10547 .collect();
10548 let read_q = Query {
10549 matches: stmt.matches,
10550 optional_clauses: vec![],
10551 where_expr: stmt.where_expr,
10552 unwinds: vec![],
10553 post_unwind_where: None,
10554 stages: vec![],
10555 returns,
10556 distinct: false,
10557 order_by: vec![],
10558 skip: None,
10559 limit: None,
10560 };
10561 let ops = plan(&read_q).map_err(|e| GraphError::QueryError {
10562 detail: format!("plan: {e}"),
10563 })?;
10564 // Role-scoped writes: mask the MATCH read phase so hidden nodes are
10565 // invisible → hidden ≡ absent ≡ zero-rows (spec §3.1, hidden ≡ absent).
10566 let match_rs = {
10567 let mask_opt = self.pending_write_authz.as_ref().map(|a| a.mask.clone());
10568 if let Some(ref mask) = mask_opt {
10569 execute(&self.view_masked(mask), &ops, &Params(params))
10570 } else {
10571 execute(&self.view(), &ops, &Params(params))
10572 }
10573 }
10574 .map_err(|e| GraphError::QueryError {
10575 detail: format!("execute: {e}"),
10576 })?;
10577
10578 // Collect (etype, src_key, dst_key) for each row × each delete target.
10579 let mut del_ops: Vec<(String, String, String)> = Vec::new();
10580 for row_i in 0..match_rs.len() {
10581 for ed in &stmt.deletes {
10582 let src_key = match match_rs.get(row_i, &ed.src_var) {
10583 Some(Value::Str(k)) => k.clone(),
10584 _ => {
10585 return Err(GraphError::QueryError {
10586 detail: format!(
10587 "DELETE src variable '{}' did not resolve to a node key",
10588 ed.src_var
10589 ),
10590 })
10591 }
10592 };
10593 let dst_key = match match_rs.get(row_i, &ed.dst_var) {
10594 Some(Value::Str(k)) => k.clone(),
10595 _ => {
10596 return Err(GraphError::QueryError {
10597 detail: format!(
10598 "DELETE dst variable '{}' did not resolve to a node key",
10599 ed.dst_var
10600 ),
10601 })
10602 }
10603 };
10604 del_ops.push((ed.etype.clone(), src_key, dst_key));
10605 }
10606 }
10607
10608 // Apply as one atomic batch.
10609 let deleted = del_ops.len();
10610 let mut batch = self.batch();
10611 for (etype, src_key, dst_key) in del_ops {
10612 batch.delete_edge(&etype, &src_key, &dst_key);
10613 }
10614 batch.commit().map_err(|e| match e {
10615 GraphError::RuleOwned { .. } => GraphError::QueryError {
10616 detail: "cannot delete derived edge; retract via the rule or change the property"
10617 .to_string(),
10618 },
10619 other => other,
10620 })?;
10621
10622 let mut rs = write_result_set();
10623 rs.push_row(vec![
10624 Some(Value::Int(0)),
10625 Some(Value::Int(0)),
10626 Some(Value::Int(deleted as i64)),
10627 ]);
10628 Ok(rs)
10629 }
10630
10631 /// Execute `MATCH … [DETACH] DELETE <node_var> [, …]`.
10632 ///
10633 /// Collects the matching node keys via an ephemeral read query, then calls
10634 /// `delete_node` on each one. When `stmt.detach` is `false` (bare DELETE)
10635 /// the executor first checks that the node has no incident edges; if any
10636 /// remain it returns a named error matching openCypher semantics.
10637 fn exec_match_delete_node(
10638 &mut self,
10639 stmt: MatchDeleteNodeStmt,
10640 params: &BTreeMap<String, Value>,
10641 ) -> Result<ResultSet> {
10642 // Build a read query returning only the node keys we need.
10643 let returns: Vec<RetItem> = stmt
10644 .node_vars
10645 .iter()
10646 .map(|v| RetItem {
10647 value: RetVal::Var(v.clone()),
10648 alias: None,
10649 })
10650 .collect();
10651 let read_q = Query {
10652 matches: stmt.matches,
10653 optional_clauses: vec![],
10654 where_expr: stmt.where_expr,
10655 unwinds: vec![],
10656 post_unwind_where: None,
10657 stages: vec![],
10658 returns,
10659 distinct: false,
10660 order_by: vec![],
10661 skip: None,
10662 limit: None,
10663 };
10664 let ops = plan(&read_q).map_err(|e| GraphError::QueryError {
10665 detail: format!("plan: {e}"),
10666 })?;
10667 // Role-scoped writes: mask the MATCH read phase so hidden nodes are
10668 // invisible → hidden ≡ absent ≡ zero-rows (spec §3.1, hidden ≡ absent).
10669 let match_rs = {
10670 let mask_opt = self.pending_write_authz.as_ref().map(|a| a.mask.clone());
10671 if let Some(ref mask) = mask_opt {
10672 execute(&self.view_masked(mask), &ops, &Params(params))
10673 } else {
10674 execute(&self.view(), &ops, &Params(params))
10675 }
10676 }
10677 .map_err(|e| GraphError::QueryError {
10678 detail: format!("execute: {e}"),
10679 })?;
10680
10681 // Collect unique node keys to delete (deduplicate across rows × vars).
10682 let mut keys: Vec<String> = Vec::new();
10683 for row_i in 0..match_rs.len() {
10684 for var in &stmt.node_vars {
10685 if let Some(Value::Str(k)) = match_rs.get(row_i, var) {
10686 if !keys.contains(k) {
10687 keys.push(k.clone());
10688 }
10689 }
10690 }
10691 }
10692
10693 if !stmt.detach {
10694 // openCypher bare DELETE: error if any matched node has incident edges.
10695 for key in &keys {
10696 if let Some(id) = self.ids.get(key) {
10697 let tv = self.topo_view();
10698 let has_edges = tv.etypes().any(|et| {
10699 !tv.neighbors(et, Direction::Out, id).is_empty()
10700 || !tv.neighbors(et, Direction::In, id).is_empty()
10701 });
10702 if has_edges {
10703 return Err(GraphError::QueryError {
10704 detail: format!(
10705 "Cannot delete node `{key}` because it still has incident edges. \
10706 Use DETACH DELETE to remove the node and all its edges."
10707 ),
10708 });
10709 }
10710 }
10711 }
10712 }
10713
10714 let mut nodes_deleted = 0i64;
10715 let mut edges_deleted = 0i64;
10716 for key in keys {
10717 match self.delete_node(&key) {
10718 Ok(report) => {
10719 nodes_deleted += 1;
10720 edges_deleted += (report.manual_edges + report.derived_edges) as i64;
10721 }
10722 Err(GraphError::KeyNotFound { .. }) => {
10723 // Node may have been deleted by an earlier iteration (e.g., via
10724 // multiple MATCH rows for the same node). Safe to skip.
10725 }
10726 Err(e) => return Err(e),
10727 }
10728 }
10729
10730 let mut rs = write_result_set();
10731 rs.push_row(vec![
10732 Some(Value::Int(0)),
10733 Some(Value::Int(0)),
10734 Some(Value::Int(nodes_deleted + edges_deleted)),
10735 ]);
10736 Ok(rs)
10737 }
10738
10739 /// Props the MERGE create arm inserts: the identifying key, plus `ns` when
10740 /// the pattern named one, or the executing role's sole namespace when it
10741 /// did not. A role bound to two or more namespaces cannot choose, and is
10742 /// refused with [`MERGE_CREATE_NEEDS_ONE_NAMESPACE`]. The authorizer still
10743 /// refuses a named `ns` the role cannot write.
10744 fn merge_create_props(
10745 &self,
10746 key_field: &str,
10747 key_value: &Value,
10748 named_ns: Option<&Value>,
10749 ) -> Result<Vec<(String, Value)>> {
10750 let mut props = vec![(key_field.to_string(), key_value.clone())];
10751 if let Some(ns) = named_ns {
10752 props.push((NS_PROP.to_string(), ns.clone()));
10753 return Ok(props);
10754 }
10755 if let Some(ns) = self.merge_create_stamp_ns()? {
10756 props.push((NS_PROP.to_string(), Value::Str(ns)));
10757 }
10758 Ok(props)
10759 }
10760
10761 /// The namespace a role-scoped MERGE create stamps when the pattern does
10762 /// not name `ns`. `None` = unscoped / full authority, so the node lands in
10763 /// `default`.
10764 fn merge_create_stamp_ns(&self) -> Result<Option<String>> {
10765 let Some(authz) = self.pending_write_authz.as_ref() else {
10766 return Ok(None);
10767 };
10768 let Some(def) = self.role_def_for(&authz.role) else {
10769 return Ok(None);
10770 };
10771 match def.namespaces.as_deref() {
10772 Some([only]) => Ok(Some(only.clone())),
10773 Some(_) => Err(GraphError::RoleWriteDenied {
10774 reason: MERGE_CREATE_NEEDS_ONE_NAMESPACE.to_string(),
10775 }),
10776 None => Ok(None),
10777 }
10778 }
10779
10780 fn exec_merge(
10781 &mut self,
10782 stmt: core_query::cypher::MergeStmt,
10783 params: &BTreeMap<String, Value>,
10784 ) -> Result<ResultSet> {
10785 // MERGE: check if a node with the given key already exists.
10786 let key = match &stmt.key_value {
10787 Value::Str(s) => s.clone(),
10788 _ => {
10789 return Err(GraphError::QueryError {
10790 detail: format!(
10791 "MERGE key value must be a string (got {:?})",
10792 stmt.key_value
10793 ),
10794 })
10795 }
10796 };
10797
10798 if let Some(var) = stmt.var.as_deref() {
10799 for sc in stmt.on_create.iter().chain(&stmt.on_match) {
10800 if sc.var != var {
10801 return Err(GraphError::QueryError {
10802 detail: format!(
10803 "SET variable '{}' does not match MERGE variable '{var}'",
10804 sc.var
10805 ),
10806 });
10807 }
10808 }
10809 }
10810
10811 // ── MERGE authz pre-check (when role-scoped) ─────────────────────────
10812 //
10813 // MERGE scope precondition: check create OR update scope for the
10814 // declared label BEFORE calling `has_node` (timing-oracle closure,
10815 // spec §6.2 "MERGE visibility oracle" item: hidden ≡ absent for
10816 // unscoped roles — the scope denial fires without touching the key store).
10817 //
10818 // Clone to avoid holding a borrow on `self.pending_write_authz` while
10819 // also calling `self.ids.get(key)`.
10820 let merge_existed: bool = if let Some(authz) = self.pending_write_authz.clone() {
10821 let has_create = authz.scope.create_labels.contains(&stmt.label);
10822 let has_update = authz.scope.update_labels.contains(&stmt.label);
10823 if !has_create && !has_update {
10824 // Scope-before-lookup: 403 without has_node call (timing oracle
10825 // closure — see test_merge_unscoped_no_key_lookup).
10826 return Err(GraphError::RoleWriteDenied {
10827 reason: format!(
10828 "role-bound token: label '{}' not in write scope (create_labels)",
10829 stmt.label
10830 ),
10831 });
10832 }
10833 // Key lookup under mask.
10834 match self.ids.get(key.as_str()) {
10835 Some(id) if authz.mask.contains_id(id) => {
10836 // Visible: must have update scope to proceed to match arm.
10837 if !has_update {
10838 return Err(GraphError::RoleWriteDenied {
10839 reason: format!(
10840 "role-bound token: label '{}' not in write scope (update_labels)",
10841 stmt.label
10842 ),
10843 });
10844 }
10845 true // existed = true → match arm
10846 }
10847 Some(_) => {
10848 // Hidden: same error as absent to the role (spec §3.1/§3.3).
10849 return Err(GraphError::RoleWriteDenied {
10850 reason: "role-bound token: target node not visible".into(),
10851 });
10852 }
10853 None => {
10854 // Absent: must have create scope to proceed to the create arm.
10855 //
10856 // Update-only roles (create_labels empty, update_labels set):
10857 // return the SAME "not visible" error as the hidden-key branch
10858 // so hidden ≡ absent — no distinguishing oracle (spec §6.1
10859 // "confirm existence of hidden nodes: No").
10860 //
10861 // Create-scoped roles (has_create=true): absent → create arm
10862 // as before. The accepted structural key-existence disclosure
10863 // (§THREAT-MODEL) applies only when the role holds create scope.
10864 if !has_create {
10865 return Err(GraphError::RoleWriteDenied {
10866 reason: "role-bound token: target node not visible".into(),
10867 });
10868 }
10869 false // existed = false → create arm
10870 }
10871 }
10872 } else {
10873 // Full authority: use the existing non-masked has_node check.
10874 self.has_node(&key)
10875 };
10876
10877 let existed = merge_existed;
10878 let create_props = if existed {
10879 None
10880 } else {
10881 Some(self.merge_create_props(&stmt.key_field, &stmt.key_value, stmt.ns.as_ref())?)
10882 };
10883 let mut created = 0i64;
10884 if create_props.is_some() || !stmt.on_match.is_empty() {
10885 let mut batch = self.batch();
10886 if let Some(props) = create_props {
10887 batch.insert_node(&stmt.label, &key, props);
10888 for sc in &stmt.on_create {
10889 let value = resolve_merge_set_value(&sc.value, params)?;
10890 batch.set_prop(&key, &sc.field, value);
10891 }
10892 created = 1;
10893 } else {
10894 for sc in &stmt.on_match {
10895 let value = resolve_merge_set_value(&sc.value, params)?;
10896 batch.set_prop(&key, &sc.field, value);
10897 }
10898 }
10899 batch.commit()?;
10900 }
10901
10902 // Refresh the role mask so the just-created node is visible to this
10903 // statement's RETURN (read-after-write). Safe: create_labels ⊆ read labels
10904 // (apply_schema subset rule), so the new node's label is already in the
10905 // role's read scope — this never widens beyond the role's declared labels.
10906 if !existed {
10907 if let Some(role) = self.pending_write_authz.as_ref().map(|a| a.role.clone()) {
10908 let new_mask = self.mask_for_role(&role)?;
10909 if let Some(a) = self.pending_write_authz.as_mut() {
10910 a.mask = new_mask;
10911 }
10912 }
10913 }
10914
10915 // Optional RETURN clause: project the node (created or matched) as a read result.
10916 if let Some(returns) = stmt.returns {
10917 let var = stmt.var.as_deref().unwrap_or("_mn0");
10918 let q = Query {
10919 matches: vec![Pattern {
10920 start: NodePat {
10921 var: Some(var.to_string()),
10922 label: Some(stmt.label.clone()),
10923 props: vec![("id".to_string(), Operand::Lit(stmt.key_value.clone()))],
10924 },
10925 chain: vec![],
10926 shortest: false,
10927 }],
10928 optional_clauses: vec![],
10929 where_expr: None,
10930 unwinds: vec![],
10931 post_unwind_where: None,
10932 stages: vec![],
10933 returns,
10934 distinct: false,
10935 order_by: vec![],
10936 skip: None,
10937 limit: None,
10938 };
10939 let ops = plan(&q).map_err(|e| GraphError::QueryError {
10940 detail: format!("plan: {e}"),
10941 })?;
10942 // Use view_masked when a role-scoped write is in flight so the
10943 // post-merge projection is consistent with the masked read phase.
10944 let mask_opt = self.pending_write_authz.as_ref().map(|a| a.mask.clone());
10945 return (if let Some(ref mask) = mask_opt {
10946 execute(&self.view_masked(mask), &ops, &Params(params))
10947 } else {
10948 execute(&self.view(), &ops, &Params(params))
10949 })
10950 .map_err(|e| GraphError::QueryError {
10951 detail: format!("execute: {e}"),
10952 });
10953 }
10954
10955 let mut rs = write_result_set();
10956 rs.push_row(vec![
10957 Some(Value::Int(created)),
10958 Some(Value::Int(0)),
10959 Some(Value::Int(0)),
10960 ]);
10961 Ok(rs)
10962 }
10963
10964 /// Return all rule-owned edges between `key_a` and `key_b` (either direction),
10965 /// annotated with rule name, edge type, direction, and weight.
10966 /// Results are sorted by (rule, edge_type).
10967 /// Returns `Err(KeyNotFound)` if either key is unknown.
10968 pub fn explain(&self, key_a: &str, key_b: &str) -> Result<Vec<Explanation>> {
10969 self.ensure_v8_base_sections_loaded();
10970 let id_a = self
10971 .ids
10972 .get(key_a)
10973 .ok_or_else(|| GraphError::KeyNotFound { key: key_a.into() })?;
10974 let id_b = self
10975 .ids
10976 .get(key_b)
10977 .ok_or_else(|| GraphError::KeyNotFound { key: key_b.into() })?;
10978
10979 let mut results = Vec::new();
10980
10981 // Walk the smaller incident set so explain is O(min(deg(a), deg(b)))
10982 // rather than O(total provenance).
10983 let scan = if self.engine.provenance_touching_len(id_a)
10984 <= self.engine.provenance_touching_len(id_b)
10985 {
10986 id_a
10987 } else {
10988 id_b
10989 };
10990 for (rule_name, etype, src, dst) in self.engine.provenance_touching(scan) {
10991 if !((src == id_a && dst == id_b) || (src == id_b && dst == id_a)) {
10992 continue;
10993 }
10994 let Some(rule_def) = self.engine.rules().find(|r| r.name == rule_name) else {
10995 continue;
10996 };
10997 let edge_type = match self.syms.resolve(etype) {
10998 Some(s) => s.to_string(),
10999 None => continue,
11000 };
11001 // Provenance (src, dst) ids come from the archived PROVENANCE section
11002 // (large, no eager CRC). A corrupt section can produce ids that are
11003 // out of range; return Corrupt rather than panic.
11004 let src_key = self
11005 .ids
11006 .key_of(src)
11007 .ok_or_else(|| GraphError::Corrupt {
11008 detail: format!("v8: provenance src id {src} not in id table"),
11009 })?
11010 .to_string();
11011 let dst_key = self
11012 .ids
11013 .key_of(dst)
11014 .ok_or_else(|| GraphError::Corrupt {
11015 detail: format!("v8: provenance dst id {dst} not in id table"),
11016 })?
11017 .to_string();
11018 let stored = rule_def.weight_prop.as_deref().and_then(|prop| {
11019 self.edge_props_view()
11020 .get(etype, src, dst, prop)
11021 .and_then(|v| {
11022 if let Value::Float(f) = v {
11023 Some(f)
11024 } else {
11025 None
11026 }
11027 })
11028 });
11029 // Rules that store no weight (KeyMatch/FieldEqual defaults, auto-FK)
11030 // still have a score: recompute it from the predicate so explain
11031 // never reports "no score" for an edge the engine scored. Via-hop
11032 // rules score over their via set, not over (src, dst), so leave
11033 // those None rather than report a number the rule did not produce.
11034 let weight = stored.or_else(|| {
11035 if rule_def.via_edge.is_some() {
11036 return None;
11037 }
11038 let props_view = build_props_view(&self.props, &self.base);
11039 let src_get = |field: &str| props_view.get(src, field).map(|vr| vr.into_value());
11040 let dst_get = |field: &str| props_view.get(dst, field).map(|vr| vr.into_value());
11041 let src_view = NodeView {
11042 key: &src_key,
11043 props: &src_get,
11044 };
11045 let dst_view = NodeView {
11046 key: &dst_key,
11047 props: &dst_get,
11048 };
11049 evaluate(&rule_def.predicate, &src_view, &dst_view)
11050 });
11051 results.push(Explanation {
11052 rule: rule_name.to_string(),
11053 edge_type,
11054 src_key,
11055 dst_key,
11056 weight,
11057 predicate: PredicateSummary {
11058 approximate: rule_def.approximate,
11059 ..PredicateSummary::from(&rule_def.predicate)
11060 },
11061 via_edge: rule_def.via_edge.clone(),
11062 });
11063 }
11064
11065 results.sort_by(|a, b| a.rule.cmp(&b.rule).then(a.edge_type.cmp(&b.edge_type)));
11066 Ok(results)
11067 }
11068
11069 /// [`explain`](Self::explain) with the scoped read contract (§5.3): both
11070 /// endpoints are subject-checked, and any explanation whose evidence runs
11071 /// through a hidden node is **dropped entirely, not redacted**.
11072 ///
11073 /// A plain two-node rule's evidence is the pair itself, so once both
11074 /// subjects are visible there is nothing left to hide. A **via-hop** rule is
11075 /// different: it fires `src → dst` because some node carrying `via_label`
11076 /// sits between them, and [`Explanation`] carries the hop's edge *type*
11077 /// (`via_edge`) and never the hop's key. There is no field to blank, so a
11078 /// redacted explanation would still say "these two are linked through
11079 /// something you cannot see" — which discloses that the something exists.
11080 /// The explanation is therefore kept only when at least one **visible** via
11081 /// node satisfies the rule on its own.
11082 ///
11083 /// The weight is the **visible corpus's** number, not the store's: a via-hop
11084 /// rule stores the max over every via it hopped through, so the stored value
11085 /// can be a score only a hidden via produced. It is recomputed over the
11086 /// visible vias alone.
11087 ///
11088 /// Hidden or unknown `key_a` or `key_b` → [`GraphError::KeyNotFound`].
11089 pub fn explain_scoped(
11090 &self,
11091 key_a: &str,
11092 key_b: &str,
11093 mask: &crate::mask::NodeMask,
11094 ) -> Result<Vec<Explanation>> {
11095 for key in [key_a, key_b] {
11096 if !mask.contains_node(self, key) {
11097 return Err(GraphError::KeyNotFound { key: key.into() });
11098 }
11099 }
11100 Ok(self
11101 .explain(key_a, key_b)?
11102 .into_iter()
11103 .filter_map(|e| self.scoped_explanation(e, mask))
11104 .collect())
11105 }
11106
11107 /// `e` as a caller limited to `mask` may have it, or `None` when it must be
11108 /// dropped entirely.
11109 ///
11110 /// Every non-via-hop explanation passes through untouched: its only nodes
11111 /// are the two subjects, which [`explain_scoped`](Self::explain_scoped) has
11112 /// already checked, and its weight is scored over that pair alone.
11113 ///
11114 /// A via-hop explanation is kept only when some via node the caller may see
11115 /// satisfies the rule on its own — and then its weight is recomputed as the
11116 /// max over exactly those vias. The engine writes the max over **all** of
11117 /// them (`core-rules::engine`, `best = prev.max(score)`), so passing the
11118 /// stored number through would let a hidden node set a figure the caller
11119 /// reads: the same disclosure dropping the explanation exists to prevent.
11120 ///
11121 /// A rule that stores no weight still reports none. The recomputed score is
11122 /// a sanitised version of a number `explain` already returned, never a new
11123 /// one — a scoped read must not say more than the unscoped read it narrows.
11124 fn scoped_explanation(
11125 &self,
11126 e: Explanation,
11127 mask: &crate::mask::NodeMask,
11128 ) -> Option<Explanation> {
11129 let Some(via_edge) = e.via_edge.clone() else {
11130 return Some(e);
11131 };
11132 let Some(rule_def) = self.engine.rules().find(|r| r.name == e.rule) else {
11133 // The rule is gone but its provenance is not; nothing can vouch for
11134 // the hop, so nothing is shown.
11135 return None;
11136 };
11137 let Some(via_label) = rule_def.via_label.as_deref() else {
11138 return Some(e);
11139 };
11140 let (Some(src), Some(dst)) = (self.ids.get(&e.src_key), self.ids.get(&e.dst_key)) else {
11141 return None;
11142 };
11143 let (Some(via_etype), Some(via_sym)) = (self.syms.get(&via_edge), self.syms.get(via_label))
11144 else {
11145 return None;
11146 };
11147 let via_dir = rule_def.via_dir.unwrap_or(Direction::Out);
11148 let props_view = build_props_view(&self.props, &self.base);
11149 let dst_get = |field: &str| props_view.get(dst, field).map(|vr| vr.into_value());
11150 let dst_view = NodeView {
11151 key: &e.dst_key,
11152 props: &dst_get,
11153 };
11154 // The rule's own namespace test, the one the engine applies to each via
11155 // candidate (`core-rules::engine::rule_sees_node`). Without it a
11156 // visible, out-of-namespace via — right label, satisfying predicate —
11157 // vouches for a hop the engine never made, and an explanation whose real
11158 // evidence is a hidden in-namespace node is kept.
11159 let rule_sees = |id: u32| match rule_def.namespace.as_deref() {
11160 None => true,
11161 Some(ns) => {
11162 let value = props_view.get(id, NS_PROP).map(|vr| vr.into_value());
11163 namespace_of_value(value.as_ref()) == ns
11164 }
11165 };
11166 let best = self
11167 .topo_view()
11168 .neighbors(via_etype, via_dir, src)
11169 .iter()
11170 .copied()
11171 .filter_map(|via| {
11172 if !mask.contains_id(via) {
11173 return None;
11174 }
11175 if self.labels.get(via as usize).copied() != Some(via_sym) {
11176 return None;
11177 }
11178 if !rule_sees(via) {
11179 return None;
11180 }
11181 let via_key = self.ids.key_of(via)?;
11182 let via_get = |field: &str| props_view.get(via, field).map(|vr| vr.into_value());
11183 let via_view = NodeView {
11184 key: via_key,
11185 props: &via_get,
11186 };
11187 evaluate(&rule_def.predicate, &via_view, &dst_view)
11188 })
11189 .fold(None::<f64>, |best, score| {
11190 Some(match best {
11191 None => score,
11192 Some(prev) => prev.max(score),
11193 })
11194 })?;
11195 let weight = e.weight.map(|_| best);
11196 Some(Explanation { weight, ..e })
11197 }
11198
11199 pub fn neighbors(&self, key: &str, edge_type: &str, dir: Direction) -> Result<Vec<String>> {
11200 let id = self
11201 .ids
11202 .get(key)
11203 .ok_or_else(|| GraphError::KeyNotFound { key: key.into() })?;
11204 let Some(sym) = self.syms.get(edge_type) else {
11205 return Ok(Vec::new());
11206 };
11207 self.topo_view()
11208 .neighbors(sym, dir, id)
11209 .iter()
11210 .map(|&n| {
11211 self.ids
11212 .key_of(n)
11213 .map(|k| k.to_string())
11214 .ok_or_else(|| GraphError::Corrupt {
11215 detail: format!("topology id {n} has no key"),
11216 })
11217 })
11218 .collect::<Result<Vec<_>>>()
11219 }
11220
11221 /// Unique directed degree of `key`. Unknown key → [`GraphError::KeyNotFound`].
11222 /// Unknown `edge_type` → 0. [`crate::algo::AlgoDir::Both`] is out + in (sum).
11223 pub fn degree(
11224 &self,
11225 key: &str,
11226 edge_type: Option<&str>,
11227 direction: crate::algo::AlgoDir,
11228 ) -> Result<u64> {
11229 let id = self
11230 .ids
11231 .get(key)
11232 .ok_or_else(|| GraphError::KeyNotFound { key: key.into() })?;
11233 let topo = self.topo_view();
11234 Ok(Self::unique_directed_degree(
11235 &topo, &self.syms, id, edge_type, direction,
11236 ))
11237 }
11238
11239 /// [`degree`](Self::degree) summing each pair's **insert count** instead of
11240 /// counting each pair once (§5.13).
11241 ///
11242 /// The unique degree asks how many neighbours there are; this asks how many
11243 /// times they were inserted. A pair with no recorded count contributes 1,
11244 /// so on a store that never called
11245 /// [`enable_multiplicity`](Self::enable_multiplicity) this returns exactly
11246 /// what [`degree`](Self::degree) returns rather than erroring — the
11247 /// distinction is a readout preference, not a demand the store cannot meet.
11248 ///
11249 /// `AlgoDir::Both` still sums out + in, so a pair visible on both sides
11250 /// still contributes twice: multiplicity changes what a pair is worth, never
11251 /// how a direction is counted.
11252 pub fn degree_multiplicity(
11253 &self,
11254 key: &str,
11255 edge_type: Option<&str>,
11256 direction: crate::algo::AlgoDir,
11257 ) -> Result<u64> {
11258 let id = self
11259 .ids
11260 .get(key)
11261 .ok_or_else(|| GraphError::KeyNotFound { key: key.into() })?;
11262 Ok(Self::multiplicity_directed_degree(
11263 &self.topo_view(),
11264 &self.edge_props_view(),
11265 &self.syms,
11266 id,
11267 edge_type,
11268 direction,
11269 None,
11270 ))
11271 }
11272
11273 /// [`degree_multiplicity`](Self::degree_multiplicity) under a scope.
11274 ///
11275 /// The sum covers **visible pairs only**. A hidden neighbour's inserts stay
11276 /// out of it for the reason
11277 /// [`degree_scoped`](Self::degree_scoped) documents, and more sharply: an
11278 /// unscoped multiplicity count discloses not only that a hidden neighbour
11279 /// exists but how often it was written.
11280 ///
11281 /// Hidden or unknown `key` → [`GraphError::KeyNotFound`].
11282 pub fn degree_scoped_multiplicity(
11283 &self,
11284 key: &str,
11285 edge_type: Option<&str>,
11286 direction: crate::algo::AlgoDir,
11287 mask: &crate::mask::NodeMask,
11288 ) -> Result<u64> {
11289 let id = self
11290 .ids
11291 .get(key)
11292 .filter(|&id| mask.contains_id(id))
11293 .ok_or_else(|| GraphError::KeyNotFound { key: key.into() })?;
11294 Ok(Self::multiplicity_directed_degree(
11295 &self.topo_view(),
11296 &self.edge_props_view(),
11297 &self.syms,
11298 id,
11299 edge_type,
11300 direction,
11301 Some(mask),
11302 ))
11303 }
11304
11305 /// [`degree`](Self::degree) counting **only neighbours the mask admits**.
11306 ///
11307 /// The filter is a correctness requirement, not an optimisation: an
11308 /// unfiltered count discloses the existence of a hidden neighbour to a
11309 /// caller who cannot see it, which is the same leak
11310 /// [`node_edges_scoped`](Self::node_edges_scoped) exists to prevent —
11311 /// reached by arithmetic instead of by name.
11312 ///
11313 /// Hidden or unknown `key` → [`GraphError::KeyNotFound`]. Unknown
11314 /// `edge_type` is still 0, as it is unscoped.
11315 pub fn degree_scoped(
11316 &self,
11317 key: &str,
11318 edge_type: Option<&str>,
11319 direction: crate::algo::AlgoDir,
11320 mask: &crate::mask::NodeMask,
11321 ) -> Result<u64> {
11322 let id = self
11323 .ids
11324 .get(key)
11325 .filter(|&id| mask.contains_id(id))
11326 .ok_or_else(|| GraphError::KeyNotFound { key: key.into() })?;
11327 let topo = self.topo_view();
11328 Ok(Self::visible_directed_degree(
11329 &topo, &self.syms, id, edge_type, direction, mask,
11330 ))
11331 }
11332
11333 /// Unique directed degree for a subset or a label scan.
11334 ///
11335 /// Unknown keys in `keys` are omitted (mask-like). `keys = Some(&[])` →
11336 /// empty `Ok(vec![])`. `limit` is applied after sorting degree desc, key
11337 /// asc, and only when `Some`. Invalid `where_` → `QueryError`.
11338 #[allow(clippy::too_many_arguments)]
11339 pub fn degrees(
11340 &self,
11341 keys: Option<&[String]>,
11342 label: Option<&str>,
11343 where_: Option<&PropPredicate>,
11344 edge_type: Option<&str>,
11345 direction: crate::algo::AlgoDir,
11346 limit: Option<usize>,
11347 ) -> Result<Vec<(String, u64)>> {
11348 self.degrees_inner(
11349 keys, label, where_, edge_type, direction, limit, None, false,
11350 )
11351 }
11352
11353 /// [`degrees`](Self::degrees) reporting each row's **insert-count** sum
11354 /// instead of its unique neighbour count, as
11355 /// [`degree_multiplicity`](Self::degree_multiplicity) does for one key.
11356 ///
11357 /// The sort is still degree descending, key ascending — over the counts this
11358 /// reading produces — and `limit` still applies after it.
11359 #[allow(clippy::too_many_arguments)]
11360 pub fn degrees_multiplicity(
11361 &self,
11362 keys: Option<&[String]>,
11363 label: Option<&str>,
11364 where_: Option<&PropPredicate>,
11365 edge_type: Option<&str>,
11366 direction: crate::algo::AlgoDir,
11367 limit: Option<usize>,
11368 ) -> Result<Vec<(String, u64)>> {
11369 self.degrees_inner(keys, label, where_, edge_type, direction, limit, None, true)
11370 }
11371
11372 /// [`degrees_scoped`](Self::degrees_scoped) reporting insert counts.
11373 ///
11374 /// Both filters apply: a hidden key stays out of the result, and every
11375 /// row's sum covers its **visible** pairs only.
11376 #[allow(clippy::too_many_arguments)]
11377 pub fn degrees_scoped_multiplicity(
11378 &self,
11379 keys: Option<&[String]>,
11380 label: Option<&str>,
11381 where_: Option<&PropPredicate>,
11382 edge_type: Option<&str>,
11383 direction: crate::algo::AlgoDir,
11384 limit: Option<usize>,
11385 mask: &crate::mask::NodeMask,
11386 ) -> Result<Vec<(String, u64)>> {
11387 self.degrees_inner(
11388 keys,
11389 label,
11390 where_,
11391 edge_type,
11392 direction,
11393 limit,
11394 Some(mask),
11395 true,
11396 )
11397 }
11398
11399 /// [`degrees`](Self::degrees) with the scope applied on both sides: a hidden
11400 /// key is omitted from the input — whether it arrived in `keys` or came out
11401 /// of the `label`/`where_` scan — and every row's count is the count of its
11402 /// **visible** neighbours, for the reason
11403 /// [`degree_scoped`](Self::degree_scoped) documents.
11404 ///
11405 /// Unlike `degree_scoped`, a hidden key here is not
11406 /// [`GraphError::KeyNotFound`]: `degrees` already drops unknown keys
11407 /// silently, so hidden and absent stay one answer by staying out of the
11408 /// result. `limit` still applies after the sort, and so counts visible rows.
11409 #[allow(clippy::too_many_arguments)]
11410 pub fn degrees_scoped(
11411 &self,
11412 keys: Option<&[String]>,
11413 label: Option<&str>,
11414 where_: Option<&PropPredicate>,
11415 edge_type: Option<&str>,
11416 direction: crate::algo::AlgoDir,
11417 limit: Option<usize>,
11418 mask: &crate::mask::NodeMask,
11419 ) -> Result<Vec<(String, u64)>> {
11420 self.degrees_inner(
11421 keys,
11422 label,
11423 where_,
11424 edge_type,
11425 direction,
11426 limit,
11427 Some(mask),
11428 false,
11429 )
11430 }
11431
11432 /// The body shared by [`degrees`](Self::degrees) and
11433 /// [`degrees_scoped`](Self::degrees_scoped). `mask = None` is the unscoped
11434 /// contract unchanged.
11435 #[allow(clippy::too_many_arguments)]
11436 fn degrees_inner(
11437 &self,
11438 keys: Option<&[String]>,
11439 label: Option<&str>,
11440 where_: Option<&PropPredicate>,
11441 edge_type: Option<&str>,
11442 direction: crate::algo::AlgoDir,
11443 limit: Option<usize>,
11444 mask: Option<&crate::mask::NodeMask>,
11445 multiplicity: bool,
11446 ) -> Result<Vec<(String, u64)>> {
11447 if let Some(pred) = where_ {
11448 pred.validate_named("where")
11449 .map_err(|detail| GraphError::QueryError { detail })?;
11450 }
11451 if matches!(keys, Some(ks) if ks.is_empty()) {
11452 return Ok(Vec::new());
11453 }
11454 let view = self.view();
11455 let ids: Vec<u32> = match keys {
11456 Some(ks) => {
11457 let mut seen = HashSet::new();
11458 let mut out = Vec::new();
11459 for k in ks {
11460 let Some(id) = view.ids.get(k) else {
11461 continue;
11462 };
11463 if !seen.insert(id) {
11464 continue;
11465 }
11466 if let Some(pred) = where_ {
11467 let holds = match view.prop(id, &pred.field) {
11468 None => pred.holds(None),
11469 Some(vr) => pred.holds(Some(vr.as_value())),
11470 };
11471 if !holds {
11472 continue;
11473 }
11474 }
11475 out.push(id);
11476 }
11477 out
11478 }
11479 None => Self::vector_candidates(&view, label, where_),
11480 };
11481 let mut out: Vec<(String, u64)> = ids
11482 .into_iter()
11483 // A hidden candidate leaves as quietly as an unknown key does.
11484 .filter(|&id| mask.is_none_or(|m| m.contains_id(id)))
11485 .filter_map(|id| {
11486 let key = self.ids.key_of(id)?.to_string();
11487 let deg = match (multiplicity, mask) {
11488 // The same `view` the unique arms read, so the per-row
11489 // rebuild F9 measured is gone and all three arms agree on
11490 // the state they are reading.
11491 (true, m) => Self::multiplicity_directed_degree(
11492 &view.topo,
11493 &view.edge_props,
11494 view.syms,
11495 id,
11496 edge_type,
11497 direction,
11498 m,
11499 ),
11500 (false, Some(m)) => Self::visible_directed_degree(
11501 &view.topo, view.syms, id, edge_type, direction, m,
11502 ),
11503 (false, None) => Self::unique_directed_degree(
11504 &view.topo, view.syms, id, edge_type, direction,
11505 ),
11506 };
11507 Some((key, deg))
11508 })
11509 .collect();
11510 out.sort_by(|a, b| b.1.cmp(&a.1).then_with(|| a.0.cmp(&b.0)));
11511 if let Some(lim) = limit {
11512 out.truncate(lim);
11513 }
11514 Ok(out)
11515 }
11516
11517 /// Unique neighbour count for `id` across `edge_type` (or all types) and
11518 /// `direction`. Unknown `edge_type` → 0. `Both` sums out + in.
11519 fn unique_directed_degree(
11520 topo: &TopologyView<'_>,
11521 syms: &Interner,
11522 id: u32,
11523 edge_type: Option<&str>,
11524 direction: crate::algo::AlgoDir,
11525 ) -> u64 {
11526 let dirs: &[Direction] = match direction {
11527 crate::algo::AlgoDir::Out => &[Direction::Out],
11528 crate::algo::AlgoDir::In => &[Direction::In],
11529 crate::algo::AlgoDir::Both => &[Direction::Out, Direction::In],
11530 };
11531 match edge_type {
11532 Some(name) => {
11533 let Some(et) = syms.get(name) else {
11534 return 0;
11535 };
11536 dirs.iter().map(|&d| topo.degree(et, d, id) as u64).sum()
11537 }
11538 None => topo
11539 .etypes()
11540 .map(|et| {
11541 dirs.iter()
11542 .map(|&d| topo.degree(et, d, id) as u64)
11543 .sum::<u64>()
11544 })
11545 .sum(),
11546 }
11547 }
11548
11549 /// [`unique_directed_degree`](Self::unique_directed_degree) counting only
11550 /// neighbours `mask` admits.
11551 ///
11552 /// Same shape, one substitution: `topo.degree` is a length, so it cannot be
11553 /// filtered; the neighbour list it measures can. `Both` still sums out + in,
11554 /// so a node visible on both sides still counts twice — the filter changes
11555 /// which neighbours are counted, never how a degree is defined.
11556 fn visible_directed_degree(
11557 topo: &TopologyView<'_>,
11558 syms: &Interner,
11559 id: u32,
11560 edge_type: Option<&str>,
11561 direction: crate::algo::AlgoDir,
11562 mask: &crate::mask::NodeMask,
11563 ) -> u64 {
11564 let dirs: &[Direction] = match direction {
11565 crate::algo::AlgoDir::Out => &[Direction::Out],
11566 crate::algo::AlgoDir::In => &[Direction::In],
11567 crate::algo::AlgoDir::Both => &[Direction::Out, Direction::In],
11568 };
11569 let visible = |et: u32| -> u64 {
11570 dirs.iter()
11571 .map(|&d| {
11572 topo.neighbors(et, d, id)
11573 .iter()
11574 .filter(|&&n| mask.contains_id(n))
11575 .count() as u64
11576 })
11577 .sum()
11578 };
11579 match edge_type {
11580 Some(name) => syms.get(name).map_or(0, visible),
11581 None => topo.etypes().map(visible).sum(),
11582 }
11583 }
11584
11585 /// Sum of the insert counts of `id`'s pairs (§5.13), over `edge_type` (or
11586 /// all types) and `direction`, restricted to what `mask` admits when one is
11587 /// given.
11588 ///
11589 /// The same neighbour lists the unique reading measures, with each entry
11590 /// worth its pair's count rather than worth 1 — so the filter decides which
11591 /// pairs are in the sum and the count decides what each contributes. A
11592 /// direction decides which way round the pair is addressed: an `In`
11593 /// neighbour `n` of `id` is the pair `(et, n, id)`.
11594 ///
11595 /// Takes its views as parameters, exactly as the unique helpers do, because
11596 /// it is called once per row from a label scan. `edge_props_view()` reaches
11597 /// into the mmap'd base's rkyv section on every call, so building the two
11598 /// views inside made an N-row `degrees(multiplicity=True)` do N section
11599 /// accesses where the unique reading does one: worth 2.57 ms of 16.68 ms
11600 /// over 20 000 rows, about 0.13 us per row (defect #29,
11601 /// `tests/f9_bench.rs`). Most of that call's cost is the per-neighbour
11602 /// count lookup and is inherent, so this is a hoist, not a rescue.
11603 ///
11604 /// The views are exactly `self.view()`'s own `topo` and `edge_props`, so a
11605 /// caller that already has a view passes its halves and reads the same
11606 /// state it reads everything else from.
11607 fn multiplicity_directed_degree(
11608 topo: &TopologyView<'_>,
11609 edge_props: &EdgePropsView<'_>,
11610 syms: &Interner,
11611 id: u32,
11612 edge_type: Option<&str>,
11613 direction: crate::algo::AlgoDir,
11614 mask: Option<&crate::mask::NodeMask>,
11615 ) -> u64 {
11616 let dirs: &[Direction] = match direction {
11617 crate::algo::AlgoDir::Out => &[Direction::Out],
11618 crate::algo::AlgoDir::In => &[Direction::In],
11619 crate::algo::AlgoDir::Both => &[Direction::Out, Direction::In],
11620 };
11621 let count_of = |et: u32, src: u32, dst: u32| -> u64 {
11622 match edge_props.get(et, src, dst, EDGE_COUNT_PROP) {
11623 Some(Value::Int(n)) if n > 0 => n as u64,
11624 _ => 1,
11625 }
11626 };
11627 let per_etype = |et: u32| -> u64 {
11628 dirs.iter()
11629 .map(|&d| {
11630 topo.neighbors(et, d, id)
11631 .iter()
11632 .filter(|&&n| mask.is_none_or(|m| m.contains_id(n)))
11633 .map(|&n| match d {
11634 Direction::Out => count_of(et, id, n),
11635 Direction::In => count_of(et, n, id),
11636 })
11637 .sum::<u64>()
11638 })
11639 .sum()
11640 };
11641 match edge_type {
11642 Some(name) => syms.get(name).map_or(0, per_etype),
11643 None => topo.etypes().map(per_etype).sum(),
11644 }
11645 }
11646
11647 /// Return the last-change commit sequence for `key`, or `None` if the node
11648 /// does not exist or has never been mutated since the last V5-V7 snapshot
11649 /// (horizon-bounded for legacy stores).
11650 ///
11651 /// The returned sequence is a monotonically increasing counter that starts
11652 /// at 1 for the first commit after `open` and increments with every
11653 /// successful write. WAL replay at open also assigns sequences (1..N for N
11654 /// replayed frames), so sequences are consistent across snapshot+WAL cycles.
11655 ///
11656 /// For V5-V7 stores opened without a V8 snapshot, nodes that were present
11657 /// in the snapshot but not touched by any WAL frame will return `None`
11658 /// (horizon-bounded: CAS against such nodes is only safe after the first
11659 /// V8 snapshot or after the node is next mutated).
11660 pub fn last_changed(&self, key: &str) -> Option<u64> {
11661 let id = self.ids.get(key)?;
11662 self.last_change.get(&id).copied()
11663 }
11664
11665 /// Which loaded store this handle is.
11666 ///
11667 /// Paired with [`commit_seq`](GraphDb::commit_seq) it identifies a graph
11668 /// state outright, which `commit_seq` alone does not: two stores of the same
11669 /// age share a sequence, and a reload can return to one. Memos of dense
11670 /// node ids that the handle does not own are stamped with both; see
11671 /// [`StoreStamp`](crate::mask::StoreStamp).
11672 pub(crate) fn store_id(&self) -> crate::mask::StoreId {
11673 self.store_id
11674 }
11675
11676 /// The current commit sequence (number of successful commits since open,
11677 /// including WAL replay frames). Useful for recording a baseline before
11678 /// a read-modify-write cycle.
11679 pub fn commit_seq(&self) -> u64 {
11680 self.commit_seq
11681 }
11682
11683 /// Check that all `preconds` are satisfied against the current db state.
11684 /// Returns `Err(GraphError::CasConflict)` on the first failing precondition.
11685 pub(crate) fn check_preconditions(&self, preconds: &[Precondition]) -> Result<()> {
11686 for precond in preconds {
11687 match precond {
11688 Precondition::NodeUnchangedSince { key, expected } => {
11689 // Missing entry means the node predates the WAL window or
11690 // does not exist; treat as 0 (before any commit).
11691 let actual = self.last_changed(key).unwrap_or_default();
11692 if actual != *expected {
11693 return Err(GraphError::CasConflict {
11694 key: key.clone(),
11695 expected: *expected,
11696 actual,
11697 });
11698 }
11699 }
11700 Precondition::NodeAbsent { key } => {
11701 // Node must not exist (not live).
11702 if self.ids.get(key).is_some() {
11703 let actual = self.last_changed(key).unwrap_or(0);
11704 return Err(GraphError::CasConflict {
11705 key: key.clone(),
11706 expected: u64::MAX,
11707 actual,
11708 });
11709 }
11710 }
11711 }
11712 }
11713 Ok(())
11714 }
11715
11716 /// Apply a batch of mutations with compare-and-set preconditions.
11717 ///
11718 /// All preconditions are checked atomically before any operation is applied.
11719 /// If any precondition fails, the entire batch is rejected with
11720 /// [`GraphError::CasConflict`] and no WAL frame is written.
11721 ///
11722 /// # Returns
11723 /// `(nodes_inserted, edges_inserted)` on success, same as [`write_batch`].
11724 ///
11725 /// # Errors
11726 /// - [`GraphError::CasConflict`] if any precondition is not satisfied.
11727 /// - Any error that [`write_batch`] would return for the ops themselves.
11728 pub fn write_batch_cas(
11729 &mut self,
11730 preconds: Vec<Precondition>,
11731 ops: Vec<BatchOp>,
11732 ) -> Result<(usize, usize)> {
11733 self.check_preconditions(&preconds)?;
11734 self.commit_logged_batch(ops, None, None).map(inserted_pair)
11735 }
11736
11737 /// Update the per-node last-change map for a WAL record at commit `seq`.
11738 ///
11739 /// Called after a successful apply to record which nodes were touched.
11740 /// For replay, called with the WAL-frame's replayed seq.
11741 ///
11742 /// Touch definition (see [`Precondition`] doc):
11743 /// - InsertNode / InsertNodeId / SetProp / SetPropId / RemoveProp → the node.
11744 /// - InsertEdge / InsertEdgeId / DeleteEdge → both src and dst.
11745 /// - DeleteNode → node tombstoned; last_changed() returns None so no update needed.
11746 /// - DerivedEdge markers, Intern, rule/view records → no-ops.
11747 /// - Batch → recurse into inner records.
11748 fn update_last_change_from_rec(&mut self, rec: &WalRecord, seq: u64) {
11749 match rec {
11750 WalRecord::InsertNode { key, .. }
11751 | WalRecord::SetProp { key, .. }
11752 | WalRecord::RemoveProp { key, .. } => {
11753 if let Some(id) = self.ids.get(key) {
11754 self.last_change.insert(id, seq);
11755 }
11756 }
11757 WalRecord::InsertNodeId { key, .. } => {
11758 if let Some(id) = self.ids.get(key) {
11759 self.last_change.insert(id, seq);
11760 }
11761 }
11762 WalRecord::SetPropId { id, .. } => {
11763 self.last_change.insert(*id, seq);
11764 }
11765 WalRecord::InsertEdge {
11766 src_key, dst_key, ..
11767 }
11768 | WalRecord::DeleteEdge {
11769 src_key, dst_key, ..
11770 } => {
11771 if let Some(src_id) = self.ids.get(src_key) {
11772 self.last_change.insert(src_id, seq);
11773 }
11774 if let Some(dst_id) = self.ids.get(dst_key) {
11775 self.last_change.insert(dst_id, seq);
11776 }
11777 }
11778 WalRecord::InsertEdgeId { src, dst, .. } => {
11779 self.last_change.insert(*src, seq);
11780 self.last_change.insert(*dst, seq);
11781 }
11782 // A count record touches the pair, so it touches both endpoints —
11783 // the same reading `InsertEdgeId` gets, because a duplicate insert
11784 // that raises the count *is* a mutation of that pair. The opt-in
11785 // declaration touches nothing.
11786 WalRecord::SetEdgeCount { src, dst, .. } if !rec.is_multiplicity_decl() => {
11787 self.last_change.insert(*src, seq);
11788 self.last_change.insert(*dst, seq);
11789 }
11790 WalRecord::SetEdgeCount { .. } => {}
11791 // DeleteNode: node is tombstoned; last_changed(key) returns None for
11792 // deleted keys (ids.get() returns None post-tombstone), so no update needed.
11793 // History markers: state no-ops; the underlying mutation already
11794 // touched the relevant nodes' last_change entries.
11795 WalRecord::DeleteNode { .. }
11796 | WalRecord::DerivedEdgeAdded { .. }
11797 | WalRecord::DerivedEdgeRetracted { .. }
11798 | WalRecord::Intern { .. }
11799 | WalRecord::CreateRule { .. }
11800 | WalRecord::DeleteRule { .. }
11801 | WalRecord::RebuildRule { .. }
11802 | WalRecord::CreateView { .. }
11803 | WalRecord::DeleteView { .. }
11804 | WalRecord::EnableFulltext { .. }
11805 | WalRecord::DisableFulltext { .. }
11806 | WalRecord::EnableIndex { .. }
11807 | WalRecord::DisableIndex { .. } => {}
11808 // RenameNode: node id is stable; update last_change via the new key.
11809 // Called after apply(), so ids already reflects new_key.
11810 WalRecord::RenameNode { new_key, .. } => {
11811 if let Some(id) = self.ids.get(new_key) {
11812 self.last_change.insert(id, seq);
11813 }
11814 }
11815 WalRecord::Batch(inner) => {
11816 for inner_rec in inner {
11817 self.update_last_change_from_rec(inner_rec, seq);
11818 }
11819 }
11820 }
11821 }
11822
11823 pub fn node_count(&self) -> usize {
11824 self.ids.len()
11825 }
11826
11827 /// Configure archive retention: keep the `N` newest WAL archives at each
11828 /// [`snapshot_with`] call when `archive_wal: true`.
11829 ///
11830 /// `Some(N)` where N > 0 → prune oldest archives keeping the newest N.
11831 /// `Some(0)` or `None` → unlimited (no pruning).
11832 ///
11833 /// Pruning only ever happens inside [`snapshot_with`]; this method only
11834 /// stores the policy. Archives below the retention limit are deleted
11835 /// oldest-first. The horizon floor is updated so that
11836 /// [`was_linked`] / history APIs return `CommitOutOfRange` for commits
11837 /// in pruned archives rather than silently returning wrong data.
11838 pub fn set_wal_archive_retention(&mut self, keep: Option<u32>) {
11839 self.wal_archive_retention = keep;
11840 }
11841
11842 /// Delete any WAL archives that are fully below the current horizon floor.
11843 ///
11844 /// Orphaned archives arise when the floor is written first during retention
11845 /// pruning and then a crash interrupts the archive-delete sequence. The
11846 /// opening cleanup ensures no subsequent read path sees stale data.
11847 ///
11848 /// Under the monotonic naming scheme, the archive name N equals the
11849 /// cumulative end-frame index of the archive in global commit space (i.e.
11850 /// the archive covers global frames `[prev_n, N)`). An archive is
11851 /// fully orphaned when `N <= wal_horizon_floor`: all of its frames fall
11852 /// below the floor and have already been counted in it.
11853 fn cleanup_orphaned_archives(&mut self) -> Result<()> {
11854 if self.wal_horizon_floor == 0 {
11855 // Floor at 0 means no pruning has ever occurred; nothing to clean.
11856 return Ok(());
11857 }
11858 let archive_ns = self.fs.list_archives()?;
11859 for n in archive_ns {
11860 if n <= self.wal_horizon_floor {
11861 // Archive N ends at global frame N; all its frames are below
11862 // the floor (floor already accounts for them) → orphaned.
11863 self.fs.delete_archive(n).map_err(GraphError::Io)?;
11864 } else {
11865 // Archives are sorted ascending; first one above floor stops scan.
11866 break;
11867 }
11868 }
11869 Ok(())
11870 }
11871
11872 /// Collect all WAL frames from surviving archives (oldest-first) then the
11873 /// live WAL into one flat list, and return the total along with the number
11874 /// of archive frames at the front of the list.
11875 ///
11876 /// Commit indices into the returned list are LOCAL (0 = first frame of
11877 /// oldest surviving archive). To obtain the GLOBAL index add
11878 /// `self.wal_horizon_floor`.
11879 fn all_frames(&self) -> Result<(Vec<WalRecord>, u64)> {
11880 let archive_ns = self.fs.list_archives()?;
11881 let mut all: Vec<WalRecord> = Vec::new();
11882 for n in archive_ns {
11883 let bytes = self.fs.read_archive(n)?;
11884 let (frames, _) = decode_all(&bytes);
11885 all.extend(frames);
11886 }
11887 let archive_count = all.len() as u64;
11888 let live_bytes = self.fs.read(FileId::Wal)?;
11889 let (live_frames, _) = decode_all(&live_bytes);
11890 all.extend(live_frames);
11891 Ok((all, archive_count))
11892 }
11893
11894 /// Return the total number of committed WAL frames visible in the current
11895 /// horizon window, including frames in surviving WAL archives.
11896 ///
11897 /// This is the exclusive upper bound for valid `at_commit` indices in
11898 /// `was_linked`. Valid indices are `wal_horizon_floor()..wal_total_commits()`.
11899 ///
11900 /// Returns the horizon floor when all surviving history is empty.
11901 pub fn wal_total_commits(&self) -> Result<u64> {
11902 let (frames, _) = self.all_frames()?;
11903 Ok(self.wal_horizon_floor + frames.len() as u64)
11904 }
11905
11906 /// The global frame index of the first commit reachable through surviving
11907 /// archives (0 when no archives have been pruned).
11908 pub fn wal_horizon_floor(&self) -> u64 {
11909 self.wal_horizon_floor
11910 }
11911
11912 /// Return the per-node change history for `key` by scanning the on-disk WAL.
11913 ///
11914 /// ## Horizon
11915 ///
11916 /// History reaches back only to the last WAL-truncating snapshot, exactly like `open_at`.
11917 /// Snapshots written with `keep_wal: true` preserve deeper history. This is the honest,
11918 /// zero-cost contract; a durable history log is out of scope.
11919 ///
11920 /// ## Derived edges
11921 ///
11922 /// Rule-created (derived) edges are **not** in the WAL and therefore do not appear in
11923 /// history. Only edges written directly by the application are recorded.
11924 ///
11925 /// ## Deleted nodes
11926 ///
11927 /// For nodes that have been deleted, dense-id records (SetPropId, InsertEdgeId) that
11928 /// predate the deletion may not resolve (the id is tombstoned in the live map). The
11929 /// string-keyed `DeleteNode` record still matches and produces a `NodeDeleted` entry.
11930 /// Prop/edge history of a deleted node may therefore be partially unresolvable.
11931 ///
11932 /// ## Dense-id edge entries and tombstoned partners
11933 ///
11934 /// Edge entries from dense-id WAL records (`InsertEdgeId`) are omitted when the partner
11935 /// endpoint's dense id is tombstoned. As a result, a live node's history can contain an
11936 /// `EdgeRemoved` (string-keyed, always resolves) without a corresponding `EdgeAdded`.
11937 /// Build commit-bounded alias intervals for `queried_key`.
11938 ///
11939 /// Returns a list of `(key, valid_from_inclusive, valid_until_exclusive)` tuples.
11940 /// A record written under `key` at commit `c` matches the queried identity iff
11941 /// `c >= valid_from && (valid_until.is_none() || c < valid_until)`.
11942 ///
11943 /// Each alias entry carries both a lower and an upper bound so that key-reuse
11944 /// after a rename is handled correctly: if "a" is renamed to "b" at commit 5,
11945 /// then a NEW node is created as "a" at commit 7 and renamed to "c" at commit 10,
11946 /// querying "c" must NOT surface identity-1's events (commits 0–4 under "a");
11947 /// only identity-2's events (commits 7–9 under "a") are in scope.
11948 ///
11949 /// Only **forward aliasing**: querying the *new* key surfaces events written
11950 /// under the *old* key. The reverse direction is not supported.
11951 fn build_key_alias_intervals(
11952 &self,
11953 frames: &[core_storage::wal::WalRecord],
11954 queried_key: &str,
11955 ) -> Vec<(String, u64, Option<u64>)> {
11956 use core_storage::wal::WalRecord;
11957
11958 // Pre-pass: build reverse_rename and key_starts maps.
11959 let mut reverse_rename: HashMap<String, (String, u64)> = HashMap::new();
11960 let mut key_starts: HashMap<String, Vec<u64>> = HashMap::new();
11961
11962 for (local_i, frame) in frames.iter().enumerate() {
11963 let commit = self.wal_horizon_floor + local_i as u64;
11964 let records: &[WalRecord] = match frame {
11965 WalRecord::Batch(inner) => inner.as_slice(),
11966 single => std::slice::from_ref(single),
11967 };
11968 for rec in records {
11969 match rec {
11970 WalRecord::InsertNode { key, .. } | WalRecord::InsertNodeId { key, .. } => {
11971 key_starts.entry(key.clone()).or_default().push(commit);
11972 }
11973 WalRecord::RenameNode { old_key, new_key } => {
11974 // new_key came into existence at this commit.
11975 key_starts.entry(new_key.clone()).or_default().push(commit);
11976 // Record the reverse rename: new_key was introduced by renaming old_key.
11977 reverse_rename.insert(new_key.clone(), (old_key.clone(), commit));
11978 }
11979 _ => {}
11980 }
11981 }
11982 }
11983
11984 // Build alias intervals by following the reverse rename chain.
11985 let mut result: Vec<(String, u64, Option<u64>)> = Vec::new();
11986 let mut current_key = queried_key.to_string();
11987 let mut current_valid_until: Option<u64> = None;
11988
11989 loop {
11990 // valid_from: the most recent commit where current_key was assigned to this
11991 // identity. For aliases (valid_until = Some(vu)), find the last start event
11992 // for the key strictly before vu — this is where the alias's occupancy by
11993 // this identity began, correctly excluding prior identities that reused the key.
11994 let valid_from = if let Some(vu) = current_valid_until {
11995 key_starts
11996 .get(¤t_key)
11997 .and_then(|starts| starts.iter().rev().find(|&&s| s < vu).copied())
11998 .unwrap_or(self.wal_horizon_floor)
11999 } else {
12000 // Queried key — no upper bound; may have been introduced at any commit.
12001 self.wal_horizon_floor
12002 };
12003
12004 result.push((current_key.clone(), valid_from, current_valid_until));
12005
12006 match reverse_rename.get(¤t_key) {
12007 Some((old_key, rename_commit)) => {
12008 current_valid_until = Some(*rename_commit);
12009 current_key = old_key.clone();
12010 }
12011 None => break,
12012 }
12013 }
12014
12015 result
12016 }
12017
12018 /// Returns true if `record_key` matches any alias interval that covers `commit`.
12019 fn aliases_match(
12020 intervals: &[(String, u64, Option<u64>)],
12021 record_key: &str,
12022 commit: u64,
12023 ) -> bool {
12024 intervals
12025 .iter()
12026 .any(|(k, vf, vu)| k == record_key && commit >= *vf && vu.is_none_or(|u| commit < u))
12027 }
12028
12029 /// Return the change history of node `key` by scanning the on-disk WAL.
12030 ///
12031 /// ## Horizon
12032 ///
12033 /// History reaches back only as far as the retained WAL. The returned
12034 /// [`HistoryResult`](crate::history::HistoryResult) carries `total_commits`
12035 /// (the exclusive upper bound for valid commit indices) and `horizon` (the
12036 /// oldest commit still reachable). When `horizon > 0`, older events were
12037 /// pruned and are not in `items`.
12038 pub fn node_history(
12039 &self,
12040 key: &str,
12041 ) -> Result<crate::history::HistoryResult<crate::history::HistoryEntry>> {
12042 use crate::history::{HistoryChange, HistoryEntry, HistoryResult};
12043 use core_storage::wal::WalRecord;
12044
12045 let (frames, _) = self.all_frames()?;
12046 let total_commits = self.wal_horizon_floor + frames.len() as u64;
12047
12048 // Resolve commit-bounded alias intervals for `key` (handles renames in the WAL).
12049 let alias_intervals = self.build_key_alias_intervals(&frames, key);
12050
12051 let mut out: Vec<HistoryEntry> = Vec::new();
12052
12053 for (local_i, frame) in frames.iter().enumerate() {
12054 let commit = self.wal_horizon_floor + local_i as u64;
12055 // Collect the inner records to process — Batch is one commit, single records are one commit.
12056 let records: &[WalRecord] = match frame {
12057 WalRecord::Batch(inner) => inner.as_slice(),
12058 single => std::slice::from_ref(single),
12059 };
12060
12061 for rec in records {
12062 let change = match rec {
12063 WalRecord::InsertNode { label, key: k, .. }
12064 if Self::aliases_match(&alias_intervals, k, commit) =>
12065 {
12066 Some(HistoryChange::NodeInserted {
12067 label: label.clone(),
12068 })
12069 }
12070 WalRecord::InsertNodeId { label, key: k, .. }
12071 if Self::aliases_match(&alias_intervals, k, commit) =>
12072 {
12073 let label_str = match self.syms.resolve(*label) {
12074 Some(s) => s.to_string(),
12075 None => continue,
12076 };
12077 Some(HistoryChange::NodeInserted { label: label_str })
12078 }
12079 WalRecord::SetProp {
12080 key: k,
12081 field,
12082 value,
12083 } if Self::aliases_match(&alias_intervals, k, commit) => {
12084 Some(HistoryChange::PropSet {
12085 field: field.clone(),
12086 value: value.clone(),
12087 })
12088 }
12089 WalRecord::SetPropId { id, field, value } => {
12090 // Use key_of_historical (not key_of) so a node's prop_set
12091 // events remain visible after the node is later deleted:
12092 // key_of returns None for a tombstoned id, which would
12093 // silently drop every PropSet between insert and delete.
12094 // Mirrors the InsertEdgeId arm below and edge_history's
12095 // own id-keyed arms.
12096 match self.ids.key_of_historical(*id) {
12097 // key_of_historical returns the last-known (possibly
12098 // post-rename, possibly post-delete) key; compare to queried key.
12099 Some(resolved) if resolved == key => {
12100 let field_str = match self.syms.resolve(*field) {
12101 Some(s) => s.to_string(),
12102 None => continue,
12103 };
12104 Some(HistoryChange::PropSet {
12105 field: field_str,
12106 value: value.clone(),
12107 })
12108 }
12109 _ => None,
12110 }
12111 }
12112 WalRecord::RemoveProp { key: k, field }
12113 if Self::aliases_match(&alias_intervals, k, commit) =>
12114 {
12115 Some(HistoryChange::PropRemoved {
12116 field: field.clone(),
12117 })
12118 }
12119 WalRecord::InsertEdge {
12120 edge_type,
12121 src_key,
12122 dst_key,
12123 } => {
12124 if Self::aliases_match(&alias_intervals, src_key, commit) {
12125 Some(HistoryChange::EdgeAdded {
12126 edge_type: edge_type.clone(),
12127 other: dst_key.clone(),
12128 outgoing: true,
12129 })
12130 } else if Self::aliases_match(&alias_intervals, dst_key, commit) {
12131 Some(HistoryChange::EdgeAdded {
12132 edge_type: edge_type.clone(),
12133 other: src_key.clone(),
12134 outgoing: false,
12135 })
12136 } else {
12137 None
12138 }
12139 }
12140 WalRecord::InsertEdgeId { etype, src, dst } => {
12141 let etype_str = match self.syms.resolve(*etype) {
12142 Some(s) => s.to_string(),
12143 None => continue,
12144 };
12145 // key_of_historical (not key_of): an edge added before
12146 // either endpoint was later deleted must still resolve —
12147 // see the SetPropId arm above and edge_history's
12148 // InsertEdgeId arm, which use the same lookup for the
12149 // same reason.
12150 let src_key = self.ids.key_of_historical(*src);
12151 let dst_key = self.ids.key_of_historical(*dst);
12152 if src_key == Some(key) {
12153 let other = match dst_key {
12154 Some(s) => s.to_string(),
12155 None => continue,
12156 };
12157 Some(HistoryChange::EdgeAdded {
12158 edge_type: etype_str,
12159 other,
12160 outgoing: true,
12161 })
12162 } else if dst_key == Some(key) {
12163 let other = match src_key {
12164 Some(s) => s.to_string(),
12165 None => continue,
12166 };
12167 Some(HistoryChange::EdgeAdded {
12168 edge_type: etype_str,
12169 other,
12170 outgoing: false,
12171 })
12172 } else {
12173 None
12174 }
12175 }
12176 WalRecord::DeleteEdge {
12177 edge_type,
12178 src_key,
12179 dst_key,
12180 } => {
12181 if Self::aliases_match(&alias_intervals, src_key, commit) {
12182 Some(HistoryChange::EdgeRemoved {
12183 edge_type: edge_type.clone(),
12184 other: dst_key.clone(),
12185 outgoing: true,
12186 })
12187 } else if Self::aliases_match(&alias_intervals, dst_key, commit) {
12188 Some(HistoryChange::EdgeRemoved {
12189 edge_type: edge_type.clone(),
12190 other: src_key.clone(),
12191 outgoing: false,
12192 })
12193 } else {
12194 None
12195 }
12196 }
12197 WalRecord::DeleteNode { key: k }
12198 if Self::aliases_match(&alias_intervals, k, commit) =>
12199 {
12200 Some(HistoryChange::NodeDeleted)
12201 }
12202 // Skip: rule/view/fulltext/intern metadata; Batch wrapper handled above.
12203 _ => None,
12204 };
12205
12206 if let Some(change) = change {
12207 out.push(HistoryEntry { commit, change });
12208 }
12209 }
12210 }
12211
12212 Ok(HistoryResult {
12213 items: out,
12214 total_commits,
12215 horizon: self.wal_horizon_floor,
12216 })
12217 }
12218
12219 /// Return the per-edge change history between nodes `a` and `b` by scanning
12220 /// the on-disk WAL.
12221 ///
12222 /// ## Horizon
12223 ///
12224 /// History reaches back only to the last WAL-truncating snapshot, exactly
12225 /// like `node_history` and `open_at`. The returned [`HistoryResult`] carries
12226 /// `total_commits` (= number of WAL frames), which is the exclusive upper
12227 /// bound for valid commit indices.
12228 ///
12229 /// ## Derived edges
12230 ///
12231 /// Rule-derived edges appear via `DerivedEdgeAdded` / `DerivedEdgeRetracted`
12232 /// WAL markers written by `log_then_apply_with` after each rule-firing
12233 /// mutation. The `rule` field of those events carries the rule name.
12234 ///
12235 /// ## DeleteNode
12236 ///
12237 /// When a node is deleted, its manual incident edges are swept inline without
12238 /// individual `DeleteEdge` WAL records. `edge_history` detects `DeleteNode`
12239 /// events for either endpoint and synthesises `Retracted(rule:None)` events
12240 /// for each manual edge that was active at that point. Derived edges active at
12241 /// the time of deletion are handled by the `DerivedEdgeRetracted` marker that
12242 /// the engine appends immediately after the `DeleteNode` record; those events
12243 /// carry correct rule attribution and are emitted by the marker arm, not the
12244 /// synthetic sweep.
12245 ///
12246 /// ## Masks
12247 ///
12248 /// Like `node_history`, this method has no mask parameter and returns WAL
12249 /// history regardless of any role mask. For masked history semantics, apply
12250 /// the mask at the caller level.
12251 pub fn edge_history(
12252 &self,
12253 a: &str,
12254 b: &str,
12255 ) -> Result<crate::history::HistoryResult<crate::history::EdgeHistoryEvent>> {
12256 use crate::history::{EdgeEvent, EdgeHistoryEvent, HistoryResult};
12257 use core_storage::wal::WalRecord;
12258
12259 let (frames, _) = self.all_frames()?;
12260 let total_commits = self.wal_horizon_floor + frames.len() as u64;
12261
12262 // Resolve all historical names for a and b (handles RenameNode in the WAL).
12263 // Intervals are commit-bounded so recycled keys don't contaminate histories.
12264 let alias_a = self.build_key_alias_intervals(&frames, a);
12265 let alias_b = self.build_key_alias_intervals(&frames, b);
12266
12267 // Active edges between a and b tracked as (edge_type, src_key, dst_key, is_derived).
12268 // The is_derived flag is used by the DeleteNode sweep: manual edges are
12269 // swept with a synthetic Retracted(rule:None); derived edges are skipped
12270 // because the engine writes a DerivedEdgeRetracted marker immediately after
12271 // the DeleteNode record, which carries the correct rule attribution.
12272 let mut active: Vec<(String, String, String, bool)> = Vec::new();
12273 let mut out: Vec<EdgeHistoryEvent> = Vec::new();
12274
12275 for (local_i, frame) in frames.iter().enumerate() {
12276 let commit = self.wal_horizon_floor + local_i as u64;
12277 let records: &[WalRecord] = match frame {
12278 WalRecord::Batch(inner) => inner.as_slice(),
12279 single => std::slice::from_ref(single),
12280 };
12281
12282 for rec in records {
12283 match rec {
12284 WalRecord::InsertEdge {
12285 edge_type,
12286 src_key,
12287 dst_key,
12288 } => {
12289 let is_ab = Self::aliases_match(&alias_a, src_key, commit)
12290 && Self::aliases_match(&alias_b, dst_key, commit);
12291 let is_ba = Self::aliases_match(&alias_b, src_key, commit)
12292 && Self::aliases_match(&alias_a, dst_key, commit);
12293 if is_ab || is_ba {
12294 active.push((
12295 edge_type.clone(),
12296 src_key.clone(),
12297 dst_key.clone(),
12298 false,
12299 ));
12300 out.push(EdgeHistoryEvent {
12301 edge_type: edge_type.clone(),
12302 commit,
12303 event: EdgeEvent::Added,
12304 rule: None,
12305 });
12306 }
12307 }
12308 WalRecord::InsertEdgeId { etype, src, dst } => {
12309 let etype_str = match self.syms.resolve(*etype) {
12310 Some(s) => s.to_string(),
12311 None => continue,
12312 };
12313 // Use key_of_historical so tombstoned nodes (deleted
12314 // later in the WAL) still resolve during the scan.
12315 let src_key = self.ids.key_of_historical(*src);
12316 let dst_key = self.ids.key_of_historical(*dst);
12317 let is_ab = src_key == Some(a) && dst_key == Some(b);
12318 let is_ba = src_key == Some(b) && dst_key == Some(a);
12319 if is_ab || is_ba {
12320 let src_str = src_key.unwrap().to_string();
12321 let dst_str = dst_key.unwrap().to_string();
12322 active.push((etype_str.clone(), src_str, dst_str, false));
12323 out.push(EdgeHistoryEvent {
12324 edge_type: etype_str,
12325 commit,
12326 event: EdgeEvent::Added,
12327 rule: None,
12328 });
12329 }
12330 }
12331 WalRecord::DeleteEdge {
12332 edge_type,
12333 src_key,
12334 dst_key,
12335 } => {
12336 let is_ab = Self::aliases_match(&alias_a, src_key, commit)
12337 && Self::aliases_match(&alias_b, dst_key, commit);
12338 let is_ba = Self::aliases_match(&alias_b, src_key, commit)
12339 && Self::aliases_match(&alias_a, dst_key, commit);
12340 if is_ab || is_ba {
12341 // Remove the first matching active entry (flag ignored).
12342 if let Some(pos) = active.iter().position(|(et, s, d, _)| {
12343 et == edge_type && s == src_key && d == dst_key
12344 }) {
12345 active.remove(pos);
12346 }
12347 out.push(EdgeHistoryEvent {
12348 edge_type: edge_type.clone(),
12349 commit,
12350 event: EdgeEvent::Retracted,
12351 rule: None,
12352 });
12353 }
12354 }
12355 WalRecord::DeleteNode { key: k }
12356 if Self::aliases_match(&alias_a, k, commit)
12357 || Self::aliases_match(&alias_b, k, commit) =>
12358 {
12359 // Sweep: implicitly retract only MANUAL active edges.
12360 // Derived active edges are skipped here because the rule
12361 // engine appends a DerivedEdgeRetracted marker immediately
12362 // after this DeleteNode record; that marker produces the
12363 // single correctly-attributed Retracted event. Derived
12364 // entries are dropped from `active` (the marker arm's
12365 // idempotent retain finds nothing to remove).
12366 for (et, _, _, is_derived) in active.drain(..) {
12367 if !is_derived {
12368 out.push(EdgeHistoryEvent {
12369 edge_type: et,
12370 commit,
12371 event: EdgeEvent::Retracted,
12372 rule: None,
12373 });
12374 }
12375 // Derived: drop silently; marker carries the Retracted event.
12376 }
12377 }
12378 WalRecord::DerivedEdgeAdded {
12379 rule,
12380 edge_type: et,
12381 src_key,
12382 dst_key,
12383 } => {
12384 let is_ab = Self::aliases_match(&alias_a, src_key, commit)
12385 && Self::aliases_match(&alias_b, dst_key, commit);
12386 let is_ba = Self::aliases_match(&alias_b, src_key, commit)
12387 && Self::aliases_match(&alias_a, dst_key, commit);
12388 if is_ab || is_ba {
12389 active.push((et.clone(), src_key.clone(), dst_key.clone(), true));
12390 out.push(EdgeHistoryEvent {
12391 edge_type: et.clone(),
12392 commit,
12393 event: EdgeEvent::Added,
12394 rule: Some(rule.clone()),
12395 });
12396 }
12397 }
12398 WalRecord::DerivedEdgeRetracted {
12399 rule,
12400 edge_type: et,
12401 src_key,
12402 dst_key,
12403 } => {
12404 let is_ab = Self::aliases_match(&alias_a, src_key, commit)
12405 && Self::aliases_match(&alias_b, dst_key, commit);
12406 let is_ba = Self::aliases_match(&alias_b, src_key, commit)
12407 && Self::aliases_match(&alias_a, dst_key, commit);
12408 if is_ab || is_ba {
12409 // Push unconditionally: a derived edge whose Added marker
12410 // predates the history horizon has no `active` entry, but
12411 // the retraction is still a real in-window event.
12412 // Remove from active idempotently if present.
12413 active.retain(|(aet, s, d, _)| {
12414 !(aet == et && s == src_key && d == dst_key)
12415 });
12416 out.push(EdgeHistoryEvent {
12417 edge_type: et.clone(),
12418 commit,
12419 event: EdgeEvent::Retracted,
12420 rule: Some(rule.clone()),
12421 });
12422 }
12423 }
12424 // All other records (InsertNode, SetProp, CreateRule, etc.)
12425 // do not affect edges between a and b.
12426 _ => {}
12427 }
12428 }
12429 }
12430
12431 Ok(HistoryResult {
12432 items: out,
12433 total_commits,
12434 horizon: self.wal_horizon_floor,
12435 })
12436 }
12437
12438 /// Return `true` iff an edge of `edge_type` existed between `a` and `b`
12439 /// (in either direction) at the WAL commit `at_commit`.
12440 ///
12441 /// ## Horizon
12442 ///
12443 /// Valid commit indices are `0..total_commits` where `total_commits` is the
12444 /// number of WAL frames. An `at_commit >= total_commits` is outside the
12445 /// visible horizon and returns [`GraphError::CommitOutOfRange`].
12446 ///
12447 /// ## Derived edges
12448 ///
12449 /// Rule-derived edges are tracked via `DerivedEdgeAdded` / `DerivedEdgeRetracted`
12450 /// WAL markers appended at firing time (Task 1). `was_linked` reads these markers
12451 /// and therefore includes derived edges in its point-in-time evaluation,
12452 /// matching `edge_history`'s fidelity.
12453 pub fn was_linked(&self, a: &str, b: &str, edge_type: &str, at_commit: u64) -> Result<bool> {
12454 use core_storage::wal::WalRecord;
12455
12456 let (frames, _) = self.all_frames()?;
12457 let total_commits = self.wal_horizon_floor + frames.len() as u64;
12458
12459 // Horizon floor: commits in pruned archives are unreachable.
12460 if at_commit < self.wal_horizon_floor {
12461 return Err(GraphError::CommitOutOfRange {
12462 commit: at_commit,
12463 total: total_commits,
12464 floor: self.wal_horizon_floor,
12465 });
12466 }
12467 if at_commit >= total_commits {
12468 return Err(GraphError::CommitOutOfRange {
12469 commit: at_commit,
12470 total: total_commits,
12471 floor: self.wal_horizon_floor,
12472 });
12473 }
12474
12475 // Resolve all historical names for a and b (handles RenameNode in the WAL).
12476 // Intervals are commit-bounded so recycled keys don't contaminate point-in-time reads.
12477 let alias_a = self.build_key_alias_intervals(&frames, a);
12478 let alias_b = self.build_key_alias_intervals(&frames, b);
12479
12480 // Local index into surviving frames (0 = first frame of oldest archive).
12481 let local_commit = at_commit - self.wal_horizon_floor;
12482
12483 // Replay local frames 0..=local_commit, tracking active edges.
12484 let mut active: BTreeSet<(String, String, String)> = BTreeSet::new();
12485
12486 for (local_i, frame) in frames.iter().enumerate().take((local_commit + 1) as usize) {
12487 let commit = self.wal_horizon_floor + local_i as u64;
12488 let records: &[WalRecord] = match frame {
12489 WalRecord::Batch(inner) => inner.as_slice(),
12490 single => std::slice::from_ref(single),
12491 };
12492
12493 for rec in records {
12494 match rec {
12495 WalRecord::InsertEdge {
12496 edge_type: et,
12497 src_key,
12498 dst_key,
12499 } => {
12500 let is_ab = Self::aliases_match(&alias_a, src_key, commit)
12501 && Self::aliases_match(&alias_b, dst_key, commit);
12502 let is_ba = Self::aliases_match(&alias_b, src_key, commit)
12503 && Self::aliases_match(&alias_a, dst_key, commit);
12504 if is_ab || is_ba {
12505 active.insert((et.clone(), src_key.clone(), dst_key.clone()));
12506 }
12507 }
12508 WalRecord::InsertEdgeId { etype, src, dst } => {
12509 let etype_str = match self.syms.resolve(*etype) {
12510 Some(s) => s.to_string(),
12511 None => continue,
12512 };
12513 // Use key_of_historical so tombstoned nodes resolve.
12514 let src_key = self.ids.key_of_historical(*src);
12515 let dst_key = self.ids.key_of_historical(*dst);
12516 let is_ab = src_key == Some(a) && dst_key == Some(b);
12517 let is_ba = src_key == Some(b) && dst_key == Some(a);
12518 if is_ab || is_ba {
12519 active.insert((
12520 etype_str,
12521 src_key.unwrap().to_string(),
12522 dst_key.unwrap().to_string(),
12523 ));
12524 }
12525 }
12526 WalRecord::DeleteEdge {
12527 edge_type: et,
12528 src_key,
12529 dst_key,
12530 } => {
12531 let is_ab = Self::aliases_match(&alias_a, src_key, commit)
12532 && Self::aliases_match(&alias_b, dst_key, commit);
12533 let is_ba = Self::aliases_match(&alias_b, src_key, commit)
12534 && Self::aliases_match(&alias_a, dst_key, commit);
12535 if is_ab || is_ba {
12536 active.remove(&(et.clone(), src_key.clone(), dst_key.clone()));
12537 }
12538 }
12539 WalRecord::DeleteNode { key: k }
12540 if Self::aliases_match(&alias_a, k, commit)
12541 || Self::aliases_match(&alias_b, k, commit) =>
12542 {
12543 // All edges touching the deleted node are gone.
12544 active.retain(|(_, s, d)| s != k && d != k);
12545 }
12546 WalRecord::DerivedEdgeAdded {
12547 edge_type: et,
12548 src_key,
12549 dst_key,
12550 ..
12551 } => {
12552 let is_ab = Self::aliases_match(&alias_a, src_key, commit)
12553 && Self::aliases_match(&alias_b, dst_key, commit);
12554 let is_ba = Self::aliases_match(&alias_b, src_key, commit)
12555 && Self::aliases_match(&alias_a, dst_key, commit);
12556 if is_ab || is_ba {
12557 active.insert((et.clone(), src_key.clone(), dst_key.clone()));
12558 }
12559 }
12560 WalRecord::DerivedEdgeRetracted {
12561 edge_type: et,
12562 src_key,
12563 dst_key,
12564 ..
12565 } => {
12566 let is_ab = Self::aliases_match(&alias_a, src_key, commit)
12567 && Self::aliases_match(&alias_b, dst_key, commit);
12568 let is_ba = Self::aliases_match(&alias_b, src_key, commit)
12569 && Self::aliases_match(&alias_a, dst_key, commit);
12570 if is_ab || is_ba {
12571 active.remove(&(et.clone(), src_key.clone(), dst_key.clone()));
12572 }
12573 }
12574 _ => {}
12575 }
12576 }
12577 }
12578
12579 Ok(active.iter().any(|(et, _, _)| et == edge_type))
12580 }
12581
12582 /// Every edge incident to `key` — either endpoint — that existed at WAL
12583 /// commit `commit`, from ONE scan of the WAL.
12584 ///
12585 /// This is the bulk form of [`was_linked`](GraphDb::was_linked): answering
12586 /// "what did K's relationships look like at commit C" with one call instead
12587 /// of one [`edge_history`](GraphDb::edge_history) per candidate partner.
12588 /// The two agree edge for edge.
12589 ///
12590 /// Results are sorted by `(edge_type, src_key, dst_key)`.
12591 ///
12592 /// ## Horizon
12593 ///
12594 /// Valid commit indices are `wal_horizon_floor()..wal_total_commits()`;
12595 /// anything outside is [`GraphError::CommitOutOfRange`], exactly like
12596 /// `was_linked`. An unknown key is not an error — it simply had no edges.
12597 ///
12598 /// ## Derived edges
12599 ///
12600 /// `DerivedEdgeAdded` / `DerivedEdgeRetracted` markers carry rule
12601 /// attribution, so a rule-owned edge comes back with `derived: true` and
12602 /// `rule: Some(name)`.
12603 ///
12604 /// ## Renames
12605 ///
12606 /// `key` is matched through the same commit-bounded alias intervals
12607 /// `edge_history` uses, so querying a node's *current* key surfaces edges
12608 /// written under an earlier name. Endpoint keys in the result are reported
12609 /// under the name the node carries today, so they can be fed straight back
12610 /// into `node_info`, `explain` or another `edges_at`.
12611 ///
12612 /// ## Masks
12613 ///
12614 /// Like `edge_history` and `node_history`, this reads the WAL regardless of
12615 /// any role mask. Apply masking at the caller level.
12616 pub fn edges_at(&self, key: &str, commit: u64) -> Result<Vec<EdgeAt>> {
12617 use core_storage::wal::WalRecord;
12618
12619 let (frames, _) = self.all_frames()?;
12620 let total_commits = self.wal_horizon_floor + frames.len() as u64;
12621
12622 // Horizon floor: commits in pruned archives are unreachable.
12623 if commit < self.wal_horizon_floor || commit >= total_commits {
12624 return Err(GraphError::CommitOutOfRange {
12625 commit,
12626 total: total_commits,
12627 floor: self.wal_horizon_floor,
12628 });
12629 }
12630
12631 // Commit-bounded historical names of `key` (handles RenameNode).
12632 let alias = self.build_key_alias_intervals(&frames, key);
12633
12634 // Forward rename chain, for reporting endpoints under their current
12635 // names: old key → [(commit, new key)] in ascending commit order.
12636 // Built over the whole WAL, not just the prefix up to `commit`, because
12637 // a rename after `commit` still changes what the node is called today.
12638 let mut renames: HashMap<String, Vec<(u64, String)>> = HashMap::new();
12639 for (local_i, frame) in frames.iter().enumerate() {
12640 let c = self.wal_horizon_floor + local_i as u64;
12641 let records: &[WalRecord] = match frame {
12642 WalRecord::Batch(inner) => inner.as_slice(),
12643 single => std::slice::from_ref(single),
12644 };
12645 for rec in records {
12646 if let WalRecord::RenameNode { old_key, new_key } = rec {
12647 renames
12648 .entry(old_key.clone())
12649 .or_default()
12650 .push((c, new_key.clone()));
12651 }
12652 }
12653 }
12654
12655 // The name a node written as `k` at commit `from` carries today.
12656 // Follows the first rename at or after `from`, then keeps going. The
12657 // iteration cap bounds a rename cycle inside a single batch.
12658 let canon = |k: &str, from: u64| -> String {
12659 if renames.is_empty() {
12660 return k.to_string();
12661 }
12662 let mut cur = k.to_string();
12663 let mut at = from;
12664 for _ in 0..64 {
12665 match renames
12666 .get(&cur)
12667 .and_then(|v| v.iter().find(|(c, _)| *c >= at))
12668 {
12669 Some((c, new)) => {
12670 at = *c;
12671 cur = new.clone();
12672 }
12673 None => break,
12674 }
12675 }
12676 cur
12677 };
12678
12679 let local_commit = commit - self.wal_horizon_floor;
12680 // (edge_type, src_key, dst_key) → (derived, rule)
12681 let mut active: BTreeMap<(String, String, String), (bool, Option<String>)> =
12682 BTreeMap::new();
12683
12684 for (local_i, frame) in frames.iter().enumerate().take((local_commit + 1) as usize) {
12685 let c = self.wal_horizon_floor + local_i as u64;
12686 let records: &[WalRecord] = match frame {
12687 WalRecord::Batch(inner) => inner.as_slice(),
12688 single => std::slice::from_ref(single),
12689 };
12690
12691 for rec in records {
12692 match rec {
12693 WalRecord::InsertEdge {
12694 edge_type,
12695 src_key,
12696 dst_key,
12697 } => {
12698 if Self::aliases_match(&alias, src_key, c)
12699 || Self::aliases_match(&alias, dst_key, c)
12700 {
12701 active.insert(
12702 (edge_type.clone(), canon(src_key, c), canon(dst_key, c)),
12703 (false, None),
12704 );
12705 }
12706 }
12707 WalRecord::InsertEdgeId { etype, src, dst } => {
12708 let Some(etype_str) = self.syms.resolve(*etype) else {
12709 continue;
12710 };
12711 // `key_of_historical` resolves tombstoned ids too, and
12712 // already returns the node's current key — no rename
12713 // canonicalisation needed on this arm.
12714 let (Some(src_key), Some(dst_key)) = (
12715 self.ids.key_of_historical(*src),
12716 self.ids.key_of_historical(*dst),
12717 ) else {
12718 continue;
12719 };
12720 if src_key == key || dst_key == key {
12721 active.insert(
12722 (
12723 etype_str.to_string(),
12724 src_key.to_string(),
12725 dst_key.to_string(),
12726 ),
12727 (false, None),
12728 );
12729 }
12730 }
12731 WalRecord::DeleteEdge {
12732 edge_type,
12733 src_key,
12734 dst_key,
12735 } => {
12736 if Self::aliases_match(&alias, src_key, c)
12737 || Self::aliases_match(&alias, dst_key, c)
12738 {
12739 active.remove(&(
12740 edge_type.clone(),
12741 canon(src_key, c),
12742 canon(dst_key, c),
12743 ));
12744 }
12745 }
12746 WalRecord::DeleteNode { key: k } => {
12747 if active.is_empty() {
12748 continue;
12749 }
12750 if Self::aliases_match(&alias, k, c) {
12751 // Our node is gone; every incident edge goes with it.
12752 active.clear();
12753 } else {
12754 // A partner is gone; its edges to us go with it.
12755 let ck = canon(k, c);
12756 active.retain(|(_, s, d), _| *s != ck && *d != ck);
12757 }
12758 }
12759 WalRecord::DerivedEdgeAdded {
12760 rule,
12761 edge_type,
12762 src_key,
12763 dst_key,
12764 } => {
12765 if Self::aliases_match(&alias, src_key, c)
12766 || Self::aliases_match(&alias, dst_key, c)
12767 {
12768 active.insert(
12769 (edge_type.clone(), canon(src_key, c), canon(dst_key, c)),
12770 (true, Some(rule.clone())),
12771 );
12772 }
12773 }
12774 WalRecord::DerivedEdgeRetracted {
12775 edge_type,
12776 src_key,
12777 dst_key,
12778 ..
12779 } => {
12780 if Self::aliases_match(&alias, src_key, c)
12781 || Self::aliases_match(&alias, dst_key, c)
12782 {
12783 active.remove(&(
12784 edge_type.clone(),
12785 canon(src_key, c),
12786 canon(dst_key, c),
12787 ));
12788 }
12789 }
12790 // InsertNode, SetProp, CreateRule, … do not move edges.
12791 _ => {}
12792 }
12793 }
12794 }
12795
12796 // BTreeMap iteration is already (edge_type, src, dst) order.
12797 Ok(active
12798 .into_iter()
12799 .map(|((edge_type, src_key, dst_key), (derived, rule))| EdgeAt {
12800 edge_type,
12801 src_key,
12802 dst_key,
12803 derived,
12804 rule,
12805 })
12806 .collect())
12807 }
12808
12809 /// The derived edges that would be retracted and derived if `key.field`
12810 /// were set to `value` — computed WITHOUT writing anything.
12811 ///
12812 /// Nothing is committed and nothing on `self` is mutated: the rule engine's
12813 /// provenance, its candidate indexes, the topology and the property columns
12814 /// are all cloned first, the change is applied to the clone, and the real
12815 /// per-node re-derivation (`RuleEngine::on_node_changed` — the same call
12816 /// `set_prop` makes during apply) runs against it. The derived-edge deltas
12817 /// it emits are the answer, so rule semantics — predicates, top-k,
12818 /// via-hops, chaining, weights — are the engine's, not a re-implementation.
12819 ///
12820 /// Works on a read-only handle.
12821 ///
12822 /// **While a rule's vector index is still building** (`RuleStats::building`)
12823 /// the clone carries no pending-build state, so this reports the edges that
12824 /// rule would derive — which the live store will not derive until its
12825 /// backfill runs. Right about the end state, early about the timing.
12826 ///
12827 /// Returns `Err(KeyNotFound)` for an unknown or tombstoned key and
12828 /// `Err(ViewPropReadOnly)` for a field a view owns — matching
12829 /// [`set_prop`](GraphDb::set_prop)'s validation. A change with no effect
12830 /// (the node already holds `value`, or no rule watches `field`) returns
12831 /// empty lists.
12832 ///
12833 /// ## Cost
12834 ///
12835 /// One clone of the property columns, the topology overlay, the symbol
12836 /// interner, the edge properties and the provenance map, plus one candidate
12837 /// re-index (O(nodes × rules)). That is much cheaper than copying the store
12838 /// directory, but it is not free — this is an interactive "what if", not a
12839 /// hot path.
12840 pub fn what_if_set_prop(&self, key: &str, field: &str, value: Value) -> Result<WhatIf> {
12841 // The engine's provenance, HNSW and IVF state live in the mmap'd base
12842 // until something asks for them. On a store opened cold from a snapshot
12843 // this is the first ask, and without it the clone below starts from an
12844 // empty provenance map: nothing to retract, so `lost` comes back empty.
12845 self.ensure_v8_base_sections_loaded();
12846
12847 let empty = WhatIf {
12848 lost: Vec::new(),
12849 gained: Vec::new(),
12850 };
12851
12852 if let Some(view_name) = self.view_store.view_for_prop(field) {
12853 return Err(GraphError::ViewPropReadOnly {
12854 view_name: view_name.to_string(),
12855 });
12856 }
12857 MutPreview::new(self).check_live_key(key)?;
12858 let id = self
12859 .ids
12860 .get(key)
12861 .ok_or_else(|| GraphError::KeyNotFound { key: key.into() })?;
12862
12863 let rules: Vec<RuleDef> = self.engine.rules().cloned().collect();
12864 if rules.is_empty() {
12865 return Ok(empty);
12866 }
12867
12868 // No rule watches this field → no derivation can change.
12869 if !rules.iter().any(|r| r.watched_fields().contains(field)) {
12870 return Ok(empty);
12871 }
12872
12873 let old_value = build_props_view(&self.props, &self.base)
12874 .get(id, field)
12875 .map(|vr| vr.into_value());
12876 if old_value.as_ref() == Some(&value) {
12877 return Ok(empty);
12878 }
12879
12880 // --- Clone every piece of state the re-derivation writes to. ---
12881 let mut props = self.props.clone();
12882 let mut topo = self.topo.clone();
12883 let mut syms = self.syms.clone();
12884 let mut edge_props = self.edge_props.clone();
12885
12886 let mut tripped: BTreeMap<String, bool> = BTreeMap::new();
12887 let mut fires: BTreeMap<String, u64> = BTreeMap::new();
12888 for r in &rules {
12889 tripped.insert(r.name.clone(), self.engine.is_tripped(&r.name));
12890 fires.insert(r.name.clone(), self.engine.fire_count(&r.name));
12891 }
12892 // `provenance()` decodes retained snapshot bytes on first use; the
12893 // engine clone needs the real map, not an empty one.
12894 let provenance = self.engine.provenance().clone();
12895 let mut engine = core_rules::RuleEngine::from_persist(rules, provenance, tripped, fires);
12896
12897 // Build the candidate indexes from the state BEFORE the change, exactly
12898 // as apply() sees them: `on_node_changed` withdraws the node under its
12899 // old value and refiles it under the new one, so the index must not
12900 // already reflect the change.
12901 engine.reindex_all_load_state(
12902 &self.ids,
12903 &syms,
12904 &self.labels,
12905 build_props_view(&self.props, &self.base),
12906 self.engine.export_ivf_state(),
12907 self.engine.export_hnsw_state_passthrough(),
12908 );
12909 engine.set_emit_deltas(true);
12910
12911 // --- Apply the hypothetical change and re-derive. ---
12912 props.set(id, field, value);
12913 {
12914 let mut gm = make_graph_mut(
12915 &self.ids,
12916 &mut syms,
12917 &self.labels,
12918 build_props_view(&props, &self.base),
12919 &mut topo,
12920 &self.base,
12921 &mut edge_props,
12922 );
12923 engine.on_node_changed(id, Some((field, old_value)), &mut gm);
12924 }
12925
12926 let mut lost: BTreeSet<EdgeAt> = BTreeSet::new();
12927 let mut gained: BTreeSet<EdgeAt> = BTreeSet::new();
12928 for d in engine.drain_deltas() {
12929 let edge = EdgeAt {
12930 edge_type: d.edge_type,
12931 src_key: d.src_key,
12932 dst_key: d.dst_key,
12933 derived: true,
12934 rule: Some(d.rule),
12935 };
12936 if d.fired {
12937 gained.insert(edge);
12938 } else {
12939 lost.insert(edge);
12940 }
12941 }
12942 // An edge retracted and re-derived within the same re-derivation (top-k
12943 // churn) is not a change the caller would see.
12944 let churn: Vec<EdgeAt> = lost.intersection(&gained).cloned().collect();
12945 for e in churn {
12946 lost.remove(&e);
12947 gained.remove(&e);
12948 }
12949
12950 Ok(WhatIf {
12951 lost: lost.into_iter().collect(),
12952 gained: gained.into_iter().collect(),
12953 })
12954 }
12955
12956 pub fn edge_count(&self) -> u64 {
12957 self.topo_view().edge_count()
12958 }
12959
12960 /// Live/tombstone/edge counts plus per-rule provenance size, trip latch,
12961 /// and fire counter (includes rebuild evaluations). Rules are sorted by name.
12962 pub fn stats(&self) -> Stats {
12963 self.ensure_v8_base_sections_loaded();
12964 let building = self.engine.builds_in_progress();
12965 let rules: Vec<RuleStats> = self
12966 .engine
12967 .rules()
12968 .map(|r| RuleStats {
12969 name: r.name.clone(),
12970 edges: self
12971 .engine
12972 .provenance()
12973 .get(&r.name)
12974 .map(|s| s.len() as u64)
12975 .unwrap_or(0),
12976 tripped: self.engine.is_tripped(&r.name),
12977 fires: self.engine.fire_count(&r.name),
12978 approximate: r.approximate,
12979 building: building.iter().find(|b| b.rule == r.name).cloned(),
12980 })
12981 .collect();
12982 Stats {
12983 nodes_live: self.ids.live_len(),
12984 nodes_tombstoned: self.ids.len() - self.ids.live_len(),
12985 edges: self.topo_view().edge_count(),
12986 rules,
12987 chain_truncations: self.engine.chain_truncations(),
12988 history_floor: self.wal_horizon_floor,
12989 namespaces: self.namespace_stats(),
12990 }
12991 }
12992
12993 /// On-disk size of the WAL file in bytes.
12994 ///
12995 /// Reads file metadata without loading WAL contents. Returns `Err` for
12996 /// in-memory (`SimFs`) databases where no WAL file exists on disk.
12997 pub fn wal_size_bytes(&self) -> std::io::Result<u64> {
12998 let path = self.fs.wal_path().ok_or_else(|| {
12999 std::io::Error::new(
13000 std::io::ErrorKind::Unsupported,
13001 "wal_path not available for this Fs implementation",
13002 )
13003 })?;
13004 Ok(std::fs::metadata(path)?.len())
13005 }
13006
13007 /// Set the slow-query threshold. Queries whose execution time equals or
13008 /// exceeds `ms` milliseconds are logged. Pass `0` to disable.
13009 ///
13010 /// Use this setter in tests — the environment variable
13011 /// `MUSHROOMDB_SLOW_QUERY_MS` is process-global and races parallel test
13012 /// threads.
13013 pub fn set_slow_query_threshold_ms(&mut self, ms: u64) {
13014 self.slow_query_threshold_ms = ms;
13015 }
13016
13017 /// Snapshot of the slow-query ring buffer and lifetime counter.
13018 pub fn slow_query_snapshot(&self) -> SlowQuerySnapshot {
13019 let log = self.slow_queries.lock().unwrap_or_else(|e| e.into_inner());
13020 SlowQuerySnapshot {
13021 threshold_ms: self.slow_query_threshold_ms,
13022 count: log.total,
13023 last: log.entries.iter().cloned().collect(),
13024 }
13025 }
13026
13027 /// Instant the database was opened. Used by consumers (e.g. `/metrics`)
13028 /// to compute uptime.
13029 pub fn started_at(&self) -> std::time::Instant {
13030 self.started_at
13031 }
13032
13033 /// The on-disk snapshot version a store that has opted in to nothing
13034 /// writes — the **floor**, not the whole answer.
13035 ///
13036 /// It is not "the version this binary writes", and it is not "the version
13037 /// this binary reads". Since v0.6.10 this binary writes 9 **or** 10
13038 /// depending on the store — [`snapshot::version_for`] decides, and a store
13039 /// that has called [`enable_multiplicity`](Self::enable_multiplicity)
13040 /// writes 10 — and it reads 5 through 10. A caller comparing a store's
13041 /// stamp against this value must use `>=`, not `==`, or it will report an
13042 /// opted-in store as needing a migration *down*; `cli::run_migrate` is the
13043 /// worked example.
13044 ///
13045 /// The name is kept for compatibility: it is public API reachable from the
13046 /// CLI and from any embedder, and respelling it would break them for a
13047 /// doc-level clarification.
13048 ///
13049 /// [`snapshot::version_for`]: core_storage::snapshot::version_for
13050 pub fn format_version() -> u16 {
13051 core_storage::snapshot::VERSION
13052 }
13053
13054 /// Test-support: total bytes appended (SimFs only usage).
13055 pub fn fs_total_appended(&self) -> usize
13056 where
13057 F: FsIntrospect,
13058 {
13059 self.fs.total_appended()
13060 }
13061
13062 /// Test-support: successful `Fs::sync` calls (SimFs / counting fs).
13063 pub fn fs_sync_count(&self) -> usize
13064 where
13065 F: FsIntrospect,
13066 {
13067 self.fs.sync_count()
13068 }
13069
13070 /// Consume the db, returning its fs (for crash simulation).
13071 pub fn into_fs(self) -> F {
13072 self.fs
13073 }
13074
13075 pub fn snapshot(&mut self) -> Result<()> {
13076 self.snapshot_with(SnapshotOptions::default())
13077 }
13078
13079 /// Snapshot with explicit options.
13080 ///
13081 /// # `keep_wal`
13082 ///
13083 /// When `keep_wal` is `false` (the default, same as [`snapshot`]):
13084 /// - The WAL is replaced with a minimal baseline containing one
13085 /// `EnableFulltext` record per active declaration. All pre-snapshot
13086 /// history is discarded; `open_at` can only reach post-snapshot commits.
13087 ///
13088 /// When `keep_wal` is `true`:
13089 /// - The WAL is left intact. All pre-snapshot commits remain reachable
13090 /// via `open_at`. The existing WAL already contains the original
13091 /// `EnableFulltext` records, so no baseline re-write is needed; the
13092 /// recovery guards in `apply()` silently skip any duplicate records on
13093 /// replay.
13094 /// - Crash window: a crash after the snapshot write but before the next
13095 /// WAL write leaves the full pre-snapshot WAL intact. On reopen the
13096 /// snapshot is loaded and the WAL replayed idempotently over it — safe
13097 /// because every `apply()` arm is idempotent when replayed over an
13098 /// already-current snapshot.
13099 pub fn snapshot_with(&mut self, opts: SnapshotOptions) -> Result<()> {
13100 if self.read_only {
13101 return Err(GraphError::ReadOnly);
13102 }
13103 // A snapshot rewrites `wal.bin` through a tmp+rename, so a peer that is
13104 // appending ends up holding a descriptor on an unlinked inode and loses
13105 // commits it believes durable. Snapshotting therefore requires the
13106 // cross-process write lock, exactly as appending does. Unlike the WAL
13107 // append path this does not go through `log_then_apply_with`, so both
13108 // guards are repeated here.
13109 if self.degraded {
13110 return Err(GraphError::Io(std::io::Error::other(
13111 "database degraded after group-commit fsync failure; reopen required",
13112 )));
13113 }
13114 if self.lock_denied {
13115 return Err(GraphError::Busy { holder: None });
13116 }
13117 // Capture whether snapshot.bin already existed BEFORE this snapshot write.
13118 // Used by the archive path's conservative genesis-chain check: if a prior
13119 // snapshot exists but wal.truncated does not, we cannot distinguish a
13120 // legacy store (may have been truncated in an older code version) from a
13121 // new store that only used keep_wal=true. Conservative: refuse genesis in
13122 // both cases. Must be sampled here, before the snapshot write below.
13123 //
13124 // `snapshot_preserved_history` is the one case where the answer is not a
13125 // guess: a snapshot *this handle* took, on a store that had none when it
13126 // opened, and that kept the WAL. The proxy defers to it, because
13127 // otherwise `enable_multiplicity` — whose forced snapshot is exactly
13128 // that — would permanently disqualify the store from a genesis chain it
13129 // is fully entitled to (defect #23).
13130 let had_prior_snapshot = self.fs.snapshot_path().map(|p| p.exists()).unwrap_or(false)
13131 && !self.snapshot_preserved_history;
13132 // Which version this store writes. V9 unless it has opted in to
13133 // multiplicity, in which case V10 — the stamp that makes a reader which
13134 // does not know WAL discriminant 23 refuse the open instead of
13135 // truncating the WAL at the first such frame. The container is
13136 // identical either way; only these two header bytes move.
13137 let snapshot_version = core_storage::snapshot::version_for(self.multiplicity);
13138 self.ensure_v8_base_sections_loaded();
13139 // Ensure provenance is decoded before to_persist() clones it.
13140 self.engine.ensure_provenance_loaded_mut();
13141 let (rule_defs_typed, provenance, rule_tripped, rule_fires) = self.engine.to_persist();
13142 let rule_defs = rule_defs_typed
13143 .iter()
13144 .map(|r| bincode::serialize(r).expect("RuleDef serialize cannot fail"))
13145 .collect();
13146 // Collect HNSW state and IVF state. When indexes are not yet
13147 // populated (clean open, no mutation since open), pass the retained
13148 // raw bytes through directly so that migrate/snapshot does not
13149 // silently discard fitted approximate-rule indexes.
13150 let hnsw_state = self.engine.export_hnsw_state_passthrough();
13151 let ivf_bytes = if !self.engine.indexes_populated() {
13152 // Pass retained IVF bytes through unchanged (no re-encode).
13153 self.engine.retained_ivf_bytes_clone().unwrap_or_default()
13154 } else {
13155 // Indexes live: encode from current state.
13156 let raw_ivf = self.engine.export_ivf_state();
13157 let ivf_state_map: BTreeMap<String, core_storage::snapshot::PerRuleIvfState> = raw_ivf
13158 .into_iter()
13159 .map(|(name, ((sc, sa, sd), (dc, da, dd)))| {
13160 (
13161 name,
13162 core_storage::snapshot::PerRuleIvfState {
13163 src: core_storage::snapshot::SideIvfState {
13164 centroids: sc,
13165 clusters: sa,
13166 drift: sd,
13167 },
13168 dst: core_storage::snapshot::SideIvfState {
13169 centroids: dc,
13170 clusters: da,
13171 drift: dd,
13172 },
13173 },
13174 )
13175 })
13176 .collect();
13177 if ivf_state_map.is_empty() {
13178 Vec::new()
13179 } else {
13180 bincode::serialize(&ivf_state_map).expect("IVF state serialize cannot fail")
13181 }
13182 };
13183 let view_defs: Vec<Vec<u8>> = self
13184 .view_store
13185 .views()
13186 .map(|v| bincode::serialize(v).expect("ViewDef serialize cannot fail"))
13187 .collect();
13188 if self.base.is_some() {
13189 // V8 merge-snapshot path: encode base+overlay into a new V8 snapshot,
13190 // write it atomically, remap it as the new base, then clear the overlay.
13191 let meta = V8Meta {
13192 labels: self.labels.clone(),
13193 edge_props: self.edge_props.clone(),
13194 rule_defs,
13195 provenance,
13196 rule_tripped,
13197 rule_fires,
13198 ivf_bytes,
13199 view_defs,
13200 wal_truncated: !opts.keep_wal,
13201 hnsw: hnsw_state,
13202 last_change: self.last_change.clone(),
13203 };
13204 let mut buf: Vec<u8> = Vec::new();
13205 {
13206 // Clone the Arc so the old base stays alive while we encode.
13207 // The borrow of archived_csr (into old_base's mmap) is released
13208 // at the end of this block, before we replace self.base.
13209 let old_base = self.base.clone().expect("is_some checked above");
13210 let archived_csr = old_base.topology().map_err(|e| GraphError::Corrupt {
13211 detail: format!("v8 snapshot: topology section: {e:?}"),
13212 })?;
13213 let archived_cols = old_base.columns().map_err(|e| GraphError::Corrupt {
13214 detail: format!("v8 snapshot: columns section: {e:?}"),
13215 })?;
13216 // `None` when the base predates V9 — the migration path: its
13217 // string columns still carry their own tables and this snapshot
13218 // is the rewrite that collapses them into section 12.
13219 let archived_strings =
13220 old_base
13221 .string_table()
13222 .transpose()
13223 .map_err(|e| GraphError::Corrupt {
13224 detail: format!("v8 snapshot: strings section: {e:?}"),
13225 })?;
13226 let archived_edge_props =
13227 old_base
13228 .edge_props_section()
13229 .map_err(|e| GraphError::Corrupt {
13230 detail: format!("v8 snapshot: edge_props section: {e:?}"),
13231 })?;
13232 let edge_props_raw =
13233 old_base
13234 .edge_props_raw_bytes()
13235 .map_err(|e| GraphError::Corrupt {
13236 detail: format!("v8 snapshot: edge_props raw bytes: {e:?}"),
13237 })?;
13238 let prov_raw =
13239 old_base
13240 .provenance_raw_bytes()
13241 .map_err(|e| GraphError::Corrupt {
13242 detail: format!("v8 snapshot: provenance raw bytes: {e:?}"),
13243 })?;
13244 encode_v8(
13245 Some(archived_csr),
13246 Some(archived_cols),
13247 archived_strings,
13248 Some((archived_edge_props, edge_props_raw)),
13249 Some(prov_raw),
13250 &self.topo,
13251 &self.props,
13252 &self.ids,
13253 &self.syms,
13254 &meta,
13255 &mut buf,
13256 )?;
13257 }
13258 core_storage::snapshot::stamp_container_version(&mut buf, snapshot_version)?;
13259 self.fs.write_atomic(FileId::Snapshot, &buf)?;
13260 // Remap the freshly-written snapshot as the new base.
13261 // C2: use file mmap on RealFs; fall back to from_bytes on SimFs.
13262 let new_base = if let Some(snap_path) = self.fs.snapshot_path() {
13263 core_storage::v8::MappedBase::map(&snap_path)
13264 } else {
13265 core_storage::v8::MappedBase::from_bytes(buf)
13266 }
13267 .map_err(|e| GraphError::Corrupt {
13268 detail: format!("v8 snapshot: remap new base: {e:?}"),
13269 })?;
13270 self.base = Some(Arc::new(new_base));
13271 // Clear the overlay and prop tombstones — all data is now in the new base.
13272 self.topo = Topology::new();
13273 self.props = core_storage::columns::ColumnStore::new();
13274 } else {
13275 // Legacy path (V5–V7 stores without a V8 base).
13276 //
13277 // Memory-diet path: build V8Meta directly from &self — no SnapshotState
13278 // clone and no encode_v8_from_state intermediate clones. The big
13279 // structures (self.topo, self.props) are borrowed, not cloned.
13280 // self.edge_props is moved (not cloned) because we immediately clear it
13281 // when we remap the new V8 snapshot as self.base (see below).
13282 //
13283 // Eliminates from peak RSS vs. the old SnapshotState path:
13284 // • self.topo.clone() (~topology HashMap footprint)
13285 // • self.props.clone() (~column-store footprint)
13286 // • encode_v8_from_state V8Meta secondary clones (labels, edge_props, …)
13287 let meta = V8Meta {
13288 labels: self.labels.clone(),
13289 wal_truncated: !opts.keep_wal,
13290 // Move edge_props out so the large overlay is freed when meta
13291 // drops at end of this block (self.edge_props is now empty; reads
13292 // after base assignment go through the mmap'd base section).
13293 edge_props: std::mem::take(&mut self.edge_props),
13294 rule_defs,
13295 provenance,
13296 rule_tripped,
13297 rule_fires,
13298 ivf_bytes,
13299 view_defs,
13300 hnsw: hnsw_state,
13301 last_change: self.last_change.clone(),
13302 };
13303 let mut buf = Vec::new();
13304 encode_v8(
13305 None,
13306 None,
13307 None,
13308 None,
13309 None,
13310 &self.topo,
13311 &self.props,
13312 &self.ids,
13313 &self.syms,
13314 &meta,
13315 &mut buf,
13316 )?;
13317 // meta (and the moved edge_props inside it) is no longer needed;
13318 // drop it before the write to keep the peak window narrow.
13319 drop(meta);
13320 core_storage::snapshot::stamp_container_version(&mut buf, snapshot_version)?;
13321 self.fs.write_atomic(FileId::Snapshot, &buf)?;
13322 // Remap the freshly-written V8 snapshot as self.base.
13323 // On RealFs: drop the encode buffer before mmap to recover ~1.9 GiB.
13324 // On SimFs (tests): pass buf to from_bytes.
13325 let new_base = if let Some(snap_path) = self.fs.snapshot_path() {
13326 drop(buf);
13327 core_storage::v8::MappedBase::map(&snap_path)
13328 } else {
13329 core_storage::v8::MappedBase::from_bytes(buf)
13330 }
13331 .map_err(|e| GraphError::Corrupt {
13332 detail: format!("v8 snapshot: remap new base (legacy path): {e:?}"),
13333 })?;
13334 self.base = Some(Arc::new(new_base));
13335 // Free the large heap-allocated decoded state — all data is now in the
13336 // mmap'd base. Mirrors the V8 merge-snapshot path (see above).
13337 // self.edge_props was already moved into meta and is effectively empty.
13338 self.topo = Topology::new();
13339 self.props = core_storage::columns::ColumnStore::new();
13340 }
13341
13342 if opts.archive_wal {
13343 // History-preserving snapshot (Task 4):
13344 // 1. Snapshot already written above (write_atomic → fsynced).
13345 // 2. Rename WAL → wal.<commit_seq>.archive (atomic, same fs).
13346 // Crash window B: crash here leaves archive present, WAL
13347 // absent. Reopen: snapshot loaded (full state), no WAL
13348 // replay. Archive is NOT replayed into live state — it is
13349 // pre-snapshot by construction. Safe.
13350 // 3. Optionally write genesis marker (first archive only, no
13351 // prior WAL truncation).
13352 // 4. Prune old archives (retention), update horizon floor.
13353 // Pruning invalidates the genesis chain; delete marker.
13354 // 5. Write new minimal baseline WAL (write_atomic).
13355 // Crash window C: crash here leaves new archive plus no live
13356 // WAL. Same as window B — handled above.
13357 //
13358 // Sample existing archives BEFORE the rename so we can detect
13359 // whether this is the first archive.
13360 let existing_archives = self.fs.list_archives()?;
13361 let is_first_archive = existing_archives.is_empty();
13362
13363 // Compute a globally-monotonic archive name: the name equals the
13364 // cumulative end-frame index of the archive in global commit space.
13365 //
13366 // Using `commit_seq` directly is UNSOUND across sessions: on reopen
13367 // commit_seq is seeded from max(last_change), which underestimates
13368 // the WAL depth when trailing commits (e.g. insert_edge) do not
13369 // update last_change. A session-2 archive could then receive a name
13370 // ≤ the session-1 archive, causing incorrect sort order or collision.
13371 //
13372 // Instead: read and decode the live WAL here (before the rename) to
13373 // get its exact frame count, then add it to the last known global
13374 // end-frame index (the name of the most recent existing archive, or
13375 // wal_horizon_floor if no archives exist). This is O(WAL size) but
13376 // snapshot is already serialising the full graph state, so the cost
13377 // is dominated.
13378 let live_wal_bytes_for_name = self.fs.read(FileId::Wal)?;
13379 let (live_frames_for_name, _) = decode_all(&live_wal_bytes_for_name);
13380 let archive_n = existing_archives
13381 .last()
13382 .copied()
13383 .unwrap_or(self.wal_horizon_floor)
13384 + live_frames_for_name.len() as u64;
13385 self.fs.archive_wal(archive_n)?;
13386
13387 // The replacement WAL goes in **immediately**, with no fallible call
13388 // between it and the rename above.
13389 //
13390 // The rename is what removes the store's live declarations — the
13391 // multiplicity opt-in, and every `EnableFulltext` / `EnableIndex` —
13392 // and this write is what puts them back. Every call that used to sit
13393 // in between (the genesis marker, the retention sweep's reads, the
13394 // floor write, the archive deletes) was a `?` that could leave the
13395 // store with neither, so a single transient `Err` was enough to lose
13396 // a declaration that no rebuild can recover (defect #22).
13397 //
13398 // Ordering alone cannot close the crash window between two
13399 // filesystem calls; for the multiplicity declaration the V10 stamp
13400 // does that on the open path. What ordering does close is the much
13401 // wider window in which an ordinary I/O error did it — and that half
13402 // covers all three declarations, not just the one with a stamp.
13403 let mut baseline_wal: Vec<u8> = Vec::new();
13404 // The multiplicity opt-in is a declaration like the two below it,
13405 // and it is re-emitted for the same reason: truncation must not
13406 // silently opt the store back out and stop counting.
13407 if self.multiplicity {
13408 baseline_wal
13409 .extend_from_slice(&encode_record(&core_storage::wal::MULTIPLICITY_ENABLED));
13410 }
13411 for (label, field) in self.fulltext.enabled_pairs() {
13412 let rec = WalRecord::EnableFulltext {
13413 label: label.clone(),
13414 field: field.clone(),
13415 };
13416 baseline_wal.extend_from_slice(&encode_record(&rec));
13417 }
13418 for (label, field) in self.prop_index.enabled_pairs() {
13419 let rec = WalRecord::EnableIndex {
13420 label: label.clone(),
13421 field: field.clone(),
13422 };
13423 baseline_wal.extend_from_slice(&encode_record(&rec));
13424 }
13425 self.fs.write_atomic(FileId::Wal, &baseline_wal)?;
13426
13427 // Genesis marker: written once when the first archive is taken
13428 // from a store that has never undergone a WAL-truncating snapshot.
13429 // When present, `open_at` may replay archive-resident commits from
13430 // empty state (the archive chain covers from global index 0).
13431 //
13432 // Two conditions must ALL hold:
13433 // 1. This is the first archive (existing_archives was empty).
13434 // 2. No snapshot.bin existed before this operation (had_prior_snapshot=false).
13435 // A WAL-truncating snapshot (keep_wal=false) always writes snapshot.bin
13436 // before truncating the WAL, so if any prior truncating snapshot was taken
13437 // — even in a previous session — snapshot.bin is present and this condition
13438 // is false. This subsumes the cross-session truncation case without
13439 // requiring a separate wal.truncated sidecar file.
13440 // For legacy stores (snapshot.bin written by an older code version that
13441 // may have truncated the WAL), the same conservative refusal applies:
13442 // we cannot prove the chain is complete, so we refuse genesis (cost =
13443 // no as-of-through-archives; never silent wrong data).
13444 // The one exception is a snapshot this handle took itself, on a store
13445 // that had none when it opened, with the WAL kept: there the answer is
13446 // known rather than guessed, and `snapshot_preserved_history` says so.
13447 // Without that exception `enable_multiplicity`'s forced keep_wal
13448 // snapshot would disqualify the store forever (defect #23).
13449 // On SimFs (snapshot_path() == None) had_prior_snapshot is always false,
13450 // so SimFs always passes this check.
13451 if is_first_archive && !had_prior_snapshot {
13452 self.fs.write_genesis_marker()?;
13453 self.archive_genesis_chain = true;
13454 }
13455
13456 // Retention pruning: keep newest `keep` archives; delete oldest.
13457 // Pruning is the ONLY deletion site for archives.
13458 //
13459 // Crash-safety ordering (C1 fix):
13460 // 1. Count frames in surplus archives (reads only — no mutation).
13461 // 2. Advance and PERSIST the horizon floor FIRST via write-then-
13462 // rename (atomic). A crash after this point leaves orphaned
13463 // archives on disk, but the floor is correct. The opening
13464 // cleanup sweep (`cleanup_orphaned_archives`) removes them on
13465 // the next open, so the store is always safe to reopen.
13466 // 3. Delete the genesis marker (floor > 0 already blocks open_at
13467 // via the conjunctive gate; marker cleanup is belt-and-suspenders).
13468 // 4. Delete surplus archives. A crash between any two deletes
13469 // leaves the floor committed and orphaned archives cleaned at
13470 // next open — never a stale floor with a missing archive prefix.
13471 if let Some(keep) = self.wal_archive_retention {
13472 if keep > 0 {
13473 let archives = self.fs.list_archives()?;
13474 // archives is sorted ascending (oldest first)
13475 if archives.len() as u32 > keep {
13476 let surplus = archives.len() - keep as usize;
13477 // Step 1: count pruned frames (reads, no mutation).
13478 let mut pruned_frames = 0u64;
13479 for &n in &archives[..surplus] {
13480 let bytes = self.fs.read_archive(n)?;
13481 let (frames, _) = decode_all(&bytes);
13482 pruned_frames += frames.len() as u64;
13483 }
13484 // Step 2: advance and persist floor FIRST.
13485 self.wal_horizon_floor += pruned_frames;
13486 self.fs.write_horizon_floor(self.wal_horizon_floor)?;
13487 // Step 3: delete genesis marker (floor > 0 already
13488 // blocks open_at; this is belt-and-suspenders cleanup).
13489 if pruned_frames > 0 && self.archive_genesis_chain {
13490 self.fs.delete_genesis_marker()?;
13491 self.archive_genesis_chain = false;
13492 }
13493 // Step 4: delete surplus archives. Crash here →
13494 // orphaned archives; cleaned at next open.
13495 for &n in &archives[..surplus] {
13496 self.fs.delete_archive(n)?;
13497 }
13498 }
13499 }
13500 }
13501 } else if opts.keep_wal {
13502 // keep_wal=true: WAL is left untouched. The existing WAL already
13503 // contains the EnableFulltext records from the original enable calls;
13504 // replay is idempotent (guards in apply() skip already-live entries).
13505 // No baseline re-write is needed or safe here — the full WAL history
13506 // must remain intact for open_at to reach pre-snapshot commits.
13507 } else {
13508 // keep_wal=false (default): truncate by replacing the WAL with a
13509 // minimal baseline of one EnableFulltext record per active pair.
13510 //
13511 // Crash-ordering: write_atomic is atomic.
13512 // • Crash before snapshot write → WAL unchanged. Safe.
13513 // • Crash after snapshot write but before this WAL write → full
13514 // pre-snapshot WAL still present; open_with replays idempotently.
13515 // • Crash after both writes → normal post-snapshot state.
13516 //
13517 // Genesis chain: a WAL-truncating snapshot breaks the archive chain
13518 // for any archives taken AFTER this point (their WAL slices would
13519 // not start at genesis). Delete any existing genesis marker so that
13520 // open_at refuses archive-resident commits. Future sessions are
13521 // covered by had_prior_snapshot: snapshot.bin written here persists
13522 // across sessions and prevents a later archiving session from
13523 // incorrectly claiming a complete genesis chain.
13524 if self.archive_genesis_chain {
13525 self.fs.delete_genesis_marker()?;
13526 self.archive_genesis_chain = false;
13527 }
13528 // And this handle can no longer prove the WAL is whole: it is about
13529 // to truncate it itself. Same-session archives after this point get
13530 // the conservative answer, exactly as cross-session ones do.
13531 self.snapshot_preserved_history = false;
13532 let mut baseline_wal: Vec<u8> = Vec::new();
13533 // The multiplicity opt-in is a declaration like the two below it,
13534 // and it is re-emitted for the same reason: truncation must not
13535 // silently opt the store back out and stop counting.
13536 if self.multiplicity {
13537 baseline_wal
13538 .extend_from_slice(&encode_record(&core_storage::wal::MULTIPLICITY_ENABLED));
13539 }
13540 for (label, field) in self.fulltext.enabled_pairs() {
13541 let rec = WalRecord::EnableFulltext {
13542 label: label.clone(),
13543 field: field.clone(),
13544 };
13545 baseline_wal.extend_from_slice(&encode_record(&rec));
13546 }
13547 for (label, field) in self.prop_index.enabled_pairs() {
13548 let rec = WalRecord::EnableIndex {
13549 label: label.clone(),
13550 field: field.clone(),
13551 };
13552 baseline_wal.extend_from_slice(&encode_record(&rec));
13553 }
13554 self.fs.write_atomic(FileId::Wal, &baseline_wal)?;
13555 }
13556 // After snapshot the overlay may have changed (V8 merge path clears
13557 // self.topo and self.props). Refresh the MVCC fold so future readers
13558 // see the post-snapshot state rather than stale overlay data.
13559 self.fold_now();
13560 // We wrote the snapshot and (unless keep_wal) replaced the WAL, so both
13561 // markers this handle uses to detect other processes' work must be
13562 // re-taken from disk. Skipping this would make our own snapshot look
13563 // like a peer's on the next staleness check and force a needless
13564 // reload.
13565 self.wal_consumed = self.fs.wal_len().map_err(GraphError::Io)?;
13566 self.snapshot_ident = self.fs.snapshot_ident().map_err(GraphError::Io)?;
13567 Ok(())
13568 }
13569}
13570
13571/// What a batch node insert does when its key is already taken.
13572///
13573/// A mirror rebuild writes a frame onto a store that already has content, so
13574/// "the key exists" is a routine answer rather than a failure. The decision is
13575/// made during the batch's existing validate pass, from one id-map lookup per
13576/// row, so the frame stays atomic and re-ingest stays O(n).
13577#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)]
13578pub enum OnConflict {
13579 /// Refuse the whole frame with [`GraphError::DuplicateKey`]. The default,
13580 /// and the only behaviour before v0.6.10.
13581 #[default]
13582 Error,
13583 /// Leave the stored node exactly as it is — properties, label and edges —
13584 /// and count it in [`BatchOutcome::skipped`].
13585 Skip,
13586 /// Keep the key and make the node's properties **exactly** the supplied
13587 /// props: supplied fields are set, fields absent from the supplied props
13588 /// are removed. A supplied label that differs from the stored one, and a
13589 /// supplied `ns` that would move the node, are row errors — relabelling is
13590 /// [`GraphDb::rename_node`], not a side effect of a rebuild.
13591 ///
13592 /// Two properties are outside "exactly", both because they are not the
13593 /// caller's to supply:
13594 ///
13595 /// - `ns` is immutable, so an omitted `ns` leaves the node where it is
13596 /// rather than moving it to `default`;
13597 /// - a property a **view** owns is kept, not removed. Supplying one is a
13598 /// row error, so omitting it cannot be a request to delete it, and
13599 /// refusing the row instead would make `Replace` impossible for the whole
13600 /// population a view has written to. Each such field kept is counted in
13601 /// [`BatchOutcome::kept_view_owned`] — the row is still `replaced` and
13602 /// still raises no row error, so that count is the only signal a caller
13603 /// gets that the stored node carries a field its frame did not describe.
13604 Replace,
13605}
13606
13607/// What one [`OnConflict::Replace`] row resolves to.
13608///
13609/// The property writes that make the node exactly the supplied props —
13610/// `Some(value)` is a set, `None` a removal — paired with how many view-owned
13611/// fields the row kept instead of removing, which is the one way the result is
13612/// not exactly the supplied props. See [`MutPreview::plan_replace`].
13613type ReplacePlan = (Vec<(String, Option<Value>)>, usize);
13614
13615/// What one committed batch did.
13616///
13617/// [`BatchBuilder::commit`] returns the first two fields as a tuple; the rest
13618/// exist for [`OnConflict`] and are always zero / empty without it.
13619#[derive(Clone, Debug, Default, PartialEq, Eq)]
13620pub struct BatchOutcome {
13621 /// Node records actually written.
13622 pub nodes_inserted: usize,
13623 /// Edge records actually written. A duplicate edge is a silent no-op under
13624 /// every policy — adjacency is a set — and is not counted.
13625 pub edges_inserted: usize,
13626 /// Rows whose key was taken and whose policy was [`OnConflict::Skip`].
13627 pub skipped: usize,
13628 /// Rows whose key was taken and whose policy was [`OnConflict::Replace`].
13629 pub replaced: usize,
13630 /// View-owned properties an [`OnConflict::Replace`] row **kept** although
13631 /// the caller did not supply them — counted per field, so one row that
13632 /// keeps two contributes two.
13633 ///
13634 /// This is the one respect in which `Replace` does not make a node's props
13635 /// exactly the supplied ones (see [`OnConflict::Replace`]). Those rows
13636 /// still count in `replaced` and still raise no `row_errors`, because
13637 /// nothing went wrong: a view's property is not the caller's to supply or
13638 /// to remove. A mirror rebuild that needs its copy to be byte-exact reads
13639 /// this to learn that the store kept fields its frame did not describe.
13640 pub kept_view_owned: usize,
13641 /// `(row, why)` for rows an [`OnConflict::Replace`] refused. `row` counts
13642 /// node-insert ops in this batch from zero, which for a caller that queues
13643 /// its nodes in order is the index of the offending node. The rest of the
13644 /// frame still commits; the refused row changes nothing.
13645 pub row_errors: Vec<(usize, String)>,
13646}
13647
13648/// The `(nodes_inserted, edges_inserted)` pair every pre-0.6.10 commit entry
13649/// point returns. Keeps those signatures unchanged now that the validate pass
13650/// produces a [`BatchOutcome`].
13651fn inserted_pair(outcome: BatchOutcome) -> (usize, usize) {
13652 (outcome.nodes_inserted, outcome.edges_inserted)
13653}
13654
13655/// One entry of a frame the validate pass has decided on, before
13656/// [`GraphDb::rewrite_wal_dense_planned`] turns it into dense-id records.
13657///
13658/// Almost every entry is already a finished [`WalRecord`]. The exception is a
13659/// duplicate edge insert: its count names a dense triple, and on the batch path
13660/// the endpoints and the edge type may all be created by earlier records in the
13661/// *same* frame, so no id for them exists until the dense rewrite allocates it.
13662/// Carrying the keys this far and resolving them there is what lets the count
13663/// survive the shape a mirror rebuild writes (defect #24).
13664enum PlannedRec {
13665 Rec(WalRecord),
13666 DuplicateCount {
13667 edge_type: String,
13668 src_key: String,
13669 dst_key: String,
13670 },
13671}
13672
13673/// Queued mutation for a [`BatchBuilder`] or [`GraphDb::commit_group`].
13674///
13675/// The `submit_batch` / `commit_group` APIs accept `Vec<BatchOp>` so that
13676/// callers can build a set of mutations without holding `&mut GraphDb` and
13677/// hand them off to the group-committing writer for durable, batched I/O.
13678pub enum BatchOp {
13679 InsertNode {
13680 label: String,
13681 key: String,
13682 props: Vec<(String, Value)>,
13683 },
13684 InsertEdge {
13685 edge_type: String,
13686 src_key: String,
13687 dst_key: String,
13688 },
13689 SetProp {
13690 key: String,
13691 field: String,
13692 value: Value,
13693 },
13694 RemoveProp {
13695 key: String,
13696 field: String,
13697 },
13698 DeleteEdge {
13699 edge_type: String,
13700 src_key: String,
13701 dst_key: String,
13702 },
13703 DeleteNode {
13704 key: String,
13705 },
13706 CreateRule(RuleDef),
13707 DeleteRule {
13708 name: String,
13709 },
13710 /// Rename a node's key. Validated: old must exist, new must not.
13711 RenameNode {
13712 old_key: String,
13713 new_key: String,
13714 },
13715 /// Insert an edge, auto-creating any missing endpoint as a plain node with
13716 /// `placeholder_label` and no props. Rules fire and last-change is updated
13717 /// for each created endpoint (normal InsertNode semantics in the batch frame).
13718 InsertEdgeUpsert {
13719 edge_type: String,
13720 src_key: String,
13721 dst_key: String,
13722 placeholder_label: String,
13723 },
13724 /// Insert `key`, or — when the key is already taken — do what `on_conflict`
13725 /// says. Queued by [`BatchBuilder::insert_node_on_conflict`]; `Error`
13726 /// queues a plain [`BatchOp::InsertNode`] instead, so this variant only
13727 /// ever carries `Skip` or `Replace`.
13728 InsertNodeOnConflict {
13729 label: String,
13730 key: String,
13731 props: Vec<(String, Value)>,
13732 on_conflict: OnConflict,
13733 },
13734}
13735
13736/// Three-way node visibility status used by `check_single_op_authz`.
13737enum NodeAuthzStatus {
13738 /// Node exists in the store and is in the role's read mask.
13739 Visible(String), // carries the node's label
13740 /// Node exists in the store but is NOT in the role's read mask.
13741 Hidden,
13742 /// Node does not exist in the store.
13743 Absent,
13744}
13745
13746/// Overlay of ops already accepted earlier in the same batch. Never written
13747/// back to the database — validation only.
13748#[derive(Default)]
13749struct Overlay {
13750 extra_keys: BTreeSet<String>,
13751 /// Label of each node inserted earlier in this batch. The store does not
13752 /// have these keys yet, so `label_of` cannot answer for them, and
13753 /// `OnConflict::Replace` has to compare labels.
13754 extra_labels: BTreeMap<String, String>,
13755 deleted_keys: BTreeSet<String>,
13756 extra_props: BTreeMap<(String, String), Value>,
13757 removed_props: BTreeSet<(String, String)>,
13758 extra_edges: BTreeSet<(String, String, String)>,
13759 deleted_edges: BTreeSet<(String, String, String)>,
13760 extra_rules: BTreeSet<String>,
13761 deleted_rules: BTreeSet<String>,
13762 /// `rule name → (via_edge, edge_type)` for every via-hop rule accepted
13763 /// earlier in this batch. Feeds the rule-chain cycle check, which otherwise
13764 /// sees only the rules already committed to the engine. Keyed by name so a
13765 /// later `DeleteRule` in the same batch drops the arc with the rule.
13766 extra_rule_arcs: BTreeMap<String, (String, String)>,
13767}
13768
13769/// Read-only view of live db state plus a batch overlay. Shared by single-op
13770/// public methods (empty overlay) and `commit_batch`.
13771struct MutPreview<'a, F: Fs> {
13772 db: &'a GraphDb<F>,
13773 overlay: Overlay,
13774}
13775
13776/// Shortest path from `start` to `target` following `arcs` (`from → to`), or
13777/// `None` if `target` is unreachable.
13778///
13779/// Used for rule-chain cycle detection, where an arc is "a rule hops over
13780/// `from` and writes `to`". Breadth-first over BTree-ordered adjacency, so the
13781/// reported path is stable for a given rule set, and iterative so a pathological
13782/// rule graph cannot overflow the stack.
13783fn find_cycle_through(arcs: &[(String, String)], start: &str, target: &str) -> Option<Vec<String>> {
13784 let mut adj: BTreeMap<&str, BTreeSet<&str>> = BTreeMap::new();
13785 for (from, to) in arcs {
13786 adj.entry(from.as_str()).or_default().insert(to.as_str());
13787 }
13788 let mut parent: BTreeMap<&str, &str> = BTreeMap::new();
13789 let mut visited: BTreeSet<&str> = BTreeSet::new();
13790 let mut queue: std::collections::VecDeque<&str> = std::collections::VecDeque::new();
13791 visited.insert(start);
13792 queue.push_back(start);
13793 while let Some(node) = queue.pop_front() {
13794 if node == target {
13795 let mut path = vec![node.to_string()];
13796 let mut cur = node;
13797 while let Some(&p) = parent.get(cur) {
13798 path.push(p.to_string());
13799 cur = p;
13800 }
13801 path.reverse();
13802 return Some(path);
13803 }
13804 for &next in adj.get(node).into_iter().flatten() {
13805 if visited.insert(next) {
13806 parent.insert(next, node);
13807 queue.push_back(next);
13808 }
13809 }
13810 }
13811 None
13812}
13813
13814impl<'a, F: Fs> MutPreview<'a, F> {
13815 fn new(db: &'a GraphDb<F>) -> Self {
13816 Self {
13817 db,
13818 overlay: Overlay::default(),
13819 }
13820 }
13821
13822 fn has_key(&self, key: &str) -> bool {
13823 if self.overlay.extra_keys.contains(key) {
13824 return true;
13825 }
13826 if self.overlay.deleted_keys.contains(key) {
13827 return false;
13828 }
13829 self.db.ids.get(key).is_some()
13830 }
13831
13832 fn has_prop(&self, key: &str, field: &str) -> bool {
13833 if !self.has_key(key) {
13834 return false;
13835 }
13836 let k = (key.to_string(), field.to_string());
13837 if self.overlay.removed_props.contains(&k) {
13838 return false;
13839 }
13840 if self.overlay.extra_props.contains_key(&k) {
13841 return true;
13842 }
13843 // Fresh identity (first insert in this batch, or delete+reinsert):
13844 // ignore props still sitting on the soon-to-be-tombstoned slot.
13845 if self.overlay.extra_keys.contains(key) {
13846 return false;
13847 }
13848 self.db.get_prop(key, field).is_some()
13849 }
13850
13851 fn has_edge(&self, edge_type: &str, src_key: &str, dst_key: &str) -> bool {
13852 let k = (
13853 edge_type.to_string(),
13854 src_key.to_string(),
13855 dst_key.to_string(),
13856 );
13857 if self.overlay.deleted_edges.contains(&k) {
13858 return false;
13859 }
13860 if self.overlay.extra_edges.contains(&k) {
13861 return true;
13862 }
13863 // A key created in this batch (including reinsert) has no db edges.
13864 if self.overlay.extra_keys.contains(src_key) || self.overlay.extra_keys.contains(dst_key) {
13865 return false;
13866 }
13867 if self.overlay.deleted_keys.contains(src_key)
13868 || self.overlay.deleted_keys.contains(dst_key)
13869 {
13870 return false;
13871 }
13872 let Some(src) = self.db.ids.get(src_key) else {
13873 return false;
13874 };
13875 let Some(dst) = self.db.ids.get(dst_key) else {
13876 return false;
13877 };
13878 let Some(sym) = self.db.syms.get(edge_type) else {
13879 return false;
13880 };
13881 self.db
13882 .topo_view()
13883 .neighbors(sym, Direction::Out, src)
13884 .binary_search(&dst)
13885 .is_ok()
13886 }
13887
13888 fn has_rule(&self, name: &str) -> bool {
13889 if self.overlay.extra_rules.contains(name) {
13890 return true;
13891 }
13892 if self.overlay.deleted_rules.contains(name) {
13893 return false;
13894 }
13895 self.db.engine.rules().any(|r| r.name == name)
13896 }
13897
13898 fn is_rule_owned(&self, edge_type: &str, src_key: &str, dst_key: &str) -> bool {
13899 if self.overlay.extra_keys.contains(src_key) || self.overlay.extra_keys.contains(dst_key) {
13900 return false;
13901 }
13902 if self.overlay.deleted_keys.contains(src_key)
13903 || self.overlay.deleted_keys.contains(dst_key)
13904 {
13905 return false;
13906 }
13907 let Some(src) = self.db.ids.get(src_key) else {
13908 return false;
13909 };
13910 let Some(dst) = self.db.ids.get(dst_key) else {
13911 return false;
13912 };
13913 let Some(et) = self.db.syms.get(edge_type) else {
13914 return false;
13915 };
13916 // extra_rules is deliberately not consulted: a CreateRule earlier in
13917 // this batch has not fired, so it contributes no provenance. That is
13918 // the documented rule-window gap (see GraphDb::batch).
13919 if self.overlay.deleted_rules.is_empty() {
13920 return self.db.engine.is_owned(et, src, dst);
13921 }
13922 for (rule, triples) in self.db.engine.provenance() {
13923 if self.overlay.deleted_rules.contains(rule) {
13924 continue;
13925 }
13926 if triples.contains(&(et, src, dst)) {
13927 return true;
13928 }
13929 }
13930 false
13931 }
13932
13933 /// The refusals a node creation makes, in the order it makes them.
13934 ///
13935 /// A view owns its property, and creating a node that carries one is a
13936 /// write to it exactly as `set_prop` is — so it is refused here, at the one
13937 /// choke-point `GraphDb::insert_node`, `BatchOp::InsertNode` and the
13938 /// no-conflict arm of `BatchOp::InsertNodeOnConflict` all pass through.
13939 ///
13940 /// Leaving creation exempt was not harmless. The value was stored and
13941 /// served: a created node the view has no reason to revisit keeps the
13942 /// caller's number for the life of the handle, and the backfill at the next
13943 /// open overwrites it — so the store answered `deg = 777` before a restart
13944 /// and `deg = 0` after, for a property every other surface calls read-only.
13945 /// It also split one op two ways: supplying a view-owned field under
13946 /// `OnConflict::Replace` was already a row error on a taken key while the
13947 /// same field on a fresh key was accepted.
13948 ///
13949 /// Checked before the key, like [`MutPreview::prepare_remove_prop`], so the
13950 /// answer does not depend on whether the key exists.
13951 fn check_insert_node(&self, key: &str, props: &[(String, Value)]) -> Result<()> {
13952 for (field, _) in props {
13953 if let Some(view_name) = self.db.view_store.view_for_prop(field) {
13954 return Err(GraphError::ViewPropReadOnly {
13955 view_name: view_name.to_string(),
13956 });
13957 }
13958 }
13959 if self.has_key(key) {
13960 Err(GraphError::DuplicateKey { key: key.into() })
13961 } else {
13962 Ok(())
13963 }
13964 }
13965
13966 fn check_live_key(&self, key: &str) -> Result<()> {
13967 if self.has_key(key) {
13968 Ok(())
13969 } else {
13970 Err(GraphError::KeyNotFound { key: key.into() })
13971 }
13972 }
13973
13974 fn prepare_insert_edge(&self, edge_type: &str, src_key: &str, dst_key: &str) -> Result<bool> {
13975 for k in [src_key, dst_key] {
13976 if !self.has_key(k) {
13977 return Err(GraphError::KeyNotFound { key: k.into() });
13978 }
13979 }
13980 if self.is_rule_owned(edge_type, src_key, dst_key) {
13981 return Err(GraphError::RuleOwned {
13982 detail: format!("edge {edge_type} {src_key}→{dst_key} is rule-owned"),
13983 });
13984 }
13985 // A user-written edge stays inside one namespace. Derived edges do not
13986 // come through here — the engine adds them directly — and the rule
13987 // scoping check is what keeps those pure.
13988 let src_ns = self.namespace_in_batch(src_key);
13989 let dst_ns = self.namespace_in_batch(dst_key);
13990 if src_ns != dst_ns {
13991 return Err(GraphError::CrossNamespace {
13992 src: src_key.to_string(),
13993 src_ns,
13994 dst: dst_key.to_string(),
13995 dst_ns,
13996 });
13997 }
13998 Ok(!self.has_edge(edge_type, src_key, dst_key))
13999 }
14000
14001 fn prepare_remove_prop(&self, key: &str, field: &str) -> Result<bool> {
14002 // A view owns its property, and the refusal has to live here rather
14003 // than on `GraphDb::remove_prop`: `BatchOp::RemoveProp` never meets
14004 // that one, and it is what the HTTP `DELETE /node/{key}/prop/{field}`
14005 // route, `Batch::remove_prop` and the CLI all submit. This is the one
14006 // choke-point every removal passes, exactly as it is for `ns` below.
14007 // Checked before the key, so the answer does not depend on whether the
14008 // key exists — which is also what `GraphDb::remove_prop` answered when
14009 // it carried the only copy of this guard.
14010 if let Some(view_name) = self.db.view_store.view_for_prop(field) {
14011 return Err(GraphError::ViewPropReadOnly {
14012 view_name: view_name.to_string(),
14013 });
14014 }
14015 self.check_live_key(key)?;
14016 // Removing `ns` is changing the namespace — to `default`, the namespace
14017 // an absent property names. It goes through this one choke-point and NOT
14018 // through `rewrite_wal_dense` (a `RemoveProp` needs no dense rewrite), so
14019 // the immutability rule has to be stated here as well. Without it the
14020 // node silently lands in `default` on the next open: the cross-namespace
14021 // edge guard is defeated and a default-bound role reads a tenant's node.
14022 if field == NS_PROP {
14023 let from = self.namespace_in_batch(key);
14024 if from != NS_DEFAULT {
14025 return Err(GraphError::NamespaceImmutable {
14026 key: key.to_string(),
14027 from,
14028 to: NS_DEFAULT.to_string(),
14029 });
14030 }
14031 // Already in `default`: the removal changes no namespace. It is the
14032 // no-op `set_prop` to the current namespace is, not an error.
14033 return Ok(false);
14034 }
14035 Ok(self.has_prop(key, field))
14036 }
14037
14038 fn prepare_delete_edge(&self, edge_type: &str, src_key: &str, dst_key: &str) -> Result<bool> {
14039 for k in [src_key, dst_key] {
14040 if !self.has_key(k) {
14041 return Err(GraphError::KeyNotFound { key: k.into() });
14042 }
14043 }
14044 // Provenance-owned OR a live rule would derive this pair. User-first
14045 // edges that a later rule matches are not in `owned`, but deleting
14046 // them would leave a hole `rebuild_rule` immediately fills.
14047 if self.is_rule_owned(edge_type, src_key, dst_key) {
14048 return Err(GraphError::RuleOwned {
14049 detail: format!(
14050 "edge {edge_type} {src_key}→{dst_key} is rule-owned; \
14051 delete or change the owning rule"
14052 ),
14053 });
14054 }
14055 if self.would_derive(edge_type, src_key, dst_key) {
14056 return Err(GraphError::RuleOwned {
14057 detail: format!(
14058 "edge {edge_type} {src_key}→{dst_key} is rule-owned; \
14059 delete or change the owning rule, or a live rule would re-derive it"
14060 ),
14061 });
14062 }
14063 Ok(self.has_edge(edge_type, src_key, dst_key))
14064 }
14065
14066 /// True if any live rule (minus overlay-deleted names) would derive
14067 /// `(edge_type, src, dst)` from current overlay-visible props/labels.
14068 /// CreateRule names in `extra_rules` are ignored — same documented
14069 /// same-batch rule-window as [`Self::is_rule_owned`].
14070 fn would_derive(&self, edge_type: &str, src_key: &str, dst_key: &str) -> bool {
14071 if src_key == dst_key {
14072 return false;
14073 }
14074 let Some(src_label) = self.label_of(src_key) else {
14075 return false;
14076 };
14077 let Some(dst_label) = self.label_of(dst_key) else {
14078 return false;
14079 };
14080 for rule in self.db.engine.rules() {
14081 if self.overlay.deleted_rules.contains(&rule.name) {
14082 continue;
14083 }
14084 if rule.edge_type != edge_type {
14085 continue;
14086 }
14087 if rule.src_label != src_label || rule.dst_label != dst_label {
14088 continue;
14089 }
14090 let src_props = |f: &str| self.prop_value(src_key, f);
14091 let dst_props = |f: &str| self.prop_value(dst_key, f);
14092 let src_view = NodeView {
14093 key: src_key,
14094 props: &src_props,
14095 };
14096 let dst_view = NodeView {
14097 key: dst_key,
14098 props: &dst_props,
14099 };
14100 if evaluate(&rule.predicate, &src_view, &dst_view).is_some() {
14101 return true;
14102 }
14103 }
14104 false
14105 }
14106
14107 fn label_of(&self, key: &str) -> Option<String> {
14108 if self.overlay.deleted_keys.contains(key) {
14109 return None;
14110 }
14111 // Fresh identities created in this batch have no stored label in the
14112 // overlay; they cannot be provenance-owned yet either.
14113 let id = self.db.ids.get(key)?;
14114 let sym = self.db.labels.get(id as usize).copied()?;
14115 if sym == u32::MAX {
14116 return None;
14117 }
14118 self.db.syms.resolve(sym).map(str::to_string)
14119 }
14120
14121 /// The label `key` carries as this batch sees it — including a node
14122 /// inserted earlier in the same batch, which the store does not have yet.
14123 fn label_in_batch(&self, key: &str) -> Option<String> {
14124 if self.overlay.deleted_keys.contains(key) {
14125 return None;
14126 }
14127 if let Some(label) = self.overlay.extra_labels.get(key) {
14128 return Some(label.clone());
14129 }
14130 self.label_of(key)
14131 }
14132
14133 /// The property writes that make `key`'s props exactly `props`, or why the
14134 /// row is refused.
14135 ///
14136 /// `Some(value)` is a set and `None` is a removal. `store_fields` is every
14137 /// field name the store knows, hoisted by the caller so a frame of N
14138 /// replaces reads the field list once rather than N times.
14139 ///
14140 /// The second half of the pair is how many view-owned fields this row kept
14141 /// rather than removed — the one part of "exactly the supplied props" that
14142 /// does not hold, and the caller's only signal that it did not.
14143 ///
14144 /// The refusals are row errors, not frame errors: a mirror rebuild should
14145 /// learn which of its rows disagree with the store without losing the rows
14146 /// that agree.
14147 fn plan_replace(
14148 &self,
14149 label: &str,
14150 key: &str,
14151 props: &[(String, Value)],
14152 store_fields: &[String],
14153 ) -> std::result::Result<ReplacePlan, String> {
14154 // A different label is a relabel, and a rebuild does not relabel: that
14155 // is `rename_node` or an explicit write, never a side effect here.
14156 let stored = self.label_in_batch(key).unwrap_or_default();
14157 if stored != label {
14158 return Err(format!(
14159 "node {key}: on_conflict=\"replace\" will not relabel {stored:?} to {label:?}; \
14160 relabelling is rename_node or an explicit write"
14161 ));
14162 }
14163 // `ns` is immutable. Replace removes what the supplied props omit, so
14164 // an omitted `ns` is a move to `default` exactly as a different `ns` is
14165 // a move to that one; both are the same refusal.
14166 let from = self.namespace_in_batch(key);
14167 let to = match props.iter().find(|(field, _)| field == NS_PROP) {
14168 Some((_, Value::Str(ns))) => ns.clone(),
14169 Some((_, value)) => {
14170 return Err(format!(
14171 "node {key}: {NS_PROP} must be a string naming a namespace, got {value:?}"
14172 ));
14173 }
14174 None => NS_DEFAULT.to_string(),
14175 };
14176 if to != from {
14177 return Err(format!(
14178 "node {key}: {NS_PROP} is immutable; on_conflict=\"replace\" cannot move it \
14179 from {from:?} to {to:?}"
14180 ));
14181 }
14182
14183 if let Some(why) = self.supplied_view_owned_prop(key, props) {
14184 return Err(why);
14185 }
14186
14187 let supplied: BTreeSet<&str> = props.iter().map(|(field, _)| field.as_str()).collect();
14188 let mut writes = Vec::new();
14189 for (field, value) in props {
14190 // `ns` names the namespace the node is already in, so the write is
14191 // the no-op the dense-rewrite seam would drop anyway.
14192 if field == NS_PROP {
14193 continue;
14194 }
14195 // Already exactly this value: a rebuild of an unchanged row should
14196 // cost no WAL record.
14197 if self.prop_value(key, field).as_ref() == Some(value) {
14198 continue;
14199 }
14200 writes.push((field.clone(), Some(value.clone())));
14201 }
14202 // Everything the node still carries that the supplied props do not.
14203 // `ns` is never removed: it is immutable, and the check above has
14204 // already established the node stays where it is.
14205 let overlay_fields = self
14206 .overlay
14207 .extra_props
14208 .keys()
14209 .filter(|(k, _)| k == key)
14210 .map(|(_, field)| field.as_str());
14211 //
14212 // A view-owned field is filtered out rather than refused. It is not the
14213 // caller's to supply (supplying one is still the row error above) and
14214 // so it is not part of what "exactly the supplied ones" ranges over:
14215 // omitting it is not a request to delete it. Refusing here instead
14216 // would make `replace` impossible for every node a view has written to
14217 // — which on a store carrying a view is the whole population a mirror
14218 // rebuild has to cover.
14219 let omitted: BTreeSet<&str> = store_fields
14220 .iter()
14221 .map(String::as_str)
14222 .chain(overlay_fields)
14223 .filter(|field| {
14224 *field != NS_PROP && !supplied.contains(field) && self.has_prop(key, field)
14225 })
14226 .collect();
14227 // The view-owned half is kept, and counted: the row still commits and
14228 // still reports no error, so without this number a mirror rebuild is
14229 // told it got exactly what it asked for when it did not (defect #18).
14230 let (stale, kept): (Vec<&str>, Vec<&str>) = omitted
14231 .into_iter()
14232 .partition(|field| self.db.view_store.view_for_prop(field).is_none());
14233 writes.extend(stale.into_iter().map(|field| (field.to_string(), None)));
14234 Ok((writes, kept.len()))
14235 }
14236
14237 /// The row error a supplied view-owned field earns, or `None`.
14238 ///
14239 /// Shared by [`MutPreview::plan_replace`] and the no-conflict arm of
14240 /// `BatchOp::InsertNodeOnConflict` so that one op answers a supplied
14241 /// view-owned field the same way whether or not the key was already taken.
14242 fn supplied_view_owned_prop(&self, key: &str, props: &[(String, Value)]) -> Option<String> {
14243 props.iter().find_map(|(field, _)| {
14244 self.db.view_store.view_for_prop(field).map(|view_name| {
14245 format!(
14246 "node {key}: property {field:?} is owned by view {view_name:?} and is \
14247 read-only"
14248 )
14249 })
14250 })
14251 }
14252
14253 /// The namespace `key` is in as this batch sees it — including a node
14254 /// inserted earlier in the same batch, which the store does not have yet.
14255 fn namespace_in_batch(&self, key: &str) -> String {
14256 namespace_of_value(self.prop_value(key, NS_PROP).as_ref()).to_string()
14257 }
14258
14259 fn prop_value(&self, key: &str, field: &str) -> Option<Value> {
14260 if !self.has_key(key) {
14261 return None;
14262 }
14263 let k = (key.to_string(), field.to_string());
14264 if self.overlay.removed_props.contains(&k) {
14265 return None;
14266 }
14267 if let Some(v) = self.overlay.extra_props.get(&k) {
14268 return Some(v.clone());
14269 }
14270 if self.overlay.extra_keys.contains(key) {
14271 return None;
14272 }
14273 self.db.get_prop(key, field)
14274 }
14275
14276 fn check_create_rule(&self, def: &RuleDef) -> Result<()> {
14277 def.validate()
14278 .map_err(|e| GraphError::RuleInvalid { detail: e })?;
14279 if self.has_rule(&def.name) {
14280 return Err(GraphError::RuleInvalid {
14281 detail: format!("rule {:?} already exists", def.name),
14282 });
14283 }
14284 // Rule-chain cycle rejection. Derived edges feed via-hop rules, so a
14285 // rule set forms a graph whose arcs are "hops over `via_edge`, writes
14286 // `edge_type`". A cycle in that graph is a rule set that would re-fire
14287 // itself forever; the engine's depth cap would silently truncate it
14288 // instead, leaving an arbitrary partial result. Reject it here, the one
14289 // place that sees the whole rule set.
14290 //
14291 // Rules accepted earlier in the same batch count too: the overlay
14292 // carries their arcs, so a cycle cannot be assembled one op at a time.
14293 if let Some(via) = def.via_edge.as_deref() {
14294 if via == def.edge_type {
14295 return Err(GraphError::RuleInvalid {
14296 detail: format!("rule chain cycle: {} -> {}", via, def.edge_type),
14297 });
14298 }
14299 let mut arcs: Vec<(String, String)> = self
14300 .db
14301 .engine
14302 .rules()
14303 .filter(|r| !self.overlay.deleted_rules.contains(&r.name))
14304 .filter_map(|r| r.via_edge.clone().map(|v| (v, r.edge_type.clone())))
14305 .collect();
14306 arcs.extend(self.overlay.extra_rule_arcs.values().cloned());
14307 arcs.push((via.to_string(), def.edge_type.clone()));
14308 if let Some(path) = find_cycle_through(&arcs, &def.edge_type, via) {
14309 return Err(GraphError::RuleInvalid {
14310 detail: format!("rule chain cycle: {} -> {}", via, path.join(" -> ")),
14311 });
14312 }
14313 }
14314 Ok(())
14315 }
14316
14317 fn check_delete_rule(&self, name: &str) -> Result<()> {
14318 if self.has_rule(name) {
14319 Ok(())
14320 } else {
14321 Err(GraphError::RuleNotFound { name: name.into() })
14322 }
14323 }
14324
14325 fn note_insert_node(&mut self, label: &str, key: &str, props: &[(String, Value)]) {
14326 self.overlay.deleted_keys.remove(key);
14327 self.overlay.extra_keys.insert(key.to_string());
14328 self.overlay
14329 .extra_labels
14330 .insert(key.to_string(), label.to_string());
14331 self.overlay.extra_props.retain(|(k, _), _| k != key);
14332 self.overlay.removed_props.retain(|(k, _)| k != key);
14333 for (field, value) in props {
14334 self.overlay
14335 .extra_props
14336 .insert((key.to_string(), field.clone()), value.clone());
14337 }
14338 }
14339
14340 fn note_insert_edge(&mut self, edge_type: &str, src_key: &str, dst_key: &str) {
14341 let k = (
14342 edge_type.to_string(),
14343 src_key.to_string(),
14344 dst_key.to_string(),
14345 );
14346 self.overlay.deleted_edges.remove(&k);
14347 self.overlay.extra_edges.insert(k);
14348 }
14349
14350 fn note_set_prop(&mut self, key: &str, field: &str, value: &Value) {
14351 let k = (key.to_string(), field.to_string());
14352 self.overlay.removed_props.remove(&k);
14353 self.overlay.extra_props.insert(k, value.clone());
14354 }
14355
14356 fn note_remove_prop(&mut self, key: &str, field: &str) {
14357 let k = (key.to_string(), field.to_string());
14358 self.overlay.extra_props.remove(&k);
14359 self.overlay.removed_props.insert(k);
14360 }
14361
14362 fn note_delete_edge(&mut self, edge_type: &str, src_key: &str, dst_key: &str) {
14363 let k = (
14364 edge_type.to_string(),
14365 src_key.to_string(),
14366 dst_key.to_string(),
14367 );
14368 self.overlay.extra_edges.remove(&k);
14369 self.overlay.deleted_edges.insert(k);
14370 }
14371
14372 fn note_delete_node(&mut self, key: &str) {
14373 self.overlay.extra_keys.remove(key);
14374 self.overlay.deleted_keys.insert(key.to_string());
14375 self.overlay.extra_props.retain(|(k, _), _| k != key);
14376 self.overlay.removed_props.retain(|(k, _)| k != key);
14377 self.overlay
14378 .extra_edges
14379 .retain(|(_, s, d)| s != key && d != key);
14380 self.overlay
14381 .deleted_edges
14382 .retain(|(_, s, d)| s != key && d != key);
14383 }
14384
14385 fn note_create_rule(&mut self, def: &RuleDef) {
14386 self.overlay.deleted_rules.remove(&def.name);
14387 self.overlay.extra_rules.insert(def.name.clone());
14388 // Rules accepted earlier in this batch are not in the engine yet, so
14389 // the cycle check would not see their arcs. Keep the arc, not just the
14390 // name, so a batch cannot smuggle in a cycle one op at a time.
14391 if let Some(via) = def.via_edge.clone() {
14392 self.overlay
14393 .extra_rule_arcs
14394 .insert(def.name.clone(), (via, def.edge_type.clone()));
14395 }
14396 }
14397
14398 fn check_rename_node(&self, old: &str, new: &str) -> Result<()> {
14399 if !self.has_key(old) {
14400 return Err(GraphError::KeyNotFound { key: old.into() });
14401 }
14402 if self.has_key(new) {
14403 return Err(GraphError::DuplicateKey { key: new.into() });
14404 }
14405 Ok(())
14406 }
14407
14408 fn note_rename_node(&mut self, old: &str, new: &str) {
14409 // Mark old as deleted so subsequent batch ops cannot reference it.
14410 self.overlay.extra_keys.remove(old);
14411 self.overlay.deleted_keys.insert(old.to_string());
14412 // Mark new as extra so subsequent batch ops can reference it.
14413 self.overlay.deleted_keys.remove(new);
14414 self.overlay.extra_keys.insert(new.to_string());
14415 // Migrate any overlay props from old key to new key.
14416 let new_str = new.to_string();
14417 let transferred: Vec<((String, String), Value)> = self
14418 .overlay
14419 .extra_props
14420 .iter()
14421 .filter(|((k, _), _)| k.as_str() == old)
14422 .map(|((_, f), v)| ((new_str.clone(), f.clone()), v.clone()))
14423 .collect();
14424 self.overlay
14425 .extra_props
14426 .retain(|(k, _), _| k.as_str() != old);
14427 for (k, v) in transferred {
14428 self.overlay.extra_props.insert(k, v);
14429 }
14430 // Migrate removed_props.
14431 let transferred_removed: Vec<(String, String)> = self
14432 .overlay
14433 .removed_props
14434 .iter()
14435 .filter(|(k, _)| k.as_str() == old)
14436 .map(|(_, f)| (new_str.clone(), f.clone()))
14437 .collect();
14438 self.overlay
14439 .removed_props
14440 .retain(|(k, _)| k.as_str() != old);
14441 for k in transferred_removed {
14442 self.overlay.removed_props.insert(k);
14443 }
14444 }
14445
14446 fn note_delete_rule(&mut self, name: &str) {
14447 self.overlay.extra_rules.remove(name);
14448 // Drop its chain arc too: a rule created and then deleted in the same
14449 // batch must not make a later, legal rule look like a cycle.
14450 self.overlay.extra_rule_arcs.remove(name);
14451 self.overlay.deleted_rules.insert(name.to_string());
14452 // Treat the deleted rule's current provenance as gone so a later
14453 // delete_edge of those triples is a no-op (matches sequential).
14454 if let Some(triples) = self.db.engine.provenance().get(name) {
14455 for &(et, s, d) in triples {
14456 let Some(etype) = self.db.syms.resolve(et) else {
14457 continue;
14458 };
14459 let Some(src) = self.db.ids.key_of(s) else {
14460 continue;
14461 };
14462 let Some(dst) = self.db.ids.key_of(d) else {
14463 continue;
14464 };
14465 let k = (etype.to_string(), src.to_string(), dst.to_string());
14466 self.overlay.extra_edges.remove(&k);
14467 self.overlay.deleted_edges.insert(k);
14468 }
14469 }
14470 }
14471}
14472
14473/// Collects mutations and commits them as one WAL `Batch` frame.
14474///
14475/// Holds `&mut GraphDb` for its lifetime. Queue with the same method names
14476/// as [`GraphDb`]; call [`commit`](Self::commit) to validate, log, and apply.
14477/// See [`GraphDb::batch`] for validation and atomicity rules.
14478pub struct BatchBuilder<'a, F: Fs> {
14479 db: &'a mut GraphDb<F>,
14480 ops: Vec<BatchOp>,
14481}
14482
14483impl<'a, F: Fs> BatchBuilder<'a, F> {
14484 pub fn insert_node(
14485 &mut self,
14486 label: &str,
14487 key: &str,
14488 props: Vec<(String, Value)>,
14489 ) -> &mut Self {
14490 self.ops.push(BatchOp::InsertNode {
14491 label: label.into(),
14492 key: key.into(),
14493 props,
14494 });
14495 self
14496 }
14497
14498 /// Queue a node insert whose answer to a taken key is `on_conflict`.
14499 ///
14500 /// [`OnConflict::Error`] queues exactly the op [`insert_node`](Self::insert_node)
14501 /// does, so the default path is unchanged.
14502 pub fn insert_node_on_conflict(
14503 &mut self,
14504 label: &str,
14505 key: &str,
14506 props: Vec<(String, Value)>,
14507 on_conflict: OnConflict,
14508 ) -> &mut Self {
14509 self.ops.push(match on_conflict {
14510 OnConflict::Error => BatchOp::InsertNode {
14511 label: label.into(),
14512 key: key.into(),
14513 props,
14514 },
14515 on_conflict => BatchOp::InsertNodeOnConflict {
14516 label: label.into(),
14517 key: key.into(),
14518 props,
14519 on_conflict,
14520 },
14521 });
14522 self
14523 }
14524
14525 pub fn insert_edge(&mut self, edge_type: &str, src_key: &str, dst_key: &str) -> &mut Self {
14526 self.ops.push(BatchOp::InsertEdge {
14527 edge_type: edge_type.into(),
14528 src_key: src_key.into(),
14529 dst_key: dst_key.into(),
14530 });
14531 self
14532 }
14533
14534 pub fn set_prop(&mut self, key: &str, field: &str, value: Value) -> &mut Self {
14535 self.ops.push(BatchOp::SetProp {
14536 key: key.into(),
14537 field: field.into(),
14538 value,
14539 });
14540 self
14541 }
14542
14543 pub fn remove_prop(&mut self, key: &str, field: &str) -> &mut Self {
14544 self.ops.push(BatchOp::RemoveProp {
14545 key: key.into(),
14546 field: field.into(),
14547 });
14548 self
14549 }
14550
14551 pub fn delete_edge(&mut self, edge_type: &str, src_key: &str, dst_key: &str) -> &mut Self {
14552 self.ops.push(BatchOp::DeleteEdge {
14553 edge_type: edge_type.into(),
14554 src_key: src_key.into(),
14555 dst_key: dst_key.into(),
14556 });
14557 self
14558 }
14559
14560 pub fn delete_node(&mut self, key: &str) -> &mut Self {
14561 self.ops.push(BatchOp::DeleteNode { key: key.into() });
14562 self
14563 }
14564
14565 pub fn create_rule(&mut self, def: RuleDef) -> &mut Self {
14566 self.ops.push(BatchOp::CreateRule(def));
14567 self
14568 }
14569
14570 pub fn delete_rule(&mut self, name: &str) -> &mut Self {
14571 self.ops.push(BatchOp::DeleteRule { name: name.into() });
14572 self
14573 }
14574
14575 /// Queue a node-rename in this batch.
14576 ///
14577 /// Validation (old exists, new not taken) runs at commit time.
14578 pub fn rename_node(&mut self, old_key: &str, new_key: &str) -> &mut Self {
14579 self.ops.push(BatchOp::RenameNode {
14580 old_key: old_key.into(),
14581 new_key: new_key.into(),
14582 });
14583 self
14584 }
14585
14586 /// Queue an edge insert with endpoint auto-creation.
14587 ///
14588 /// Any missing endpoint is created as a plain node `{key, label:
14589 /// placeholder_label, no props}` inside this batch frame. Rules fire and
14590 /// last-change is updated for each auto-created node.
14591 pub fn insert_edge_upsert(
14592 &mut self,
14593 edge_type: &str,
14594 src_key: &str,
14595 dst_key: &str,
14596 placeholder_label: &str,
14597 ) -> &mut Self {
14598 self.ops.push(BatchOp::InsertEdgeUpsert {
14599 edge_type: edge_type.into(),
14600 src_key: src_key.into(),
14601 dst_key: dst_key.into(),
14602 placeholder_label: placeholder_label.into(),
14603 });
14604 self
14605 }
14606
14607 /// Validate every queued op, then log one `Batch` frame and apply.
14608 /// Empty / all-noop batches return `Ok(())` without writing the WAL.
14609 /// A second `commit()` after a successful one is an empty-batch no-op
14610 /// (queued ops were taken).
14611 /// Takes `&mut self` so it chains after the queue methods (`b.insert_node(..).commit()`)
14612 /// and also works as `let mut b = db.batch(); b.insert_node(..); b.commit()`.
14613 ///
14614 /// **Rule-window limitation:** batch validation cannot see edges that a
14615 /// rule created earlier in the *same* batch will derive at apply time, so
14616 /// a `delete_edge` / `insert_edge` in that window is silently no-oped
14617 /// where sequential calls would return `Err(RuleOwned)`. State integrity
14618 /// is unaffected (idempotent apply, provenance intact). Create rules in
14619 /// their own batch, or sequentially, when later ops may touch derived
14620 /// edges.
14621 /// Validate every queued op and commit atomically.
14622 ///
14623 /// Returns `(nodes_inserted, edges_inserted)` — the counts of node and edge
14624 /// WAL records actually written (duplicate edges are silent no-ops and are
14625 /// NOT counted). Both are 0 when the batch is empty or all-noop.
14626 pub fn commit(&mut self) -> Result<(usize, usize)> {
14627 let ops = std::mem::take(&mut self.ops);
14628 self.db.commit_batch(ops)
14629 }
14630
14631 /// [`commit`](Self::commit) with the full [`BatchOutcome`] — the counts a
14632 /// caller needs when its rows carry an [`OnConflict`] policy.
14633 pub fn commit_outcome(&mut self) -> Result<BatchOutcome> {
14634 let ops = std::mem::take(&mut self.ops);
14635 self.db.commit_logged_batch(ops, None, None)
14636 }
14637
14638 /// Same as [`commit`](Self::commit) but tail the inner events with
14639 /// [`MutationEvent::Ingested`] instead of [`MutationEvent::BatchApplied`].
14640 pub(crate) fn commit_ingest(&mut self, label: &str, inserted: usize) -> Result<(usize, usize)> {
14641 let ops = std::mem::take(&mut self.ops);
14642 self.db
14643 .commit_logged_batch(ops, Some((label.to_string(), inserted)), None)
14644 .map(inserted_pair)
14645 }
14646}
14647
14648pub struct NodeRef<'a, F: Fs> {
14649 db: &'a GraphDb<F>,
14650 id: u32,
14651}
14652
14653impl<'a, F: Fs> NodeRef<'a, F> {
14654 pub fn key(&self) -> &str {
14655 self.db.ids.key_of(self.id).expect("dense ids")
14656 }
14657
14658 pub fn label(&self) -> &str {
14659 let sym = self
14660 .db
14661 .labels
14662 .get(self.id as usize)
14663 .copied()
14664 .filter(|&s| s != u32::MAX)
14665 .expect("real nodes always have a label; u32::MAX sentinel cannot occur");
14666 self.db.syms.resolve(sym).expect("interned label symbol")
14667 }
14668
14669 pub fn prop(&self, field: &str) -> Option<Value> {
14670 self.db
14671 .props_view()
14672 .get(self.id, field)
14673 .map(|vr| vr.into_value())
14674 }
14675
14676 /// All stored fields for this node, sorted by field name.
14677 ///
14678 /// Reads from the full base+overlay view so that props stored only in the
14679 /// V8 snapshot base (i.e. before any post-snapshot WAL writes) are visible.
14680 pub fn props(&self) -> BTreeMap<String, Value> {
14681 let mut out = BTreeMap::new();
14682 let pv = self.db.props_view();
14683 for field in pv.field_names() {
14684 if let Some(vr) = pv.get(self.id, &field) {
14685 out.insert(field, vr.into_value());
14686 }
14687 }
14688 out
14689 }
14690
14691 /// depth-N BFS as a ResultSet: columns ["key","label","depth"], BFS order.
14692 pub fn neighborhood(&self, depth: u32, edge_types: Option<&[&str]>, dir: Dir) -> ResultSet {
14693 let view = self.db.view();
14694 let resolved: Option<Vec<u32>> = edge_types.map(|names| {
14695 names
14696 .iter()
14697 .filter_map(|name| view.syms.get(name))
14698 .collect()
14699 });
14700 let nb = neighborhood(&view, self.id, depth, resolved.as_deref(), dir);
14701 let mut rs = ResultSet::new(vec!["key".into(), "label".into(), "depth".into()]);
14702 for (nid, d) in nb.nodes {
14703 let key = view.key_of(nid);
14704 let label = view
14705 .label_of(nid)
14706 .expect("real nodes always have a label; u32::MAX sentinel cannot occur");
14707 rs.push_row(vec![
14708 Some(Value::Str(key.to_string())),
14709 Some(Value::Str(label.to_string())),
14710 Some(Value::Int(d as i64)),
14711 ]);
14712 }
14713 rs
14714 }
14715
14716 /// 1-hop, Both directions: edge-type name → sorted unique neighbor keys.
14717 pub fn grouped_by_edge_type(&self) -> BTreeMap<String, Vec<String>> {
14718 let view = self.db.view();
14719 let mut groups: BTreeMap<String, BTreeSet<String>> = BTreeMap::new();
14720 for e in expand(&view, self.id, None, Dir::Both) {
14721 // Skip edges with unknown etypes (only possible from corrupt large
14722 // TOPOLOGY section; function returns BTreeMap not Result).
14723 let Some(etype) = view.syms.resolve(e.etype) else {
14724 continue;
14725 };
14726 let etype = etype.to_string();
14727 let nbr = if e.src == self.id { e.dst } else { e.src };
14728 groups
14729 .entry(etype)
14730 .or_default()
14731 .insert(view.key_of(nbr).to_string());
14732 }
14733 groups
14734 .into_iter()
14735 .map(|(k, v)| (k, v.into_iter().collect()))
14736 .collect()
14737 }
14738}
14739
14740#[cfg(test)]
14741mod tests {
14742 use super::*;
14743 use core_rules::Predicate;
14744
14745 fn tmp_dir(name: &str) -> std::path::PathBuf {
14746 let d =
14747 std::env::temp_dir().join(format!("graphdb-db-unit-{}-{}", name, std::process::id()));
14748 let _ = std::fs::remove_dir_all(&d);
14749 d
14750 }
14751
14752 fn fk_rule() -> RuleDef {
14753 RuleDef {
14754 name: "works_at".into(),
14755 src_label: "Person".into(),
14756 dst_label: "Org".into(),
14757 predicate: Predicate::KeyMatch {
14758 field: "org_id".into(),
14759 },
14760 edge_type: "WORKS_AT".into(),
14761 weight_prop: None,
14762 max_edges: None,
14763 approximate: false,
14764 via_label: None,
14765 via_edge: None,
14766 via_dir: None,
14767 namespace: None,
14768 }
14769 }
14770
14771 /// Regression guard for the no-views delta-copy fast path.
14772 ///
14773 /// When no views are defined, `pending_deltas_since().to_vec()` must never
14774 /// be called — even during a large CreateRule backfill. The DELTA_COPY_COUNT
14775 /// thread-local is incremented inside every `if !view_store.is_empty()` block;
14776 /// a count of 0 after the entire sequence proves the guard fires correctly.
14777 #[test]
14778 fn no_delta_copy_when_no_views() {
14779 DELTA_COPY_COUNT.with(|c| c.set(0));
14780 let dir = tmp_dir("no-delta-copy");
14781 {
14782 let mut db = GraphDb::open(&dir).unwrap();
14783 // Insert 50 Org + 50 Person nodes with FK links.
14784 for i in 0..50u32 {
14785 db.insert_node("Org", &format!("o{i}"), vec![]).unwrap();
14786 }
14787 for i in 0..50u32 {
14788 db.insert_node(
14789 "Person",
14790 &format!("p{i}"),
14791 vec![("org_id".into(), Value::Str(format!("o{i}")))],
14792 )
14793 .unwrap();
14794 }
14795 // CreateRule backfill should NOT invoke to_vec() when no views are defined.
14796 db.create_rule(fk_rule()).unwrap();
14797
14798 // Counter must stay 0 — no views, no copies.
14799 let copies = DELTA_COPY_COUNT.with(|c| c.get());
14800 assert_eq!(
14801 copies, 0,
14802 "pending_deltas_since().to_vec() called despite no views"
14803 );
14804
14805 // Derived edges must still be correct (the guard skips only the
14806 // empty delta propagation loop, not the rule application itself).
14807 let nbrs = db.neighbors("p0", "WORKS_AT", Direction::Out).unwrap();
14808 assert_eq!(
14809 nbrs,
14810 vec!["o0"],
14811 "rule must derive edges even with no views"
14812 );
14813 }
14814 let _ = std::fs::remove_dir_all(&dir);
14815 }
14816
14817 /// Gating regression: subscribe AFTER a backfill must see no stale events.
14818 /// subscribe BEFORE a backfill must see every edge-fire event.
14819 #[test]
14820 fn subscribe_after_backfill_no_stale_events() {
14821 let dir = tmp_dir("sub-after-backfill");
14822 {
14823 let mut db = GraphDb::open(&dir).unwrap();
14824 for i in 0..10u32 {
14825 db.insert_node("Org", &format!("o{i}"), vec![]).unwrap();
14826 db.insert_node(
14827 "Person",
14828 &format!("p{i}"),
14829 vec![("org_id".into(), Value::Str(format!("o{i}")))],
14830 )
14831 .unwrap();
14832 }
14833 // Create rule BEFORE subscribing — emit_deltas is false during backfill.
14834 db.create_rule(fk_rule()).unwrap();
14835
14836 // Subscribe AFTER the backfill — queue must be empty (no stale events).
14837 let sub = db.subscribe_all_rules().unwrap();
14838 // No events should have queued for the prior backfill.
14839 assert!(
14840 sub.try_recv().is_none(),
14841 "subscribe after backfill must see no stale events"
14842 );
14843
14844 // Inserting a new node now should fire an event (emit_deltas is now true).
14845 db.insert_node("Org", "o_new", vec![]).unwrap();
14846 db.insert_node(
14847 "Person",
14848 "p_new",
14849 vec![("org_id".into(), Value::Str("o_new".into()))],
14850 )
14851 .unwrap();
14852 let ev = sub.recv_timeout(std::time::Duration::from_millis(200));
14853 assert!(
14854 ev.is_some(),
14855 "edge-fire event must arrive after subscribe (emit_deltas=true)"
14856 );
14857 }
14858 let _ = std::fs::remove_dir_all(&dir);
14859 }
14860
14861 /// Gating regression: subscribe BEFORE a backfill → events flow.
14862 #[test]
14863 fn subscribe_before_backfill_events_flow() {
14864 let dir = tmp_dir("sub-before-backfill");
14865 {
14866 let mut db = GraphDb::open(&dir).unwrap();
14867 // Subscribe FIRST — emit_deltas becomes true.
14868 let sub = db.subscribe_all_rules().unwrap();
14869
14870 for i in 0..5u32 {
14871 db.insert_node("Org", &format!("o{i}"), vec![]).unwrap();
14872 db.insert_node(
14873 "Person",
14874 &format!("p{i}"),
14875 vec![("org_id".into(), Value::Str(format!("o{i}")))],
14876 )
14877 .unwrap();
14878 }
14879 // Backfill fires with emit_deltas=true → events queued.
14880 db.create_rule(fk_rule()).unwrap();
14881
14882 // Should receive at least one edge-fired event from the backfill.
14883 let mut received = 0usize;
14884 while sub.try_recv().is_some() {
14885 received += 1;
14886 }
14887 assert!(
14888 received > 0,
14889 "subscribe before backfill must receive edge-fire events (got 0)"
14890 );
14891 }
14892 let _ = std::fs::remove_dir_all(&dir);
14893 }
14894
14895 /// Companion: when a view IS defined, the delta path fires and view values update.
14896 #[test]
14897 fn delta_copy_fires_when_view_exists() {
14898 use core_rules::ViewSource;
14899 DELTA_COPY_COUNT.with(|c| c.set(0));
14900 let dir = tmp_dir("delta-copy-with-view");
14901 {
14902 let mut db = GraphDb::open(&dir).unwrap();
14903 db.insert_node("Org", "o1", vec![]).unwrap();
14904 db.insert_node(
14905 "Person",
14906 "p1",
14907 vec![("org_id".into(), Value::Str("o1".into()))],
14908 )
14909 .unwrap();
14910 // Declare a Degree view so is_empty() returns false.
14911 db.create_view(ViewDef {
14912 name: "degree_out".into(),
14913 label: "Person".into(),
14914 view_prop: "degree_out".into(),
14915 source: ViewSource::Degree {
14916 edge_type: "WORKS_AT".into(),
14917 direction: Direction::Out,
14918 },
14919 })
14920 .unwrap();
14921 db.create_rule(fk_rule()).unwrap();
14922
14923 // At least one delta copy should have happened (CreateRule backfill).
14924 let copies = DELTA_COPY_COUNT.with(|c| c.get());
14925 assert!(
14926 copies > 0,
14927 "expected delta copy to fire when a view is defined"
14928 );
14929
14930 // View value should be computed: p1 has one WORKS_AT out-edge.
14931 let info = db.node_info("p1").unwrap();
14932 let degree = info.props.get("degree_out");
14933 assert!(
14934 degree.is_some(),
14935 "view prop should be written to node props"
14936 );
14937 }
14938 let _ = std::fs::remove_dir_all(&dir);
14939 }
14940
14941 /// Regression: `open_at_with` must call `rebuild_all` after WAL replay so
14942 /// derived-edge-driven view values reflect the as-of state rather than just
14943 /// the initial backfill written at `CreateView` time.
14944 ///
14945 /// Base WAL frames (indices 0..=5 before history markers):
14946 /// 0: insert Org "o1"
14947 /// 1: create_view "employee_count" (Degree / WORKS_AT / In) on Org
14948 /// 2: create_rule fk_rule (WORKS_AT, Person→Org via org_id)
14949 /// 3: insert Person "p1" → rule fires WORKS_AT p1→o1 (degree = 1) ← mid
14950 /// 4: insert Person "p2" → rule fires WORKS_AT p2→o1 (degree = 2)
14951 /// 5: insert Person "p3" → rule fires WORKS_AT p3→o1 (degree = 3) ← latest
14952 ///
14953 /// Each rule-fire also appends a DerivedEdgeAdded history-marker frame (state
14954 /// no-op), so the total commit count is higher than the base frame count.
14955 /// The "latest" open_at commit is computed dynamically via `wal_commit_count_at`.
14956 ///
14957 /// Without `rebuild_all`, the as-of instance's "emp" view stays at the
14958 /// initial backfill value (0) instead of reflecting the replayed derived edges.
14959 #[test]
14960 fn open_at_derived_edge_view_values_correct() {
14961 use core_rules::ViewSource;
14962 let dir = tmp_dir("open-at-view-rebuild");
14963 {
14964 let mut db = GraphDb::open(&dir).unwrap();
14965 // frame 0
14966 db.insert_node("Org", "o1", vec![]).unwrap();
14967 // frame 1: create view — initial backfill sees 0 derived edges (none fired yet)
14968 db.create_view(ViewDef {
14969 name: "employee_count".into(),
14970 label: "Org".into(),
14971 view_prop: "emp".into(),
14972 source: ViewSource::Degree {
14973 edge_type: "WORKS_AT".into(),
14974 direction: Direction::In,
14975 },
14976 })
14977 .unwrap();
14978 // frame 2: create rule — no Persons yet; backfill is a no-op
14979 db.create_rule(fk_rule()).unwrap();
14980 // frame 3: p1 — rule fires WORKS_AT p1→o1; degree = 1
14981 db.insert_node(
14982 "Person",
14983 "p1",
14984 vec![("org_id".into(), Value::Str("o1".into()))],
14985 )
14986 .unwrap();
14987 // frame 4: p2 — degree = 2
14988 db.insert_node(
14989 "Person",
14990 "p2",
14991 vec![("org_id".into(), Value::Str("o1".into()))],
14992 )
14993 .unwrap();
14994 // frame 5: p3 — degree = 3
14995 db.insert_node(
14996 "Person",
14997 "p3",
14998 vec![("org_id".into(), Value::Str("o1".into()))],
14999 )
15000 .unwrap();
15001 // Sanity: normal open sees degree = 3.
15002 assert_eq!(
15003 db.get_view_prop("o1", "emp"),
15004 Some(Value::Int(3)),
15005 "normal db must show degree 3 after 3 derived edges"
15006 );
15007 } // WAL flushed
15008
15009 // Re-open normally to get the authoritative reference value.
15010 let normal_db = GraphDb::open(&dir).unwrap();
15011 let normal_emp = normal_db.get_view_prop("o1", "emp");
15012 assert_eq!(
15013 normal_emp,
15014 Some(Value::Int(3)),
15015 "re-opened normal db must show degree 3"
15016 );
15017
15018 // Latest as-of (last WAL commit): must match the normal open.
15019 // History-marker frames are appended after each rule-fire, so the total
15020 // commit count is computed dynamically rather than hardcoded.
15021 let total = crate::wal_commit_count_at(&dir).unwrap();
15022 let aof_latest = GraphDb::open_at(&dir, total - 1).unwrap();
15023 assert_eq!(
15024 aof_latest.get_view_prop("o1", "emp"),
15025 normal_emp,
15026 "open_at latest: derived-edge view must equal normal open (rebuild_all required)"
15027 );
15028
15029 // Mid-history as-of (commit 3 = p1 insert Batch frame): only p1; degree = 1.
15030 // The DerivedEdgeAdded marker for p1 is at frame 4 (state no-op on replay),
15031 // so replaying 0..=3 correctly re-derives only the p1→o1 edge.
15032 let aof_mid = GraphDb::open_at(&dir, 3).unwrap();
15033 assert_eq!(
15034 aof_mid.get_view_prop("o1", "emp"),
15035 Some(Value::Int(1)),
15036 "open_at mid-history: only p1 exists at frame 3, degree must be 1"
15037 );
15038
15039 let _ = std::fs::remove_dir_all(&dir);
15040 }
15041
15042 /// Pin: subscribe_* on an as-of instance must return Err(ReadOnly) —
15043 /// as-of instances never commit, so distribute_events never runs and any
15044 /// subscription would wait forever.
15045 #[test]
15046 fn subscribe_on_as_of_returns_read_only_error() {
15047 let dir = tmp_dir("sub-as-of-read-only");
15048 {
15049 let mut db = GraphDb::open(&dir).unwrap();
15050 db.insert_node("Org", "o1", vec![]).unwrap();
15051 db.create_rule(fk_rule()).unwrap();
15052 }
15053 let mut aof = GraphDb::open_at(&dir, 0).unwrap();
15054
15055 assert!(
15056 matches!(
15057 aof.subscribe_all_rules(),
15058 Err(core_storage::GraphError::ReadOnly)
15059 ),
15060 "subscribe_all_rules on as-of must return ReadOnly"
15061 );
15062 assert!(
15063 matches!(
15064 aof.subscribe_writes(),
15065 Err(core_storage::GraphError::ReadOnly)
15066 ),
15067 "subscribe_writes on as-of must return ReadOnly"
15068 );
15069 assert!(
15070 matches!(
15071 aof.subscribe_rule("works_at"),
15072 Err(core_storage::GraphError::ReadOnly)
15073 ),
15074 "subscribe_rule on as-of must return ReadOnly"
15075 );
15076 let _ = std::fs::remove_dir_all(&dir);
15077 }
15078
15079 /// Regression: a failed dense WAL rewrite must not leave speculative
15080 /// interns in `syms`. If it does, the next successful mutation logs an
15081 /// `Intern` record with an inflated id; replay (which never saw the
15082 /// orphans) assigns a smaller id and the WAL becomes unreplayable.
15083 #[test]
15084 fn dense_rewrite_error_rolls_back_speculative_interns() {
15085 let dir = tmp_dir("dense-rewrite-rollback");
15086 {
15087 let mut db = GraphDb::open(&dir).unwrap();
15088 db.insert_node("Person", "a", vec![]).unwrap();
15089
15090 // Bypass MutPreview validation to hit the rewrite's own error path
15091 // (same shape as an id-exhaustion failure mid-rewrite). The
15092 // InsertEdge arm interns the edge type before it resolves keys.
15093 let err = db.rewrite_wal_dense(vec![WalRecord::InsertEdge {
15094 edge_type: "ORPHAN_TYPE".into(),
15095 src_key: "missing".into(),
15096 dst_key: "a".into(),
15097 }]);
15098 assert!(err.is_err(), "rewrite of a missing src key must fail");
15099 assert_eq!(
15100 db.syms.get("ORPHAN_TYPE"),
15101 None,
15102 "failed rewrite must roll back speculative interns"
15103 );
15104
15105 // A later successful mutation must produce a replayable WAL.
15106 db.set_prop("a", "later_field", Value::Int(2)).unwrap();
15107 }
15108 let db = GraphDb::open(&dir).expect("WAL must replay after failed rewrite");
15109 assert_eq!(db.get_prop("a", "later_field"), Some(Value::Int(2)));
15110 let _ = std::fs::remove_dir_all(&dir);
15111 }
15112}