core_api/db.rs
1use crate::ingest::{IngestOptions, IngestReport};
2use crate::roles::{PropPredicate, RoleDef, RolesFile, WriteScope};
3use crate::subscription::{
4 event_matches, DbEvent, SubEntry, SubFilter, SubInner, Subscription, DEFAULT_SUB_CAPACITY,
5};
6use core_query::cypher::ast::{ret_val_label, ArithOp};
7use core_query::cypher::{
8 execute, execute_union, is_subscribable, is_write_tokens, lex, parse, parse_read, parse_write,
9 plan, Expr, MatchDeleteNodeStmt, NodePat, Operand, Params, Pattern, PlanOp, Query, RetItem,
10 RetVal, WriteStatement,
11};
12use core_query::{eval_cmp, eval_filter, expand, neighborhood, Dir, Filter, GraphView, ResultSet};
13use core_rules::{
14 decode_rule_def, ef_max, evaluate, BuildProgress, EngineEdgeDelta, GraphMut, NodeView,
15 Predicate, RuleDef, RuleEngine, ViewDef, ViewStore,
16};
17use core_storage::fs::{FileId, Fs, FsIntrospect, RealFs};
18use core_storage::fulltext::FulltextIndex;
19use core_storage::property_index::PropertyIndex;
20use core_storage::v8::encode::{
21 archived_hnsw_to_owned, archived_rules_meta_to_owned, archived_to_idmap, archived_to_interner,
22 archived_views_to_owned, decode_last_change_bytes, decode_meta, encode_v8, V8Meta,
23};
24use core_storage::v8::seam::TopologyView;
25use core_storage::wal::{decode_all, encode_record, WalRecord};
26use core_storage::EdgePropsView;
27use core_storage::{
28 namespace_of_value, ColumnStore, Direction, EdgeProps, GraphError, IdMap, Interner, Result,
29 Topology, Value,
30};
31pub use core_storage::{valid_namespace, NS_DEFAULT, NS_MAX_LEN, NS_PROP};
32
33/// Index of [`NS_DEFAULT`] in `GraphDb::ns_names` — always zero, so the
34/// open-time pass over a store with no `ns` column fills `node_ns` with one
35/// constant and allocates no names.
36const NS_DEFAULT_IDX: u32 = 0;
37
38/// The reserved edge property holding a pair's insert count (§5.13).
39///
40/// Absent means 1 — the count is written only from the second insert of a
41/// triple onward, and only on a store that called
42/// [`GraphDb::enable_multiplicity`]. The engine owns the name: Cypher `SET` on
43/// it is refused, as the other reserved names are.
44pub const EDGE_COUNT_PROP: &str = "count";
45use serde::{Deserialize, Serialize};
46use std::collections::{BTreeMap, BTreeSet, HashMap, HashSet};
47use std::sync::Arc;
48
49/// Print a timing checkpoint when MUSHROOMDB_TRACE_OPEN is set.
50/// Zero-cost when the env var is absent (the var check is O(1) after first call).
51macro_rules! trace_open {
52 ($phase:literal, $t:expr) => {
53 if std::env::var("MUSHROOMDB_TRACE_OPEN").is_ok() {
54 eprintln!(
55 "[MUSHROOMDB_TRACE_OPEN] {:40} {:>9.3?}",
56 $phase,
57 $t.elapsed()
58 );
59 }
60 };
61}
62
63/// Print a migration phase checkpoint when MUSHROOMDB_TRACE_MIGRATE is set.
64/// Zero-cost when the env var is absent (the var check is O(1) after first call).
65macro_rules! trace_migrate {
66 ($phase:literal, $t:expr) => {
67 if std::env::var("MUSHROOMDB_TRACE_MIGRATE").is_ok() {
68 eprintln!(
69 "[MUSHROOMDB_TRACE_MIGRATE] {:40} {:>9.3?}",
70 $phase,
71 $t.elapsed()
72 );
73 }
74 };
75}
76
77// Test-only: counts how many times `pending_deltas_since().to_vec()` actually
78// executes (i.e., at least one view is defined). Used to verify the fast-path
79// guard skips the allocation when `view_store.is_empty()`.
80#[cfg(test)]
81thread_local! {
82 static DELTA_COPY_COUNT: std::cell::Cell<usize> = const { std::cell::Cell::new(0) };
83}
84
85// Per-thread count of query-subscription `execute` calls in `distribute_events`.
86//
87// Incremented each time a query subscription actually runs its plan (i.e.,
88// the label-skip fast-path did not fire). Because `distribute_events` is
89// called synchronously on the writer thread, this thread-local correctly
90// isolates each test thread's count even when integration tests run in
91// parallel. Read via [`query_sub_exec_count`].
92thread_local! {
93 static QUERY_SUB_EXECS_TL: std::cell::Cell<usize> = const { std::cell::Cell::new(0) };
94}
95
96/// Return the number of query-subscription re-executions logged on this
97/// thread since the process started (or since last reset via
98/// [`reset_query_sub_exec_count`]).
99///
100/// Primarily for integration tests that verify the label-skip fast-path.
101#[doc(hidden)]
102pub fn query_sub_exec_count() -> usize {
103 QUERY_SUB_EXECS_TL.with(|c| c.get())
104}
105
106/// Reset the per-thread query-subscription execution counter to zero.
107#[doc(hidden)]
108pub fn reset_query_sub_exec_count() {
109 QUERY_SUB_EXECS_TL.with(|c| c.set(0));
110}
111
112// Exact-versus-approximate warnings emitted on this thread. Thread-local for
113// the same reason [`QUERY_SUB_EXECS_TL`] is: integration tests run in parallel
114// and each gets its own thread, so a neighbour's masked search cannot be
115// mistaken for this test's.
116thread_local! {
117 static AMBIGUOUS_EXACTNESS_WARNS: std::cell::Cell<u64> = const { std::cell::Cell::new(0) };
118 static AMBIGUOUS_EXACTNESS_LAST: std::cell::RefCell<Option<String>> =
119 const { std::cell::RefCell::new(None) };
120}
121
122/// How many times a masked, non-exact vector search has explained itself on
123/// this thread since the last [`ambiguous_exactness_warns_reset`].
124///
125/// The line itself is the product; this counter exists so a test can assert it
126/// is printed **once per index** rather than once per call.
127///
128/// **Single-threaded assertions only.** The suppression set this counts is a
129/// `Mutex<HashSet<_>>` on the `GraphDb` — shared by every thread — while the
130/// counter is thread-local. Under a concurrent caller (`serve`, which is the
131/// deployment the warning exists for) the thread that prints the line is not
132/// necessarily the thread that asked, so a zero here does not mean the line was
133/// not printed and a one does not mean it was printed once. It answers
134/// "once per index" only in a test that owns the store.
135#[doc(hidden)]
136pub fn ambiguous_exactness_warns() -> u64 {
137 AMBIGUOUS_EXACTNESS_WARNS.with(|c| c.get())
138}
139
140/// The most recent exactness warning printed on this thread, verbatim.
141///
142/// The advice a caller reads has to be advice that caller can act on, which the
143/// counter alone cannot witness — see
144/// `the_hybrid_path_advises_a_call_the_hybrid_caller_can_make`. Carries the
145/// same single-threaded caveat as [`ambiguous_exactness_warns`].
146#[doc(hidden)]
147pub fn ambiguous_exactness_last_warning() -> Option<String> {
148 AMBIGUOUS_EXACTNESS_LAST.with(|c| c.borrow().clone())
149}
150
151/// Reset this thread's exactness-warning counter and recorded line.
152#[doc(hidden)]
153pub fn ambiguous_exactness_warns_reset() {
154 AMBIGUOUS_EXACTNESS_WARNS.with(|c| c.set(0));
155 AMBIGUOUS_EXACTNESS_LAST.with(|c| *c.borrow_mut() = None);
156}
157
158/// Which signature reached the approximate masked vector leg.
159///
160/// One code path, two entry points, and the advice cannot be the same: a
161/// warning that names an argument the caller's function does not take sends
162/// them looking for a parameter that is not there. `search_hybrid` and
163/// `search_hybrid_scoped` take `(text_field, query_text, vector_field,
164/// query_vec, label, k[, mask])` — no `exact`, no `where`.
165///
166/// The warning is not suppressed on the hybrid path. The approximation is the
167/// same one, and a caller who read `mask=` as a promise of exhaustiveness is
168/// the reader it was written for whichever door they came in by; only the
169/// remedy differs, so only the remedy changes.
170#[derive(Clone, Copy, PartialEq, Eq, Hash, Debug)]
171enum ExactnessCaller {
172 /// `find_similar` / `find_similar_vector_*` — `exact` and `where` are its
173 /// own parameters.
174 Vector,
175 /// `search_hybrid` / `search_hybrid_scoped` — neither argument exists, and
176 /// the leg is one half of a fusion.
177 Hybrid,
178}
179
180impl ExactnessCaller {
181 fn subject(self) -> &'static str {
182 match self {
183 Self::Vector => "a masked vector search",
184 Self::Hybrid => "the vector leg of a masked hybrid search",
185 }
186 }
187
188 fn advice(self) -> &'static str {
189 match self {
190 Self::Vector => "pass exact=True or a where= predicate.",
191 // Names the call that does take the argument, because this one
192 // does not: the caller's own next step, not a parameter hunt.
193 Self::Hybrid => {
194 "run the vector leg on its own with find_similar(field, vector, mask=…, \
195 exact=True) and fuse it with search() yourself — search_hybrid itself \
196 takes no exactness argument."
197 }
198 }
199 }
200}
201
202/// Internal state for a single `subscribe_query` subscription.
203///
204/// On every commit, `distribute_events` re-executes `ops` against the current
205/// graph state, diffs the result against `prev_rows`, and pushes
206/// `DbEvent::QueryRowAdded` / `QueryRowRemoved` events to `inner`.
207///
208/// **Full re-run per commit; use LIMIT to bound execution cost.**
209/// (Differential evaluation is roadmap / Phase 5.)
210pub(crate) struct QuerySubEntry {
211 /// Compiled plan for the subscribed Cypher query.
212 ops: Vec<PlanOp>,
213 /// Column names from the first execution (fixed for the subscription lifetime).
214 columns: Vec<String>,
215 /// Serialized (JSON) row key → row data, representing the result set at
216 /// the end of the last commit. Used to diff against the new result.
217 prev_row_map: std::collections::HashMap<String, Vec<Option<Value>>>,
218 /// Weak pointer to the subscriber queue; dead Weak → subscription dropped.
219 inner: std::sync::Weak<SubInner>,
220 /// Interned label sym captured at subscribe time from the plan's leading scan
221 /// (`ScanLabel`, `IndexScan`, or `IndexIntersect` with a concrete label).
222 ///
223 /// `None` means the plan has an `Expand` op (or no recognizable leading scan
224 /// with a concrete label), and this subscription must re-execute on every
225 /// commit without skipping. This is the conservative v0.4.3 boundary: Expand
226 /// queries are never skipped because edges can alter join results regardless
227 /// of which node labels were written.
228 scan_label: Option<u32>,
229}
230
231/// A post-commit mutation notification.
232///
233/// Emitted from `log_then_apply` after the WAL append, fsync, and
234/// in-memory `apply` all succeed. Never emitted for rejected operations
235/// (validation errors, [`GraphError::RuleOwned`], duplicate keys, no-op
236/// deletes/removes). Event payloads carry user keys and rule names, never
237/// internal ids.
238///
239/// **Replay:** [`GraphDb::open`] / [`GraphDb::open_with`] replay the WAL via
240/// `apply` only. Emission lives exclusively in `log_then_apply`, so
241/// recovery is silent even if a sink were installed (it cannot be: the
242/// sink is in-memory and set after open).
243///
244/// **Ordering:** a `Batch` WAL frame emits one event per inner record, then
245/// [`MutationEvent::BatchApplied`]. An ingest commit emits those same inner
246/// events, then [`MutationEvent::Ingested`] (not `BatchApplied`). An empty
247/// or all-noop batch writes no WAL and emits nothing (including no summary).
248///
249/// **Derived edges:** rule-created or retracted edges are not individually
250/// evented — they are recoverable from the triggering mutation plus the live
251/// rule set. Only the triggering record is emitted.
252///
253/// **Wire form:** externally tagged snake_case JSON
254/// (`{"node_inserted":{"label":"A","key":"k"}}`).
255#[derive(Debug, Clone, PartialEq, Eq, Serialize)]
256#[serde(rename_all = "snake_case")]
257pub enum MutationEvent {
258 NodeInserted {
259 label: String,
260 key: String,
261 },
262 PropSet {
263 key: String,
264 field: String,
265 },
266 PropRemoved {
267 key: String,
268 field: String,
269 },
270 EdgeInserted {
271 edge_type: String,
272 src: String,
273 dst: String,
274 },
275 EdgeDeleted {
276 edge_type: String,
277 src: String,
278 dst: String,
279 },
280 NodeDeleted {
281 key: String,
282 },
283 RuleCreated {
284 name: String,
285 },
286 RuleDeleted {
287 name: String,
288 },
289 RuleRebuilt {
290 name: String,
291 },
292 BatchApplied {
293 ops: usize,
294 },
295 Ingested {
296 label: String,
297 inserted: usize,
298 },
299}
300
301fn event_from_record(rec: &WalRecord, intern: &Interner, ids: &IdMap) -> Option<MutationEvent> {
302 match rec {
303 WalRecord::InsertNode { label, key, .. } => Some(MutationEvent::NodeInserted {
304 label: label.clone(),
305 key: key.clone(),
306 }),
307 WalRecord::InsertNodeId { label, key, .. } => Some(MutationEvent::NodeInserted {
308 label: intern.resolve(*label)?.to_string(),
309 key: key.clone(),
310 }),
311 WalRecord::SetProp { key, field, .. } => Some(MutationEvent::PropSet {
312 key: key.clone(),
313 field: field.clone(),
314 }),
315 WalRecord::SetPropId { id, field, .. } => Some(MutationEvent::PropSet {
316 key: ids.key_of(*id)?.to_string(),
317 field: intern.resolve(*field)?.to_string(),
318 }),
319 WalRecord::RemoveProp { key, field } => Some(MutationEvent::PropRemoved {
320 key: key.clone(),
321 field: field.clone(),
322 }),
323 WalRecord::InsertEdge {
324 edge_type,
325 src_key,
326 dst_key,
327 } => Some(MutationEvent::EdgeInserted {
328 edge_type: edge_type.clone(),
329 src: src_key.clone(),
330 dst: dst_key.clone(),
331 }),
332 WalRecord::InsertEdgeId { etype, src, dst } => Some(MutationEvent::EdgeInserted {
333 edge_type: intern.resolve(*etype)?.to_string(),
334 src: ids.key_of(*src)?.to_string(),
335 dst: ids.key_of(*dst)?.to_string(),
336 }),
337 WalRecord::DeleteEdge {
338 edge_type,
339 src_key,
340 dst_key,
341 } => Some(MutationEvent::EdgeDeleted {
342 edge_type: edge_type.clone(),
343 src: src_key.clone(),
344 dst: dst_key.clone(),
345 }),
346 WalRecord::DeleteNode { key } => Some(MutationEvent::NodeDeleted { key: key.clone() }),
347 WalRecord::CreateRule { def_bytes } => {
348 let def: RuleDef = decode_rule_def(def_bytes).ok()?;
349 Some(MutationEvent::RuleCreated { name: def.name })
350 }
351 WalRecord::DeleteRule { name } => Some(MutationEvent::RuleDeleted { name: name.clone() }),
352 WalRecord::RebuildRule { name } => Some(MutationEvent::RuleRebuilt { name: name.clone() }),
353 WalRecord::Batch(_)
354 | WalRecord::CreateView { .. }
355 | WalRecord::DeleteView { .. }
356 | WalRecord::EnableFulltext { .. }
357 | WalRecord::DisableFulltext { .. }
358 | WalRecord::EnableIndex { .. }
359 | WalRecord::DisableIndex { .. }
360 | WalRecord::Intern { .. }
361 // History markers are no-ops for mutation events — they carry no new
362 // state and rules re-derive deterministically on replay.
363 | WalRecord::DerivedEdgeAdded { .. }
364 | WalRecord::DerivedEdgeRetracted { .. }
365 // A count changes neither the node nor the edge population: the pair it
366 // counts was already there, which is why it is written at all.
367 | WalRecord::SetEdgeCount { .. }
368 // RenameNode carries no node/edge count change; no special event.
369 | WalRecord::RenameNode { .. } => None,
370 }
371}
372
373/// Database-wide counters plus per-rule budget/fire stats.
374#[derive(Debug, Clone, PartialEq, Eq, Serialize)]
375pub struct Stats {
376 pub nodes_live: usize,
377 pub nodes_tombstoned: usize,
378 pub edges: u64,
379 pub rules: Vec<RuleStats>,
380 /// How many writes hit the rule-chaining depth cap with work still pending,
381 /// since this handle was opened. Non-zero means some derived edges beyond
382 /// the cap are stale and no single later write will repair them: split the
383 /// rule chain or shorten it. Never persisted, so it resets on reopen.
384 #[serde(default)]
385 pub chain_truncations: u64,
386 /// The oldest commit index history still reaches (the WAL horizon floor).
387 /// `0` means nothing has been pruned and history is complete; a non-zero
388 /// value means events before that commit were pruned and are gone.
389 #[serde(default)]
390 pub history_floor: u64,
391 /// Live node counts per namespace, in name order. Always carries
392 /// `default` — a store is at least its default namespace — so a
393 /// single-tenant store reads `[{"name":"default", …}]` and a reader can
394 /// tell "no namespaces in use" from one entry.
395 #[serde(default)]
396 pub namespaces: Vec<NamespaceStats>,
397}
398
399/// Live node count for one namespace; one entry of [`Stats::namespaces`].
400#[derive(Debug, Clone, PartialEq, Eq, Serialize)]
401pub struct NamespaceStats {
402 pub name: String,
403 pub nodes_live: usize,
404}
405
406/// One rule's provenance size, trip latch, and fire counter.
407///
408/// `tripped` is a one-way latch: once set, the engine adds no new edges for
409/// that rule until [`GraphDb::rebuild_rule`] (and only if the full desired
410/// set then fits). `fires` counts `on_node_changed` evaluations plus
411/// backfill/rebuild participant ticks (rebuild counts even when it is a
412/// provenance no-op).
413#[derive(Debug, Clone, PartialEq, Eq, Serialize)]
414pub struct RuleStats {
415 pub name: String,
416 pub edges: u64,
417 pub tripped: bool,
418 pub fires: u64,
419 /// Whether this rule uses the approximate IVF-Flat candidate path.
420 pub approximate: bool,
421 /// `Some` while this rule's vector index is still being built.
422 ///
423 /// The rule derives **no** edges until it is `None`: the backfill is one
424 /// commit that runs after the index is whole, so a caller never sees a
425 /// partial edge set. Absent from the JSON when the rule is not building,
426 /// which is every rule created over a corpus at or below
427 /// [`core_rules::HNSW_BUILD_BATCH`] vectors.
428 #[serde(default, skip_serializing_if = "Option::is_none")]
429 pub building: Option<BuildProgress>,
430}
431
432/// One entry in the slow-query ring buffer.
433#[derive(Debug, Clone, Serialize)]
434pub struct SlowQueryEntry {
435 /// Execution time in whole milliseconds.
436 pub ms: u64,
437 /// The Cypher query string that was slow.
438 pub query: String,
439 /// The commit sequence number at the time the query ran.
440 pub at_commit: u64,
441}
442
443/// Snapshot of the slow-query log returned by [`GraphDb::slow_query_snapshot`].
444#[derive(Debug, Clone, Serialize)]
445pub struct SlowQuerySnapshot {
446 /// Current threshold in milliseconds (0 = disabled).
447 pub threshold_ms: u64,
448 /// Total number of slow queries ever recorded (not capped by ring size).
449 pub count: u64,
450 /// Most-recent slow queries (up to 16), oldest first.
451 pub last: Vec<SlowQueryEntry>,
452}
453
454/// Internal ring-buffer state protected by a `Mutex` so `query(&self)` can
455/// write to it without a mutable borrow.
456struct SlowQueryLog {
457 entries: std::collections::VecDeque<SlowQueryEntry>,
458 total: u64,
459}
460
461/// Maximum number of entries kept in the slow-query ring buffer.
462const SLOW_QUERY_RING_CAP: usize = 16;
463
464/// Wire summary of a [`Predicate`]. JSON only — `Explanation` is never
465/// bincode-persisted (WAL/snapshots store `RuleDef` bytes, not this type).
466#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
467pub struct PredicateSummary {
468 pub kind: String,
469 pub fields: Vec<String>,
470 pub min: Option<f64>,
471 pub tolerance: Option<f64>,
472 pub km: Option<f64>,
473 pub parts: Option<Vec<PredicateSummary>>,
474 /// True when the owning rule has `approximate=true` (IVF-Flat candidate path).
475 /// Always false for predicates reported without rule context (sub-predicates in `parts`).
476 #[serde(default)]
477 pub approximate: bool,
478}
479
480impl From<&Predicate> for PredicateSummary {
481 fn from(p: &Predicate) -> Self {
482 match p {
483 Predicate::KeyMatch { field } => PredicateSummary {
484 kind: "key_match".into(),
485 fields: vec![field.clone()],
486 min: None,
487 tolerance: None,
488 km: None,
489 parts: None,
490 approximate: false,
491 },
492 Predicate::FieldEqual { field } => PredicateSummary {
493 kind: "field_equal".into(),
494 fields: vec![field.clone()],
495 min: None,
496 tolerance: None,
497 km: None,
498 parts: None,
499 approximate: false,
500 },
501 Predicate::Overlap { field, min } => PredicateSummary {
502 kind: "overlap".into(),
503 fields: vec![field.clone()],
504 min: Some(*min),
505 tolerance: None,
506 km: None,
507 parts: None,
508 approximate: false,
509 },
510 Predicate::NumericWithin { field, tolerance } => PredicateSummary {
511 kind: "numeric_within".into(),
512 fields: vec![field.clone()],
513 min: None,
514 tolerance: Some(*tolerance),
515 km: None,
516 parts: None,
517 approximate: false,
518 },
519 Predicate::GeoRadius { field, km } => PredicateSummary {
520 kind: "geo_radius".into(),
521 fields: vec![field.clone()],
522 min: None,
523 tolerance: None,
524 km: Some(*km),
525 parts: None,
526 approximate: false,
527 },
528 Predicate::VectorSimilar { field, min } => PredicateSummary {
529 kind: "vector_similar".into(),
530 fields: vec![field.clone()],
531 min: Some(*min),
532 tolerance: None,
533 km: None,
534 parts: None,
535 approximate: false,
536 },
537 Predicate::All(inner) => {
538 let parts: Vec<PredicateSummary> = inner.iter().map(Self::from).collect();
539 let mut fields = Vec::new();
540 for part in &parts {
541 for f in &part.fields {
542 if !fields.contains(f) {
543 fields.push(f.clone());
544 }
545 }
546 }
547 PredicateSummary {
548 kind: "all".into(),
549 fields,
550 min: None,
551 tolerance: None,
552 km: None,
553 parts: Some(parts),
554 approximate: false,
555 }
556 }
557 Predicate::Any(inner) => {
558 let parts: Vec<PredicateSummary> = inner.iter().map(Self::from).collect();
559 let mut fields = Vec::new();
560 for part in &parts {
561 for f in &part.fields {
562 if !fields.contains(f) {
563 fields.push(f.clone());
564 }
565 }
566 }
567 PredicateSummary {
568 kind: "any".into(),
569 fields,
570 min: None,
571 tolerance: None,
572 km: None,
573 parts: Some(parts),
574 approximate: false,
575 }
576 }
577 }
578 }
579}
580
581/// Snapshot of a live node's key, label, and columnar properties.
582///
583/// `props` is a [`BTreeMap`] so field order is deterministic (sorted by name)
584/// regardless of insert order or the columnar store's `HashMap` iteration.
585///
586/// Deliberately does not derive `Serialize`: `Value`'s serde form is
587/// internally tagged. Wire JSON is built by `value_to_json` in the server.
588#[derive(Debug, Clone, PartialEq)]
589pub struct NodeInfo {
590 pub key: String,
591 pub label: String,
592 pub props: BTreeMap<String, Value>,
593}
594
595/// Counts returned by [`GraphDb::delete_node`].
596#[derive(Debug, Clone, PartialEq, Eq, Default)]
597pub struct DeleteReport {
598 /// Number of manual (user-inserted) edges removed.
599 pub manual_edges: u64,
600 /// Number of derived (rule-owned) edges retracted.
601 pub derived_edges: u64,
602}
603
604/// One directed edge incident on a node, with provenance membership.
605///
606/// `derived` is true iff `(edge_type, src, dst)` is in the rule engine's
607/// Plan-8 `by_node` provenance index.
608#[derive(Debug, Clone, PartialEq, Eq, Serialize)]
609pub struct EdgeInfo {
610 pub edge_type: String,
611 pub src_key: String,
612 pub dst_key: String,
613 pub derived: bool,
614}
615
616/// One directed edge incident on a node at a point in WAL history, with the
617/// rule that derived it when it is rule-owned.
618///
619/// Returned by [`GraphDb::edges_at`] (sorted by `(edge_type, src_key, dst_key)`)
620/// and by [`GraphDb::what_if_set_prop`].
621#[derive(Debug, Clone, PartialEq, Eq, PartialOrd, Ord, Serialize)]
622pub struct EdgeAt {
623 pub edge_type: String,
624 pub src_key: String,
625 pub dst_key: String,
626 /// `true` when a rule wrote the edge (`DerivedEdgeAdded` in the WAL, or a
627 /// live provenance entry).
628 pub derived: bool,
629 /// The rule that derived the edge. `None` for a manual edge.
630 pub rule: Option<String>,
631}
632
633/// The derived edges a hypothetical property change would retract and derive.
634///
635/// Returned by [`GraphDb::what_if_set_prop`]. Both lists are sorted by
636/// `(edge_type, src_key, dst_key)` and every entry is rule-derived.
637#[derive(Debug, Clone, PartialEq, Eq, Serialize)]
638pub struct WhatIf {
639 /// Derived edges that exist now and would be retracted.
640 pub lost: Vec<EdgeAt>,
641 /// Derived edges that do not exist now and would be derived.
642 pub gained: Vec<EdgeAt>,
643}
644
645/// An edge with mask-aware endpoint visibility.
646///
647/// Returned by [`GraphDb::node_edges_masked`] in [`crate::mask::MaskMode::Stub`]
648/// mode — hidden endpoints carry `*_restricted: true`.
649#[derive(Debug, Clone, PartialEq, Eq)]
650pub struct MaskedEdge {
651 pub edge_type: String,
652 pub src_key: String,
653 /// `true` when `src_key` is in the DB but hidden from the mask.
654 pub src_restricted: bool,
655 pub dst_key: String,
656 /// `true` when `dst_key` is in the DB but hidden from the mask.
657 pub dst_restricted: bool,
658 pub derived: bool,
659}
660
661/// Result of a mask-aware node lookup via [`GraphDb::node_info_masked`].
662///
663/// `None` from that method means the key does not exist (→ 404).
664/// `Some(Restricted)` is only produced when `mask.mode() == MaskMode::Stub`.
665#[derive(Debug, PartialEq)]
666pub enum MaskedNodeResult {
667 Visible(NodeInfo),
668 /// Node exists in the DB but is hidden from this mask.
669 Restricted,
670}
671
672/// One rule-owned edge between two nodes, with the rule name, edge type,
673/// direction (src_key → dst_key), and weight if the rule stores one.
674#[derive(Debug, Clone, PartialEq, Serialize)]
675pub struct Explanation {
676 pub rule: String,
677 pub edge_type: String,
678 pub src_key: String,
679 pub dst_key: String,
680 pub weight: Option<f64>,
681 pub predicate: PredicateSummary,
682 /// For a via-hop rule, the edge type the rule hops over to reach its
683 /// candidates. `None` for a plain two-node rule. A via-hop rule whose
684 /// `via_edge` is itself rule-derived is the chaining case: the hop edge
685 /// was written by another rule in the same commit.
686 #[serde(default)]
687 pub via_edge: Option<String>,
688}
689
690/// Report returned by [`GraphDb::backup_to`].
691#[derive(Debug, Clone)]
692pub struct BackupReport {
693 /// Filenames copied into the destination directory (sorted ascending).
694 pub files: Vec<String>,
695 /// Total bytes written across all copied files.
696 pub bytes: u64,
697 /// `true` when the destination opened cleanly and passed post-copy checks.
698 ///
699 /// For stores that have a `snapshot.bin` this means: all V8 section CRCs
700 /// matched **and** the destination opened without error.
701 ///
702 /// For WAL-only stores (no `snapshot.bin`) there is no snapshot to
703 /// CRC-check; `verified` is `true` when the destination opened and
704 /// replayed the WAL without error (record-level checksums in the WAL
705 /// provide the integrity signal, not section CRCs).
706 pub verified: bool,
707}
708
709/// One directed edge in export form, with optional rule attribution for derived edges.
710///
711/// Returned by [`GraphDb::all_edges_for_export`].
712///
713/// Does not derive `Eq`/`Ord`: `weight` is an `f64` and NaN breaks a total
714/// order. Callers that need a stable edge ordering already sort by
715/// `(edge_type, src, dst)` explicitly (see `all_edges_for_export`).
716#[derive(Debug, Clone, PartialEq, PartialOrd)]
717pub struct ExportEdge {
718 pub edge_type: String,
719 pub src: String,
720 pub dst: String,
721 pub derived: bool,
722 /// Rule name that created this edge, if derived. `None` for manual edges.
723 pub rule: Option<String>,
724 /// The creating rule's declared `weight_prop`, read off this edge, when
725 /// derived and numeric (`Int`/`Float`). `None` for manual edges, derived
726 /// edges whose rule declares no `weight_prop`, or a non-numeric value.
727 pub weight: Option<f64>,
728}
729
730/// One edge type's shape, as [`GraphDb::edge_type_census`] counts it.
731///
732/// Deliberately per *type* and not per edge: everything here is a summary a
733/// caller can print in one line, and none of it costs a record per edge.
734#[derive(Debug, Clone, PartialEq, Eq)]
735pub struct EdgeTypeCensus {
736 pub edge_type: String,
737 /// Directed edges of this type. Counted the way
738 /// [`GraphDb::edge_count`] counts: each edge once, from its source.
739 pub edges: u64,
740 /// Every label seen on a source of this type, sorted.
741 pub src_labels: Vec<String>,
742 /// Every label seen on a destination of this type, sorted.
743 pub dst_labels: Vec<String>,
744 /// The rules that declare this `edge_type`, sorted. Empty for a type
745 /// written by hand.
746 pub rules: Vec<String>,
747 /// `(src key, dst key)` of the first edge of this type in the store's own
748 /// id order — a real pair to quote in an example.
749 pub sample: Option<(String, String)>,
750}
751
752/// Construct the standard write-query result set (columns: created, properties_set, deleted).
753fn write_result_set() -> ResultSet {
754 ResultSet::new(vec![
755 "created".into(),
756 "properties_set".into(),
757 "deleted".into(),
758 ])
759}
760
761fn resolve_merge_set_value(op: &Operand, params: &BTreeMap<String, Value>) -> Result<Value> {
762 match op {
763 Operand::Lit(v) => Ok(v.clone()),
764 Operand::Param(name) => params
765 .get(name)
766 .cloned()
767 .ok_or_else(|| GraphError::QueryError {
768 detail: format!("missing parameter `{name}`"),
769 }),
770 _ => Err(GraphError::QueryError {
771 detail: "ON CREATE/ON MATCH SET value must be a literal or $parameter".into(),
772 }),
773 }
774}
775
776fn operand_node_vars(op: &Operand, out: &mut Vec<String>) {
777 match op {
778 Operand::Prop { var, .. } | Operand::Var(var) => {
779 if !out.contains(var) {
780 out.push(var.clone());
781 }
782 }
783 Operand::FuncCall { args, .. } => {
784 for arg in args {
785 operand_node_vars(arg, out);
786 }
787 }
788 Operand::BinArith { left, right, .. } => {
789 operand_node_vars(left, out);
790 operand_node_vars(right, out);
791 }
792 Operand::Case { branches, default } => {
793 // Branch conditions reference vars already bound (and mask-filtered)
794 // by the MATCH phase, so collecting from the value operands + ELSE
795 // is sufficient for RETURN-projection var discovery.
796 for (_, value) in branches {
797 operand_node_vars(value, out);
798 }
799 if let Some(d) = default {
800 operand_node_vars(d, out);
801 }
802 }
803 Operand::Index { base, index } => {
804 operand_node_vars(base, out);
805 operand_node_vars(index, out);
806 }
807 Operand::Lit(_) | Operand::Param(_) => {}
808 }
809}
810
811fn ret_node_vars(items: &[RetItem]) -> Vec<String> {
812 let mut out = Vec::new();
813 for item in items {
814 match &item.value {
815 RetVal::Var(v) | RetVal::Prop { var: v, .. } => {
816 if !out.contains(v) {
817 out.push(v.clone());
818 }
819 }
820 RetVal::FuncCall { args, .. } => {
821 for arg in args {
822 operand_node_vars(arg, &mut out);
823 }
824 }
825 RetVal::ScalarExpr(op) => operand_node_vars(op, &mut out),
826 RetVal::Agg { .. } => {}
827 }
828 }
829 out
830}
831
832fn add_var(out: &mut Vec<String>, v: &str) {
833 if !out.iter().any(|x| x == v) {
834 out.push(v.to_string());
835 }
836}
837
838fn pattern_node_vars(pats: &[Pattern]) -> Vec<String> {
839 let mut out = Vec::new();
840 for p in pats {
841 if let Some(v) = &p.start.var {
842 add_var(&mut out, v);
843 }
844 for (_, dest) in &p.chain {
845 if let Some(v) = &dest.var {
846 add_var(&mut out, v);
847 }
848 }
849 }
850 out
851}
852
853fn pattern_rel_vars(pats: &[Pattern]) -> Vec<String> {
854 let mut out = Vec::new();
855 for p in pats {
856 for (rel, _) in &p.chain {
857 if rel.hops.is_none() {
858 if let Some(v) = &rel.var {
859 add_var(&mut out, v);
860 }
861 }
862 }
863 }
864 out
865}
866
867fn rel_type_alias(var: &str) -> String {
868 format!("__rt_{var}")
869}
870
871fn ret_column_name(item: &RetItem) -> String {
872 if let Some(alias) = &item.alias {
873 return alias.clone();
874 }
875 // The same naming rule the planner and the executor use, so a
876 // write-statement RETURN names its columns exactly as a read query does.
877 // An aggregate is not legal in a write-statement RETURN; it keeps the
878 // placeholder it always had.
879 ret_val_label(&item.value).unwrap_or_else(|| "<agg>".to_string())
880}
881
882fn eval_set_return_operand<F: Fs>(
883 db: &GraphDb<F>,
884 match_rs: &ResultSet,
885 row: usize,
886 rel_vars: &[String],
887 op: &Operand,
888 params: &BTreeMap<String, Value>,
889) -> Result<Option<Value>> {
890 match op {
891 Operand::Lit(v) => Ok(Some(v.clone())),
892 Operand::Param(name) => params.get(name).cloned().ok_or_else(|| GraphError::QueryError {
893 detail: format!("missing parameter `{name}`"),
894 }).map(Some),
895 Operand::Var(name) if rel_vars.iter().any(|r| r == name) => Err(GraphError::QueryError {
896 detail: format!(
897 "cannot return relationship variable '{name}' bare; return its properties ({name}.field) instead"
898 ),
899 }),
900 Operand::Var(name) => Ok(match_rs.get(row, name).cloned()),
901 Operand::Prop { var, field } => {
902 if rel_vars.iter().any(|r| r == var) {
903 return Ok(None);
904 }
905 let Some(Value::Str(key)) = match_rs.get(row, var) else {
906 return Ok(None);
907 };
908 if let Some(v) = db.get_prop(key, field) {
909 return Ok(Some(v));
910 }
911 // Same stored-wins identity fallback as the read path:
912 // n.key / n.id / n.label, not only get_prop.
913 Ok(match field.as_str() {
914 "key" | "id" => Some(Value::Str(key.clone())),
915 "label" => db
916 .node_ref(key)
917 .map(|n| Value::Str(n.label().to_owned())),
918 _ => None,
919 })
920 }
921 Operand::FuncCall { name, args } => {
922 eval_set_return_func(db, match_rs, row, rel_vars, name, args, params)
923 }
924 Operand::BinArith { op, left, right } => {
925 let lv = eval_set_return_operand(db, match_rs, row, rel_vars, left, params)?;
926 let rv = eval_set_return_operand(db, match_rs, row, rel_vars, right, params)?;
927 eval_set_return_arith(op, lv, rv)
928 }
929 Operand::Case { branches, default } => {
930 for (cond, value) in branches {
931 if eval_set_return_expr(db, match_rs, row, rel_vars, cond, params, 0)? {
932 return eval_set_return_operand(db, match_rs, row, rel_vars, value, params);
933 }
934 }
935 match default {
936 Some(d) => eval_set_return_operand(db, match_rs, row, rel_vars, d, params),
937 None => Ok(None),
938 }
939 }
940 Operand::Index { base, index } => {
941 let base_val = eval_set_return_operand(db, match_rs, row, rel_vars, base, params)?;
942 let idx_val = eval_set_return_operand(db, match_rs, row, rel_vars, index, params)?;
943 Ok(core_query::value_ops::index_list(base_val, idx_val))
944 }
945 }
946}
947
948fn eval_set_return_expr<F: Fs>(
949 db: &GraphDb<F>,
950 match_rs: &ResultSet,
951 row: usize,
952 rel_vars: &[String],
953 expr: &Expr,
954 params: &BTreeMap<String, Value>,
955 depth: u32,
956) -> Result<bool> {
957 if depth > 256 {
958 return Err(GraphError::QueryError {
959 detail: "expression nesting too deep".into(),
960 });
961 }
962 match expr {
963 Expr::And(lhs, rhs) => {
964 let l = eval_set_return_expr(db, match_rs, row, rel_vars, lhs, params, depth + 1)?;
965 let r = eval_set_return_expr(db, match_rs, row, rel_vars, rhs, params, depth + 1)?;
966 Ok(l && r)
967 }
968 Expr::Or(lhs, rhs) => {
969 let l = eval_set_return_expr(db, match_rs, row, rel_vars, lhs, params, depth + 1)?;
970 let r = eval_set_return_expr(db, match_rs, row, rel_vars, rhs, params, depth + 1)?;
971 Ok(l || r)
972 }
973 Expr::Not(inner) => Ok(!eval_set_return_expr(
974 db,
975 match_rs,
976 row,
977 rel_vars,
978 inner,
979 params,
980 depth + 1,
981 )?),
982 Expr::Cmp { lhs, op, rhs } => {
983 let l = eval_set_return_operand(db, match_rs, row, rel_vars, lhs, params)?;
984 let r = eval_set_return_operand(db, match_rs, row, rel_vars, rhs, params)?;
985 match (l, r) {
986 (Some(a), Some(b)) => Ok(eval_cmp(op, &a, &b)),
987 _ => Ok(false),
988 }
989 }
990 Expr::Truthy(op) => {
991 let val = eval_set_return_operand(db, match_rs, row, rel_vars, op, params)?;
992 Ok(match val {
993 None => false,
994 Some(Value::Bool(b)) => b,
995 Some(Value::Int(n)) => n != 0,
996 Some(Value::Float(f)) => f != 0.0,
997 Some(Value::Str(s)) => !s.is_empty(),
998 Some(Value::List(v)) => !v.is_empty(),
999 Some(Value::Map(m)) => !m.is_empty(),
1000 })
1001 }
1002 Expr::IsNull(op) => {
1003 let val = eval_set_return_operand(db, match_rs, row, rel_vars, op, params)?;
1004 Ok(val.is_none())
1005 }
1006 Expr::IsNotNull(op) => {
1007 let val = eval_set_return_operand(db, match_rs, row, rel_vars, op, params)?;
1008 Ok(val.is_some())
1009 }
1010 Expr::In { expr, list } => {
1011 let Some(needle) = eval_set_return_operand(db, match_rs, row, rel_vars, expr, params)?
1012 else {
1013 return Ok(false);
1014 };
1015 for item_op in list {
1016 match eval_set_return_operand(db, match_rs, row, rel_vars, item_op, params)? {
1017 None => {}
1018 Some(Value::List(items)) => {
1019 for item in items {
1020 if eval_cmp(&core_query::CmpOp::Eq, &needle, &item) {
1021 return Ok(true);
1022 }
1023 }
1024 }
1025 Some(item) if eval_cmp(&core_query::CmpOp::Eq, &needle, &item) => {
1026 return Ok(true);
1027 }
1028 Some(_) => {}
1029 }
1030 }
1031 Ok(false)
1032 }
1033 }
1034}
1035
1036fn eval_set_return_arith(
1037 op: &ArithOp,
1038 lv: Option<Value>,
1039 rv: Option<Value>,
1040) -> Result<Option<Value>> {
1041 match (lv, rv) {
1042 (None, _) | (_, None) => Ok(None),
1043 (Some(Value::Int(a)), Some(Value::Int(b))) => {
1044 let result = match op {
1045 ArithOp::Sub => a.saturating_sub(b),
1046 ArithOp::Mul => a.saturating_mul(b),
1047 ArithOp::Add => a.saturating_add(b),
1048 ArithOp::Div => {
1049 if b == 0 {
1050 return Err(GraphError::QueryError {
1051 detail: "division by zero".into(),
1052 });
1053 }
1054 a.checked_div(b).unwrap_or(i64::MAX)
1055 }
1056 };
1057 Ok(Some(Value::Int(result)))
1058 }
1059 (Some(lv), Some(rv)) => {
1060 let a = match &lv {
1061 Value::Float(f) => *f,
1062 Value::Int(i) => *i as f64,
1063 _ => {
1064 return Err(GraphError::QueryError {
1065 detail: format!("arithmetic operand must be numeric, got {lv:?}"),
1066 })
1067 }
1068 };
1069 let b = match &rv {
1070 Value::Float(f) => *f,
1071 Value::Int(i) => *i as f64,
1072 _ => {
1073 return Err(GraphError::QueryError {
1074 detail: format!("arithmetic operand must be numeric, got {rv:?}"),
1075 })
1076 }
1077 };
1078 let result = match op {
1079 ArithOp::Sub => a - b,
1080 ArithOp::Mul => a * b,
1081 ArithOp::Add => a + b,
1082 ArithOp::Div => {
1083 if b == 0.0 {
1084 return Err(GraphError::QueryError {
1085 detail: "division by zero".into(),
1086 });
1087 }
1088 a / b
1089 }
1090 };
1091 Ok(Some(Value::Float(result)))
1092 }
1093 }
1094}
1095
1096fn eval_set_return_func<F: Fs>(
1097 db: &GraphDb<F>,
1098 match_rs: &ResultSet,
1099 row: usize,
1100 rel_vars: &[String],
1101 name: &str,
1102 args: &[Operand],
1103 params: &BTreeMap<String, Value>,
1104) -> Result<Option<Value>> {
1105 let norm = name.to_ascii_lowercase();
1106 if norm == "type" {
1107 if args.len() != 1 {
1108 return Err(GraphError::QueryError {
1109 detail: format!("type() requires exactly 1 argument, got {}", args.len()),
1110 });
1111 }
1112 let Operand::Var(rel) = &args[0] else {
1113 return Err(GraphError::QueryError {
1114 detail: "type() argument must be a relationship variable (e.g. type(r))".into(),
1115 });
1116 };
1117 return Ok(match_rs.get(row, &rel_type_alias(rel)).cloned());
1118 }
1119 if norm == "key" || norm == "id" {
1120 let fname = if norm == "id" { "id" } else { "key" };
1121 if args.len() != 1 {
1122 return Err(GraphError::QueryError {
1123 detail: format!("{fname}() requires exactly 1 argument, got {}", args.len()),
1124 });
1125 }
1126 let Operand::Var(var) = &args[0] else {
1127 return Err(GraphError::QueryError {
1128 detail: format!("{fname}() argument must be a node variable (e.g. {fname}(n))"),
1129 });
1130 };
1131 if rel_vars.iter().any(|r| r == var) {
1132 return Err(GraphError::QueryError {
1133 detail: format!("{fname}() argument `{var}` is a relationship, not a node"),
1134 });
1135 }
1136 // MATCH rows bind node variables to their key string, so the column
1137 // value *is* the key. `id()` aliases `key()`.
1138 return Ok(match_rs.get(row, var).cloned());
1139 }
1140 let mut vals = Vec::with_capacity(args.len());
1141 for arg in args {
1142 vals.push(eval_set_return_operand(
1143 db, match_rs, row, rel_vars, arg, params,
1144 )?);
1145 }
1146 match norm.as_str() {
1147 "tolower" => {
1148 if vals.len() != 1 {
1149 return Err(GraphError::QueryError {
1150 detail: format!("toLower() requires exactly 1 argument, got {}", vals.len()),
1151 });
1152 }
1153 Ok(vals[0].clone().map(|val| match val {
1154 Value::Str(s) => Value::Str(s.to_ascii_lowercase()),
1155 other => other,
1156 }))
1157 }
1158 "toupper" => {
1159 if vals.len() != 1 {
1160 return Err(GraphError::QueryError {
1161 detail: format!("toUpper() requires exactly 1 argument, got {}", vals.len()),
1162 });
1163 }
1164 Ok(vals[0].clone().map(|val| match val {
1165 Value::Str(s) => Value::Str(s.to_ascii_uppercase()),
1166 other => other,
1167 }))
1168 }
1169 "size" => match vals.first().cloned().flatten() {
1170 None => Ok(None),
1171 Some(Value::Str(s)) => Ok(Some(Value::Int(s.len() as i64))),
1172 Some(Value::List(items)) => Ok(Some(Value::Int(items.len() as i64))),
1173 Some(_) => Ok(None),
1174 },
1175 "coalesce" => Ok(vals.into_iter().flatten().next()),
1176 "abs" => match vals.first().cloned().flatten() {
1177 None => Ok(None),
1178 Some(Value::Int(n)) => Ok(Some(Value::Int(n.saturating_abs()))),
1179 Some(Value::Float(f)) => Ok(Some(Value::Float(f.abs()))),
1180 Some(_) => Ok(None),
1181 },
1182 "round" => match vals.first().cloned().flatten() {
1183 None => Ok(None),
1184 Some(Value::Float(f)) => Ok(Some(Value::Float(f.round()))),
1185 Some(Value::Int(n)) => Ok(Some(Value::Int(n))),
1186 Some(_) => Ok(None),
1187 },
1188 "decay" => {
1189 if vals.len() != 3 {
1190 return Err(GraphError::QueryError {
1191 detail: format!("decay() requires exactly 3 arguments, got {}", vals.len()),
1192 });
1193 }
1194 match (vals[0].clone(), vals[1].clone(), vals[2].clone()) {
1195 (None, _, _) | (_, None, _) | (_, _, None) => Ok(None),
1196 (Some(b), Some(a), Some(h)) => {
1197 let numeric = |v: Value| -> Result<f64> {
1198 match v {
1199 Value::Int(n) => Ok(n as f64),
1200 Value::Float(f) => Ok(f),
1201 other => Err(GraphError::QueryError {
1202 detail: format!(
1203 "decay() requires numeric arguments, got {other:?}"
1204 ),
1205 }),
1206 }
1207 };
1208 let b = numeric(b)?;
1209 let a = numeric(a)?;
1210 let h = numeric(h)?;
1211 if h <= 0.0 {
1212 return Err(GraphError::QueryError {
1213 detail: "decay() requires halflife > 0".into(),
1214 });
1215 }
1216 Ok(Some(Value::Float(b * 0.5f64.powf(a / h))))
1217 }
1218 }
1219 }
1220 _ => Err(GraphError::QueryError {
1221 detail: format!(
1222 "unknown function `{name}`; supported: toLower, toUpper, size, coalesce, type, abs, round, decay, key, id"
1223 ),
1224 }),
1225 }
1226}
1227
1228fn eval_set_return_item<F: Fs>(
1229 db: &GraphDb<F>,
1230 match_rs: &ResultSet,
1231 row: usize,
1232 rel_vars: &[String],
1233 item: &RetItem,
1234 params: &BTreeMap<String, Value>,
1235) -> Result<Option<Value>> {
1236 match &item.value {
1237 RetVal::Var(v) => eval_set_return_operand(
1238 db,
1239 match_rs,
1240 row,
1241 rel_vars,
1242 &Operand::Var(v.clone()),
1243 params,
1244 ),
1245 RetVal::Prop { var, field } => eval_set_return_operand(
1246 db,
1247 match_rs,
1248 row,
1249 rel_vars,
1250 &Operand::Prop {
1251 var: var.clone(),
1252 field: field.clone(),
1253 },
1254 params,
1255 ),
1256 RetVal::FuncCall { name, args } => {
1257 eval_set_return_func(db, match_rs, row, rel_vars, name, args, params)
1258 }
1259 RetVal::ScalarExpr(op) => eval_set_return_operand(db, match_rs, row, rel_vars, op, params),
1260 RetVal::Agg { .. } => Err(GraphError::QueryError {
1261 detail: "aggregates are not supported in MATCH … SET … RETURN".into(),
1262 }),
1263 }
1264}
1265
1266/// Project user RETURN from original MATCH rows after SET. No rematch.
1267fn project_set_return_rows<F: Fs>(
1268 db: &GraphDb<F>,
1269 rel_vars: &[String],
1270 match_rs: &ResultSet,
1271 returns: &[RetItem],
1272 params: &BTreeMap<String, Value>,
1273) -> Result<ResultSet> {
1274 let columns: Vec<String> = returns.iter().map(ret_column_name).collect();
1275 let mut out = ResultSet::new(columns);
1276 for row in 0..match_rs.len() {
1277 let mut cells = Vec::with_capacity(returns.len());
1278 for item in returns {
1279 cells.push(eval_set_return_item(
1280 db, match_rs, row, rel_vars, item, params,
1281 )?);
1282 }
1283 out.push_row(cells);
1284 }
1285 Ok(out)
1286}
1287
1288/// Single construction point for a `GraphMut` view over the split-borrowed graph fields.
1289/// Callers use `std::mem::take` on the engine before calling this, then restore it after.
1290/// Extract a `Vec<f64>` from a `Value::List` whose items are all numeric.
1291/// Returns `None` for non-list values or lists with non-numeric elements.
1292/// Extra candidates pulled from an approximate index before re-scoring, over and
1293/// above the `k` asked for.
1294///
1295/// The index orders candidates by `f32` distances, which agree with the exact
1296/// `f64` cosine to about 1e-6. Re-scoring can therefore only reshuffle
1297/// candidates inside a band that narrow — it cannot move a hit past one that is
1298/// further away by more than 1e-6 — so the only way a true top-`k` member can be
1299/// lost is if the index ranked it just outside `k` on the `f32` order. Fetching
1300/// `k + 16` covers any such band up to 16 members wide, which at 1e-6 means 16
1301/// vectors within a millionth of each other in cosine: a duplicate cluster, and
1302/// then the members are interchangeable anyway. `min` is applied to the exact
1303/// score, never to the index's, so a hit sitting on the threshold is decided
1304/// exactly.
1305const VECTOR_RESCORE_MARGIN: usize = 16;
1306
1307/// Cosine similarity between an already-unit query and node `id`'s `field`
1308/// vector, read from the **`f64`** properties. `None` when the node has no
1309/// numeric-list vector there, or its norm is zero.
1310///
1311/// The single definition of the score this API reports. Both the brute-force
1312/// scan and the re-scoring step that follows an index lookup go through it, so
1313/// the two paths cannot disagree — which is the property
1314/// `index_and_brute_force_agree_on_scores` pins.
1315fn exact_vector_similarity(
1316 view: &GraphView<'_>,
1317 id: u32,
1318 field: &str,
1319 q_unit: &[f64],
1320) -> Option<f64> {
1321 let v = view.prop(id, field)?;
1322 let xs = value_as_float_list(&v.into_value())?;
1323 let v_norm: f64 = xs.iter().map(|x| x * x).sum::<f64>().sqrt();
1324 if v_norm == 0.0 {
1325 return None;
1326 }
1327 Some(
1328 q_unit
1329 .iter()
1330 .zip(xs.iter())
1331 .map(|(a, b)| a * (b / v_norm))
1332 .sum(),
1333 )
1334}
1335
1336fn value_as_float_list(v: &Value) -> Option<Vec<f64>> {
1337 match v {
1338 Value::List(items) => items
1339 .iter()
1340 .map(|item| match item {
1341 Value::Float(f) => Some(*f),
1342 Value::Int(i) => Some(*i as f64),
1343 _ => None,
1344 })
1345 .collect(),
1346 _ => None,
1347 }
1348}
1349
1350fn make_graph_mut<'a>(
1351 ids: &'a IdMap,
1352 syms: &'a mut Interner,
1353 labels: &'a [u32],
1354 props: core_storage::v8::seam::ColumnsView<'a>,
1355 topo: &'a mut Topology,
1356 base: &'a Option<std::sync::Arc<core_storage::v8::MappedBase>>,
1357 edge_props: &'a mut EdgeProps,
1358) -> GraphMut<'a> {
1359 GraphMut {
1360 ids,
1361 syms,
1362 labels,
1363 props,
1364 topo,
1365 base_topo: base_csr(base),
1366 edge_props,
1367 }
1368}
1369
1370/// The archived CSR of an open V8 snapshot, for the rule engine's graph reads.
1371///
1372/// A store opened from a snapshot keeps its edges in the mapping and its
1373/// overlay empty, so a rule that reads the graph's shape has to see both.
1374fn base_csr(
1375 base: &Option<std::sync::Arc<core_storage::v8::MappedBase>>,
1376) -> Option<&core_storage::v8::layout::ArchivedCsr> {
1377 base.as_ref().map(|b| {
1378 b.topology()
1379 .expect("base topology section bounds validated at open")
1380 })
1381}
1382
1383/// Build a `ColumnsView` from the disjoint `props` overlay and optional V8 base.
1384///
1385/// Takes explicit field references rather than `&self` so the caller can hold
1386/// simultaneous mutable borrows of other fields (e.g. `syms`, `topo`).
1387fn build_props_view<'a>(
1388 props: &'a ColumnStore,
1389 base: &'a Option<std::sync::Arc<core_storage::v8::MappedBase>>,
1390) -> core_storage::v8::seam::ColumnsView<'a> {
1391 match base {
1392 None => core_storage::v8::seam::ColumnsView::owned(props),
1393 Some(b) => {
1394 let archived = b
1395 .columns()
1396 .expect("base columns section bounds validated at open");
1397 core_storage::v8::seam::ColumnsView::with_base_cached(props, archived, b.mixed_cache())
1398 .with_shared_strings(base_string_table(b))
1399 }
1400 }
1401}
1402
1403/// The base columns section paired with the string table that resolves its
1404/// string ids — what `ViewStore` needs to read a neighbour's string property
1405/// out of a V9 snapshot.
1406fn base_columns(
1407 base: &Option<std::sync::Arc<core_storage::v8::MappedBase>>,
1408) -> Option<core_storage::v8::seam::BaseColumns<'_>> {
1409 base.as_ref().map(|b| core_storage::v8::seam::BaseColumns {
1410 cols: b
1411 .columns()
1412 .expect("base columns section bounds validated at open"),
1413 strings: base_string_table(b),
1414 })
1415}
1416
1417/// The shared string table of a V9 base, or `None` for a pre-V9 one.
1418///
1419/// Every `ColumnsView` built over a base must carry it: without it a V9
1420/// snapshot's string columns, whose own tables are empty, read back as absent.
1421fn base_string_table(
1422 base: &core_storage::v8::MappedBase,
1423) -> Option<&core_storage::v8::layout::ArchivedStringTable> {
1424 base.string_table()
1425 .transpose()
1426 .expect("base strings section bounds validated at open")
1427}
1428
1429fn build_topo_view<'a>(
1430 overlay: &'a Topology,
1431 base: &'a Option<std::sync::Arc<core_storage::v8::MappedBase>>,
1432) -> core_storage::v8::seam::TopologyView<'a> {
1433 match base {
1434 None => core_storage::v8::seam::TopologyView::owned(overlay),
1435 Some(b) => {
1436 let archived_csr = b
1437 .topology()
1438 .expect("base topology section bounds validated at open");
1439 core_storage::v8::seam::TopologyView::with_base(overlay, archived_csr)
1440 }
1441 }
1442}
1443
1444/// When [`GraphDb`] calls `Fs::sync` after a WAL append.
1445///
1446/// Default is [`Strict`](FsyncPolicy::Strict): every `log_then_apply_with`
1447/// fsyncs (single `insert_node` / `set_prop`). Ingest and `write_batch`
1448/// emit one `WalRecord::Batch` and fsync once at that frame (Batched).
1449/// [`Relaxed`](FsyncPolicy::Relaxed) skips WAL sync; [`GraphDb::snapshot`]
1450/// is still durable via `write_atomic`. Crash-recovery DST stays Strict.
1451#[derive(Clone, Copy, Debug, PartialEq, Eq, Default)]
1452pub enum FsyncPolicy {
1453 /// Every WAL commit calls `fs.sync` (today's behavior).
1454 #[default]
1455 Strict,
1456 /// Sync only at a `Batch` frame end. Single-op path stays Strict unless
1457 /// this policy is set on the database.
1458 Batched,
1459 /// Never call `fs.sync`. [`GraphDb::snapshot`] still syncs via `write_atomic`.
1460 Relaxed,
1461}
1462
1463/// A precondition for a compare-and-set batch write.
1464///
1465/// All preconditions in a [`GraphDb::write_batch_cas`] or
1466/// [`crate::SharedDb::submit_batch_cas`] call are checked atomically before
1467/// any operation in the batch is applied. If any precondition fails, the
1468/// entire batch is rejected with [`GraphError::CasConflict`] and no WAL frame
1469/// is written.
1470///
1471/// # Touch definition
1472///
1473/// A node's last-change commit (`last_changed`) is updated when any of the
1474/// following state-changing WAL records touch it:
1475///
1476/// - `InsertNode` / `InsertNodeId` — the newly-inserted node.
1477/// - `SetProp` / `SetPropId` / `RemoveProp` — the property-bearing node.
1478/// - `InsertEdge` / `InsertEdgeId` / `DeleteEdge` — **both** src and dst
1479/// endpoints (an edge change touches both sides).
1480/// - `DeleteNode` — the node is tombstoned; `last_changed` returns `None`
1481/// for deleted keys so the pre-deletion entry is never observed.
1482///
1483/// History markers (`DerivedEdgeAdded` / `DerivedEdgeRetracted`) are
1484/// state no-ops. The underlying mutation that triggered rule firing already
1485/// updated the relevant nodes' last-change entries. Rule-management records
1486/// (`CreateRule`, `DeleteRule`, `RebuildRule`) and view/full-text declarations
1487/// do not touch any node's last-change.
1488#[derive(Debug, Clone, PartialEq, Eq)]
1489pub enum Precondition {
1490 /// The node's last-change commit must equal `expected`.
1491 ///
1492 /// Fails with [`GraphError::CasConflict`] when:
1493 /// - The node does not exist (`last_changed` returns `None`), or
1494 /// - The recorded commit seq does not match `expected`.
1495 NodeUnchangedSince { key: String, expected: u64 },
1496 /// The node must not exist (not inserted, or already deleted).
1497 ///
1498 /// Fails with [`GraphError::CasConflict`] (expected=`u64::MAX`,
1499 /// actual=`last_changed(key).unwrap_or(0)`) when the node is live.
1500 NodeAbsent { key: String },
1501}
1502
1503pub struct GraphDb<F: Fs> {
1504 fs: F,
1505 ids: Arc<IdMap>,
1506 syms: Arc<Interner>,
1507 topo: Arc<Topology>,
1508 props: Arc<ColumnStore>,
1509 labels: Arc<Vec<u32>>, // node id -> label symbol
1510 /// Namespace names by index; index [`NS_DEFAULT_IDX`] is always
1511 /// [`NS_DEFAULT`]. Derived beside [`Self::node_ns`], never persisted.
1512 ///
1513 /// A private table rather than the shared [`Interner`]: interning
1514 /// `"default"` at open would add a symbol to the store's symbol table and
1515 /// change the bytes of the next snapshot of a store that has no namespaces
1516 /// at all.
1517 ns_names: Vec<String>,
1518 /// Namespace index per dense node id, into [`Self::ns_names`];
1519 /// [`NS_DEFAULT_IDX`] for a node with no `ns` property.
1520 ///
1521 /// Derived: built by one pass over the `ns` column at open (which reads
1522 /// nothing when the column does not exist) and maintained at every node
1523 /// insert. Never written to a snapshot or the WAL, because the property it
1524 /// mirrors already is. A namespace cannot change, so no other record shape
1525 /// can move a node between namespaces.
1526 node_ns: Vec<u32>,
1527 edge_props: Arc<EdgeProps>,
1528 engine: RuleEngine,
1529 view_store: ViewStore,
1530 /// Incremental inverted index for full-text-lite search.
1531 /// Rebuild-on-open: populated from WAL replay + rebuild_all at open end.
1532 fulltext: Arc<FulltextIndex>,
1533 /// Opt-in equality index over scalar node properties.
1534 /// Rebuild-on-open: declarations replay from the WAL, postings rebuild at
1535 /// open end (mirrors `fulltext`).
1536 prop_index: PropertyIndex,
1537 /// Whether this store records insert-count multiplicity (§5.13).
1538 ///
1539 /// Declared like `prop_index`'s enabled pairs — a WAL record replayed at
1540 /// open, re-emitted into the baseline by a truncating snapshot — but it
1541 /// gates a *format* step rather than an index: `WalRecord::SetEdgeCount`
1542 /// (discriminant 23) is written only when this is `true`, so a store that
1543 /// never opts in stays readable by a binary that predates the record.
1544 multiplicity: bool,
1545 event_sink: Option<Box<dyn Fn(MutationEvent) + Send + Sync>>,
1546 /// WAL fsync cadence. Default [`FsyncPolicy::Strict`].
1547 fsync: FsyncPolicy,
1548 /// Monotonically increasing per-commit counter. A single `log_then_apply_with`
1549 /// call increments this once; all events emitted from that call share the same
1550 /// `commit_seq` value.
1551 commit_seq: u64,
1552 /// Commit → wall-clock map, loaded from the `commit_times.bin` sidecar at
1553 /// open and appended to by `log_then_apply_with` — the one place a commit
1554 /// is born. Replay does **not** stamp: `apply_frames` re-applies commits
1555 /// that already happened, and `SystemTime::now()` there would record replay
1556 /// time as commit time. Empty on a store written before v0.6.11, which
1557 /// makes every date query answer `NoRecordedTime` rather than guess.
1558 commit_times: core_storage::commit_times::CommitTimes,
1559 /// When set, subsequent commits are recorded at this instant instead of the
1560 /// system clock.
1561 ///
1562 /// Sticky on purpose. A backfill replays history that happened over months,
1563 /// and a day's worth of rows genuinely share one instant — a one-shot flag
1564 /// would mean setting it before every row of a bulk load, and forgetting one
1565 /// would stamp that row "now" in the middle of 2026-06. Sticky makes the
1566 /// failure visible instead: forget to move it and every commit carries the
1567 /// same timestamp, which a date query answers oddly and an inspection shows
1568 /// at once.
1569 commit_time_override: Option<i64>,
1570 /// `true` only while the open path is replaying, where `load_from_disk`
1571 /// calls `fulltext.rebuild_all` unconditionally afterwards.
1572 ///
1573 /// Replaying an `EnableFulltext` record backfills its pair with a full
1574 /// `0..ids.len()` scan, and every snapshot re-emits one such record per
1575 /// enabled pair — so on a snapshotted store the open does that scan once per
1576 /// pair and then `rebuild_all` clears every posting and does it all again.
1577 /// The backfill is pure waste *when a rebuild follows*, which is true of the
1578 /// open path and **false** of `refresh()`: refresh applies peer frames and
1579 /// then only folds, so its backfill is the only thing that indexes them.
1580 fulltext_rebuild_follows: bool,
1581 /// `true` when `commit_times.bin` was present but would not decode.
1582 ///
1583 /// Mirrors `roles: None`: a damaged map must not read as "this store
1584 /// records no times", because that is also what an honest pre-v0.6.11 store
1585 /// says. Date queries answer `Corrupt` instead, and nothing is appended to
1586 /// a file already known to be damaged.
1587 commit_times_poisoned: bool,
1588 /// RBAC role definitions loaded from `roles.json` at open.
1589 ///
1590 /// `Some(roles)` — loaded successfully (may be empty when no roles are defined).
1591 /// `None` — `roles.json` was present but corrupt; `mask_for_role` returns
1592 /// `Err` for any request (fail-loud, never silently grant empty visibility).
1593 roles: Option<Vec<RoleDef>>,
1594 /// Memo for [`mask_for_role`](GraphDb::mask_for_role), keyed by
1595 /// `(role, commit_seq)` — a scoped reader between two writes resolves once.
1596 ///
1597 /// Shared by `Arc` with every [`ReaderSnapshot`](crate::reader::ReaderSnapshot)
1598 /// taken from this handle. Replaced (not cleared) whenever the role
1599 /// definitions change or the store is reloaded, which `commit_seq` does not
1600 /// record; see [`RoleMaskCache`](crate::mask::RoleMaskCache).
1601 role_masks: Arc<crate::mask::RoleMaskCache>,
1602 /// Which loaded store this handle is, for memos that outlive it.
1603 ///
1604 /// `role_masks` needs no such thing — the handle owns it and replaces it —
1605 /// but a [`Scope`](crate::mask::Scope) is the caller's, so its resolved key
1606 /// leg is stamped with this alongside `commit_seq`. Minted fresh here and
1607 /// again in [`reset_for_reload`](GraphDb::reset_for_reload), at exactly the
1608 /// two points a fresh `RoleMaskCache` is installed; see
1609 /// [`StoreStamp`](crate::mask::StoreStamp) for the invariant.
1610 store_id: crate::mask::StoreId,
1611 /// Live subscriptions. Entries with a dead `Weak` are pruned on the next
1612 /// distribute_events call.
1613 subscriptions: Vec<SubEntry>,
1614 /// Live query subscriptions. Re-executed on every commit when non-empty.
1615 /// Dead `Weak` entries are pruned inside `distribute_events`.
1616 query_subscriptions: Vec<QuerySubEntry>,
1617 /// Queue capacity for new subscriptions created by this db. Default is
1618 /// [`DEFAULT_SUB_CAPACITY`]; can be overridden via [`set_sub_capacity`]
1619 /// to test Lagged behaviour with small queues.
1620 sub_capacity: usize,
1621 /// True for as-of instances opened via [`GraphDb::open_at`].
1622 /// Every mutation method and `snapshot()` returns [`GraphError::ReadOnly`]
1623 /// when this flag is set.
1624 read_only: bool,
1625 /// Total WAL commit count at the time [`open_at`] was called.
1626 /// 0 for normal (non-as-of) instances.
1627 total_wal_commits: u64,
1628 /// Immutable mmap-backed base snapshot (V8). When `Some`, `self.topo` is
1629 /// the WAL-replay overlay (empty at open time, populated by apply()) and
1630 /// reads go through a merged `TopologyView`. `self.props` is always
1631 /// fully materialized (base + WAL replay) for HNSW/IVF and view compat.
1632 base: Option<Arc<core_storage::v8::MappedBase>>,
1633 // ── MVCC epoch reader state ───────────────────────────────────────────────
1634 /// Most-recent full overlay clone. Initialized at end of `open_with` /
1635 /// `open_at_with`; refreshed every `FOLD_EVERY_K` commits.
1636 /// `None` only between struct creation and the first fold.
1637 fold_overlay: Option<Arc<crate::reader::FrozenOverlay>>,
1638 /// Per-commit deltas accumulated since the last fold.
1639 delta_tail: Vec<Arc<crate::reader::CommitDelta>>,
1640 /// How many commits have occurred since the last fold.
1641 commits_since_fold: usize,
1642 /// When true, `log_then_apply_with` buffers event notifications instead of
1643 /// firing them immediately. Used by the group-commit drain thread to defer
1644 /// events until after the group fsync (R2: durability before notification).
1645 /// Cleared to false once the drain thread flushes or discards the buffer.
1646 defer_events: bool,
1647 /// Buffered events accumulated while `defer_events` is true.
1648 deferred_events: Vec<DeferredEvent>,
1649 /// Set to true by the group-commit drain thread when a group fsync fails
1650 /// after WAL truncation. All subsequent mutation attempts return an IO
1651 /// error until the database is reopened.
1652 degraded: bool,
1653 /// Set to `true` after `ensure_v8_base_sections_loaded` has read provenance,
1654 /// HNSW, and IVF sections from the mmap base into the engine's retained
1655 /// fields. `false` on all opens until first use; always `true` for non-V8
1656 /// opens (base is None, fast-path sets flag immediately).
1657 v8_sections_loaded: std::sync::atomic::AtomicBool,
1658 /// Serializes the one-time section population in `ensure_v8_base_sections_loaded`.
1659 v8_sections_mutex: std::sync::Mutex<()>,
1660 /// Per-node last-change commit sequence. `last_change[node_id] = seq` means
1661 /// the node was last modified by commit `seq`.
1662 ///
1663 /// Loaded from V8 section 11 at open; updated on every state-changing commit
1664 /// and WAL replay frame. V5-V7 stores start with an empty map; pre-WAL-horizon
1665 /// nodes return `None` from `last_changed` until they are next mutated.
1666 ///
1667 /// See [`Precondition`] for the full touch definition.
1668 last_change: HashMap<u32, u64>,
1669 /// WAL archive retention policy set by [`set_wal_archive_retention`].
1670 /// `None` = unlimited (keep all archives); `Some(N)` = keep N newest archives,
1671 /// pruning older ones at snapshot time. 0 is treated as unlimited.
1672 wal_archive_retention: Option<u32>,
1673 /// Global frame index of the first commit that is still reachable through
1674 /// surviving archives. Persisted to `wal.floor` sidecar when pruning occurs.
1675 /// Default 0 = all history reachable.
1676 wal_horizon_floor: u64,
1677 /// True when the surviving archive chain forms a continuous WAL history
1678 /// starting from the store's first commit (the genesis chain).
1679 ///
1680 /// `open_at` may replay archive-resident commits from empty state only when
1681 /// this flag is true AND `wal_horizon_floor == 0`. Cleared whenever:
1682 /// - a WAL-truncating snapshot (`keep_wal=false`) is taken after archives
1683 /// already exist (breaks the chain for subsequent archives), or
1684 /// - any archive is pruned (floor advances past zero).
1685 ///
1686 /// Persisted via the `wal.genesis` marker file; loaded from it at open.
1687 archive_genesis_chain: bool,
1688 /// True when this handle can *prove* the live WAL has never been truncated:
1689 /// there was no `snapshot.bin` when it opened the store, and it has taken no
1690 /// truncating snapshot since.
1691 ///
1692 /// The archive path's genesis check asks "did a snapshot exist before this
1693 /// one?" as a proxy for "was the WAL ever truncated". The proxy is sound
1694 /// across sessions — this binary cannot tell a history-preserving snapshot
1695 /// from a truncating one once the handle that took it is gone — but inside
1696 /// one session it is not, and `enable_multiplicity` made that visible: its
1697 /// forced `keep_wal` snapshot left the WAL entirely intact and yet
1698 /// permanently disqualified the store from ever receiving a genesis marker
1699 /// (defect #23). This flag is what the proxy defers to when the answer is
1700 /// actually known.
1701 snapshot_preserved_history: bool,
1702 /// Transient write-authz context set by `write_batch_authz` /
1703 /// `query_write_authz` for the duration of ONE mutation call.
1704 /// Always `None` at rest. Never serialized, never WAL-replayed.
1705 pending_write_authz: Option<WriteAuthz>,
1706 /// Slow-query threshold in milliseconds. 0 = disabled.
1707 /// Seeded from `MUSHROOMDB_SLOW_QUERY_MS` at open; override via
1708 /// [`GraphDb::set_slow_query_threshold_ms`] (tests must use the setter
1709 /// — env vars are process-global and race parallel test threads).
1710 slow_query_threshold_ms: u64,
1711 /// Ring buffer of recent slow queries (interior-mutable so `query(&self)`
1712 /// can record entries without requiring `&mut self`).
1713 slow_queries: std::sync::Mutex<SlowQueryLog>,
1714 /// `(field, label, caller)` triples whose exact-versus-approximate
1715 /// ambiguity this handle has already explained once. See
1716 /// [`note_ambiguous_exactness`](GraphDb::note_ambiguous_exactness).
1717 /// The caller shape is part of the key because the two shapes give
1718 /// different advice — silencing one with the other would leave a caller
1719 /// reading advice meant for a signature it does not have.
1720 /// Advice bookkeeping, not graph state: a reload keeps it, as the
1721 /// slow-query log does.
1722 warned_ambiguous_exactness: std::sync::Mutex<HashSet<(String, String, ExactnessCaller)>>,
1723 /// Instant at which the database was opened (used by `/metrics` uptime).
1724 started_at: std::time::Instant,
1725 // ── Multi-process state (cross-process lock + WAL tailing) ────────────────
1726 /// Byte offset of the WAL prefix already applied to in-memory state.
1727 ///
1728 /// Advanced by exactly the encoded length of every frame this handle
1729 /// appends, and by the decoded byte count of every tail
1730 /// [`refresh`](GraphDb::refresh) absorbs. Rewound by
1731 /// [`set_wal_consumed`](GraphDb::set_wal_consumed) when the group-commit
1732 /// drain thread truncates a failed group. Compared against the WAL's
1733 /// on-disk length to decide staleness.
1734 wal_consumed: u64,
1735 /// The **global 0-based frame index the next appended WAL frame will
1736 /// occupy** — `wal_horizon_floor` plus every frame currently reachable
1737 /// through archives and the live WAL.
1738 ///
1739 /// This is the space every history surface addresses: `edges_at`,
1740 /// `was_linked`, both history readouts and `open_at` all index the sequence
1741 /// [`all_frames`](GraphDb::all_frames) returns, and
1742 /// [`wal_total_commits`](GraphDb::wal_total_commits) counts it.
1743 ///
1744 /// It exists because **a commit is not a frame**. `commit_seq` counts
1745 /// commits; a commit whose rules fire appends a *second* frame — the
1746 /// derived-edge history marker — that no counter of commits ever sees. The
1747 /// two diverge by one frame per rule-firing commit, cumulatively, so
1748 /// deriving a frame index from `commit_seq` under-reports by more and more
1749 /// as history grows and resolves every date to an earlier graph. Silently:
1750 /// an older graph is a plausible answer, not an error.
1751 ///
1752 /// Maintained in lockstep with [`wal_consumed`](GraphDb::wal_consumed) —
1753 /// the same appends advance both, one in frames and one in bytes — so the
1754 /// two are seeded and rewound at exactly the same places. Keep it that way.
1755 wal_frames_written: u64,
1756 /// Identity of the snapshot this handle's base state came from, as
1757 /// `(len, mtime_nanos)`. A different value means another process replaced
1758 /// the snapshot and the WAL no longer continues our state: refresh reloads.
1759 snapshot_ident: Option<(u64, u64)>,
1760 /// The options this handle was opened with. Replayed verbatim when
1761 /// `refresh` has to rebuild from disk.
1762 open_opts: OpenOptions,
1763 /// True when this handle holds the cross-process write lock for its whole
1764 /// lifetime (a plain read-write open). Per-write lock acquisition is a
1765 /// no-op on such a handle, and never releases the lock.
1766 holds_lifetime_lock: bool,
1767 /// True between a failed lock acquisition and the end of the write scope
1768 /// that failed. Makes every WAL-appending mutation in that scope return
1769 /// [`GraphError::Busy`] instead of writing.
1770 lock_denied: bool,
1771 /// True for an as-of view opened via [`GraphDb::open_at`]. Such a view is
1772 /// pinned to one commit, so it is never stale and never refreshes — later
1773 /// commits by any process are deliberately invisible to it.
1774 pinned: bool,
1775}
1776
1777/// One group of deferred event notifications, held until the group fsync
1778/// completes. Replayed by [`GraphDb::flush_deferred_events`].
1779struct DeferredEvent {
1780 rec: core_storage::WalRecord,
1781 engine_deltas: Vec<EngineEdgeDelta>,
1782 seq: u64,
1783 ingest: Option<(String, usize)>,
1784}
1785
1786/// Options for [`GraphDb::open_with_options`].
1787#[derive(Clone, Copy, Debug)]
1788pub struct OpenOptions {
1789 /// Rewrite an old-format snapshot to the current VERSION after a
1790 /// successful load (default `true`). The old snapshot is kept as
1791 /// `snapshot.bin.bak` until the next clean open at the current version,
1792 /// at which point the `.bak` is deleted.
1793 ///
1794 /// Set to `false` to open a store without touching any on-disk files
1795 /// (useful for read-only inspection of a store at an older format).
1796 pub auto_migrate: bool,
1797
1798 /// Write the valid WAL prefix back over a torn tail on open (default
1799 /// `true`). Truncating a genuinely torn tail is correct crash recovery.
1800 ///
1801 /// Set to `false` for an unattended reader. The valid prefix is still
1802 /// decoded and replayed in memory, but nothing is written: a reader that
1803 /// opens while another process is mid-append would otherwise discard a
1804 /// frame that writer believes durable. `mushroomdb recall`, which runs on
1805 /// every prompt, passes `false` for exactly this reason.
1806 pub repair_wal: bool,
1807
1808 /// Open without ever writing to the store (default `false`).
1809 ///
1810 /// A read-only handle:
1811 /// - returns [`GraphError::ReadOnly`] from every mutation and from
1812 /// `snapshot()`;
1813 /// - performs no disk write at open — no WAL repair write-back and no
1814 /// auto-migration rewrite, whatever the other two flags say;
1815 /// - never takes the cross-process write lock, so it opens immediately even
1816 /// while another process is writing, and never makes a writer wait.
1817 ///
1818 /// [`refresh`](GraphDb::refresh) and [`is_stale`](GraphDb::is_stale) work
1819 /// normally, so a read-only handle can follow another process's commits.
1820 pub read_only: bool,
1821}
1822
1823impl Default for OpenOptions {
1824 fn default() -> Self {
1825 Self {
1826 auto_migrate: true,
1827 repair_wal: true,
1828 read_only: false,
1829 }
1830 }
1831}
1832
1833/// How long a writer polls for the cross-process write lock before giving up
1834/// with [`GraphError::Busy`].
1835///
1836/// Long enough to ride out another process's commit (a batch apply plus one
1837/// fsync), short enough that a stuck peer surfaces as an error rather than a
1838/// hang.
1839pub const WRITE_LOCK_WAIT: std::time::Duration = std::time::Duration::from_secs(2);
1840
1841/// Refusal when a `MERGE` create cannot choose a namespace.
1842///
1843/// A role bound to two or more namespaces cannot have its create arm land in
1844/// `default`, and the statement did not name `ns`. The role must name one.
1845pub const MERGE_CREATE_NEEDS_ONE_NAMESPACE: &str =
1846 "role-bound token: MERGE create requires the role to name one namespace";
1847
1848/// Interval between poll attempts while waiting for the cross-process lock.
1849pub(crate) const LOCK_POLL_INTERVAL: std::time::Duration = std::time::Duration::from_millis(10);
1850
1851/// Why `load_from_disk` is running, which decides whether it may repair.
1852#[derive(Clone, Copy, Debug, PartialEq, Eq)]
1853enum LoadOrigin {
1854 /// A fresh open. Crash recovery is this handle's job: a torn WAL tail is
1855 /// the signature of a crash and truncating it is correct, and archives
1856 /// orphaned by an interrupted prune can be swept.
1857 Open,
1858 /// A reload driven by [`GraphDb::refresh`], because another process
1859 /// replaced the snapshot. Nothing here is crash recovery — the store is
1860 /// live and someone else is writing it — so this origin writes nothing.
1861 Reload,
1862}
1863
1864/// Authorization context carried by `write_batch_authz` / `query_write_authz`.
1865///
1866/// `None` at the call site = full authority (today's zero-cost behavior).
1867/// `Some(WriteAuthz)` = role-scoped: the decision table (plan §"authz decision
1868/// table") is evaluated per-op inside `commit_logged_batch` BEFORE any WAL
1869/// record is built. A denial returns an error with no WAL frame written.
1870///
1871/// The mask is ALWAYS `Omit`-mode: role-token paths must never acknowledge
1872/// hidden-node existence to callers.
1873#[derive(Clone, Debug)]
1874pub struct WriteAuthz {
1875 pub role: String,
1876 pub scope: WriteScope,
1877 /// Resolved by `mask_for_role` under the same write guard as the mutation.
1878 /// Always `Omit`-mode — never `Stub`.
1879 pub mask: crate::mask::NodeMask,
1880}
1881
1882/// The error every role surface gives when `roles.json` did not parse at open.
1883///
1884/// One text, so `mask_for_role` and [`GraphDb::roles_checked`] cannot drift
1885/// apart on the same cause.
1886fn roles_poisoned() -> GraphError {
1887 GraphError::Corrupt {
1888 detail: "roles.json was corrupt at open; fix the file and re-open to restore role access"
1889 .into(),
1890 }
1891}
1892
1893/// Write `bytes` to `snapshot.bin.bak` atomically with full fsync.
1894///
1895/// Uses [`RealFs::write_atomic`] which applies `F_FULLFSYNC` on macOS and
1896/// `sync_all` on other platforms, then renames the `.tmp` file into place and
1897/// syncs the directory entry. This is the only correct path for writing the
1898/// `.bak` — plain `std::fs::write + sync_all` misses both `F_FULLFSYNC` and
1899/// the directory sync.
1900pub fn write_snapshot_bak(dir: &std::path::Path, bytes: &[u8]) -> crate::Result<()> {
1901 use core_storage::fs::{FileId, Fs as _};
1902 RealFs::new(dir)
1903 .map_err(core_storage::GraphError::Io)?
1904 .write_atomic(FileId::SnapshotBak, bytes)
1905 .map_err(core_storage::GraphError::Io)
1906}
1907
1908/// Return the on-disk snapshot format version without decoding the full snapshot.
1909///
1910/// Reads only the 6-byte header (magic + version LE). Returns `None` when no
1911/// snapshot file exists (WAL-only store). Returns an error if the header is
1912/// malformed.
1913pub fn snapshot_version_at(dir: &std::path::Path) -> crate::Result<Option<u16>> {
1914 use std::io::Read as _;
1915 let path = dir.join("snapshot.bin");
1916 let mut header = [0u8; 6];
1917 let n = match std::fs::File::open(&path) {
1918 Ok(mut f) => f.read(&mut header).map_err(core_storage::GraphError::Io)?,
1919 Err(e) if e.kind() == std::io::ErrorKind::NotFound => return Ok(None),
1920 Err(e) => return Err(core_storage::GraphError::Io(e)),
1921 };
1922 core_storage::snapshot::peek_version(&header[..n])
1923}
1924
1925/// Options for [`GraphDb::snapshot_with`].
1926#[derive(Debug, Clone, Default)]
1927pub struct SnapshotOptions {
1928 /// When `true`, the WAL is preserved after the snapshot write.
1929 /// Pre-snapshot commits remain reachable via [`GraphDb::open_at`].
1930 /// When `false` (the default), the WAL is truncated to a minimal
1931 /// baseline so cold-start replay stays fast.
1932 pub keep_wal: bool,
1933 /// When `true`, the current WAL is renamed to `wal.<commit_seq>.archive`
1934 /// before a fresh WAL baseline is written (history-preserving snapshot).
1935 ///
1936 /// This is the feature opt-in: `false` (the default) leaves the existing
1937 /// truncation / keep-wal behaviour byte-identical. `archive_wal` takes
1938 /// precedence over `keep_wal` when both are set.
1939 ///
1940 /// Archives can be scanned by [`GraphDb::node_history`],
1941 /// [`GraphDb::edge_history`], [`GraphDb::was_linked`], and
1942 /// [`GraphDb::open_at`], extending the reachable history horizon across
1943 /// snapshot boundaries.
1944 pub archive_wal: bool,
1945}
1946
1947/// Derive the scan-label sym for the commit-skip fast-path.
1948///
1949/// Walks `ops` to find the plan's leading scan op (`ScanLabel`, `IndexScan`,
1950/// or `IndexIntersect`) with a concrete label string, then interns it.
1951///
1952/// Returns `None` in all cases where skipping is unsafe:
1953/// - Any `Expand` op is present (edge traversal; edges change results regardless
1954/// of node labels).
1955/// - The leading scan has no label (`ScanLabel { label: None }` — full scan).
1956/// - No recognizable leading scan op is found.
1957///
1958/// This is the conservative v0.4.3 boundary. The caller stores the result in
1959/// [`QuerySubEntry::scan_label`] at subscribe time; `None` means always execute.
1960fn extract_scan_label(ops: &[PlanOp], syms: &mut Interner) -> Option<u32> {
1961 // Any Expand → must always re-execute (edges can change join results).
1962 if ops.iter().any(|op| matches!(op, PlanOp::Expand { .. })) {
1963 return None;
1964 }
1965 for op in ops {
1966 match op {
1967 PlanOp::ScanLabel {
1968 label: Some(label), ..
1969 } => return Some(syms.intern(label)),
1970 PlanOp::IndexScan {
1971 label: Some(label), ..
1972 } => return Some(syms.intern(label)),
1973 PlanOp::IndexIntersect {
1974 label: Some(label), ..
1975 } => return Some(syms.intern(label)),
1976 _ => {}
1977 }
1978 }
1979 None
1980}
1981
1982/// How an as-of read is restricted — the argument to
1983/// [`GraphDb::query_at_scoped`].
1984///
1985/// Every variant is resolved against the graph **as it was at the requested
1986/// commit**, not against the current graph.
1987#[derive(Debug, Clone, Copy)]
1988pub enum AsOfScope<'a> {
1989 /// Everything the named role may see. The role *definition* is the current
1990 /// one — `roles.json` is a sidecar and has no past version — but its
1991 /// `keys` and `labels` are resolved against the as-of graph.
1992 Role(&'a str),
1993 /// An explicit node-key allow-list. Keys that did not exist at that commit
1994 /// resolve to nothing.
1995 Keys(&'a [String]),
1996 /// A role intersected with a client-supplied allow-list. The intersection
1997 /// is the never-widen rule: a client mask can only narrow a role.
1998 RoleAndKeys(&'a str, &'a [String]),
1999 /// Every live node in one namespace, as the graph was at that commit.
2000 ///
2001 /// A namespace cannot change — it is set at insert and immutable — so the
2002 /// answer is simply "the nodes that existed then and are in this
2003 /// namespace". A name no node uses resolves to nothing, never to
2004 /// everything.
2005 Namespace(&'a str),
2006}
2007
2008impl GraphDb<RealFs> {
2009 /// Open the database at `dir` with default options.
2010 ///
2011 /// Equivalent to `open_with_options(dir, OpenOptions::default())`.
2012 /// Old-format snapshots (V5, V6) are automatically migrated to the
2013 /// current version on a successful load (see [`OpenOptions::auto_migrate`]).
2014 pub fn open(dir: &std::path::Path) -> Result<Self> {
2015 Self::open_with_options(dir, OpenOptions::default())
2016 }
2017
2018 /// Open the database at `dir` with explicit options.
2019 ///
2020 /// When `opts.auto_migrate` is `true` (the default) and the on-disk
2021 /// snapshot is an older format version, this function:
2022 /// 1. Copies the current `snapshot.bin` to `snapshot.bin.bak` (atomic
2023 /// + fsynced) before any modification.
2024 /// 2. Rewrites `snapshot.bin` at the current format version via
2025 /// [`GraphDb::snapshot_with`] with `keep_wal: true` (WAL preserved).
2026 ///
2027 /// If migration fails the error is returned and the original files are
2028 /// intact (the `.bak` was written before the new snapshot was attempted).
2029 ///
2030 /// A clean open that finds the snapshot already at the current version
2031 /// deletes any leftover `.bak` file.
2032 ///
2033 /// WAL-only stores (no snapshot) are never auto-migrated on open.
2034 ///
2035 /// `opts.repair_wal` controls the other write this function can make; see
2036 /// [`OpenOptions::repair_wal`]. With both flags `false` the open touches
2037 /// no file on disk.
2038 pub fn open_with_options(dir: &std::path::Path, opts: OpenOptions) -> Result<Self> {
2039 Self::open_dir(dir, opts, true)
2040 }
2041
2042 /// Open without taking the cross-process write lock for the handle's
2043 /// lifetime.
2044 ///
2045 /// Only [`SharedDb`](crate::SharedDb) uses this: a long-lived server holds
2046 /// its handle open indefinitely, so it takes the lock per write instead of
2047 /// keeping every other process out of the store for as long as it runs.
2048 pub(crate) fn open_unlocked(dir: &std::path::Path) -> Result<Self> {
2049 Self::open_dir(dir, OpenOptions::default(), false)
2050 }
2051
2052 fn open_dir(dir: &std::path::Path, opts: OpenOptions, hold_lock: bool) -> Result<Self> {
2053 // Header-only peek — 6 bytes, no full decode.
2054 let snap_version = snapshot_version_at(dir)?;
2055
2056 // Full load: decode snapshot + replay WAL + rebuild indexes.
2057 let mut db = Self::open_generic(RealFs::new(dir)?, opts, hold_lock)?;
2058
2059 // A read-only handle writes nothing at open, so it never migrates —
2060 // the old-format snapshot is loaded and left exactly as it is.
2061 if opts.auto_migrate && !opts.read_only {
2062 match snap_version {
2063 Some(ver) if ver < core_storage::snapshot::VERSION => {
2064 let _tm = std::time::Instant::now();
2065 // Copy the original snapshot to .bak at OS level — no in-memory
2066 // buffer required for a 2+ GiB file.
2067 //
2068 // Crash-safety: snapshot.bin remains intact (write_atomic inside
2069 // snapshot_with uses a .tmp+rename) until the V8 write succeeds.
2070 // A torn .bak on crash is acceptable because the original
2071 // snapshot.bin is the authoritative source until after the rename.
2072 std::fs::copy(dir.join("snapshot.bin"), dir.join("snapshot.bin.bak"))
2073 .map_err(core_storage::GraphError::Io)?;
2074 trace_migrate!("bak copy done", _tm);
2075 // Rewrite snapshot at current version; keep WAL intact.
2076 db.snapshot_with(SnapshotOptions {
2077 keep_wal: true,
2078 ..SnapshotOptions::default()
2079 })?;
2080 trace_migrate!("snapshot_with done", _tm);
2081 }
2082 Some(_) => {
2083 // Already current version: remove any leftover .bak.
2084 let bak = dir.join("snapshot.bin.bak");
2085 if bak.exists() {
2086 std::fs::remove_file(&bak).map_err(core_storage::GraphError::Io)?;
2087 }
2088 }
2089 None => {
2090 // WAL-only store — nothing to migrate on open.
2091 }
2092 }
2093 }
2094
2095 Ok(db)
2096 }
2097
2098 /// Open a read-only view of the database as it existed after `commit`.
2099 ///
2100 /// Commit indices are 0-based over the current WAL: commit 0 is the state
2101 /// after the first WAL frame, commit N-1 is the state after the N-th (most
2102 /// recent) frame. Call [`GraphDb::open`] to read the full current state.
2103 ///
2104 /// **Replay base.** [`GraphDb::snapshot`] truncates the WAL when it runs,
2105 /// so as-of can only reach commits recorded in the current WAL (those
2106 /// written after the most recent snapshot, or all commits if no snapshot
2107 /// was ever taken). Commit 0 in `open_at` always refers to the first
2108 /// frame in the WAL that exists on disk, not the first ever write to the
2109 /// database. When the on-disk snapshot recorded that it truncated the
2110 /// WAL (V7, default `keep_wal: false`), it is loaded as the base state
2111 /// before frame replay, so the as-of view includes all pre-snapshot data.
2112 /// Snapshots written with `keep_wal: true` (and legacy V5/V6 snapshots)
2113 /// are ignored and replay is WAL-only, as before.
2114 ///
2115 /// **Read-only.** Every mutation method and `snapshot()` on the returned
2116 /// instance returns [`GraphError::ReadOnly`]. Queries, `explain()`, and
2117 /// `stats()` work normally.
2118 ///
2119 /// # Errors
2120 /// - [`GraphError::CommitOutOfRange`] if `commit >= wal_commit_count` (including
2121 /// when the WAL is empty after a snapshot).
2122 pub fn open_at(dir: &std::path::Path, commit: u64) -> Result<Self> {
2123 Self::open_at_with(RealFs::new(dir)?, commit)
2124 }
2125
2126 /// Run a **read-only** Cypher query against the graph as it existed at
2127 /// `commit` — the "time-travel" / agent-replay query. Opens a temporal view
2128 /// of this store's directory at that commit and executes the read there.
2129 ///
2130 /// The current instance is unaffected. Write statements are rejected (the
2131 /// temporal view is read-only). `commit` is a 0-based WAL commit index;
2132 /// `commit == wal_commit_count` (or `open_at`'s range) yields the newest
2133 /// state. Prefer this over holding many historical instances open.
2134 ///
2135 /// # Errors
2136 /// - [`GraphError::CommitOutOfRange`] if `commit` is past the WAL horizon.
2137 /// - A query error for a malformed or write query.
2138 pub fn query_at(
2139 &self,
2140 commit: u64,
2141 cypher: &str,
2142 params: &std::collections::BTreeMap<String, Value>,
2143 ) -> Result<ResultSet> {
2144 let temporal = self.open_at_for_read(commit, cypher)?;
2145 temporal.query(cypher, params)
2146 }
2147
2148 /// Run a **read-only** Cypher query at `commit`, restricted by `scope`.
2149 ///
2150 /// The **graph** is as of `commit`; the **role definition** is as it is
2151 /// now, because `roles.json` is a sidecar and is never a WAL record — it
2152 /// has no past version to read. A role's `keys` and `labels` are resolved
2153 /// against the commit-`commit` graph, so a role that may see a label sees
2154 /// exactly the nodes that carried it then, and an explicit key that did
2155 /// not exist yet resolves to nothing.
2156 ///
2157 /// [`AsOfScope::RoleAndKeys`] intersects the two: a client allow-list can
2158 /// only narrow what a role may see, never widen it.
2159 ///
2160 /// Write statements are rejected, exactly as [`GraphDb::query_at`] rejects
2161 /// them.
2162 ///
2163 /// # Errors
2164 /// - [`GraphError::CommitOutOfRange`] if `commit` is outside the retained
2165 /// range; the error carries that range.
2166 /// - [`GraphError::KeyNotFound`] with a `role:` prefix for an unknown role,
2167 /// or [`GraphError::Corrupt`] when `roles.json` was corrupt at open.
2168 /// - A query error for a malformed or write query.
2169 pub fn query_at_scoped(
2170 &self,
2171 commit: u64,
2172 cypher: &str,
2173 params: &std::collections::BTreeMap<String, Value>,
2174 scope: AsOfScope<'_>,
2175 ) -> Result<ResultSet> {
2176 let temporal = self.open_at_for_read(commit, cypher)?;
2177 let mask = temporal.mask_at_scope(scope)?;
2178 temporal.query_masked(cypher, params, &mask)
2179 }
2180
2181 /// As [`GraphDb::query_at_scoped`], with `namespace` intersected into
2182 /// whatever `scope` resolves to.
2183 ///
2184 /// This is what a surface needs when a caller passes `namespace` beside a
2185 /// `role` or a client mask on a time-travel read: [`AsOfScope`] names one
2186 /// restriction, and the namespace is a second one that composes with it
2187 /// rather than replacing it. The intersection is the never-widen rule — a
2188 /// namespace can only narrow what the scope already allows — and both legs
2189 /// are resolved against the graph as it was at `commit`.
2190 ///
2191 /// `AsOfScope::Namespace(ns)` is still the way to ask for a namespace alone.
2192 pub fn query_at_scoped_in_namespace(
2193 &self,
2194 commit: u64,
2195 cypher: &str,
2196 params: &std::collections::BTreeMap<String, Value>,
2197 scope: AsOfScope<'_>,
2198 namespace: &str,
2199 ) -> Result<ResultSet> {
2200 let temporal = self.open_at_for_read(commit, cypher)?;
2201 let mask = temporal
2202 .mask_at_scope(scope)?
2203 .intersect(&temporal.mask_for_namespace(namespace));
2204 temporal.query_masked(cypher, params, &mask)
2205 }
2206
2207 /// Run a **read-only** Cypher query at `commit`, restricted by a
2208 /// [`Scope`](crate::mask::Scope).
2209 ///
2210 /// [`AsOfScope`] names *one* restriction — a role, a key list, a namespace,
2211 /// or a role-and-keys pair. A `Scope` is the general shape a handle carries,
2212 /// and nesting can give it several role or namespace legs at once, so it
2213 /// cannot be spelled as an `AsOfScope`. This is the entry point a scoped
2214 /// handle uses for time travel; `query_at_scoped` stays the way to ask for
2215 /// one named restriction.
2216 ///
2217 /// Both the graph and the scope's key and namespace legs are resolved
2218 /// against `commit`; a role's *definition* is the current one, because
2219 /// `roles.json` is a sidecar with no past version — the same split
2220 /// [`GraphDb::query_at_scoped`] documents.
2221 ///
2222 /// The scope resolves **cold** here: a temporal handle is its own store, so
2223 /// its ids could never be served to a live read, but filling the scope's
2224 /// one-entry key memo from a handle thrown away at the end of this call
2225 /// would evict the live entry for nothing. See
2226 /// [`Scope::resolve_uncached`](crate::mask::Scope::resolve_uncached).
2227 ///
2228 /// # Errors
2229 /// - [`GraphError::CommitOutOfRange`] if `commit` is outside the retained
2230 /// range; the error carries that range.
2231 /// - [`GraphError::KeyNotFound`] with a `role:` prefix for an unknown role,
2232 /// or [`GraphError::Corrupt`] when `roles.json` was corrupt at open.
2233 /// - A query error for a malformed or write query.
2234 pub fn query_at_with_scope(
2235 &self,
2236 commit: u64,
2237 cypher: &str,
2238 params: &std::collections::BTreeMap<String, Value>,
2239 scope: &crate::mask::Scope,
2240 ) -> Result<ResultSet> {
2241 let temporal = self.open_at_for_read(commit, cypher)?;
2242 let mask = scope.resolve_uncached(&temporal)?;
2243 temporal.query_masked(cypher, params, &mask)
2244 }
2245
2246 /// Open the temporal view for a time-travel read and refuse write Cypher.
2247 ///
2248 /// Shared by [`GraphDb::query_at`] and [`GraphDb::query_at_scoped`] so both
2249 /// resolve the commit and reject writes identically.
2250 fn open_at_for_read(&self, commit: u64, cypher: &str) -> Result<Self> {
2251 let dir = self.fs.dir().to_path_buf();
2252 let temporal = Self::open_at(&dir, commit)?;
2253 if is_write_tokens(&lex(cypher).map_err(|e| GraphError::QueryError {
2254 detail: format!("lex: {e}"),
2255 })?) {
2256 return Err(GraphError::QueryError {
2257 detail: "query_at is read-only: write statements are not permitted in a \
2258 time-travel query"
2259 .into(),
2260 });
2261 }
2262 Ok(temporal)
2263 }
2264}
2265
2266impl<F: Fs> GraphDb<F> {
2267 /// Open over an arbitrary [`Fs`], repairing a torn WAL tail as usual.
2268 pub fn open_with(fs: F) -> Result<Self> {
2269 Self::open_with_repair(fs, true)
2270 }
2271
2272 /// As [`GraphDb::open_with`], but `repair_wal: false` decodes the valid WAL
2273 /// prefix without writing the truncation back. See
2274 /// [`OpenOptions::repair_wal`].
2275 pub fn open_with_repair(fs: F, repair_wal: bool) -> Result<Self> {
2276 Self::open_generic(
2277 fs,
2278 OpenOptions {
2279 repair_wal,
2280 ..OpenOptions::default()
2281 },
2282 true,
2283 )
2284 }
2285
2286 /// Shared open path.
2287 ///
2288 /// `hold_lock` requests the cross-process write lock for the whole handle
2289 /// lifetime — the right behaviour for a plain read-write `GraphDb`, whose
2290 /// owner writes through it directly. [`SharedDb`](crate::SharedDb) passes
2291 /// `false` and takes the lock per write instead, so that a long-lived
2292 /// server does not keep every other process out of the store.
2293 ///
2294 /// A read-only open never takes the lock regardless of `hold_lock`.
2295 fn open_generic(fs: F, opts: OpenOptions, hold_lock: bool) -> Result<Self> {
2296 let mut db = Self::new_empty(fs, opts);
2297 db.read_only = opts.read_only;
2298 if hold_lock && !opts.read_only {
2299 if !db.poll_lock(WRITE_LOCK_WAIT)? {
2300 return Err(GraphError::Busy { holder: None });
2301 }
2302 db.holds_lifetime_lock = true;
2303 }
2304 db.load_from_disk(LoadOrigin::Open)?;
2305 Ok(db)
2306 }
2307
2308 /// A handle with no state loaded: every field at its empty value, the
2309 /// filesystem and options in place. Only [`load_from_disk`] makes it
2310 /// usable.
2311 fn new_empty(fs: F, opts: OpenOptions) -> Self {
2312 Self {
2313 fs,
2314 ids: Arc::new(IdMap::new()),
2315 syms: Arc::new(Interner::new()),
2316 topo: Arc::new(Topology::new()),
2317 props: Arc::new(ColumnStore::new()),
2318 labels: Arc::new(Vec::new()),
2319 ns_names: vec![NS_DEFAULT.to_string()],
2320 node_ns: Vec::new(),
2321 edge_props: Arc::new(EdgeProps::new()),
2322 engine: RuleEngine::new(),
2323 view_store: ViewStore::new(),
2324 fulltext: Arc::new(FulltextIndex::new()),
2325 prop_index: PropertyIndex::new(),
2326 multiplicity: false,
2327 event_sink: None,
2328 fsync: FsyncPolicy::Strict,
2329 commit_seq: 0,
2330 commit_times: core_storage::commit_times::CommitTimes::default(),
2331 commit_times_poisoned: false,
2332 fulltext_rebuild_follows: false,
2333 commit_time_override: None,
2334 roles: Some(vec![]),
2335 role_masks: Arc::new(crate::mask::RoleMaskCache::new()),
2336 store_id: crate::mask::StoreId::next(),
2337 subscriptions: Vec::new(),
2338 query_subscriptions: Vec::new(),
2339 sub_capacity: DEFAULT_SUB_CAPACITY,
2340 read_only: false,
2341 total_wal_commits: 0,
2342 base: None,
2343 fold_overlay: None,
2344 delta_tail: Vec::new(),
2345 commits_since_fold: 0,
2346 defer_events: false,
2347 deferred_events: Vec::new(),
2348 degraded: false,
2349 v8_sections_loaded: std::sync::atomic::AtomicBool::new(false),
2350 v8_sections_mutex: std::sync::Mutex::new(()),
2351 last_change: HashMap::new(),
2352 wal_archive_retention: None,
2353 wal_horizon_floor: 0,
2354 archive_genesis_chain: false,
2355 // Nothing is proven until `load_from_disk` has looked at the store.
2356 snapshot_preserved_history: false,
2357 pending_write_authz: None,
2358 slow_query_threshold_ms: std::env::var("MUSHROOMDB_SLOW_QUERY_MS")
2359 .ok()
2360 .and_then(|v| v.parse().ok())
2361 .unwrap_or(100),
2362 slow_queries: std::sync::Mutex::new(SlowQueryLog {
2363 entries: std::collections::VecDeque::new(),
2364 total: 0,
2365 }),
2366 warned_ambiguous_exactness: std::sync::Mutex::new(HashSet::new()),
2367 started_at: std::time::Instant::now(),
2368 wal_consumed: 0,
2369 wal_frames_written: 0,
2370 snapshot_ident: None,
2371 open_opts: opts,
2372 holds_lifetime_lock: false,
2373 lock_denied: false,
2374 pinned: false,
2375 }
2376 }
2377
2378 /// Return every field describing stored graph state to its empty value,
2379 /// leaving this handle's own identity alone.
2380 ///
2381 /// Preserved on purpose: the filesystem, open options, lock ownership, the
2382 /// event sink and subscriptions, fsync policy, degraded flag, and the
2383 /// slow-query configuration and log. A caller that registered a sink or a
2384 /// subscription keeps it across a reload.
2385 fn reset_for_reload(&mut self) {
2386 self.ids = Arc::new(IdMap::new());
2387 self.syms = Arc::new(Interner::new());
2388 self.topo = Arc::new(Topology::new());
2389 self.props = Arc::new(ColumnStore::new());
2390 self.labels = Arc::new(Vec::new());
2391 self.ns_names = vec![NS_DEFAULT.to_string()];
2392 self.node_ns = Vec::new();
2393 self.edge_props = Arc::new(EdgeProps::new());
2394 self.engine = RuleEngine::new();
2395 self.view_store = ViewStore::new();
2396 self.fulltext = Arc::new(FulltextIndex::new());
2397 self.prop_index = PropertyIndex::new();
2398 // Cleared like every other declaration: a reload replays the store's own
2399 // WAL, and the opt-in comes back from it or not at all.
2400 self.multiplicity = false;
2401 self.commit_seq = 0;
2402 self.commit_times = core_storage::commit_times::CommitTimes::default();
2403 self.commit_times_poisoned = false;
2404 self.fulltext_rebuild_follows = false;
2405 self.commit_time_override = None;
2406 self.roles = Some(vec![]);
2407 // A fresh cache, not a cleared one: any reader snapshot still holding
2408 // the old `Arc` keeps it to itself, so nothing it memoised against the
2409 // pre-reload store can be read back through this handle.
2410 self.role_masks = Arc::new(crate::mask::RoleMaskCache::new());
2411 // The same move for memos this handle does not own. `commit_seq` is
2412 // zeroed just above and reseeded from `max(last_change)`, which a
2413 // delete-only commit leaves where it was — so a reload can land back on
2414 // a sequence a caller's `Scope` already cached a mask at. A new id is
2415 // what makes that entry stop matching.
2416 self.store_id = crate::mask::StoreId::next();
2417 self.total_wal_commits = 0;
2418 self.base = None;
2419 self.fold_overlay = None;
2420 self.delta_tail = Vec::new();
2421 self.commits_since_fold = 0;
2422 self.deferred_events = Vec::new();
2423 self.v8_sections_loaded
2424 .store(false, std::sync::atomic::Ordering::Release);
2425 self.last_change = HashMap::new();
2426 self.wal_horizon_floor = 0;
2427 self.archive_genesis_chain = false;
2428 // Re-derived by `load_from_disk` from the store it is about to read.
2429 self.snapshot_preserved_history = false;
2430 self.pending_write_authz = None;
2431 self.wal_consumed = 0;
2432 self.wal_frames_written = 0;
2433 self.snapshot_ident = None;
2434 }
2435
2436 /// Load the snapshot base and replay the WAL into an empty handle — the
2437 /// whole of what opening a store does after the struct exists.
2438 ///
2439 /// Split out of the open path so that [`refresh`](GraphDb::refresh) can
2440 /// rebuild a handle in place, without ownership of `F`, when another
2441 /// process replaces the snapshot underneath it.
2442 ///
2443 /// `origin` decides whether the two repair writes this function can make
2444 /// are appropriate; see [`LoadOrigin`].
2445 fn load_from_disk(&mut self, origin: LoadOrigin) -> Result<usize> {
2446 // Both writes below are crash recovery, and only an open is entitled to
2447 // perform them. A read-only handle promises to touch nothing, and a
2448 // reload driven by `refresh` is looking at a store another process is
2449 // actively writing: what looks like a torn tail there is a peer
2450 // mid-append, and what looks like an orphaned archive may be one that
2451 // peer is about to reference.
2452 let may_repair = origin == LoadOrigin::Open && !self.open_opts.read_only;
2453 let repair_wal = self.open_opts.repair_wal && may_repair;
2454 let db = self;
2455 db.wal_horizon_floor = db.fs.read_horizon_floor()?;
2456 db.archive_genesis_chain = db.fs.has_genesis_marker();
2457 // Opening cleanup: remove orphaned archives — archives whose frames all
2458 // fall below the horizon floor. Orphans arise when a crash interrupted
2459 // the retention-prune sequence after the floor was written but before
2460 // all surplus archives were deleted. Safe to delete: floor already
2461 // accounts for their frames.
2462 if may_repair {
2463 db.cleanup_orphaned_archives()?;
2464 }
2465 let _t0 = std::time::Instant::now();
2466 // Peek 6 bytes to determine snapshot version without reading the full
2467 // file. For RealFs this is a true partial read (O(1)); for SimFs the
2468 // default impl reads all bytes and truncates (still correct).
2469 let snap_header = db.fs.read_prefix(FileId::Snapshot, 6)?;
2470 // V8, V9 and V10 share the mmap-able container; V9 only adds section 12
2471 // and V10 adds nothing but its version stamp. A version outside that set
2472 // falls through to the full-read path below, where `snapshot::decode`
2473 // either handles it (V5–V7) or refuses it by name — which is what stops
2474 // an older binary before it reaches the WAL.
2475 let is_v8 = snap_header.len() >= 6
2476 && &snap_header[0..4] == b"GDB1"
2477 && core_storage::snapshot::is_mmap_container(u16::from_le_bytes([
2478 snap_header[4],
2479 snap_header[5],
2480 ]));
2481 // The version this store is stamped with, or `None` when it has never
2482 // been snapshotted. Read from the same six bytes, with no second read.
2483 let snapshot_version = if snap_header.len() >= 6 && &snap_header[0..4] == b"GDB1" {
2484 Some(u16::from_le_bytes([snap_header[4], snap_header[5]]))
2485 } else {
2486 None
2487 };
2488 // No snapshot means no snapshot has ever truncated the WAL, so this
2489 // handle can prove the history is whole. Once a snapshot exists that
2490 // this handle did not take, it cannot: see `snapshot_preserved_history`.
2491 db.snapshot_preserved_history = snap_header.is_empty();
2492 if is_v8 {
2493 // V8: map the file zero-copy (RealFs) or read full bytes (SimFs).
2494 // No 2.4GB heap Vec is allocated on RealFs.
2495 let mapped = Arc::new(
2496 if let Some(snap_path) = db.fs.snapshot_path() {
2497 core_storage::v8::MappedBase::map(&snap_path)
2498 } else {
2499 let snap_bytes = db.fs.read(FileId::Snapshot)?;
2500 core_storage::v8::MappedBase::from_bytes(snap_bytes)
2501 }
2502 .map_err(|e| GraphError::Corrupt {
2503 detail: format!("v8: mmap open: {e:?}"),
2504 })?,
2505 );
2506 db.restore_v8_base(Arc::clone(&mapped))?;
2507 trace_open!("restore_v8_base", _t0);
2508 db.base = Some(mapped);
2509 trace_open!("base assigned", _t0);
2510 } else if !snap_header.is_empty() {
2511 // Legacy V5-V7: full read required for decode.
2512 let snap_bytes = db.fs.read(FileId::Snapshot)?;
2513 if let Some(state) = core_storage::snapshot::decode(&snap_bytes)? {
2514 db.restore_snapshot_state(state)?;
2515 }
2516 }
2517 // else: snap_header is empty = no snapshot file, fresh store.
2518 //
2519 // Seed commit_seq from the highest seq persisted in last_change so that
2520 // WAL-replay frames (which start at commit_seq+1) always exceed any seq
2521 // already stored in the snapshot. Without this, a db with one snapshot
2522 // commit would save last_change["a"]=1, then on reopen the first WAL
2523 // frame would replay at seq=1 again — colliding and making WAL-tail
2524 // mutations indistinguishable from the snapshot baseline.
2525 //
2526 // Safety invariant (seq-recycling):
2527 // Recycled seqs (those below the seeded baseline) were NEVER stored in
2528 // last_change because they belonged to a previous db lifetime — a new
2529 // db starts at commit_seq=0 with an empty last_change. Therefore no
2530 // CAS precondition can carry a recycled seq as its `expected` value
2531 // and accidentally match a live node's last_change entry.
2532 //
2533 // `expected:0` on a deleted-then-reinserted node:
2534 // After deletion, last_changed() returns None; callers that call
2535 // last_changed() and then use NodeUnchangedSince get None.unwrap_or(0)
2536 // = 0. The reinserted node gets seq > 0, so a subsequent CAS with
2537 // expected=0 correctly conflicts. The only way to observe actual=0 in
2538 // a CasConflict would be a caller that invented expected=0 without ever
2539 // calling last_changed() — unreachable via the documented API contract.
2540 if let Some(&max_seq) = db.last_change.values().max() {
2541 db.commit_seq = db.commit_seq.max(max_seq);
2542 }
2543 let bytes = db.fs.read(FileId::Wal)?;
2544 let (records, valid_len) = decode_all(&bytes);
2545 // The valid prefix is replayed either way; `repair_wal` only decides
2546 // whether the truncation is written back. A reader that races a live
2547 // appender must not persist a truncation the writer never asked for.
2548 if valid_len < bytes.len() && repair_wal {
2549 db.fs.write_atomic(FileId::Wal, &bytes[..valid_len])?;
2550 }
2551 // WAL-present path: build indexes eagerly BEFORE replay so that the
2552 // first replayed record does not trigger the lazy-init guard (which
2553 // would call reindex_all_load_state on an empty graph, defeating the
2554 // point of restoring IVF/HNSW blobs from the snapshot).
2555 if !records.is_empty() {
2556 db.ensure_v8_base_sections_loaded();
2557 trace_open!("lazy sections loaded (WAL path)", _t0);
2558 }
2559 // Scoped to this call: `refresh()` also replays frames and is *not*
2560 // followed by a rebuild, so its backfill must still run.
2561 db.fulltext_rebuild_follows = true;
2562 // Every decoded frame occupies an index, replayed or not: markers
2563 // are state no-ops but they are not index no-ops.
2564 let decoded_frames = records.len() as u64;
2565 let replayed = db.apply_frames(records);
2566 db.fulltext_rebuild_follows = false;
2567 let replayed = replayed?;
2568 // ── The multiplicity declaration, recovered from the stamp ───────────
2569 //
2570 // The opt-in is re-emitted into every baseline WAL a snapshot writes, so
2571 // ordinarily the replay above has already found it. But
2572 // `snapshot_with(archive_wal)` renames the live WAL away and writes its
2573 // replacement afterwards, and between those two points the store holds
2574 // no live declaration at all. A crash there — or a single `Err` from any
2575 // call in between — used to opt the store back out on the next open
2576 // (defect #22): it would stop counting and write a **V9** snapshot while
2577 // the archives still carried discriminant 23, which is the exact state
2578 // the V10 stamp exists to prevent.
2579 //
2580 // The V10 stamp is what carries the conclusion. The archive clause is a
2581 // scope restriction, not a second proof — an earlier version of this
2582 // comment, and defect #22, claimed otherwise, and defect #33 corrects
2583 // it. Taking the two in order:
2584 //
2585 // **The stamp.** `snapshot_with` stamps the snapshot from
2586 // `self.multiplicity` *before* it touches the WAL, and nothing rewrites
2587 // a V10 snapshot at V9 while the store believes it is opted in. So a
2588 // V10 stamp says this store reached `enable_multiplicity` far enough to
2589 // write the snapshot — and, decisively, that every older binary already
2590 // refuses this store by name. Opting in here can cost such a reader
2591 // nothing it was not already being told.
2592 //
2593 // **What the archive clause does not prove.** It is *not* evidence that
2594 // the archive was taken while the store was opted in. A store can
2595 // archive at V9 and opt in afterwards, leaving a V10 snapshot standing
2596 // beside an archive whose WAL carries no declaration at all — see
2597 // `a_failed_opt_in_beside_an_archive_comes_back_opted_in`. The inference
2598 // held in the success case by coincidence, not by construction.
2599 //
2600 // **What it does buy: scope.** Without it the recovery would also fire
2601 // on a store that reached the V10 snapshot write and then failed with no
2602 // archive in sight. That store must stay opted out, and can: no WAL was
2603 // renamed away, nothing carries discriminant 23, and its next snapshot
2604 // rewrites at V9, which puts it back within reach of every older reader.
2605 // An archive is the marker for the one state that is not recoverable
2606 // that way — a WAL renamed away that may hold the only copy of the
2607 // declaration. `no_crash_leaves_discriminant_23_unguarded` pins that
2608 // line: it sweeps a workload with no archives at all and refuses a
2609 // V10-implies-enabled rule.
2610 //
2611 // **The invariant, whichever way the clause goes:** the recovery never
2612 // opts in a store whose snapshot is not V10. A V9 store has made no
2613 // promise to an older reader, so opting it in would start writing
2614 // discriminant 23 behind a stamp that does not guard it. Pinned by
2615 // `the_recovery_never_opts_in_a_store_whose_snapshot_is_not_v10` and
2616 // `the_recovery_does_not_opt_a_store_in_by_itself`.
2617 //
2618 // What this recovery cannot do is make the opt-in atomic; it is not,
2619 // and `enable_multiplicity` says so. See defects #32-#34.
2620 if !db.multiplicity
2621 && snapshot_version == Some(core_storage::snapshot::VERSION_10)
2622 && !db.fs.list_archives()?.is_empty()
2623 {
2624 db.multiplicity = true;
2625 }
2626 // The cursor sits at the end of the valid prefix, not the end of the
2627 // file: a torn or still-being-written tail is unconsumed by definition
2628 // and stays visible to `is_stale` until it decodes.
2629 db.wal_consumed = valid_len as u64;
2630 // The frame cursor counts the same sequence `all_frames` returns:
2631 // surviving archives first, then the live WAL, offset by the floor.
2632 // Counting the archives separately rather than calling
2633 // `wal_total_commits` keeps the live WAL from being decoded twice on
2634 // every open, and costs nothing on a store that has never archived.
2635 db.wal_frames_written = db.wal_horizon_floor + db.archive_frame_count()? + decoded_frames;
2636 db.snapshot_ident = db.fs.snapshot_ident().map_err(GraphError::Io)?;
2637 trace_open!("wal replay done", _t0);
2638 // Rebuild view values after WAL replay only when there is no V8 base.
2639 // With a V8 base, view values are correct in the snapshot and are updated
2640 // incrementally during WAL replay (on_edge_changed / on_prop_changed).
2641 // A full rebuild would read overlay-only props (empty after restore_v8_base)
2642 // and overwrite correct base values with wrong results (e.g. NeighborAgg
2643 // Sum reads no "score" in overlay → writes 0.0, shadowing the correct
2644 // base value).
2645 if db.base.is_none() {
2646 let topo_view = TopologyView::owned(&db.topo);
2647 db.view_store.rebuild_all(
2648 Arc::make_mut(&mut db.props),
2649 &topo_view,
2650 &db.ids,
2651 &db.syms,
2652 &db.labels,
2653 );
2654 }
2655 // Rebuild full-text index after WAL replay. Corrects drift from
2656 // per-record incremental apply during replay.
2657 Arc::make_mut(&mut db.fulltext).rebuild_all(
2658 &db.ids,
2659 &db.labels,
2660 &db.syms,
2661 build_props_view(&db.props, &db.base),
2662 );
2663 db.prop_index.rebuild_all(
2664 &db.ids,
2665 &db.labels,
2666 &db.syms,
2667 build_props_view(&db.props, &db.base),
2668 );
2669 // Namespaces: one pass over the `ns` column, after the snapshot is
2670 // restored and the WAL replayed. Replay maintains `node_ns` record by
2671 // record as well; this pass is what makes a snapshot-only open right,
2672 // and it reads nothing on a store with no `ns` column.
2673 db.rebuild_node_ns();
2674 // A mid-build snapshot's HNSW blob carries `complete == false`.
2675 // Register it so `serve`'s ticker sees work without waiting for a write.
2676 db.register_outstanding_index_builds();
2677 // Load roles sidecar. Missing file = no roles (Some(vec![])).
2678 // Corrupt/unparseable = poisoned (None); mask_for_role will fail-loud.
2679 db.roles = Self::load_roles_from_fs(&db.fs)?;
2680 // The time sidecar. Absent is the normal case for any store written
2681 // before v0.6.11 and is not an error; unreadable is recorded so date
2682 // queries can say "damaged" rather than "none recorded".
2683 db.load_commit_times_from_fs();
2684 // Capture the initial MVCC fold so reader() is ready immediately.
2685 db.fold_now();
2686 trace_open!("open_with complete", _t0);
2687 Ok(replayed)
2688 }
2689
2690 /// Apply decoded WAL frames to in-memory state, exactly as the open-path
2691 /// replay does — same `apply` calls, same per-frame delta drain, same
2692 /// commit-seq and last-change bookkeeping. Rules therefore fire and derived
2693 /// edges appear identically whether a frame arrives at open, from a local
2694 /// commit, or from another process by way of [`refresh`](GraphDb::refresh).
2695 ///
2696 /// Returns the number of frames applied.
2697 ///
2698 /// Deltas are drained and discarded per frame: replayed frames are already
2699 /// reflected on disk, so they are not news to a subscriber, and draining
2700 /// inside the loop keeps `pending_deltas` O(1) over a large WAL (I-2).
2701 fn apply_frames(&mut self, records: Vec<WalRecord>) -> Result<usize> {
2702 if records.is_empty() {
2703 return Ok(0);
2704 }
2705 // Materialize any state retained in the mmap base before the first
2706 // frame lands, so a replayed record cannot trip the lazy-init guard and
2707 // rebuild indexes from an empty graph. Both calls are idempotent.
2708 self.ensure_v8_base_sections_loaded();
2709 self.engine.consume_retained_state_eager(
2710 &self.ids,
2711 &self.syms,
2712 &self.labels,
2713 build_props_view(&self.props, &self.base),
2714 );
2715 let applied = records.len();
2716 for rec in records {
2717 self.apply(&rec)?;
2718 let _ = self.engine.drain_deltas();
2719 // Track commit_seq during replay so last_change entries are
2720 // consistent with the seqs assigned by log_then_apply_with on
2721 // subsequent live commits. After N replayed frames, commit_seq=N;
2722 // live commits begin at N+1.
2723 self.commit_seq += 1;
2724 let replay_seq = self.commit_seq;
2725 self.update_last_change_from_rec(&rec, replay_seq);
2726 }
2727 // Enforce I-2: if the per-frame drain above is ever removed or skipped,
2728 // this assert catches the regression in debug builds immediately.
2729 debug_assert_eq!(
2730 self.engine.pending_delta_count(),
2731 0,
2732 "pending_deltas non-empty after replay — \
2733 per-frame drain must run inside the loop to keep memory O(1)"
2734 );
2735 // T2 note: the per-frame drain IS the suppression seam for replay.
2736 // Any future as-of replay path (Plan-15 T2) must drain here to feed
2737 // replaying subscribers; the mechanism is already in place.
2738 let _ = self.engine.drain_deltas(); // belt-and-braces no-op after loop drain
2739 Ok(applied)
2740 }
2741
2742 // ── Multi-process safety: cross-process write lock + WAL tailing ──────────
2743 //
2744 // mushroomdb is many-readers / one-writer across processes. Writers take an
2745 // advisory exclusive lock on the store's `LOCK` file; readers never do.
2746 // Every handle tracks how much of the WAL it has consumed, so it can pick
2747 // up another process's commits by decoding only the new tail rather than
2748 // reopening. See `docs/site/concurrency.md`.
2749
2750 /// Whether the store on disk has moved ahead of (or out from under) this
2751 /// handle's in-memory state.
2752 ///
2753 /// True when the WAL's length differs from this handle's cursor — another
2754 /// process committed, or is mid-append — or when the snapshot file's
2755 /// identity changed. Costs two metadata lookups and reads no file contents,
2756 /// so it is cheap enough for a read path to call.
2757 ///
2758 /// Always false for an as-of view from [`GraphDb::open_at`]: such a view is
2759 /// pinned to one commit and later commits are deliberately invisible to it.
2760 pub fn is_stale(&self) -> Result<bool> {
2761 if self.pinned {
2762 return Ok(false);
2763 }
2764 if self.fs.wal_len().map_err(GraphError::Io)? != self.wal_consumed {
2765 return Ok(true);
2766 }
2767 Ok(self.fs.snapshot_ident().map_err(GraphError::Io)? != self.snapshot_ident)
2768 }
2769
2770 /// Bring this handle up to date with every commit other processes have made,
2771 /// and return how many frames were applied.
2772 ///
2773 /// The WAL tail is decoded from this handle's cursor and applied through the
2774 /// same path the open replay uses, so rules fire and derived edges appear
2775 /// exactly as they would on a fresh open. Interners, id maps and indexes
2776 /// stay valid for the same reason.
2777 ///
2778 /// A frame another process is still writing is left alone: a trailing
2779 /// partial frame is a wait, not a corruption, and the handle stays stale
2780 /// until that frame is complete. Nothing is written to disk, so a read-only
2781 /// handle can refresh freely.
2782 ///
2783 /// When the snapshot file's identity changed, or the WAL is shorter than
2784 /// this handle's cursor, the WAL no longer continues our state — another
2785 /// process snapshotted or archived. The handle is then rebuilt from disk
2786 /// with the options it was opened with, and the return value is the number
2787 /// of frames in the new WAL.
2788 ///
2789 /// Returns 0 for an as-of view, which never follows later commits.
2790 ///
2791 /// # Errors
2792 ///
2793 /// An error here leaves the handle **degraded**: it got partway through
2794 /// applying the tail, or partway through a reload, so its in-memory state
2795 /// no longer matches any point on disk. Further mutations are refused and
2796 /// the handle must be reopened. Nothing on disk was damaged — the store
2797 /// itself is fine, and a fresh open recovers it.
2798 pub fn refresh(&mut self) -> Result<u64> {
2799 if self.pinned {
2800 return Ok(0);
2801 }
2802 let disk_ident = self.fs.snapshot_ident().map_err(GraphError::Io)?;
2803 let wal_len = self.fs.wal_len().map_err(GraphError::Io)?;
2804 if disk_ident != self.snapshot_ident || wal_len < self.wal_consumed {
2805 // The WAL no longer continues our state: rebuild from disk. State
2806 // is cleared first, so a failed load leaves an empty handle — mark
2807 // it degraded rather than let a caller read an empty graph as if
2808 // it were the store's contents.
2809 self.reset_for_reload();
2810 return match self.load_from_disk(LoadOrigin::Reload) {
2811 Ok(frames) => Ok(frames as u64),
2812 Err(e) => {
2813 self.degraded = true;
2814 Err(e)
2815 }
2816 };
2817 }
2818 if wal_len == self.wal_consumed {
2819 return Ok(0);
2820 }
2821 let tail = self
2822 .fs
2823 .read_range(FileId::Wal, self.wal_consumed)
2824 .map_err(GraphError::Io)?;
2825 let (records, valid_len) = decode_all(&tail);
2826 let decoded_frames = records.len() as u64;
2827 let applied = match self.apply_frames(records) {
2828 Ok(n) => n,
2829 Err(e) => {
2830 // Some frames landed and some did not, and the cursor cannot
2831 // say how many. Advancing it would skip the rest; leaving it
2832 // would replay what already applied. Neither is recoverable in
2833 // place, so refuse further writes and require a reopen.
2834 self.degraded = true;
2835 return Err(e);
2836 }
2837 };
2838 // Advance by the bytes actually decoded, never by the file length: an
2839 // incomplete trailing frame stays unconsumed for the next refresh.
2840 self.wal_consumed += valid_len as u64;
2841 self.wal_frames_written += decoded_frames;
2842 // The peer that wrote those frames also stamped them. Absorbing the
2843 // frames without the stamps leaves this handle resolving dates from a
2844 // prefix of the store's history, and — while our own map is still
2845 // empty — one commit away from rewriting the peer's file out of
2846 // existence (`first` below decides on the map, and the map is what we
2847 // just brought up to date).
2848 self.load_commit_times_from_fs();
2849 if applied > 0 {
2850 // Peer commits must reach `reader()` snapshots taken from here on.
2851 // A full fold is what open does; refresh does not build per-commit
2852 // deltas, so there is nothing cheaper that stays correct.
2853 self.fold_now();
2854 }
2855 Ok(applied as u64)
2856 }
2857
2858 /// Byte offset of the WAL prefix this handle has applied.
2859 ///
2860 /// Exposed for tests that assert the cursor tracks appended bytes exactly.
2861 #[doc(hidden)]
2862 pub fn wal_consumed(&self) -> u64 {
2863 self.wal_consumed
2864 }
2865
2866 /// Rewind the WAL cursor after the group-commit drain thread truncated a
2867 /// failed group off the tail, so the cursor still describes the file.
2868 pub(crate) fn set_wal_consumed(&mut self, len: u64) {
2869 self.wal_consumed = len;
2870 }
2871
2872 /// One non-blocking attempt at the cross-process write lock.
2873 ///
2874 /// Takes `&self` so a caller can poll for the lock *before* it acquires the
2875 /// in-process write guard. That ordering is what keeps a busy peer in
2876 /// another process from stalling this process's readers.
2877 ///
2878 /// A handle that owns the lock for its lifetime always succeeds.
2879 pub(crate) fn try_cross_process_lock(&self) -> Result<bool> {
2880 if self.holds_lifetime_lock {
2881 return Ok(true);
2882 }
2883 self.fs.try_lock_exclusive().map_err(GraphError::Io)
2884 }
2885
2886 /// Poll for the cross-process write lock until `wait` elapses.
2887 ///
2888 /// One attempt is always made, so a zero wait is a single try. Returns
2889 /// `false` when the lock is still held elsewhere at the deadline; nothing
2890 /// has been written and retrying later is safe.
2891 ///
2892 /// Only the plain-`GraphDb` open path uses this, where the caller owns the
2893 /// handle outright. [`SharedDb`](crate::SharedDb) polls
2894 /// [`try_cross_process_lock`](GraphDb::try_cross_process_lock) itself so
2895 /// that it holds no in-process guard while it waits.
2896 fn poll_lock(&self, wait: std::time::Duration) -> Result<bool> {
2897 let deadline = std::time::Instant::now() + wait;
2898 loop {
2899 if self.try_cross_process_lock()? {
2900 return Ok(true);
2901 }
2902 let now = std::time::Instant::now();
2903 if now >= deadline {
2904 return Ok(false);
2905 }
2906 std::thread::sleep(LOCK_POLL_INTERVAL.min(deadline.saturating_duration_since(now)));
2907 }
2908 }
2909
2910 /// Open a cross-process write scope, given the outcome of an already-made
2911 /// lock attempt.
2912 ///
2913 /// The caller polls for the lock first — outside any in-process guard — and
2914 /// passes what it got. On success this refreshes, so the writes about to
2915 /// happen land on top of every other process's commits. On failure the
2916 /// handle refuses WAL-appending mutations and `snapshot()` with
2917 /// [`GraphError::Busy`] until [`end_write_lock`](GraphDb::end_write_lock)
2918 /// closes the scope, so a caller holding a guard cannot write behind
2919 /// another process's back.
2920 ///
2921 /// A handle that already owns the lock for its lifetime skips the refresh:
2922 /// no other process can have written, so there is nothing to pick up.
2923 pub(crate) fn enter_write_scope(&mut self, acquired: bool) -> Result<()> {
2924 self.lock_denied = !acquired;
2925 if !acquired || self.holds_lifetime_lock {
2926 return Ok(());
2927 }
2928 if let Err(e) = self.refresh() {
2929 // Do not hold a lock we cannot use: release it and let the caller
2930 // see the underlying failure.
2931 let _ = self.fs.unlock();
2932 self.lock_denied = true;
2933 return Err(e);
2934 }
2935 Ok(())
2936 }
2937
2938 /// Close a cross-process write scope opened by
2939 /// [`enter_write_scope`](GraphDb::enter_write_scope): release the lock and
2940 /// clear the Busy latch. Safe to call when the lock was never taken.
2941 pub(crate) fn end_write_lock(&mut self) {
2942 self.lock_denied = false;
2943 if !self.holds_lifetime_lock {
2944 // Releasing a lock we do not hold is a no-op; a failure to release
2945 // is reported by the OS closing the descriptor at handle drop.
2946 let _ = self.fs.unlock();
2947 }
2948 }
2949
2950 /// As-of replay for [`GraphDb::open_at`]: snapshot base (only when the
2951 /// snapshot truncated the WAL) plus the first `commit + 1` WAL frames;
2952 /// see [`GraphDb::open_at`] for the semantics. The per-frame drain
2953 /// mirrors `open_with` exactly so pending_delta_count is 0 on exit.
2954 /// Restore all persisted state from a decoded snapshot. Shared by
2955 /// `open_with` and (when the snapshot truncated the WAL) `open_at_with`.
2956 fn restore_snapshot_state(
2957 &mut self,
2958 state: core_storage::snapshot::SnapshotState,
2959 ) -> Result<()> {
2960 self.ids = Arc::new(state.ids);
2961 self.syms = Arc::new(state.syms);
2962 self.topo = Arc::new(state.topo);
2963 self.props = Arc::new(state.props);
2964 self.labels = Arc::new(state.labels);
2965 self.edge_props = Arc::new(state.edge_props);
2966 // Cross-section label integrity for V5/V7 snapshots: same invariants as
2967 // restore_v8_base. A crafted bincode snapshot with a short `labels` vec,
2968 // out-of-range sym ids, or a sentinel label on a live node would otherwise
2969 // open successfully and panic later in `NodeRef::label()` or
2970 // `neighborhood_masked()`. Catching it here turns those into typed
2971 // `GraphError::Corrupt` at open time.
2972 {
2973 let ids_len = self.ids.len();
2974 if self.labels.len() != ids_len {
2975 return Err(GraphError::Corrupt {
2976 detail: format!(
2977 "snapshot: labels vec has {} entries but id table has {} total slots",
2978 self.labels.len(),
2979 ids_len,
2980 ),
2981 });
2982 }
2983 let syms_len = self.syms.len() as u32;
2984 for (i, &sym) in self.labels.iter().enumerate() {
2985 let is_tombstoned = self.ids.is_tombstoned(i as u32);
2986 if sym == u32::MAX {
2987 if !is_tombstoned {
2988 return Err(GraphError::Corrupt {
2989 detail: format!(
2990 "snapshot: live node at id slot {i} has sentinel label (u32::MAX)"
2991 ),
2992 });
2993 }
2994 } else if sym >= syms_len {
2995 return Err(GraphError::Corrupt {
2996 detail: format!(
2997 "snapshot: label at id slot {i} references sym {sym} \
2998 which is out of interner range ({syms_len})"
2999 ),
3000 });
3001 }
3002 }
3003 }
3004 let defs: Vec<RuleDef> = state
3005 .rule_defs
3006 .iter()
3007 .map(|b| {
3008 decode_rule_def(b).map_err(|e| GraphError::Corrupt {
3009 detail: format!("snapshot rule_def deserialize: {e}"),
3010 })
3011 })
3012 .collect::<Result<Vec<_>>>()?;
3013 self.engine =
3014 RuleEngine::from_persist(defs, state.provenance, state.rule_tripped, state.rule_fires);
3015 // Candidate indexes are rebuilt lazily on the first mutation (see
3016 // RuleEngine::on_node_changed). HNSW blobs and IVF centroids from the
3017 // snapshot are retained without deserializing so that:
3018 // - clean-open (empty WAL): indexes stay empty; blobs load on first
3019 // ANN query via ensure_hnsw_loaded, or on first mutation via the
3020 // lazy-init guard which calls reindex_all_load_state (the scan
3021 // skips the HNSW build for every side the blob supplies).
3022 // - WAL-present: open_with calls consume_retained_state_eager before
3023 // replay so HNSW/IVF are live before any record fires the hooks.
3024 let ivf_bytes = if state.ivf_state.is_empty() {
3025 Vec::new()
3026 } else {
3027 bincode::serialize(&state.ivf_state).expect("IVF state serialize cannot fail")
3028 };
3029 // Store blobs without eagerly deserializing them.
3030 // `self.ids` is the snapshot's id table at this point — WAL replay has
3031 // not run — so its length is the line an interrupted build is detected
3032 // against.
3033 let snapshot_ids = self.ids.len() as u32;
3034 self.engine
3035 .store_snapshot_state(state.hnsw_state, ivf_bytes, snapshot_ids);
3036 // Restore view defs from snapshot (V5).
3037 // The ColumnStore already contains view values from the snapshot;
3038 // use restore_view (no collision check, no backfill) so the store
3039 // is aware of the definitions. rebuild_all runs after WAL replay.
3040 for def_bytes in &state.view_defs {
3041 let def: ViewDef =
3042 bincode::deserialize(def_bytes).map_err(|e| GraphError::Corrupt {
3043 detail: format!("snapshot view_def deserialize: {e}"),
3044 })?;
3045 self.view_store
3046 .restore_view(def)
3047 .map_err(|e| GraphError::Corrupt {
3048 detail: format!("snapshot view restore: {e}"),
3049 })?;
3050 }
3051 Ok(())
3052 }
3053
3054 /// Restore all persisted state from a V8 `MappedBase` snapshot, **except**
3055 /// topology (`self.topo` stays empty and serves as the WAL-replay overlay).
3056 ///
3057 /// `self.props` IS fully materialised from the base so that HNSW/IVF blob
3058 /// deserialization and view rebuild have access to all column data.
3059 fn restore_v8_base(&mut self, mapped: Arc<core_storage::v8::MappedBase>) -> Result<()> {
3060 self.ids = Arc::new(archived_to_idmap(mapped.ids().map_err(|e| {
3061 GraphError::Corrupt {
3062 detail: format!("v8: ids section: {e:?}"),
3063 }
3064 })?));
3065 self.syms = Arc::new(archived_to_interner(mapped.syms().map_err(|e| {
3066 GraphError::Corrupt {
3067 detail: format!("v8: syms section: {e:?}"),
3068 }
3069 })?));
3070
3071 // C1: self.props is left as an empty overlay. Column reads go through
3072 // props_view() (ColumnsView::with_base), which consults the archived base
3073 // section zero-copy. This avoids the O(columns) heap copy at every open.
3074
3075 // self.topo deliberately left as Topology::new() — overlay path.
3076
3077 let meta = decode_meta(mapped.meta_bytes().map_err(|e| GraphError::Corrupt {
3078 detail: format!("v8: meta section: {e:?}"),
3079 })?)
3080 .map_err(|e| GraphError::Corrupt {
3081 detail: format!("v8: meta decode: {e:?}"),
3082 })?;
3083 self.labels = Arc::new(meta.labels);
3084 // Cross-section label integrity: labels must cover every id slot (live
3085 // and tombstoned), every non-sentinel sym must be within the interner's
3086 // bound, and no live (non-tombstoned) node may carry the u32::MAX
3087 // sentinel label. Without this check, a crafted snapshot where the META
3088 // section (small, CRC-validated) holds a short `labels` vec, out-of-range
3089 // sym ids, or a sentinel label on a live node, would open successfully
3090 // and then panic in `NodeRef::label()`, `neighborhood_masked()`, and
3091 // related read paths. Catching the inconsistency here converts those
3092 // panics into typed `GraphError::Corrupt` at open time.
3093 {
3094 let ids_len = self.ids.len();
3095 if self.labels.len() != ids_len {
3096 return Err(GraphError::Corrupt {
3097 detail: format!(
3098 "v8: labels section has {} entries but id table has {} total slots",
3099 self.labels.len(),
3100 ids_len,
3101 ),
3102 });
3103 }
3104 let syms_len = self.syms.len() as u32;
3105 for (i, &sym) in self.labels.iter().enumerate() {
3106 let is_tombstoned = self.ids.is_tombstoned(i as u32);
3107 if sym == u32::MAX {
3108 // Sentinel is only valid for tombstoned slots.
3109 if !is_tombstoned {
3110 return Err(GraphError::Corrupt {
3111 detail: format!(
3112 "v8: live node at id slot {i} has sentinel label (u32::MAX)"
3113 ),
3114 });
3115 }
3116 } else if sym >= syms_len {
3117 return Err(GraphError::Corrupt {
3118 detail: format!(
3119 "v8: label at id slot {i} references sym {sym} \
3120 which is out of interner range ({syms_len})"
3121 ),
3122 });
3123 }
3124 }
3125 }
3126 // C3: self.edge_props stays as an empty overlay. Reads go through
3127 // edge_props_view() which consults the mmap'd base section zero-copy
3128 // via EdgePropsView::with_base. No heap decode at open time.
3129
3130 // Restore rule engine.
3131 let (rule_def_bytes, rule_tripped, rule_fires) =
3132 archived_rules_meta_to_owned(mapped.rules_meta_section().map_err(|e| {
3133 GraphError::Corrupt {
3134 detail: format!("v8: rules_meta section: {e:?}"),
3135 }
3136 })?);
3137 let defs: Vec<RuleDef> = rule_def_bytes
3138 .iter()
3139 .map(|b| {
3140 decode_rule_def(b).map_err(|e| GraphError::Corrupt {
3141 detail: format!("v8: rule_def deserialize: {e}"),
3142 })
3143 })
3144 .collect::<Result<Vec<_>>>()?;
3145 self.engine = RuleEngine::from_persist(defs, BTreeMap::new(), rule_tripped, rule_fires);
3146 // C4+C5: provenance, HNSW, and IVF sections are NOT read here.
3147 // `ensure_v8_base_sections_loaded` reads them on first use from
3148 // `self.base` (set by the caller immediately after this returns).
3149 // A clean open touches only: header + IDS + SYMS + META + RULES_META.
3150
3151 // Restore view definitions.
3152 let view_defs =
3153 archived_views_to_owned(mapped.views_section().map_err(|e| GraphError::Corrupt {
3154 detail: format!("v8: views section: {e:?}"),
3155 })?);
3156 for def_bytes in &view_defs {
3157 let def: ViewDef =
3158 bincode::deserialize(def_bytes).map_err(|e| GraphError::Corrupt {
3159 detail: format!("v8: view_def deserialize: {e}"),
3160 })?;
3161 self.view_store
3162 .restore_view(def)
3163 .map_err(|e| GraphError::Corrupt {
3164 detail: format!("v8: view restore: {e}"),
3165 })?;
3166 }
3167 // Load the last-change map from section 11 (small section; load eagerly).
3168 // Pre-Task-3 snapshots lack this section; `last_change_bytes` returns &[]
3169 // in that case and `decode_last_change_bytes` returns an empty map.
3170 let last_change_raw = mapped
3171 .last_change_bytes()
3172 .map_err(|e| GraphError::Corrupt {
3173 detail: format!("v8: last_change section: {e:?}"),
3174 })?;
3175 self.last_change = decode_last_change_bytes(last_change_raw);
3176
3177 // Validate that all deferred sections (provenance, HNSW, IVF) fit within
3178 // the file. Pure bounds check — no bytes read, no page faults triggered.
3179 // Catches truncated snapshots at open time before the lazy deferred reads.
3180 mapped.validate_section_bounds().map_err(|e| match e {
3181 GraphError::Corrupt { detail } => GraphError::Corrupt {
3182 detail: format!("v8: section bounds: {detail}"),
3183 },
3184 other => other,
3185 })?;
3186 Ok(())
3187 }
3188
3189 /// Read provenance, HNSW, and IVF sections from the mmap base into the
3190 /// engine's retained fields on first call. Subsequent calls are a no-op
3191 /// (AtomicBool fast-path).
3192 ///
3193 /// Must be called before any code path that reads or mutates engine
3194 /// provenance, HNSW, or IVF state:
3195 /// - WAL replay (before `consume_retained_state_eager`)
3196 /// - First mutation (`log_then_apply_with`)
3197 /// - Read-only paths (`stats`, `explain`, `node_edges`)
3198 /// - Snapshot (`snapshot_with`)
3199 ///
3200 /// No-op for fresh stores and V5-V7 opens (`self.base` is `None`).
3201 fn ensure_v8_base_sections_loaded(&self) {
3202 use std::sync::atomic::Ordering;
3203 if self.v8_sections_loaded.load(Ordering::Acquire) {
3204 return;
3205 }
3206 let _guard = self
3207 .v8_sections_mutex
3208 .lock()
3209 .expect("v8 sections mutex poisoned");
3210 if self.v8_sections_loaded.load(Ordering::Acquire) {
3211 return; // another caller populated while we waited
3212 }
3213 let _t = std::time::Instant::now();
3214 if let Some(base) = &self.base {
3215 // Provenance: raw rkyv bytes; CRC validated inside section_bytes.
3216 // Bounds are already validated at open time (restore_v8_base →
3217 // validate_section_bounds) — unreachable post-validate_section_bounds;
3218 // unwrap_or_default is a safety belt against impossible errors.
3219 let prov_bytes = base
3220 .provenance_raw_bytes()
3221 .map(|b| b.to_vec())
3222 .unwrap_or_default();
3223 self.engine.store_provenance_bytes(prov_bytes);
3224 // HNSW: decode rkyv blobs into owned map.
3225 let hnsw_state = base
3226 .hnsw_section()
3227 .map(archived_hnsw_to_owned)
3228 .unwrap_or_default();
3229 // IVF: raw bincode bytes; deserialized on first mutation/query.
3230 let ivf_bytes = base.ivf_bytes().map(|b| b.to_vec()).unwrap_or_default();
3231 // Called before WAL replay on a WAL-present open (`open_with`) and
3232 // before any write on a clean one, so this is the snapshot's count.
3233 let snapshot_ids = self.ids.len() as u32;
3234 self.engine
3235 .store_snapshot_state(hnsw_state, ivf_bytes, snapshot_ids);
3236 }
3237 self.v8_sections_loaded.store(true, Ordering::Release);
3238 if std::env::var("MUSHROOMDB_TRACE_OPEN").is_ok() {
3239 eprintln!(
3240 "[MUSHROOMDB_TRACE_OPEN] ensure_v8_base_sections_loaded: {:>9.3?}",
3241 _t.elapsed()
3242 );
3243 }
3244 }
3245
3246 /// Return a `TopologyView` that merges the mmap'd base (when present) with
3247 /// the in-memory WAL overlay. Used by all read paths in db.rs that need
3248 /// the full merged topology without going through `self.view()`.
3249 fn topo_view(&self) -> TopologyView<'_> {
3250 match self.base {
3251 None => TopologyView::owned(&self.topo),
3252 Some(ref base) => {
3253 // SAFETY: base lives as long as self; section bounds validated at open.
3254 // topology() uses access_unchecked; all field reads are bounds-checked in seam.rs.
3255 let archived = base
3256 .topology()
3257 .expect("base topology section bounds validated at open");
3258 TopologyView::with_base(&self.topo, archived)
3259 }
3260 }
3261 }
3262
3263 /// Return a `ColumnsView` that merges the mmap'd base columns (when a V8
3264 /// snapshot is open) with the in-memory WAL overlay. Reads consult the
3265 /// overlay first, then fall through to the archived base section zero-copy.
3266 fn props_view(&self) -> core_storage::v8::seam::ColumnsView<'_> {
3267 match self.base {
3268 None => core_storage::v8::seam::ColumnsView::owned(&self.props),
3269 Some(ref base) => {
3270 // columns() uses access_unchecked; field reads are bounds-checked in seam.rs.
3271 let archived = base
3272 .columns()
3273 .expect("base columns section bounds validated at open");
3274 core_storage::v8::seam::ColumnsView::with_base_cached(
3275 &self.props,
3276 archived,
3277 base.mixed_cache(),
3278 )
3279 .with_shared_strings(base_string_table(base))
3280 }
3281 }
3282 }
3283
3284 /// Return an `EdgePropsView` that merges the mmap'd base edge-props section
3285 /// (when a V8 snapshot is open) with the in-memory WAL overlay.
3286 ///
3287 /// Reads consult the overlay first (for post-snapshot mutations), then fall
3288 /// through to the archived base section zero-copy. Tombstones in the
3289 /// overlay mask deleted-from-base entries.
3290 fn edge_props_view(&self) -> EdgePropsView<'_> {
3291 match self.base {
3292 None => EdgePropsView::owned(&self.edge_props),
3293 Some(ref base) => {
3294 // edge_props_section() uses access_unchecked; field reads bounds-checked in seam.rs.
3295 let archived = base
3296 .edge_props_section()
3297 .expect("base edge_props section bounds validated at open");
3298 EdgePropsView::with_base(&self.edge_props, archived)
3299 }
3300 }
3301 }
3302
3303 fn open_at_with(fs: F, commit: u64) -> Result<Self> {
3304 // An as-of view never writes and is pinned to one commit: it takes no
3305 // cross-process lock and does not follow later commits.
3306 let mut db = Self::new_empty(
3307 fs,
3308 OpenOptions {
3309 repair_wal: false,
3310 auto_migrate: false,
3311 read_only: true,
3312 },
3313 );
3314 db.pinned = true; // read_only is set after replay, but pinning is immediate
3315 db.wal_horizon_floor = db.fs.read_horizon_floor()?;
3316 db.archive_genesis_chain = db.fs.has_genesis_marker();
3317 // Same orphaned-archive cleanup as open_with: floor was written first
3318 // during pruning, so a crash may have left stale archives below floor.
3319 db.cleanup_orphaned_archives()?;
3320 // Collect archive frames (oldest-first) and live WAL frames.
3321 // Archives represent pre-snapshot history; the snapshot captures the
3322 // cumulative state at the time of archiving. Crash-window guarantee:
3323 // A: crash before rename → WAL intact, no archive. Reopen: normal.
3324 // B: crash after rename, before new WAL → archive present, WAL
3325 // absent. Reopen: snapshot loaded (full state), no WAL replay.
3326 // C: crash after new baseline WAL written → normal post-archive.
3327 let archive_ns = db.fs.list_archives()?;
3328 let mut archive_frames_all: Vec<WalRecord> = Vec::new();
3329 for n in &archive_ns {
3330 let arc_bytes = db.fs.read_archive(*n)?;
3331 let (arc_frames, _) = decode_all(&arc_bytes);
3332 archive_frames_all.extend(arc_frames);
3333 }
3334 let total_archive_frames = archive_frames_all.len() as u64;
3335
3336 let live_bytes = db.fs.read(FileId::Wal)?;
3337 let (live_records, _valid_len) = decode_all(&live_bytes);
3338 let total_surviving = total_archive_frames + live_records.len() as u64;
3339 // Global total including any pruned history below the horizon floor.
3340 let total = db.wal_horizon_floor + total_surviving;
3341
3342 // Horizon and range check.
3343 if commit < db.wal_horizon_floor {
3344 return Err(GraphError::CommitOutOfRange {
3345 commit,
3346 total,
3347 floor: db.wal_horizon_floor,
3348 });
3349 }
3350 if commit >= total {
3351 return Err(GraphError::CommitOutOfRange {
3352 commit,
3353 total,
3354 floor: db.wal_horizon_floor,
3355 });
3356 }
3357
3358 // Local index into surviving frames (0 = first frame of oldest archive).
3359 let local = commit - db.wal_horizon_floor;
3360
3361 if local < total_archive_frames {
3362 // Target commit is in an archive. Correct replay from empty state
3363 // is only possible when the archive chain is an uninterrupted
3364 // genesis chain (first archive taken from a fresh store, no prior
3365 // WAL truncation) and no archives have been pruned (floor == 0).
3366 //
3367 // If either condition is violated the prefix needed to reconstruct
3368 // the requested state is gone; refuse rather than return wrong data.
3369 if db.wal_horizon_floor > 0 || !db.archive_genesis_chain {
3370 return Err(GraphError::CommitOutOfRange {
3371 commit,
3372 total,
3373 floor: db.wal_horizon_floor,
3374 });
3375 }
3376 // Replay all archive frames up to and including the target commit
3377 // from an empty database state. Archives must be replayed in order
3378 // so that dense-id intern tables are built up correctly.
3379 for rec in archive_frames_all.into_iter().take((local + 1) as usize) {
3380 db.apply(&rec)?;
3381 let _ = db.engine.drain_deltas();
3382 }
3383 } else {
3384 // Target commit is in the live WAL: load snapshot as base, then
3385 // replay the needed live WAL prefix.
3386 //
3387 // Base state: a truncating snapshot (wal_truncated=true) compacts
3388 // all pre-truncation / pre-archive commits. Dense-id records in
3389 // the live WAL reference ids/interns that the snapshot provides.
3390 // Peek 6 bytes (same pattern as open_with).
3391 let snap_header = db.fs.read_prefix(FileId::Snapshot, 6)?;
3392 let is_v8 = snap_header.len() >= 6
3393 && &snap_header[0..4] == b"GDB1"
3394 && core_storage::snapshot::is_mmap_container(u16::from_le_bytes([
3395 snap_header[4],
3396 snap_header[5],
3397 ]));
3398 if is_v8 {
3399 let state = if let Some(snap_path) = db.fs.snapshot_path() {
3400 let mapped = core_storage::v8::MappedBase::map(&snap_path).map_err(|e| {
3401 GraphError::Corrupt {
3402 detail: format!("v8: open_at mmap: {e:?}"),
3403 }
3404 })?;
3405 core_storage::snapshot::decode_v8_from_mapped(&mapped)?
3406 } else {
3407 let snap_bytes = db.fs.read(FileId::Snapshot)?;
3408 core_storage::snapshot::decode(&snap_bytes)?
3409 };
3410 if let Some(state) = state {
3411 if state.wal_truncated {
3412 db.restore_snapshot_state(state)?;
3413 }
3414 }
3415 } else if !snap_header.is_empty() {
3416 let snap_bytes = db.fs.read(FileId::Snapshot)?;
3417 if let Some(state) = core_storage::snapshot::decode(&snap_bytes)? {
3418 if state.wal_truncated {
3419 db.restore_snapshot_state(state)?;
3420 }
3421 }
3422 }
3423 // else: snap_header empty = no snapshot file.
3424 let live_local = local - total_archive_frames;
3425 for rec in live_records.into_iter().take((live_local + 1) as usize) {
3426 db.apply(&rec)?;
3427 let _ = db.engine.drain_deltas();
3428 }
3429 }
3430 // Pin: pending_delta_count must be 0 after as-of replay, mirroring T1's
3431 // post-loop assert in open_with.
3432 debug_assert_eq!(
3433 db.engine.pending_delta_count(),
3434 0,
3435 "pending_deltas non-empty after open_at replay — \
3436 per-frame drain must run inside the loop to keep memory O(1)"
3437 );
3438 let _ = db.engine.drain_deltas(); // belt-and-braces no-op
3439 // Rebuild view values after WAL replay so derived-edge-driven views
3440 // reflect the as-of state. open_at always uses the legacy path (no V8
3441 // base), so topo_view is always owned.
3442 {
3443 let topo_view = TopologyView::owned(&db.topo);
3444 db.view_store.rebuild_all(
3445 Arc::make_mut(&mut db.props),
3446 &topo_view,
3447 &db.ids,
3448 &db.syms,
3449 &db.labels,
3450 );
3451 }
3452 // Rebuild full-text index for as-of view (mirrors open_with pattern).
3453 Arc::make_mut(&mut db.fulltext).rebuild_all(
3454 &db.ids,
3455 &db.labels,
3456 &db.syms,
3457 build_props_view(&db.props, &db.base),
3458 );
3459 db.prop_index.rebuild_all(
3460 &db.ids,
3461 &db.labels,
3462 &db.syms,
3463 build_props_view(&db.props, &db.base),
3464 );
3465 // Namespaces on the temporal handle, built by the same pass the live
3466 // open uses, so an as-of mask narrows by the namespaces of that commit.
3467 db.rebuild_node_ns();
3468 // Load roles sidecar (current roles, not point-in-time).
3469 db.roles = Self::load_roles_from_fs(&db.fs)?;
3470 db.read_only = true;
3471 db.total_wal_commits = total;
3472 // Capture initial fold so reader() is immediately usable.
3473 db.fold_now();
3474 Ok(db)
3475 }
3476
3477 /// Whether this instance is a read-only as-of view.
3478 pub fn is_read_only(&self) -> bool {
3479 self.read_only
3480 }
3481
3482 // ── MVCC epoch reader ─────────────────────────────────────────────────────
3483
3484 /// Clone the current overlay state into a new `FrozenOverlay` and reset
3485 /// the delta tail. Called automatically every `FOLD_EVERY_K` commits and at
3486 /// the end of `open_with` / `open_at_with` to prime the reader.
3487 fn fold_now(&mut self) {
3488 // Eight `Arc::clone`s — refcount bumps, O(1). This used to deep-copy the
3489 // whole overlay: `IdMap` alone is a `HashMap<String, u32>` plus a
3490 // `Vec<String>`, so every node key was copied twice, on every open,
3491 // after every snapshot, every FOLD_EVERY_K commits on the write path,
3492 // and — with no commit threshold — on every `refresh()` that applied a
3493 // peer commit. A refreshing reader now pays nothing for a fold.
3494 //
3495 // `props` and `topo` were already cheap for a different reason: on a
3496 // snapshotted store `restore_v8_base` leaves them as empty overlays over
3497 // the zero-copy mmap. `ids`, `syms` and `fulltext` were not, and that
3498 // inconsistency was the defect.
3499 let frozen = crate::reader::FrozenOverlay {
3500 ids: std::sync::Arc::clone(&self.ids),
3501 syms: std::sync::Arc::clone(&self.syms),
3502 topo: std::sync::Arc::clone(&self.topo),
3503 props: std::sync::Arc::clone(&self.props),
3504 labels: std::sync::Arc::clone(&self.labels),
3505 edge_props: std::sync::Arc::clone(&self.edge_props),
3506 roles: self.roles.clone().map(std::sync::Arc::new),
3507 fulltext: std::sync::Arc::clone(&self.fulltext),
3508 };
3509 self.fold_overlay = Some(Arc::new(frozen));
3510 self.delta_tail.clear();
3511 self.commits_since_fold = 0;
3512 }
3513
3514 /// Capture a lock-free reader snapshot of the current db state.
3515 ///
3516 /// The read lock is held only for the duration of this call (to clone a
3517 /// handful of `Arc` handles). Subsequent query operations run without any
3518 /// lock.
3519 pub fn reader(&self) -> crate::reader::ReaderSnapshot {
3520 crate::reader::ReaderSnapshot::new(
3521 self.fold_overlay
3522 .clone()
3523 .expect("fold_overlay is always Some after open_with; call reader() after open"),
3524 self.base.clone(),
3525 self.delta_tail.clone(),
3526 // The snapshot's effective state is exactly this handle's state at
3527 // this commit, so it shares the memo and its version key.
3528 self.commit_seq,
3529 Arc::clone(&self.role_masks),
3530 )
3531 }
3532
3533 /// Append a delta the reader cannot apply, so that a corrupt overlay is
3534 /// reachable from a test.
3535 ///
3536 /// Compiled only under `test-hooks`, which the server's dev-dependency on
3537 /// this crate turns on. One call permanently corrupts every
3538 /// [`ReaderSnapshot`](crate::reader::ReaderSnapshot) taken from the handle,
3539 /// so it must not be in the published surface: `#[doc(hidden)]` hides it
3540 /// from rustdoc and from nothing else. The feature gate — not
3541 /// `#[cfg(test)]` — because its only callers are in `crates/server/tests`,
3542 /// a different crate, exactly as `core_rules`'s index counters are.
3543 ///
3544 /// [`ReaderSnapshot::effective`](crate::reader::ReaderSnapshot) folds the
3545 /// delta tail into a clone of the frozen overlay and answers
3546 /// [`GraphError::Corrupt`] when a record will not apply. Nothing a caller
3547 /// can do produces that state — `apply_one`'s failures are disagreements
3548 /// between the tail and the fold it is applied to, which the write path
3549 /// cannot create — so the `Corrupt` arm of every scoped reader method was
3550 /// reachable only by inspection until this hook existed. An `Intern` record
3551 /// claiming an id the frozen interner will not hand back is the smallest
3552 /// such disagreement.
3553 ///
3554 /// Only the tail is touched. This handle's own state is untouched and
3555 /// `commit_seq` does not move, so a role mask already memoised at this
3556 /// version stays memoised — which is exactly the state in which the HTTP
3557 /// role branches reach a scoped read with a corrupt overlay under them.
3558 #[cfg(any(test, feature = "test-hooks"))]
3559 #[doc(hidden)]
3560 pub fn push_unapplyable_delta_for_test(&mut self) {
3561 self.delta_tail.push(Arc::new(crate::reader::CommitDelta {
3562 records: vec![WalRecord::Intern {
3563 id: u32::MAX,
3564 text: "delta-tail-corruption".into(),
3565 }],
3566 derived_inserts: Vec::new(),
3567 derived_deletes: Vec::new(),
3568 }));
3569 }
3570
3571 /// Total number of WAL commits at the time [`open_at`] was called.
3572 /// Returns 0 for normal (non-as-of) instances.
3573 pub fn total_wal_commits(&self) -> u64 {
3574 self.total_wal_commits
3575 }
3576
3577 /// Apply a record to in-memory state. Used by both live writes and replay,
3578 /// so replay is definitionally identical to the original execution.
3579 fn apply(&mut self, rec: &WalRecord) -> Result<()> {
3580 // Before the record mutates anything: a store restored from a snapshot
3581 // defers building its candidate indexes until the first write, and that
3582 // build is a full node scan. Left where it used to fire — inside the
3583 // engine hook, after `props.set` and the label assignment — the scan
3584 // read the half-applied record and took the in-flight node's vector for
3585 // one the snapshot should have carried, which read as an interrupted
3586 // vector-index build and cost a full `RebuildRule` on the first
3587 // embedded write after every reopen. Hoisted here the scan sees exactly
3588 // the persisted state; the record's own hook then files its vector
3589 // through the ordinary insert path a line later.
3590 self.populate_indexes_before_write();
3591 match rec {
3592 WalRecord::InsertNode { label, key, props } => {
3593 let id = Arc::make_mut(&mut self.ids).try_insert(key)?;
3594 let sym = Arc::make_mut(&mut self.syms).intern(label);
3595 if self.labels.len() <= id as usize {
3596 // gap slots are sentinels, never valid label symbols
3597 Arc::make_mut(&mut self.labels).resize(id as usize + 1, u32::MAX);
3598 }
3599 Arc::make_mut(&mut self.labels)[id as usize] = sym;
3600 let mut ns_name = NS_DEFAULT.to_string();
3601 for (field, value) in props {
3602 if field == NS_PROP {
3603 ns_name = namespace_of_value(Some(value)).to_string();
3604 }
3605 Arc::make_mut(&mut self.props).set(id, field, value.clone());
3606 }
3607 self.set_node_ns(id, &ns_name);
3608 // Initialize view values for the new node before the engine runs so
3609 // delta-based increments start from a known zero baseline.
3610 self.view_store.init_node_views(
3611 id,
3612 Arc::make_mut(&mut self.props),
3613 &self.syms,
3614 &self.labels,
3615 );
3616 // Fire rules for the newly inserted node.
3617 let cursor = self.engine.pending_delta_count();
3618 let mut eng = std::mem::take(&mut self.engine);
3619 {
3620 let mut gm = make_graph_mut(
3621 &self.ids,
3622 Arc::make_mut(&mut self.syms),
3623 &self.labels,
3624 build_props_view(&self.props, &self.base),
3625 Arc::make_mut(&mut self.topo),
3626 &self.base,
3627 Arc::make_mut(&mut self.edge_props),
3628 );
3629 eng.on_node_changed(id, None, &mut gm);
3630 }
3631 self.engine = eng;
3632 // Process derived-edge deltas for view maintenance.
3633 // Fast path: skip the O(delta_count) allocation when no views exist.
3634 if !self.view_store.is_empty() {
3635 #[cfg(test)]
3636 DELTA_COPY_COUNT.with(|c| c.set(c.get() + 1));
3637 let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
3638 for d in &new_deltas {
3639 self.view_store.on_edge_changed(
3640 d.etype_sym,
3641 d.src_id,
3642 d.dst_id,
3643 d.fired,
3644 Arc::make_mut(&mut self.props),
3645 &build_topo_view(&self.topo, &self.base),
3646 &self.ids,
3647 &self.syms,
3648 &self.labels,
3649 base_columns(&self.base),
3650 );
3651 }
3652 }
3653 // Full-text index maintenance: index enabled fields for this label.
3654 if self.fulltext.has_label(label) {
3655 for (field, value) in props {
3656 if self.fulltext.is_enabled(label, field) {
3657 Arc::make_mut(&mut self.fulltext).add_tokens(id, field, value);
3658 }
3659 }
3660 }
3661 // Property (equality) index maintenance.
3662 if self.prop_index.has_label(label) {
3663 for (field, value) in props {
3664 self.prop_index.set(label, field, id, value);
3665 }
3666 }
3667 }
3668 WalRecord::InsertEdge {
3669 edge_type,
3670 src_key,
3671 dst_key,
3672 } => {
3673 let src = self.ids.get(src_key).ok_or_else(|| GraphError::Corrupt {
3674 detail: format!("wal replay references unknown key {src_key}"),
3675 })?;
3676 let dst = self.ids.get(dst_key).ok_or_else(|| GraphError::Corrupt {
3677 detail: format!("wal replay references unknown key {dst_key}"),
3678 })?;
3679 let etype = Arc::make_mut(&mut self.syms).intern(edge_type);
3680 // Skip if the edge is already visible in the merged base+overlay
3681 // view. This keeps WAL replay idempotent when the WAL contains
3682 // pre-snapshot records that are already encoded in a V8 base
3683 // (keep_wal=true opens and crash-before-truncation scenarios).
3684 if self.base.is_some()
3685 && self
3686 .topo_view()
3687 .neighbors(etype, Direction::Out, src)
3688 .contains(&dst)
3689 {
3690 return Ok(());
3691 }
3692 Arc::make_mut(&mut self.topo).add_edge(etype, src, dst);
3693 // View maintenance for manual edge insert.
3694 self.view_store.on_edge_changed(
3695 etype,
3696 src,
3697 dst,
3698 true,
3699 Arc::make_mut(&mut self.props),
3700 &build_topo_view(&self.topo, &self.base),
3701 &self.ids,
3702 &self.syms,
3703 &self.labels,
3704 base_columns(&self.base),
3705 );
3706 // Rule engine: via-hop rules must update when user edges change.
3707 let cursor = self.engine.pending_delta_count();
3708 let mut eng = std::mem::take(&mut self.engine);
3709 {
3710 let mut gm = make_graph_mut(
3711 &self.ids,
3712 Arc::make_mut(&mut self.syms),
3713 &self.labels,
3714 build_props_view(&self.props, &self.base),
3715 Arc::make_mut(&mut self.topo),
3716 &self.base,
3717 Arc::make_mut(&mut self.edge_props),
3718 );
3719 eng.on_edge_changed(edge_type, src, dst, &mut gm);
3720 }
3721 self.engine = eng;
3722 if !self.view_store.is_empty() {
3723 let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
3724 for d in &new_deltas {
3725 self.view_store.on_edge_changed(
3726 d.etype_sym,
3727 d.src_id,
3728 d.dst_id,
3729 d.fired,
3730 Arc::make_mut(&mut self.props),
3731 &build_topo_view(&self.topo, &self.base),
3732 &self.ids,
3733 &self.syms,
3734 &self.labels,
3735 base_columns(&self.base),
3736 );
3737 }
3738 }
3739 }
3740 WalRecord::SetProp { key, field, value } => {
3741 let id = self.ids.get(key).ok_or_else(|| GraphError::Corrupt {
3742 detail: format!("wal replay references unknown key {key}"),
3743 })?;
3744 let old_value = build_props_view(&self.props, &self.base)
3745 .get(id, field)
3746 .map(|vr| vr.into_value());
3747 Arc::make_mut(&mut self.props).set(id, field, value.clone());
3748 // Fire rules for the changed field.
3749 let cursor = self.engine.pending_delta_count();
3750 let mut eng = std::mem::take(&mut self.engine);
3751 {
3752 let mut gm = make_graph_mut(
3753 &self.ids,
3754 Arc::make_mut(&mut self.syms),
3755 &self.labels,
3756 build_props_view(&self.props, &self.base),
3757 Arc::make_mut(&mut self.topo),
3758 &self.base,
3759 Arc::make_mut(&mut self.edge_props),
3760 );
3761 eng.on_node_changed(id, Some((field, old_value)), &mut gm);
3762 }
3763 self.engine = eng;
3764 // Derived-edge deltas → view updates.
3765 if !self.view_store.is_empty() {
3766 #[cfg(test)]
3767 DELTA_COPY_COUNT.with(|c| c.set(c.get() + 1));
3768 let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
3769 for d in &new_deltas {
3770 self.view_store.on_edge_changed(
3771 d.etype_sym,
3772 d.src_id,
3773 d.dst_id,
3774 d.fired,
3775 Arc::make_mut(&mut self.props),
3776 &build_topo_view(&self.topo, &self.base),
3777 &self.ids,
3778 &self.syms,
3779 &self.labels,
3780 base_columns(&self.base),
3781 );
3782 }
3783 }
3784 // Neighbor-aggregate views that read `field` must also update.
3785 self.view_store.on_prop_changed(
3786 id,
3787 field,
3788 Arc::make_mut(&mut self.props),
3789 &build_topo_view(&self.topo, &self.base),
3790 &self.ids,
3791 &self.syms,
3792 &self.labels,
3793 base_columns(&self.base),
3794 );
3795 // Full-text index maintenance: update tokens for this field if indexed.
3796 if self.fulltext.field_indexed(field) {
3797 let label_opt = self.labels.get(id as usize).and_then(|&sym| {
3798 if sym == u32::MAX {
3799 None
3800 } else {
3801 self.syms.resolve(sym)
3802 }
3803 });
3804 if let Some(label) = label_opt {
3805 if self.fulltext.is_enabled(label, field) {
3806 Arc::make_mut(&mut self.fulltext).remove_node_field(id, field);
3807 Arc::make_mut(&mut self.fulltext).add_tokens(id, field, value);
3808 }
3809 }
3810 }
3811 // Property (equality) index maintenance: re-key this node's value.
3812 if self.prop_index.field_indexed(field) {
3813 let label_opt = self.labels.get(id as usize).and_then(|&sym| {
3814 if sym == u32::MAX {
3815 None
3816 } else {
3817 self.syms.resolve(sym)
3818 }
3819 });
3820 if let Some(label) = label_opt {
3821 self.prop_index.set(label, field, id, value);
3822 }
3823 }
3824 }
3825 WalRecord::Intern { id, text } => {
3826 if let Some(existing) = self.syms.get(text) {
3827 if existing != *id {
3828 return Err(GraphError::Corrupt {
3829 detail: format!(
3830 "wal intern mismatch for {text:?}: have {existing}, record {id}"
3831 ),
3832 });
3833 }
3834 } else {
3835 let got = Arc::make_mut(&mut self.syms).intern(text);
3836 if got != *id {
3837 return Err(GraphError::Corrupt {
3838 detail: format!(
3839 "wal intern assigned {got} for {text:?}, record wanted {id}"
3840 ),
3841 });
3842 }
3843 }
3844 }
3845 WalRecord::InsertNodeId { label, key, props } => {
3846 let id = Arc::make_mut(&mut self.ids).try_insert(key)?;
3847 if self.labels.len() <= id as usize {
3848 Arc::make_mut(&mut self.labels).resize(id as usize + 1, u32::MAX);
3849 }
3850 Arc::make_mut(&mut self.labels)[id as usize] = *label;
3851 let label_str = self
3852 .syms
3853 .resolve(*label)
3854 .ok_or_else(|| GraphError::Corrupt {
3855 detail: format!("wal InsertNodeId unknown label intern {label}"),
3856 })?
3857 .to_string();
3858 let mut ns_name = NS_DEFAULT.to_string();
3859 for (field_sym, value) in props {
3860 let field =
3861 self.syms
3862 .resolve(*field_sym)
3863 .ok_or_else(|| GraphError::Corrupt {
3864 detail: format!(
3865 "wal InsertNodeId unknown field intern {field_sym}"
3866 ),
3867 })?;
3868 if field == NS_PROP {
3869 ns_name = namespace_of_value(Some(value)).to_string();
3870 }
3871 Arc::make_mut(&mut self.props).set(id, field, value.clone());
3872 }
3873 self.set_node_ns(id, &ns_name);
3874 self.view_store.init_node_views(
3875 id,
3876 Arc::make_mut(&mut self.props),
3877 &self.syms,
3878 &self.labels,
3879 );
3880 let cursor = self.engine.pending_delta_count();
3881 let mut eng = std::mem::take(&mut self.engine);
3882 {
3883 let mut gm = make_graph_mut(
3884 &self.ids,
3885 Arc::make_mut(&mut self.syms),
3886 &self.labels,
3887 build_props_view(&self.props, &self.base),
3888 Arc::make_mut(&mut self.topo),
3889 &self.base,
3890 Arc::make_mut(&mut self.edge_props),
3891 );
3892 eng.on_node_changed(id, None, &mut gm);
3893 }
3894 self.engine = eng;
3895 if !self.view_store.is_empty() {
3896 #[cfg(test)]
3897 DELTA_COPY_COUNT.with(|c| c.set(c.get() + 1));
3898 let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
3899 for d in &new_deltas {
3900 self.view_store.on_edge_changed(
3901 d.etype_sym,
3902 d.src_id,
3903 d.dst_id,
3904 d.fired,
3905 Arc::make_mut(&mut self.props),
3906 &build_topo_view(&self.topo, &self.base),
3907 &self.ids,
3908 &self.syms,
3909 &self.labels,
3910 base_columns(&self.base),
3911 );
3912 }
3913 }
3914 if self.fulltext.has_label(&label_str) {
3915 for (field_sym, value) in props {
3916 let Some(field) = self.syms.resolve(*field_sym) else {
3917 continue;
3918 };
3919 if self.fulltext.is_enabled(&label_str, field) {
3920 Arc::make_mut(&mut self.fulltext).add_tokens(id, field, value);
3921 }
3922 }
3923 }
3924 if self.prop_index.has_label(&label_str) {
3925 for (field_sym, value) in props {
3926 let Some(field) = self.syms.resolve(*field_sym) else {
3927 continue;
3928 };
3929 self.prop_index.set(&label_str, field, id, value);
3930 }
3931 }
3932 }
3933 WalRecord::InsertEdgeId { etype, src, dst } => {
3934 // Replay-over-snapshot: dense ids in the pre-snapshot WAL may
3935 // already be tombstoned. Skip rather than attaching edges to
3936 // dead ids (DeleteNode keys the live re-insert, not the old id).
3937 if self.ids.is_tombstoned(*src)
3938 || self.ids.is_tombstoned(*dst)
3939 || self.ids.key_of(*src).is_none()
3940 || self.ids.key_of(*dst).is_none()
3941 {
3942 return Ok(());
3943 }
3944 // Skip if already visible in the merged view (same idempotency
3945 // guard as InsertEdge above: prevents double-counting when
3946 // pre-snapshot WAL records are replayed over a V8 base).
3947 if self.base.is_some()
3948 && self
3949 .topo_view()
3950 .neighbors(*etype, Direction::Out, *src)
3951 .contains(dst)
3952 {
3953 return Ok(());
3954 }
3955 Arc::make_mut(&mut self.topo).add_edge(*etype, *src, *dst);
3956 self.view_store.on_edge_changed(
3957 *etype,
3958 *src,
3959 *dst,
3960 true,
3961 Arc::make_mut(&mut self.props),
3962 &build_topo_view(&self.topo, &self.base),
3963 &self.ids,
3964 &self.syms,
3965 &self.labels,
3966 base_columns(&self.base),
3967 );
3968 // Rule engine: via-hop rules fire when user via-edges are inserted.
3969 // Resolve etype back to string so on_edge_changed can match rules by name.
3970 if let Some(etype_str) = self.syms.resolve(*etype).map(|s| s.to_string()) {
3971 let cursor = self.engine.pending_delta_count();
3972 let mut eng = std::mem::take(&mut self.engine);
3973 {
3974 let mut gm = make_graph_mut(
3975 &self.ids,
3976 Arc::make_mut(&mut self.syms),
3977 &self.labels,
3978 build_props_view(&self.props, &self.base),
3979 Arc::make_mut(&mut self.topo),
3980 &self.base,
3981 Arc::make_mut(&mut self.edge_props),
3982 );
3983 eng.on_edge_changed(&etype_str, *src, *dst, &mut gm);
3984 }
3985 self.engine = eng;
3986 if !self.view_store.is_empty() {
3987 let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
3988 for d in &new_deltas {
3989 self.view_store.on_edge_changed(
3990 d.etype_sym,
3991 d.src_id,
3992 d.dst_id,
3993 d.fired,
3994 Arc::make_mut(&mut self.props),
3995 &build_topo_view(&self.topo, &self.base),
3996 &self.ids,
3997 &self.syms,
3998 &self.labels,
3999 base_columns(&self.base),
4000 );
4001 }
4002 }
4003 }
4004 }
4005 WalRecord::SetPropId { id, field, value } => {
4006 if self.ids.is_tombstoned(*id) || self.ids.key_of(*id).is_none() {
4007 return Ok(());
4008 }
4009 let field_str = self
4010 .syms
4011 .resolve(*field)
4012 .ok_or_else(|| GraphError::Corrupt {
4013 detail: format!("wal SetPropId unknown field intern {field}"),
4014 })?
4015 .to_string();
4016 let old_value = build_props_view(&self.props, &self.base)
4017 .get(*id, &field_str)
4018 .map(|vr| vr.into_value());
4019 Arc::make_mut(&mut self.props).set(*id, &field_str, value.clone());
4020 let cursor = self.engine.pending_delta_count();
4021 let mut eng = std::mem::take(&mut self.engine);
4022 {
4023 let mut gm = make_graph_mut(
4024 &self.ids,
4025 Arc::make_mut(&mut self.syms),
4026 &self.labels,
4027 build_props_view(&self.props, &self.base),
4028 Arc::make_mut(&mut self.topo),
4029 &self.base,
4030 Arc::make_mut(&mut self.edge_props),
4031 );
4032 eng.on_node_changed(*id, Some((field_str.as_str(), old_value)), &mut gm);
4033 }
4034 self.engine = eng;
4035 if !self.view_store.is_empty() {
4036 #[cfg(test)]
4037 DELTA_COPY_COUNT.with(|c| c.set(c.get() + 1));
4038 let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
4039 for d in &new_deltas {
4040 self.view_store.on_edge_changed(
4041 d.etype_sym,
4042 d.src_id,
4043 d.dst_id,
4044 d.fired,
4045 Arc::make_mut(&mut self.props),
4046 &build_topo_view(&self.topo, &self.base),
4047 &self.ids,
4048 &self.syms,
4049 &self.labels,
4050 base_columns(&self.base),
4051 );
4052 }
4053 }
4054 self.view_store.on_prop_changed(
4055 *id,
4056 &field_str,
4057 Arc::make_mut(&mut self.props),
4058 &build_topo_view(&self.topo, &self.base),
4059 &self.ids,
4060 &self.syms,
4061 &self.labels,
4062 base_columns(&self.base),
4063 );
4064 if self.fulltext.field_indexed(&field_str) {
4065 let label_opt = self.labels.get(*id as usize).and_then(|&sym| {
4066 if sym == u32::MAX {
4067 None
4068 } else {
4069 self.syms.resolve(sym)
4070 }
4071 });
4072 if let Some(label) = label_opt {
4073 if self.fulltext.is_enabled(label, &field_str) {
4074 Arc::make_mut(&mut self.fulltext).remove_node_field(*id, &field_str);
4075 Arc::make_mut(&mut self.fulltext).add_tokens(*id, &field_str, value);
4076 }
4077 }
4078 }
4079 if self.prop_index.field_indexed(&field_str) {
4080 let label_opt = self.labels.get(*id as usize).and_then(|&sym| {
4081 if sym == u32::MAX {
4082 None
4083 } else {
4084 self.syms.resolve(sym)
4085 }
4086 });
4087 if let Some(label) = label_opt {
4088 self.prop_index.set(label, &field_str, *id, value);
4089 }
4090 }
4091 }
4092 WalRecord::CreateRule { def_bytes } => {
4093 let def: RuleDef = decode_rule_def(def_bytes).map_err(|e| GraphError::Corrupt {
4094 detail: format!("CreateRule def_bytes deserialize failed: {e}"),
4095 })?;
4096 // Replay-over-snapshot idempotency: the rule was captured in the snapshot
4097 // so the engine already has it; silently skip to avoid a spurious
4098 // RuleInvalid error in the crash window between snapshot write and WAL
4099 // truncation.
4100 if self.engine.rules().any(|r| r.name == def.name) {
4101 return Ok(());
4102 }
4103 let cursor = self.engine.pending_delta_count();
4104 let mut eng = std::mem::take(&mut self.engine);
4105 let result = {
4106 let mut gm = make_graph_mut(
4107 &self.ids,
4108 Arc::make_mut(&mut self.syms),
4109 &self.labels,
4110 build_props_view(&self.props, &self.base),
4111 Arc::make_mut(&mut self.topo),
4112 &self.base,
4113 Arc::make_mut(&mut self.edge_props),
4114 );
4115 eng.create_rule(def, &mut gm)
4116 };
4117 self.engine = eng;
4118 result.map_err(|e| GraphError::RuleInvalid { detail: e })?;
4119 // Derived-edge fires from backfill → view updates.
4120 // Fast path: skip O(edge_count) allocation when no views exist.
4121 if !self.view_store.is_empty() {
4122 #[cfg(test)]
4123 DELTA_COPY_COUNT.with(|c| c.set(c.get() + 1));
4124 let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
4125 for d in &new_deltas {
4126 self.view_store.on_edge_changed(
4127 d.etype_sym,
4128 d.src_id,
4129 d.dst_id,
4130 d.fired,
4131 Arc::make_mut(&mut self.props),
4132 &build_topo_view(&self.topo, &self.base),
4133 &self.ids,
4134 &self.syms,
4135 &self.labels,
4136 base_columns(&self.base),
4137 );
4138 }
4139 }
4140 }
4141 WalRecord::DeleteRule { name } => {
4142 // Replay-over-snapshot idempotency: the snapshot already captured the
4143 // post-delete state so the rule is absent; silently skip to avoid a
4144 // spurious RuleNotFound error in the crash window between snapshot write
4145 // and WAL truncation.
4146 if !self.engine.rules().any(|r| r.name == *name) {
4147 return Ok(());
4148 }
4149 let cursor = self.engine.pending_delta_count();
4150 let mut eng = std::mem::take(&mut self.engine);
4151 let result = {
4152 let mut gm = make_graph_mut(
4153 &self.ids,
4154 Arc::make_mut(&mut self.syms),
4155 &self.labels,
4156 build_props_view(&self.props, &self.base),
4157 Arc::make_mut(&mut self.topo),
4158 &self.base,
4159 Arc::make_mut(&mut self.edge_props),
4160 );
4161 eng.delete_rule(name, &mut gm)
4162 };
4163 self.engine = eng;
4164 result.map_err(|_| GraphError::RuleNotFound { name: name.clone() })?;
4165 // Derived-edge retractions → view updates.
4166 if !self.view_store.is_empty() {
4167 #[cfg(test)]
4168 DELTA_COPY_COUNT.with(|c| c.set(c.get() + 1));
4169 let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
4170 for d in &new_deltas {
4171 self.view_store.on_edge_changed(
4172 d.etype_sym,
4173 d.src_id,
4174 d.dst_id,
4175 d.fired,
4176 Arc::make_mut(&mut self.props),
4177 &build_topo_view(&self.topo, &self.base),
4178 &self.ids,
4179 &self.syms,
4180 &self.labels,
4181 base_columns(&self.base),
4182 );
4183 }
4184 }
4185 }
4186 WalRecord::RemoveProp { key, field } => {
4187 // Recovery-safe: unknown key or already-absent field is a
4188 // clean no-op. Crash-window replay over a snapshot that
4189 // already applied this record must not Err.
4190 let Some(id) = self.ids.get(key) else {
4191 return Ok(());
4192 };
4193 // Read old value through the seam for rule retraction.
4194 let old = build_props_view(&self.props, &self.base)
4195 .get(id, field)
4196 .map(|vr| vr.into_value());
4197 Arc::make_mut(&mut self.props).remove(id, field);
4198 // If the base still supplies the value after the overlay removal,
4199 // record a tombstone so ColumnsView::get does not resurrect it.
4200 // This covers both the base-only case AND the both-resident case:
4201 // base-only (in_overlay=false): old prop was only in base, remove
4202 // is a no-op on overlay, base still visible → tombstone needed.
4203 // both-resident (in_overlay=true): overlay had v2, base has v1;
4204 // removing overlay uncovers v1 → tombstone needed.
4205 // Idempotent on double-replay: second pass sees the tombstone →
4206 // get() returns None → condition is false → no duplicate tombstone.
4207 if build_props_view(&self.props, &self.base)
4208 .get(id, field)
4209 .is_some()
4210 {
4211 Arc::make_mut(&mut self.props).record_prop_tombstone(id, field);
4212 }
4213 let cursor = self.engine.pending_delta_count();
4214 let mut eng = std::mem::take(&mut self.engine);
4215 {
4216 let mut gm = make_graph_mut(
4217 &self.ids,
4218 Arc::make_mut(&mut self.syms),
4219 &self.labels,
4220 build_props_view(&self.props, &self.base),
4221 Arc::make_mut(&mut self.topo),
4222 &self.base,
4223 Arc::make_mut(&mut self.edge_props),
4224 );
4225 eng.on_node_changed(id, Some((field, old)), &mut gm);
4226 }
4227 self.engine = eng;
4228 // Derived-edge deltas → view updates.
4229 if !self.view_store.is_empty() {
4230 #[cfg(test)]
4231 DELTA_COPY_COUNT.with(|c| c.set(c.get() + 1));
4232 let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
4233 for d in &new_deltas {
4234 self.view_store.on_edge_changed(
4235 d.etype_sym,
4236 d.src_id,
4237 d.dst_id,
4238 d.fired,
4239 Arc::make_mut(&mut self.props),
4240 &build_topo_view(&self.topo, &self.base),
4241 &self.ids,
4242 &self.syms,
4243 &self.labels,
4244 base_columns(&self.base),
4245 );
4246 }
4247 }
4248 // Neighbor-aggregate views that read `field` must also update.
4249 self.view_store.on_prop_changed(
4250 id,
4251 field,
4252 Arc::make_mut(&mut self.props),
4253 &build_topo_view(&self.topo, &self.base),
4254 &self.ids,
4255 &self.syms,
4256 &self.labels,
4257 base_columns(&self.base),
4258 );
4259 // Full-text index maintenance: remove tokens for this field.
4260 if self.fulltext.field_indexed(field) {
4261 Arc::make_mut(&mut self.fulltext).remove_node_field(id, field);
4262 }
4263 // Property (equality) index maintenance: drop this node's entry.
4264 if self.prop_index.field_indexed(field) {
4265 if let Some(label) = self.labels.get(id as usize).and_then(|&sym| {
4266 (sym != u32::MAX).then(|| self.syms.resolve(sym)).flatten()
4267 }) {
4268 self.prop_index.remove_node(label, field, id);
4269 }
4270 }
4271 }
4272 WalRecord::DeleteEdge {
4273 edge_type,
4274 src_key,
4275 dst_key,
4276 } => {
4277 // Recovery-safe: unknown keys, unknown etype, or already-
4278 // absent edge is a clean no-op (remove_edge returns false).
4279 let Some(src) = self.ids.get(src_key) else {
4280 return Ok(());
4281 };
4282 let Some(dst) = self.ids.get(dst_key) else {
4283 return Ok(());
4284 };
4285 let Some(etype) = self.syms.get(edge_type) else {
4286 return Ok(());
4287 };
4288 // I3: phantom-tombstone guard. When a V8 base is present, a
4289 // DeleteEdge WAL record for an edge that was already absorbed into
4290 // the new base (i.e. neither in overlay nor in base) must be skipped.
4291 // Without this guard, remove_edge records a tombstone for an edge
4292 // that no longer exists, incorrectly understating edge_count.
4293 if self.base.is_some()
4294 && !self
4295 .topo_view()
4296 .neighbors(etype, core_storage::topology::Direction::Out, src)
4297 .contains(&dst)
4298 {
4299 return Ok(());
4300 }
4301 Arc::make_mut(&mut self.topo).remove_edge(etype, src, dst);
4302 Arc::make_mut(&mut self.edge_props).remove_edge(etype, src, dst);
4303 // View maintenance for manual edge delete (topo already updated above).
4304 self.view_store.on_edge_changed(
4305 etype,
4306 src,
4307 dst,
4308 false,
4309 Arc::make_mut(&mut self.props),
4310 &build_topo_view(&self.topo, &self.base),
4311 &self.ids,
4312 &self.syms,
4313 &self.labels,
4314 base_columns(&self.base),
4315 );
4316 // Rule engine: via-hop rules must retract when user via-edges are deleted.
4317 let cursor = self.engine.pending_delta_count();
4318 let mut eng = std::mem::take(&mut self.engine);
4319 {
4320 let mut gm = make_graph_mut(
4321 &self.ids,
4322 Arc::make_mut(&mut self.syms),
4323 &self.labels,
4324 build_props_view(&self.props, &self.base),
4325 Arc::make_mut(&mut self.topo),
4326 &self.base,
4327 Arc::make_mut(&mut self.edge_props),
4328 );
4329 eng.on_edge_changed(edge_type, src, dst, &mut gm);
4330 }
4331 self.engine = eng;
4332 if !self.view_store.is_empty() {
4333 let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
4334 for d in &new_deltas {
4335 self.view_store.on_edge_changed(
4336 d.etype_sym,
4337 d.src_id,
4338 d.dst_id,
4339 d.fired,
4340 Arc::make_mut(&mut self.props),
4341 &build_topo_view(&self.topo, &self.base),
4342 &self.ids,
4343 &self.syms,
4344 &self.labels,
4345 base_columns(&self.base),
4346 );
4347 }
4348 }
4349 }
4350 WalRecord::DeleteNode { key } => {
4351 // Recovery-safe: already-tombstoned / unknown key is a clean
4352 // no-op. Crash-window replay over a snapshot that already
4353 // applied this record cannot recover the retired id from the
4354 // key (`IdMap::get` is None), so every subsequent step is
4355 // skipped. Each step is independently idempotent if invoked
4356 // twice on a still-live id: retraction is a no-op on empty
4357 // provenance, `remove_edge` returns false, `remove_all` is a
4358 // no-op, `ids.delete` returns None, label sentinel is sticky.
4359 let Some(n) = self.ids.get(key) else {
4360 return Ok(());
4361 };
4362
4363 // (1) Retract derived edges + de-index while props/labels live.
4364 let cursor = self.engine.pending_delta_count();
4365 let mut eng = std::mem::take(&mut self.engine);
4366 {
4367 let mut gm = make_graph_mut(
4368 &self.ids,
4369 Arc::make_mut(&mut self.syms),
4370 &self.labels,
4371 build_props_view(&self.props, &self.base),
4372 Arc::make_mut(&mut self.topo),
4373 &self.base,
4374 Arc::make_mut(&mut self.edge_props),
4375 );
4376 eng.on_node_removed(n, &mut gm);
4377 }
4378 self.engine = eng;
4379 // Derived-edge retractions → view updates for neighbors.
4380 if !self.view_store.is_empty() {
4381 #[cfg(test)]
4382 DELTA_COPY_COUNT.with(|c| c.set(c.get() + 1));
4383 let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
4384 for d in &new_deltas {
4385 self.view_store.on_edge_changed(
4386 d.etype_sym,
4387 d.src_id,
4388 d.dst_id,
4389 d.fired,
4390 Arc::make_mut(&mut self.props),
4391 &build_topo_view(&self.topo, &self.base),
4392 &self.ids,
4393 &self.syms,
4394 &self.labels,
4395 base_columns(&self.base),
4396 );
4397 }
4398 }
4399
4400 // (2) Sweep ALL remaining edges incident to n, both directions,
4401 // every etype. This cascade is intentionally mask-independent:
4402 // topology integrity requires removing every edge touching the
4403 // deleted node regardless of the caller's visibility scope.
4404 // (The mask limits which nodes a role's read phase can return;
4405 // the WAL delete always executes with full storage authority.)
4406 // Collect then remove so neighbor slices stay valid during
4407 // iteration. Remove from topo first, then call view maintenance
4408 // so Avg/Min/Max recompute sees the correct (reduced) neighbor set.
4409 let etypes: Vec<u32> = self.topo.etypes().collect();
4410 let mut doomed = Vec::new();
4411 for et in &etypes {
4412 for &dst in self.topo.neighbors(*et, Direction::Out, n).as_ref() {
4413 doomed.push((*et, n, dst));
4414 }
4415 for &src in self.topo.neighbors(*et, Direction::In, n).as_ref() {
4416 doomed.push((*et, src, n));
4417 }
4418 }
4419 for (et, s, d) in doomed {
4420 Arc::make_mut(&mut self.topo).remove_edge(et, s, d);
4421 Arc::make_mut(&mut self.edge_props).remove_edge(et, s, d);
4422 // View maintenance: n's own view values will be cleared by
4423 // remove_all below; only update surviving neighbors.
4424 self.view_store.on_edge_changed(
4425 et,
4426 s,
4427 d,
4428 false,
4429 Arc::make_mut(&mut self.props),
4430 &build_topo_view(&self.topo, &self.base),
4431 &self.ids,
4432 &self.syms,
4433 &self.labels,
4434 base_columns(&self.base),
4435 );
4436 }
4437
4438 // (3) Drop every remaining prop (`ColumnStore::remove_all`).
4439 Arc::make_mut(&mut self.props).remove_all(n);
4440 // Full-text index maintenance: remove all tokens for this node.
4441 Arc::make_mut(&mut self.fulltext).remove_node(n);
4442 // Property (equality) index maintenance: drop all entries for n.
4443 self.prop_index.remove_node_all(n);
4444
4445 // (4) Retire the dense id and stamp the label sentinel.
4446 Arc::make_mut(&mut self.ids).delete(key);
4447 if let Some(slot) = Arc::make_mut(&mut self.labels).get_mut(n as usize) {
4448 *slot = u32::MAX;
4449 }
4450 }
4451 WalRecord::Batch(inner) => {
4452 // Apply each inner record in order through the same apply path.
4453 // Inner records are validated free of nested Batch by encode_record.
4454 for rec in inner {
4455 self.apply(rec)?;
4456 }
4457 }
4458 WalRecord::RebuildRule { name } => {
4459 // Replay-over-snapshot idempotency: the snapshot may already
4460 // reflect a later delete_rule, so the rule is absent; skip.
4461 if !self.engine.rules().any(|r| r.name == *name) {
4462 return Ok(());
4463 }
4464 let cursor = self.engine.pending_delta_count();
4465 let mut eng = std::mem::take(&mut self.engine);
4466 let result = {
4467 let mut gm = make_graph_mut(
4468 &self.ids,
4469 Arc::make_mut(&mut self.syms),
4470 &self.labels,
4471 build_props_view(&self.props, &self.base),
4472 Arc::make_mut(&mut self.topo),
4473 &self.base,
4474 Arc::make_mut(&mut self.edge_props),
4475 );
4476 eng.rebuild(name, &mut gm)
4477 };
4478 self.engine = eng;
4479 result.map_err(|_| GraphError::RuleNotFound { name: name.clone() })?;
4480 // Derived-edge delta changes → view updates.
4481 if !self.view_store.is_empty() {
4482 #[cfg(test)]
4483 DELTA_COPY_COUNT.with(|c| c.set(c.get() + 1));
4484 let new_deltas: Vec<_> = self.engine.pending_deltas_since(cursor).to_vec();
4485 for d in &new_deltas {
4486 self.view_store.on_edge_changed(
4487 d.etype_sym,
4488 d.src_id,
4489 d.dst_id,
4490 d.fired,
4491 Arc::make_mut(&mut self.props),
4492 &build_topo_view(&self.topo, &self.base),
4493 &self.ids,
4494 &self.syms,
4495 &self.labels,
4496 base_columns(&self.base),
4497 );
4498 }
4499 }
4500 }
4501 WalRecord::CreateView { def_bytes } => {
4502 let def: ViewDef =
4503 bincode::deserialize(def_bytes).map_err(|e| GraphError::Corrupt {
4504 detail: format!("CreateView def_bytes deserialize failed: {e}"),
4505 })?;
4506 // Replay-over-snapshot idempotency: view already present → skip.
4507 if self.view_store.has_view(&def.name) {
4508 return Ok(());
4509 }
4510 self.view_store
4511 .create_view(
4512 def,
4513 Arc::make_mut(&mut self.props),
4514 &build_topo_view(&self.topo, &self.base),
4515 &self.ids,
4516 &self.syms,
4517 &self.labels,
4518 )
4519 .map_err(|e| GraphError::RuleInvalid { detail: e })?;
4520 }
4521 WalRecord::DeleteView { name } => {
4522 // Replay-over-snapshot idempotency: view already absent → skip.
4523 if !self.view_store.has_view(name) {
4524 return Ok(());
4525 }
4526 self.view_store
4527 .delete_view(
4528 name,
4529 Arc::make_mut(&mut self.props),
4530 &self.ids,
4531 &self.labels,
4532 &self.syms,
4533 )
4534 .map_err(|_| GraphError::RuleNotFound { name: name.clone() })?;
4535 }
4536 WalRecord::EnableFulltext { label, field } => {
4537 // Replay-over-snapshot idempotency: already enabled → skip.
4538 if self.fulltext.is_enabled(label, field) {
4539 return Ok(());
4540 }
4541 Arc::make_mut(&mut self.fulltext).enable(label, field);
4542 if self.fulltext_rebuild_follows {
4543 // The open path rebuilds the whole index after replay, which
4544 // clears every posting this scan would write. Doing it twice
4545 // costs a full tokenise-and-stem pass over the corpus per
4546 // enabled pair: measured at 456 ms against 3.8 ms for the
4547 // same 30,000-entity store with no pair enabled.
4548 return Ok(());
4549 }
4550 // Backfill: index all live nodes of this label that have the field.
4551 let n = self.ids.len() as u32;
4552 for id in 0..n {
4553 let Some(&sym) = self.labels.get(id as usize) else {
4554 continue;
4555 };
4556 if sym == u32::MAX {
4557 continue; // tombstoned
4558 }
4559 let Some(lbl) = self.syms.resolve(sym) else {
4560 continue;
4561 };
4562 if lbl != label {
4563 continue;
4564 }
4565 if let Some(value) = build_props_view(&self.props, &self.base)
4566 .get(id, field)
4567 .map(|vr| vr.into_value())
4568 {
4569 Arc::make_mut(&mut self.fulltext).add_tokens(id, field, &value);
4570 }
4571 }
4572 }
4573 WalRecord::DisableFulltext { label, field } => {
4574 // Replay-over-snapshot idempotency: already disabled → skip.
4575 if !self.fulltext.is_enabled(label, field) {
4576 return Ok(());
4577 }
4578 // If another label still indexes this field, the postings column
4579 // is kept — but it must not contain node_ids from the now-disabled
4580 // label. Remove them before calling disable() so the field_indexed
4581 // guard inside disable() sees the correct post-removal state.
4582 if self.fulltext.field_indexed_by_other(label, field) {
4583 if let Some(label_sym) = self.syms.get(label) {
4584 for (node_id, &lsym) in self.labels.iter().enumerate() {
4585 if lsym == label_sym {
4586 Arc::make_mut(&mut self.fulltext)
4587 .remove_node_field(node_id as u32, field);
4588 }
4589 }
4590 }
4591 }
4592 Arc::make_mut(&mut self.fulltext).disable(label, field);
4593 }
4594 WalRecord::EnableIndex { label, field } => {
4595 // Replay-over-snapshot idempotency: already enabled → skip.
4596 if self.prop_index.is_enabled(label, field) {
4597 return Ok(());
4598 }
4599 self.prop_index.enable(label, field);
4600 // Backfill: index all live nodes of this label that have the field.
4601 let n = self.ids.len() as u32;
4602 for id in 0..n {
4603 let Some(&sym) = self.labels.get(id as usize) else {
4604 continue;
4605 };
4606 if sym == u32::MAX {
4607 continue; // tombstoned
4608 }
4609 let Some(lbl) = self.syms.resolve(sym) else {
4610 continue;
4611 };
4612 if lbl != label {
4613 continue;
4614 }
4615 if let Some(value) = build_props_view(&self.props, &self.base)
4616 .get(id, field)
4617 .map(|vr| vr.into_value())
4618 {
4619 self.prop_index.set(label, field, id, &value);
4620 }
4621 }
4622 }
4623 WalRecord::DisableIndex { label, field } => {
4624 self.prop_index.disable(label, field);
4625 }
4626 // ── insert-count multiplicity (§5.13) ────────────────────────────
4627 //
4628 // Two shapes, told apart by `count`: the opt-in declaration, and an
4629 // absolute count for one triple. Absolute is what makes this
4630 // idempotent over a snapshot base — a pre-snapshot frame replayed
4631 // over a base that already folded it in lands on the same number
4632 // rather than adding to it, which is the failure a delta (or a count
4633 // derived from `InsertEdgeId` records) would have.
4634 WalRecord::SetEdgeCount {
4635 etype,
4636 src,
4637 dst,
4638 count,
4639 } => {
4640 if rec.is_multiplicity_decl() {
4641 self.multiplicity = true;
4642 } else {
4643 Arc::make_mut(&mut self.edge_props).set(
4644 *etype,
4645 *src,
4646 *dst,
4647 EDGE_COUNT_PROP,
4648 Value::Int(*count as i64),
4649 );
4650 }
4651 }
4652 // History markers carry no replay state — rules re-derive edges
4653 // deterministically on open/replay. Skip unconditionally.
4654 WalRecord::DerivedEdgeAdded { .. } | WalRecord::DerivedEdgeRetracted { .. } => {}
4655 // ── rename_node ──────────────────────────────────────────────────
4656 WalRecord::RenameNode { old_key, new_key } => {
4657 // Recovery-safe: if old_key is already gone (key was renamed
4658 // by a snapshot or a prior replay frame), skip cleanly.
4659 if self.ids.get(old_key).is_none() {
4660 return Ok(());
4661 }
4662 // The rename only updates the key-table; the dense id, all
4663 // topo edges, props, labels, and rule state are id-indexed and
4664 // require no change.
4665 Arc::make_mut(&mut self.ids)
4666 .rename(old_key, new_key)
4667 .map_err(|e| GraphError::Corrupt {
4668 detail: format!("wal replay RenameNode {old_key}→{new_key}: {e}"),
4669 })?;
4670 }
4671 }
4672 Ok(())
4673 }
4674
4675 /// Intern `s` in `syms` and emit a WAL `Intern` record so `*Id` records
4676 /// replay on WAL-only `open_at` (no snapshot intern table). Apply is
4677 /// idempotent when the string is already bound. Always emit: after
4678 /// `snapshot()` the WAL is truncated and live intern is not on disk.
4679 fn intern_wal(&mut self, s: &str) -> (u32, WalRecord) {
4680 let id = if let Some(id) = self.syms.get(s) {
4681 id
4682 } else {
4683 Arc::make_mut(&mut self.syms).intern(s)
4684 };
4685 (
4686 id,
4687 WalRecord::Intern {
4688 id,
4689 text: s.to_string(),
4690 },
4691 )
4692 }
4693
4694 /// Rewrite user-facing records into dense-id records. On `Err`, no live
4695 /// state is left mutated: speculative interns made while building the
4696 /// output are rolled back, so a later successful mutation cannot log an
4697 /// `Intern` record whose id replay would never reproduce.
4698 fn rewrite_wal_dense(&mut self, recs: Vec<WalRecord>) -> Result<Vec<WalRecord>> {
4699 self.rewrite_wal_dense_planned(recs.into_iter().map(PlannedRec::Rec).collect())
4700 }
4701
4702 /// [`rewrite_wal_dense`](Self::rewrite_wal_dense) for a frame that still
4703 /// carries [`PlannedRec::DuplicateCount`] entries — the shape a batch
4704 /// produces, where a duplicate's count can only be named once this pass has
4705 /// assigned the frame's own ids.
4706 fn rewrite_wal_dense_planned(&mut self, recs: Vec<PlannedRec>) -> Result<Vec<WalRecord>> {
4707 let syms_checkpoint = self.syms.len();
4708 let result = self.rewrite_wal_dense_inner(recs);
4709 if result.is_err() {
4710 Arc::make_mut(&mut self.syms).truncate(syms_checkpoint);
4711 }
4712 result
4713 }
4714
4715 fn rewrite_wal_dense_inner(&mut self, recs: Vec<PlannedRec>) -> Result<Vec<WalRecord>> {
4716 let mut out = Vec::with_capacity(recs.len());
4717 // Node ids allocated by later apply(InsertNodeId) in this same batch.
4718 let mut pending: std::collections::HashMap<String, u32> = std::collections::HashMap::new();
4719 // Namespace of each node inserted earlier in this same frame, so a SET
4720 // on a node this frame created is measured against the namespace it was
4721 // created in rather than against the store, where it does not exist yet.
4722 let mut pending_ns: std::collections::HashMap<String, String> =
4723 std::collections::HashMap::new();
4724 let mut interned = std::collections::HashSet::<u32>::new();
4725 let mut next = u32::try_from(self.ids.len()).map_err(|_| GraphError::Corrupt {
4726 detail: "id space exhausted".into(),
4727 })?;
4728 // Insert counts this frame has already raised. `edge_insert_count`
4729 // reads committed state, which cannot see a count queued earlier in
4730 // this same frame, so N duplicates of one pair would otherwise all
4731 // compute `committed + 1` and the last would win.
4732 let mut pending_counts: HashMap<(u32, u32, u32), u64> = HashMap::new();
4733 let lookup = |ids: &IdMap,
4734 pending: &std::collections::HashMap<String, u32>,
4735 key: &str|
4736 -> Option<u32> { ids.get(key).or_else(|| pending.get(key).copied()) };
4737 for rec in recs {
4738 // A duplicate insert's count, resolved here and nowhere else.
4739 //
4740 // This is the only pass that knows the frame's own ids: a node
4741 // created earlier in the same frame has no dense id until the
4742 // `InsertNodeId` above allocates one, and an edge type first used in
4743 // this frame is not in `syms` until `intern_wal` puts it there.
4744 // Resolving the count in the batch's validate pass instead — where
4745 // it used to live — meant that a duplicate whose endpoints or type
4746 // were created in the same frame silently produced no count at all,
4747 // which is exactly the shape a mirror rebuild writes (defect #24).
4748 let rec = match rec {
4749 PlannedRec::Rec(rec) => rec,
4750 PlannedRec::DuplicateCount {
4751 edge_type,
4752 src_key,
4753 dst_key,
4754 } => {
4755 let (etype, intern) = self.intern_wal(&edge_type);
4756 if interned.insert(etype) {
4757 out.push(intern);
4758 }
4759 let src = lookup(&self.ids, &pending, &src_key).ok_or_else(|| {
4760 GraphError::Corrupt {
4761 detail: format!("dense WAL rewrite missing src {src_key}"),
4762 }
4763 })?;
4764 let dst = lookup(&self.ids, &pending, &dst_key).ok_or_else(|| {
4765 GraphError::Corrupt {
4766 detail: format!("dense WAL rewrite missing dst {dst_key}"),
4767 }
4768 })?;
4769 let count = pending_counts
4770 .get(&(etype, src, dst))
4771 .copied()
4772 .unwrap_or_else(|| self.edge_insert_count(etype, src, dst))
4773 .saturating_add(1);
4774 pending_counts.insert((etype, src, dst), count);
4775 out.push(WalRecord::SetEdgeCount {
4776 etype,
4777 src,
4778 dst,
4779 count,
4780 });
4781 continue;
4782 }
4783 };
4784 match rec {
4785 WalRecord::InsertNode { label, key, props } => {
4786 // Namespace validation and normalisation, on the one seam
4787 // every user-visible node insert passes through: insert_node,
4788 // a batch, ingest, Cypher CREATE and MERGE all arrive here
4789 // before the WAL append, and replay never does.
4790 let (props, ns_name) = Self::normalise_insert_ns(&key, props)?;
4791 pending_ns.insert(key.clone(), ns_name);
4792 let (label_id, intern) = self.intern_wal(&label);
4793 if interned.insert(label_id) {
4794 out.push(intern);
4795 }
4796 let mut props_id = Vec::with_capacity(props.len());
4797 for (field, value) in props {
4798 let (field_id, intern) = self.intern_wal(&field);
4799 if interned.insert(field_id) {
4800 out.push(intern);
4801 }
4802 props_id.push((field_id, value));
4803 }
4804 if lookup(&self.ids, &pending, &key).is_none() {
4805 pending.insert(key.clone(), next);
4806 next = next.checked_add(1).ok_or_else(|| GraphError::Corrupt {
4807 detail: "id space exhausted".into(),
4808 })?;
4809 }
4810 out.push(WalRecord::InsertNodeId {
4811 label: label_id,
4812 key,
4813 props: props_id,
4814 });
4815 }
4816 WalRecord::SetProp { key, field, value } => {
4817 // A namespace is set at insert and fixed after: the write is
4818 // refused when it would move the node, and dropped when it
4819 // names the namespace the node is already in. Checked here
4820 // so set_prop, a batch, Cypher SET/MERGE and every upsert
4821 // that merges props get the same answer.
4822 if field == NS_PROP {
4823 let Value::Str(ref to) = value else {
4824 return Err(GraphError::RuleInvalid {
4825 detail: format!(
4826 "node {key}: {NS_PROP} must be a string naming a namespace, \
4827 got {value:?}"
4828 ),
4829 });
4830 };
4831 let from = pending_ns
4832 .get(&key)
4833 .cloned()
4834 .or_else(|| self.namespace_of(&key))
4835 .unwrap_or_else(|| NS_DEFAULT.to_string());
4836 let to = to.clone();
4837 if to != from {
4838 return Err(GraphError::NamespaceImmutable {
4839 key: key.clone(),
4840 from,
4841 to,
4842 });
4843 }
4844 continue;
4845 }
4846 let id =
4847 lookup(&self.ids, &pending, &key).ok_or_else(|| GraphError::Corrupt {
4848 detail: format!("dense WAL rewrite missing key {key}"),
4849 })?;
4850 let (field_id, intern) = self.intern_wal(&field);
4851 if interned.insert(field_id) {
4852 out.push(intern);
4853 }
4854 out.push(WalRecord::SetPropId {
4855 id,
4856 field: field_id,
4857 value,
4858 });
4859 }
4860 WalRecord::InsertEdge {
4861 edge_type,
4862 src_key,
4863 dst_key,
4864 } => {
4865 let (etype, intern) = self.intern_wal(&edge_type);
4866 if interned.insert(etype) {
4867 out.push(intern);
4868 }
4869 let src = lookup(&self.ids, &pending, &src_key).ok_or_else(|| {
4870 GraphError::Corrupt {
4871 detail: format!("dense WAL rewrite missing src {src_key}"),
4872 }
4873 })?;
4874 let dst = lookup(&self.ids, &pending, &dst_key).ok_or_else(|| {
4875 GraphError::Corrupt {
4876 detail: format!("dense WAL rewrite missing dst {dst_key}"),
4877 }
4878 })?;
4879 out.push(WalRecord::InsertEdgeId { etype, src, dst });
4880 }
4881 WalRecord::RenameNode {
4882 ref old_key,
4883 ref new_key,
4884 } => {
4885 // Track the rename in `pending` so subsequent InsertEdge /
4886 // SetProp records in this batch can resolve the new key.
4887 let id = lookup(&self.ids, &pending, old_key).ok_or_else(|| {
4888 GraphError::Corrupt {
4889 detail: format!(
4890 "dense WAL rewrite: RenameNode old key {old_key} not found"
4891 ),
4892 }
4893 })?;
4894 pending.remove(old_key.as_str());
4895 pending.insert(new_key.clone(), id);
4896 out.push(rec);
4897 }
4898 // # Symbol-order invariant (load-bearing)
4899 //
4900 // Write-time and replay-time symbol assignment must agree: every
4901 // symbol in a `Batch` frame has to receive the same dense id when
4902 // the frame's records are replayed in order as it received when
4903 // the frame was written.
4904 //
4905 // A rule's backfill interns its `edge_type` lazily
4906 // (`core_rules::engine`, every `g.syms.intern(&def.edge_type)`
4907 // site), and that backfill runs from `apply` — during the
4908 // `CreateRule` record itself, and again from any later
4909 // `InsertNodeId` in the same frame that makes the rule fire. At
4910 // write time the whole batch is rewritten before any of it is
4911 // applied, so a later `InsertEdge` in the same batch would win the
4912 // lower id for its edge type; on replay the rule's lazy intern
4913 // gets there first and steals it, and the `Intern` record fails at
4914 // the `wal intern assigned …` check in `apply`.
4915 //
4916 // Pre-interning the rule's `edge_type` here, and emitting its
4917 // `Intern` record ahead of the `CreateRule` record, makes both
4918 // orders identical. `weight_prop` needs no pre-intern:
4919 // `EdgeProps::set` keys props by `String`, never through the
4920 // interner. `via_edge` needs none either: via-hop rules resolve it
4921 // with `syms.get` and skip when it is absent.
4922 //
4923 // `RebuildRule` and `DeleteRule` need no such handling here:
4924 // `RebuildRule` has no `BatchOp` variant, so it never appears
4925 // inside a `Batch` today — it is only ever issued as its own
4926 // standalone commit (`rebuild_rule`, or the auto-rebuild path
4927 // that logs it as a second commit after the triggering op).
4928 // `DeleteRule` does have a `BatchOp` variant and can appear
4929 // inside a `Batch`, but it carries only a rule `name` — no
4930 // `edge_type` or other symbol that needs pre-interning — so
4931 // only `CreateRule` needs this arm.
4932 WalRecord::CreateRule { ref def_bytes } => {
4933 let def = decode_rule_def(def_bytes).map_err(|e| GraphError::Corrupt {
4934 detail: format!("CreateRule def_bytes deserialize failed: {e}"),
4935 })?;
4936 let (etype, intern) = self.intern_wal(&def.edge_type);
4937 if interned.insert(etype) {
4938 out.push(intern);
4939 }
4940 out.push(rec);
4941 }
4942 other => out.push(other),
4943 }
4944 }
4945 Ok(out)
4946 }
4947
4948 fn log_dense(&mut self, recs: Vec<WalRecord>) -> Result<()> {
4949 let recs = self.rewrite_wal_dense(recs)?;
4950 match recs.len() {
4951 0 => Ok(()),
4952 1 => self.log_then_apply(recs.into_iter().next().unwrap()),
4953 _ => self.log_then_apply(WalRecord::Batch(recs)),
4954 }
4955 }
4956
4957 /// Durable write, then notify the event sink. Replay (`apply` during
4958 /// `open`) never enters this function, so it is the replay-silent seam.
4959 /// Record that the commit occupying `frame_index` happened now, and append
4960 /// those 16 bytes to the sidecar.
4961 ///
4962 /// `frame_index` is the **global 0-based WAL frame index** of the commit's
4963 /// own record — the space every history surface addresses — taken from
4964 /// [`wal_frames_written`](GraphDb::wal_frames_written) before the append
4965 /// that puts the record there.
4966 ///
4967 /// **It is deliberately not derived from `commit_seq`.** A commit is not a
4968 /// frame: one whose rules fire appends a second frame for the derived-edge
4969 /// history marker, so `commit_seq - 1` falls one frame further behind per
4970 /// rule-firing commit and every date resolves to an ever-earlier graph.
4971 /// That was the shipped behaviour through v0.6.11 and it failed silently,
4972 /// because an older graph is a plausible answer rather than an error.
4973 ///
4974 /// Called from exactly one place — `log_then_apply_with`, immediately after
4975 /// `commit_seq` is incremented. Every write path in the engine funnels
4976 /// through that function, and replay deliberately does not: `apply_frames`
4977 /// re-applies commits that already happened, so stamping there would record
4978 /// replay time as commit time.
4979 ///
4980 /// **This is the engine's only wall-clock read.** Everything else uses
4981 /// `Instant`, which is monotonic and not a date.
4982 ///
4983 /// Failure is swallowed on purpose. The sidecar is not part of the WAL or
4984 /// the snapshot, so a failed append must not fail a commit that is already
4985 /// durable — it costs a date, not data. The map is marked poisoned so the
4986 /// gap is reported rather than resolved across.
4987 fn stamp_commit_time(&mut self, frame_index: u64) {
4988 if self.commit_times_poisoned {
4989 return;
4990 }
4991 let unix_ms = match self.commit_time_override {
4992 Some(ms) => ms,
4993 None => {
4994 let Ok(now) = std::time::SystemTime::now().duration_since(std::time::UNIX_EPOCH)
4995 else {
4996 // A clock before 1970. Refuse to invent a timestamp.
4997 self.commit_times_poisoned = true;
4998 return;
4999 };
5000 now.as_millis() as i64
5001 }
5002 };
5003 let first = self.commit_times.is_empty();
5004 self.commit_times.push(frame_index, unix_ms);
5005
5006 let wrote = if first {
5007 self.fs.write_atomic(
5008 FileId::CommitTimes,
5009 &core_storage::commit_times::encode(&self.commit_times),
5010 )
5011 } else {
5012 self.fs.append(
5013 FileId::CommitTimes,
5014 &core_storage::commit_times::encode_entry(frame_index, unix_ms),
5015 )
5016 };
5017 if wrote.is_err() {
5018 self.commit_times_poisoned = true;
5019 }
5020 }
5021
5022 /// Read the time sidecar from disk into this handle.
5023 ///
5024 /// Absent is the normal case for any store written before v0.6.11 and is
5025 /// not an error; unreadable is recorded so date queries can say "damaged"
5026 /// rather than "none recorded".
5027 ///
5028 /// Called at open **and** by `refresh` when a peer's frames are absorbed.
5029 /// Both, because the map is a file another process appends to: a handle
5030 /// that raises its frame cursor to include a peer's commits while holding a
5031 /// stale map would answer dates from a prefix of the truth — and, if its
5032 /// own map were still empty, would rewrite the whole file with one entry
5033 /// and destroy the peer's.
5034 fn load_commit_times_from_fs(&mut self) {
5035 match self.fs.read(FileId::CommitTimes) {
5036 Ok(bytes) if bytes.is_empty() => {}
5037 Ok(bytes) => {
5038 // A map from an older format version is discarded, not
5039 // reported as damage and not reinterpreted. Its entries were
5040 // written correctly against a different meaning of the number
5041 // — see `COMMIT_TIMES_VERSION` — and reading them in this
5042 // build's space would resolve dates onto unrelated commits.
5043 // Leaving the map empty makes the store answer
5044 // `NoRecordedTime`, which is the truth: it records no times
5045 // this build can use, and the next commit starts a usable map.
5046 if core_storage::commit_times::superseded_version(&bytes).is_some() {
5047 return;
5048 }
5049 match core_storage::commit_times::decode(&bytes) {
5050 Ok(t) => self.commit_times = t,
5051 Err(_) => self.commit_times_poisoned = true,
5052 }
5053 }
5054 Err(_) => {}
5055 }
5056 }
5057
5058 /// Rewrite the sidecar from memory. Used after truncation, which is the one
5059 /// operation that cannot be expressed as an append.
5060 fn rewrite_commit_times(&mut self) {
5061 if self.commit_times_poisoned {
5062 return;
5063 }
5064 if self
5065 .fs
5066 .write_atomic(
5067 FileId::CommitTimes,
5068 &core_storage::commit_times::encode(&self.commit_times),
5069 )
5070 .is_err()
5071 {
5072 self.commit_times_poisoned = true;
5073 }
5074 }
5075
5076 /// The greatest commit whose recorded time is at or before `unix_ms`.
5077 ///
5078 /// Errors name what they can answer instead of guessing a commit:
5079 /// `Corrupt` when the sidecar would not decode, `NoRecordedTime` when the
5080 /// store records none, `TimeBeforeFloor` when the instant predates the
5081 /// oldest entry, and `CommitOutOfRange` when the answer falls below the WAL
5082 /// horizon and so cannot be replayed.
5083 ///
5084 /// The answer is a **0-based frame index**, ready to hand to `edges_at` or
5085 /// `was_linked` without adjustment.
5086 pub fn resolve_instant(&self, unix_ms: i64) -> Result<u64> {
5087 if self.commit_times_poisoned {
5088 return Err(GraphError::Corrupt {
5089 detail: "commit_times.bin will not decode; date queries are \
5090 unavailable on this store"
5091 .into(),
5092 });
5093 }
5094 let at = self
5095 .commit_times
5096 .resolve_instant(unix_ms, self.wal_horizon_floor)?;
5097 // The map outlives the history it describes. A truncating snapshot folds
5098 // the WAL and discards it, so entries can name commits the engine can no
5099 // longer replay — the floor check above catches pruning, and this catches
5100 // discarding. Returning an index the caller's next call will reject is a
5101 // two-step error where one will do, and `resolve_date` is public: it
5102 // either hands back a usable index or refuses.
5103 let total = self.wal_total_commits()?;
5104 if at >= total {
5105 return Err(GraphError::CommitOutOfRange {
5106 commit: at,
5107 total,
5108 floor: self.wal_horizon_floor,
5109 });
5110 }
5111 Ok(at)
5112 }
5113
5114 /// Record subsequent commits as having happened at `unix_ms`, or pass
5115 /// `None` to go back to the system clock.
5116 ///
5117 /// For **backfilled history**: a mirror importing rows that already carry
5118 /// their own timestamps, or a replay of events that happened months ago.
5119 /// Without this every such commit is stamped "now", so a store holding a
5120 /// year of imported history answers every date question with
5121 /// `TimeBeforeFloor` — the data is there and no date reaches it.
5122 ///
5123 /// Sticky until changed or cleared, because a day of backfilled rows
5124 /// genuinely shares one instant.
5125 ///
5126 /// **Import in chronological order.** A supplied instant earlier than
5127 /// anything already recorded is refused with
5128 /// [`GraphError::CommitTimeNotMonotonic`], because resolution walks commit
5129 /// order: a later commit carrying an earlier time would silently widen every
5130 /// answer after it. Equal is allowed — that is what a shared day means. The
5131 /// live clock is never held to this, so an NTP step backwards still commits.
5132 ///
5133 /// Deliberately **not** exposed over HTTP or MCP: asserting when a commit
5134 /// happened rewrites the store's apparent history, which is not something a
5135 /// role token models. It is an embedding-caller's operation.
5136 pub fn record_commits_at(&mut self, unix_ms: Option<i64>) -> Result<()> {
5137 if self.read_only {
5138 return Err(GraphError::ReadOnly);
5139 }
5140 if let Some(ms) = unix_ms {
5141 if self.commit_times_poisoned {
5142 return Err(GraphError::Corrupt {
5143 detail: "commit_times.bin will not decode; this store cannot \
5144 record an asserted commit time"
5145 .into(),
5146 });
5147 }
5148 if let Some(newest) = self.commit_times.max_ms() {
5149 if ms < newest {
5150 return Err(GraphError::CommitTimeNotMonotonic {
5151 supplied_ms: ms,
5152 newest_ms: newest,
5153 });
5154 }
5155 }
5156 }
5157 self.commit_time_override = unix_ms;
5158 Ok(())
5159 }
5160
5161 /// The instant subsequent commits are being recorded at, when one is set.
5162 pub fn commit_time_override(&self) -> Option<i64> {
5163 self.commit_time_override
5164 }
5165
5166 /// [`Self::edges_at`] addressed by an instant rather than a commit index.
5167 ///
5168 /// Resolves through [`Self::resolve_instant`] — the last commit at or
5169 /// before the instant — then answers exactly as the commit-indexed call
5170 /// does. A store that records no times refuses by name; it never guesses.
5171 pub fn edges_at_instant(&self, key: &str, unix_ms: i64) -> Result<Vec<EdgeAt>> {
5172 let commit = self.resolve_instant(unix_ms)?;
5173 self.edges_at(key, commit)
5174 }
5175
5176 /// [`Self::was_linked`] addressed by an instant rather than a commit index.
5177 pub fn was_linked_at_instant(
5178 &self,
5179 a: &str,
5180 b: &str,
5181 edge_type: &str,
5182 unix_ms: i64,
5183 ) -> Result<bool> {
5184 let commit = self.resolve_instant(unix_ms)?;
5185 self.was_linked(a, b, edge_type, commit)
5186 }
5187
5188 /// Parse an RFC 3339 instant (or a bare `YYYY-MM-DD`) and resolve it.
5189 ///
5190 /// The one place every caller-facing surface converts a date string, so
5191 /// HTTP, MCP, Python and the CLI cannot drift in what they accept.
5192 pub fn resolve_date(&self, s: &str) -> Result<u64> {
5193 // The **end** of what the string denotes. A bare date is a day, so it
5194 // resolves to the last commit at or before that day's end — resolving to
5195 // the midnight that starts it would exclude everything that happened on
5196 // the date the caller asked about.
5197 let ms = core_storage::commit_times::parse_rfc3339_end_ms(s).ok_or_else(|| {
5198 GraphError::QueryError {
5199 detail: format!(
5200 "could not parse {s:?} as a date; expected RFC 3339 \
5201 (2026-06-19, or 2026-06-19T12:00:00Z)"
5202 ),
5203 }
5204 })?;
5205 self.resolve_instant(ms)
5206 }
5207
5208 /// The recorded wall-clock time of `commit`, when the sidecar holds one.
5209 ///
5210 /// `commit` is a **0-based frame index** — the space `edges_at`,
5211 /// `was_linked` and the history events use, not the 1-based `commit_seq`.
5212 pub fn commit_time_ms(&self, commit: u64) -> Option<i64> {
5213 if self.commit_times_poisoned {
5214 return None;
5215 }
5216 self.commit_times.time_of(commit)
5217 }
5218
5219 fn log_then_apply(&mut self, rec: WalRecord) -> Result<()> {
5220 self.log_then_apply_with(rec, None, self.fsync)
5221 }
5222
5223 /// Whether this frame must fsync under `policy`.
5224 ///
5225 /// Batched contract: user-visible batches (>1 mutation) fsync; single
5226 /// mutations do not. The dense rewrite wraps a single mutation in a
5227 /// `Batch([Intern.., <one *Id record>])`, so `Intern` records are excluded
5228 /// from the count — removing that filter would make every single-op write
5229 /// fsync under Batched (or, if the threshold were raised instead, skip a
5230 /// needed fsync for real two-op batches).
5231 fn wal_needs_sync(policy: FsyncPolicy, rec: &WalRecord) -> bool {
5232 match policy {
5233 FsyncPolicy::Relaxed => false,
5234 FsyncPolicy::Strict => true,
5235 FsyncPolicy::Batched => match rec {
5236 // Intern + one mutation is the single-op rewrite, not a user batch.
5237 WalRecord::Batch(inner) => {
5238 inner
5239 .iter()
5240 .filter(|r| !matches!(r, WalRecord::Intern { .. }))
5241 .count()
5242 > 1
5243 }
5244 _ => false,
5245 },
5246 }
5247 }
5248
5249 /// # Apply-infallibility invariant (load-bearing)
5250 ///
5251 /// The ordering is: WAL append → fsync → apply. If `apply` returned `Err`
5252 /// for a `Batch` frame after a successful WAL write, the WAL would contain
5253 /// the full frame while in-memory state would reflect only the ops before
5254 /// the failure. On reopen, WAL replay would then apply the entire batch —
5255 /// diverging permanently from what the pre-crash process had in memory.
5256 ///
5257 /// For `Batch` frames this situation cannot arise because:
5258 /// - All validation runs via `commit_logged_batch`/`MutPreview` **before**
5259 /// the WAL write. `MutPreview` uses the same `&mut self` that apply will
5260 /// use, with no concurrent mutation between validation exit and apply entry.
5261 /// - Every `apply` arm for a validated op is either infallible by construction
5262 /// (`InsertNode`, `RemoveProp`, `DeleteEdge`, `DeleteNode`), has idempotency
5263 /// guards that return `Ok(())` (`CreateRule`, `DeleteRule`), or is
5264 /// guaranteed-present by validation (`InsertEdge`/`SetProp` key lookups).
5265 /// - `on_node_changed` and `on_node_removed` return `()` — never `Err`.
5266 ///
5267 /// A `debug_assert!` below fires in debug builds if `apply` ever returns
5268 /// `Err` for a `Batch` frame, making any future regression immediately visible
5269 /// in tests rather than silently diverging crash-recovery behaviour.
5270 fn log_then_apply_with(
5271 &mut self,
5272 rec: WalRecord,
5273 ingest: Option<(String, usize)>,
5274 policy: FsyncPolicy,
5275 ) -> Result<()> {
5276 // Read-only guard: as-of instances must never write the WAL.
5277 if self.read_only {
5278 return Err(GraphError::ReadOnly);
5279 }
5280 // Degraded guard: fsync failure left WAL truncated, or a refresh failed
5281 // partway; in-memory state is ahead of (or out of step with) the
5282 // on-disk WAL, so further mutations would deepen the divergence.
5283 // Reopen the database to recover. Checked before the lock guard: this
5284 // is the more serious condition and the more useful error.
5285 if self.degraded {
5286 return Err(GraphError::Io(std::io::Error::other(
5287 "database degraded after group-commit fsync failure; reopen required",
5288 )));
5289 }
5290 // Cross-process guard: this write scope asked for the store's write
5291 // lock and did not get it. Writing anyway would append frames on top of
5292 // a WAL another process is extending, so refuse instead.
5293 if self.lock_denied {
5294 return Err(GraphError::Busy { holder: None });
5295 }
5296 // Ensure retained provenance bytes are decoded into the live mutable
5297 // fields before any mutation touches self.engine.provenance. This is a
5298 // no-op if provenance was never stored (fresh store) or has already been
5299 // consumed (subsequent mutations). WAL replay calls apply() directly
5300 // and is covered by consume_retained_state_eager before replay.
5301 self.ensure_v8_base_sections_loaded();
5302 self.engine.ensure_provenance_loaded_mut();
5303 // Invariant (I-1): no stale deltas may enter from a previous apply.
5304 // If any engine method ever accumulates deltas before erroring, they would
5305 // contaminate the *next* commit's event stream. This assert fires in debug
5306 // builds, making any future regression visible at the earliest point.
5307 debug_assert_eq!(
5308 self.engine.pending_delta_count(),
5309 0,
5310 "stale engine deltas at log_then_apply_with entry — \
5311 a previous apply arm may have accumulated deltas before erroring; \
5312 the caller must drain_deltas() on any error path before returning"
5313 );
5314 let frame = encode_record(&rec);
5315 self.fs.append(FileId::Wal, &frame)?;
5316 // The cursor advances by exactly the bytes appended: these frames are
5317 // ours and already applied, so a later refresh must not replay them.
5318 self.wal_consumed += frame.len() as u64;
5319 self.wal_frames_written += 1;
5320 if Self::wal_needs_sync(policy, &rec) {
5321 self.fs.sync(FileId::Wal)?;
5322 }
5323 // Marker writing always needs the engine deltas, but the engine only
5324 // accumulates them when emit_deltas is true (normally gated on subscribers
5325 // or views being present). Enable emission for this apply if it is
5326 // currently off, then restore the original state unconditionally via an
5327 // RAII guard — this prevents a panic in apply() from leaking the flag.
5328 // The same guard resets the engine's transient chaining state. A panic
5329 // unwinding out of a rule hook would otherwise leave `chain_depth`
5330 // non-zero, which makes every later `begin_chain` decide chaining is
5331 // already running and silently switch it off for good.
5332 struct RestoreEmitDeltas(*mut RuleEngine, bool);
5333 impl Drop for RestoreEmitDeltas {
5334 fn drop(&mut self) {
5335 // SAFETY: pointer into self (GraphDb); guard is dropped within
5336 // this frame before log_then_apply_with returns.
5337 unsafe {
5338 (*self.0).set_emit_deltas(self.1);
5339 (*self.0).reset_chain_state();
5340 }
5341 }
5342 }
5343 let original_emit = self.engine.emit_deltas();
5344 if !original_emit {
5345 self.engine.set_emit_deltas(true);
5346 }
5347 // SAFETY: raw pointer into self; guard dropped within this frame.
5348 let _emit_guard = RestoreEmitDeltas(&mut self.engine as *mut _, original_emit);
5349
5350 let apply_result = self.apply(&rec);
5351 // For Batch frames, post-validation apply must be infallible (see above).
5352 // A debug_assert here catches any future change that makes apply fallible
5353 // before the caller notices via silent WAL/memory divergence.
5354 if matches!(&rec, WalRecord::Batch(_)) {
5355 debug_assert!(
5356 apply_result.is_ok(),
5357 "Batch apply returned Err after successful WAL write — \
5358 the validate-then-apply invariant has been violated; \
5359 see log_then_apply_with invariant doc"
5360 );
5361 }
5362 if apply_result.is_err() {
5363 // Discard any partial deltas accumulated by the failed apply.
5364 // They must not ride the next commit's event stream (I-1).
5365 // _emit_guard restores emit_deltas on drop automatically.
5366 let _ = self.engine.drain_deltas();
5367 let _ = self.engine.take_rebuild_needed();
5368 apply_result?;
5369 }
5370 self.commit_seq += 1;
5371 let seq = self.commit_seq;
5372 // Update per-node last-change map for the committed record.
5373 // Must happen after commit_seq is incremented so the seq is correct.
5374 self.update_last_change_from_rec(&rec, seq);
5375 // Drain engine deltas and distribute to subscribers before the existing
5376 // MutationEvent sink fires — both happen post-fsync, post-apply.
5377 // _emit_guard restores emit_deltas after this line when it drops.
5378 let engine_deltas = self.engine.drain_deltas();
5379
5380 // Append history-marker WAL records for any derived-edge changes so
5381 // that `edge_history` and `was_linked` can surface rule-attributed
5382 // events. Markers are STATE NO-OPS during replay; they are written
5383 // without an additional fsync (the triggering commit's sync already
5384 // happened; the next commit's sync covers these lazily).
5385 if !engine_deltas.is_empty() {
5386 let markers: Vec<WalRecord> = engine_deltas
5387 .iter()
5388 .map(|d| {
5389 if d.fired {
5390 WalRecord::DerivedEdgeAdded {
5391 rule: d.rule.clone(),
5392 edge_type: d.edge_type.clone(),
5393 src_key: d.src_key.clone(),
5394 dst_key: d.dst_key.clone(),
5395 }
5396 } else {
5397 WalRecord::DerivedEdgeRetracted {
5398 rule: d.rule.clone(),
5399 edge_type: d.edge_type.clone(),
5400 src_key: d.src_key.clone(),
5401 dst_key: d.dst_key.clone(),
5402 }
5403 }
5404 })
5405 .collect();
5406 let marker_frame = if markers.len() == 1 {
5407 markers.into_iter().next().unwrap()
5408 } else {
5409 WalRecord::Batch(markers)
5410 };
5411 // Ignore append errors: markers are best-effort history
5412 // annotations. Losing them does not affect state correctness.
5413 // The cursor only advances when the bytes actually landed.
5414 let marker_bytes = encode_record(&marker_frame);
5415 if self.fs.append(FileId::Wal, &marker_bytes).is_ok() {
5416 self.wal_consumed += marker_bytes.len() as u64;
5417 // A marker is a state no-op during replay but it is not an
5418 // index no-op: it occupies a frame that every history surface
5419 // counts. Missing this increment is the whole of defect 1.
5420 self.wal_frames_written += 1;
5421 }
5422 }
5423
5424 // Stamp the commit against the **last** frame it wrote.
5425 //
5426 // `edges_at` and `was_linked` reconstruct derived edges by reading the
5427 // history markers out of the WAL, not by re-running rules over a
5428 // prefix. A commit's complete state — its record *and* the edges its
5429 // rules derived — is therefore only reached at its marker frame, so
5430 // that is the frame a date naming this commit must resolve to. Stamping
5431 // the record's own frame would answer every date with the graph as it
5432 // was one derivation short.
5433 //
5434 // This runs after the marker append for that reason, and it is still
5435 // the single stamping site: one commit, one entry.
5436 self.stamp_commit_time(self.wal_frames_written - 1);
5437
5438 // Record MVCC CommitDelta for the epoch reader. The WAL record is
5439 // stored as-is (including any nested Batch / Intern records); the
5440 // ReaderSnapshot's apply_one function handles all variants.
5441 {
5442 let derived_inserts = engine_deltas
5443 .iter()
5444 .filter(|d| d.fired)
5445 .map(|d| (d.etype_sym, d.src_id, d.dst_id))
5446 .collect();
5447 let derived_deletes = engine_deltas
5448 .iter()
5449 .filter(|d| !d.fired)
5450 .map(|d| (d.etype_sym, d.src_id, d.dst_id))
5451 .collect();
5452 let delta = Arc::new(crate::reader::CommitDelta {
5453 records: vec![rec.clone()],
5454 derived_inserts,
5455 derived_deletes,
5456 });
5457 self.delta_tail.push(delta);
5458 self.commits_since_fold += 1;
5459 if self.commits_since_fold >= crate::reader::FOLD_EVERY_K {
5460 self.fold_now();
5461 }
5462 }
5463
5464 if self.defer_events {
5465 // Group-commit drain thread: hold events until after the group
5466 // fsync so subscribers only observe durable data (R2).
5467 self.deferred_events.push(DeferredEvent {
5468 rec: rec.clone(),
5469 engine_deltas,
5470 seq,
5471 ingest,
5472 });
5473 } else {
5474 self.distribute_events(&rec, &engine_deltas, seq);
5475 self.emit_committed(&rec, ingest);
5476 }
5477 // Drift is only known after apply, so auto-rebuild cannot join the
5478 // triggering op's WAL frame. Issue RebuildRule as a second commit.
5479 // Skip when `rec` is itself RebuildRule: rebuild resets drift, so a
5480 // retrigger loop is impossible if the fit succeeded, but we still
5481 // drain the flag so a leftover cannot re-enter.
5482 // One slice of any outstanding vector-index build rides here too, so a
5483 // store that is being written to finishes its build without anyone
5484 // calling `pump_index_build`. A rule that becomes whole joins the same
5485 // RebuildRule loop below.
5486 let mut rebuilds = self.engine.take_rebuild_needed();
5487 if !matches!(&rec, WalRecord::RebuildRule { .. }) {
5488 // Not after `CreateRule`: that record's own apply already did the
5489 // rule's first slice, and pumping again here would make one
5490 // `create_rule` call do two slices' work under one lock.
5491 // Nothing pending is the overwhelmingly common case and must cost
5492 // a map lookup, not an engine swap: a store being written to has
5493 // long since populated its indexes, so the `pump_index_build`
5494 // entry point owns the not-yet-populated case on its own.
5495 if !matches!(&rec, WalRecord::CreateRule { .. })
5496 && !self.engine.builds_in_progress().is_empty()
5497 {
5498 rebuilds.extend(self.pump_one_slice().into_iter().map(|b| b.rule));
5499 }
5500 let mut failed = Vec::new();
5501 for name in rebuilds {
5502 if self.engine.rules().any(|r| r.name == name) {
5503 // User op is already durable. A failed second commit must
5504 // not surface as the caller's error.
5505 if let Err(e) =
5506 self.log_then_apply(WalRecord::RebuildRule { name: name.clone() })
5507 {
5508 eprintln!(
5509 "auto-rebuild of rule {name:?} failed after durable user commit: {e}"
5510 );
5511 failed.push(name);
5512 }
5513 }
5514 }
5515 for name in failed {
5516 self.engine.queue_rebuild_needed(name);
5517 }
5518 }
5519 Ok(())
5520 }
5521
5522 /// Install a post-commit hook. Replaces any previous sink.
5523 ///
5524 /// The sink runs inside `log_then_apply` after a successful
5525 /// durable commit, while the caller still holds `&mut self`. When this
5526 /// database is behind a [`crate::SharedDb`], that means the **write
5527 /// guard is held**. The sink must never call `read` / `write` (or any
5528 /// other method) on the same `SharedDb` — the `RwLock` is not
5529 /// re-entrant and doing so deadlocks. The sink is `Send + Sync`;
5530 /// `std::sync::mpsc::Sender` is not `Sync` and will not type-check.
5531 /// Intended examples: `std::sync::mpsc::SyncSender`,
5532 /// `tokio::sync::mpsc::Sender`, `tokio::sync::broadcast::Sender`
5533 /// (non-blocking `send`), or `Arc<Mutex<Vec<MutationEvent>>>`.
5534 pub fn set_event_sink(&mut self, sink: Box<dyn Fn(MutationEvent) + Send + Sync>) {
5535 self.event_sink = Some(sink);
5536 }
5537
5538 /// Whether a post-commit event sink is currently installed.
5539 pub fn has_event_sink(&self) -> bool {
5540 self.event_sink.is_some()
5541 }
5542
5543 /// Set WAL fsync cadence. Default [`FsyncPolicy::Strict`].
5544 pub fn set_fsync_policy(&mut self, p: FsyncPolicy) {
5545 self.fsync = p;
5546 }
5547
5548 /// Return the current WAL fsync cadence.
5549 pub fn fsync_policy(&self) -> FsyncPolicy {
5550 self.fsync
5551 }
5552
5553 // ── Group-commit event deferral ───────────────────────────────────────────
5554
5555 /// Enable or disable deferred event mode.
5556 ///
5557 /// When `true`, event notifications (subscription `DbEvent`s and legacy
5558 /// `MutationEvent` sink calls) are buffered rather than fired immediately.
5559 /// Call [`flush_deferred_events`] after the group fsync to deliver them,
5560 /// or [`discard_deferred_events`] if the fsync failed and the group must
5561 /// be treated as lost.
5562 pub fn set_deferred_events_mode(&mut self, defer: bool) {
5563 self.defer_events = defer;
5564 }
5565
5566 /// Fire all buffered events accumulated since [`set_deferred_events_mode`]
5567 /// was set to true. Clears the buffer.
5568 ///
5569 /// Called by the drain thread AFTER a successful group fsync, so
5570 /// subscribers observe only data that is durably on disk.
5571 pub fn flush_deferred_events(&mut self) {
5572 let events = std::mem::take(&mut self.deferred_events);
5573 for de in events {
5574 self.distribute_events(&de.rec, &de.engine_deltas, de.seq);
5575 self.emit_committed(&de.rec, de.ingest);
5576 }
5577 }
5578
5579 /// Discard all buffered events without firing them.
5580 ///
5581 /// Called by the drain thread when a group fsync fails: the WAL has been
5582 /// truncated back to the pre-group offset, so the committed-but-unsynced
5583 /// ops must not be observable to subscribers.
5584 pub fn discard_deferred_events(&mut self) {
5585 self.deferred_events.clear();
5586 }
5587
5588 // ── Degraded state ────────────────────────────────────────────────────────
5589
5590 /// Mark this database as degraded.
5591 ///
5592 /// Called by the group-commit drain thread after a group fsync failure and
5593 /// WAL truncation: the in-memory state is now ahead of the on-disk WAL, so
5594 /// further mutations would deepen the divergence. All subsequent calls to
5595 /// [`log_then_apply_with`] return `Err` until the database is reopened.
5596 pub fn set_degraded(&mut self) {
5597 self.degraded = true;
5598 }
5599
5600 fn emit(&self, ev: MutationEvent) {
5601 if let Some(sink) = &self.event_sink {
5602 sink(ev);
5603 }
5604 }
5605
5606 fn emit_committed(&self, rec: &WalRecord, ingest: Option<(String, usize)>) {
5607 match rec {
5608 WalRecord::Batch(inner) => {
5609 for r in inner {
5610 if let Some(ev) = event_from_record(r, &self.syms, &self.ids) {
5611 self.emit(ev);
5612 }
5613 }
5614 match ingest {
5615 Some((label, inserted)) => {
5616 self.emit(MutationEvent::Ingested { label, inserted })
5617 }
5618 None => {
5619 let ops = inner
5620 .iter()
5621 .filter(|r| !matches!(r, WalRecord::Intern { .. }))
5622 .count();
5623 if ops > 1 {
5624 self.emit(MutationEvent::BatchApplied { ops });
5625 }
5626 }
5627 }
5628 }
5629 other => {
5630 if let Some(ev) = event_from_record(other, &self.syms, &self.ids) {
5631 self.emit(ev);
5632 }
5633 }
5634 }
5635 }
5636
5637 // -----------------------------------------------------------------------
5638 // Subscription API
5639 // -----------------------------------------------------------------------
5640
5641 /// Distribute post-commit events to all live subscribers.
5642 ///
5643 /// Build a row-key → row-data map from a [`ResultSet`].
5644 ///
5645 /// Each row is serialized to JSON to form its key; a debug fallback is used
5646 /// if serialization fails. Used by both the initial-seed path in
5647 /// [`Self::subscribe_query`] and the per-commit diff path in
5648 /// [`Self::distribute_events`] to keep the two in sync.
5649 fn result_to_row_map(
5650 result: &core_query::ResultSet,
5651 ) -> std::collections::HashMap<String, Vec<Option<Value>>> {
5652 (0..result.len())
5653 .map(|i| {
5654 let row = result.row(i).to_vec();
5655 let key = serde_json::to_string(&row).unwrap_or_else(|_| format!("{row:?}"));
5656 (key, row)
5657 })
5658 .collect()
5659 }
5660
5661 /// Collect the set of label syms touched by a WAL record.
5662 ///
5663 /// Returns `Some(set)` when every record in this commit can be attributed to
5664 /// a known label sym. Returns `None` when the commit must not be skipped:
5665 /// edge records, unresolvable key→label lookups, or any record type not in
5666 /// the explicit handled set.
5667 ///
5668 /// Handled record types and their actions:
5669 /// - `InsertNode` → look up label in interner (fails → None)
5670 /// - `InsertNodeId` → label sym is carried directly
5671 /// - `SetProp` → resolve key→id→label (fails → None)
5672 /// - `DeleteNode` → resolve key→id→label (fails → None)
5673 /// - `Batch` → recurse into every inner record
5674 /// - `InsertEdge`, `DeleteEdge`, `InsertEdgeId` → always None (edge records)
5675 /// - everything else → None (conservative)
5676 fn commit_touched_labels(
5677 rec: &WalRecord,
5678 syms: &Interner,
5679 ids: &IdMap,
5680 labels: &[u32],
5681 ) -> Option<BTreeSet<u32>> {
5682 let mut out = BTreeSet::new();
5683 if Self::collect_touched_labels(rec, syms, ids, labels, &mut out) {
5684 Some(out)
5685 } else {
5686 None
5687 }
5688 }
5689
5690 fn collect_touched_labels(
5691 rec: &WalRecord,
5692 syms: &Interner,
5693 ids: &IdMap,
5694 labels: &[u32],
5695 out: &mut BTreeSet<u32>,
5696 ) -> bool {
5697 match rec {
5698 // String-key insert: the dense rewrite converts this to
5699 // [Intern, InsertNodeId], so this arm fires only for legacy WAL
5700 // records written before the dense path was added.
5701 WalRecord::InsertNode { label, .. } => {
5702 if let Some(sym) = syms.get(label) {
5703 out.insert(sym);
5704 true
5705 } else {
5706 false
5707 }
5708 }
5709 // Dense-id insert (produced by rewrite_wal_dense for every
5710 // insert_node call in the current codebase).
5711 WalRecord::InsertNodeId { label, .. } => {
5712 out.insert(*label);
5713 true
5714 }
5715 // String-key prop set: dense path converts to [Intern, SetPropId].
5716 WalRecord::SetProp { key, .. } => {
5717 if let Some(sym) = Self::resolve_key_label_sym(key, ids, labels) {
5718 out.insert(sym);
5719 true
5720 } else {
5721 false
5722 }
5723 }
5724 // Dense-id prop set (produced by rewrite_wal_dense for set_prop).
5725 WalRecord::SetPropId { id, .. } => {
5726 if let Some(sym) = labels.get(*id as usize).copied().filter(|&s| s != u32::MAX) {
5727 out.insert(sym);
5728 true
5729 } else {
5730 false
5731 }
5732 }
5733 WalRecord::DeleteNode { key } => {
5734 if let Some(sym) = Self::resolve_key_label_sym(key, ids, labels) {
5735 out.insert(sym);
5736 true
5737 } else {
5738 false
5739 }
5740 }
5741 WalRecord::Batch(inner) => inner
5742 .iter()
5743 .all(|r| Self::collect_touched_labels(r, syms, ids, labels, out)),
5744 // Intern is a pure metadata record — it does not touch any node's
5745 // label and is safe to skip for the label-skip predicate.
5746 WalRecord::Intern { .. } => true,
5747 // Edge records: always re-execute (edges can change join results).
5748 WalRecord::InsertEdge { .. }
5749 | WalRecord::DeleteEdge { .. }
5750 | WalRecord::InsertEdgeId { .. } => false,
5751 _ => false,
5752 }
5753 }
5754
5755 /// Resolve a node key to its label sym via the dense id table.
5756 /// Returns `None` if the key is unknown or the label is a tombstone sentinel.
5757 fn resolve_key_label_sym(key: &str, ids: &IdMap, labels: &[u32]) -> Option<u32> {
5758 let id = ids.get(key)?;
5759 let sym = labels.get(id as usize).copied()?;
5760 (sym != u32::MAX).then_some(sym)
5761 }
5762
5763 /// Distribute post-commit events to all live subscribers.
5764 ///
5765 /// Called from `log_then_apply_with` after apply + fsync, before the
5766 /// legacy MutationEvent sink. Prunes dead `Weak` entries in-place.
5767 ///
5768 /// Query subscriptions (subscribe_query) re-execute their plan on every
5769 /// call and diff the result against the previous run. Zero overhead when
5770 /// no query subscriptions are active.
5771 fn distribute_events(&mut self, rec: &WalRecord, engine_deltas: &[EngineEdgeDelta], seq: u64) {
5772 if self.subscriptions.is_empty() && self.query_subscriptions.is_empty() {
5773 return;
5774 }
5775
5776 if !self.subscriptions.is_empty() {
5777 // Build write events from the WAL record.
5778 let write_events: Vec<DbEvent> =
5779 Self::write_events_from_record(rec, seq, &self.syms, &self.ids);
5780
5781 // Build edge events from engine deltas. Weight is looked up from
5782 // edge_props at distribution time (after apply), so it's always fresh.
5783 let edge_events: Vec<DbEvent> = engine_deltas
5784 .iter()
5785 .map(|d| {
5786 if d.fired {
5787 // The score lives under the rule's declared weight_prop,
5788 // which is not always the literal "weight".
5789 let prop = self
5790 .engine
5791 .rules()
5792 .find(|r| r.name == d.rule)
5793 .and_then(|r| r.weight_prop.as_deref());
5794 let weight = prop.and_then(|p| {
5795 self.edge_props
5796 .get(d.etype_sym, d.src_id, d.dst_id, p)
5797 .and_then(|v| {
5798 if let core_storage::Value::Float(f) = v {
5799 Some(*f)
5800 } else {
5801 None
5802 }
5803 })
5804 });
5805 DbEvent::EdgeFired {
5806 rule: d.rule.clone(),
5807 src_key: d.src_key.clone(),
5808 dst_key: d.dst_key.clone(),
5809 edge_type: d.edge_type.clone(),
5810 weight,
5811 commit_seq: seq,
5812 }
5813 } else {
5814 DbEvent::EdgeRetracted {
5815 rule: d.rule.clone(),
5816 src_key: d.src_key.clone(),
5817 dst_key: d.dst_key.clone(),
5818 edge_type: d.edge_type.clone(),
5819 commit_seq: seq,
5820 }
5821 }
5822 })
5823 .collect();
5824
5825 // Prune dead entries; push matching events to live ones.
5826 self.subscriptions.retain(|entry| {
5827 let Some(inner) = entry.inner.upgrade() else {
5828 return false;
5829 };
5830 for ev in &write_events {
5831 if event_matches(ev, &entry.filter) {
5832 inner.push(ev.clone());
5833 }
5834 }
5835 for ev in &edge_events {
5836 if event_matches(ev, &entry.filter) {
5837 inner.push(ev.clone());
5838 }
5839 }
5840 true
5841 });
5842
5843 // Turn off delta accumulation if all subscribers dropped and no views remain.
5844 if self.subscriptions.is_empty() && self.view_store.is_empty() {
5845 self.engine.set_emit_deltas(false);
5846 }
5847 }
5848
5849 // Query subscriptions: full re-run per commit, then diff rows.
5850 // IMPORTANT: full re-execution on every commit — use LIMIT to bound cost.
5851 // Differential evaluation is roadmap / Phase 5.
5852 if !self.query_subscriptions.is_empty() {
5853 // Take the list out so we can call self.view() without borrow conflict.
5854 let mut query_subs = std::mem::take(&mut self.query_subscriptions);
5855 let empty_params = BTreeMap::new();
5856 query_subs.retain_mut(|entry| {
5857 let Some(inner) = entry.inner.upgrade() else {
5858 return false; // subscriber dropped — prune
5859 };
5860 // Label-skip: if the plan has a known scan label and this commit
5861 // can be proven to touch only different labels (and no rule-derived
5862 // edge deltas fired), the result set cannot have changed — skip.
5863 if let Some(scan_sym) = entry.scan_label {
5864 if engine_deltas.is_empty() {
5865 let touched =
5866 Self::commit_touched_labels(rec, &self.syms, &self.ids, &self.labels);
5867 if touched.map(|t| !t.contains(&scan_sym)).unwrap_or(false) {
5868 return true; // safe to skip — result set unchanged
5869 }
5870 }
5871 }
5872 QUERY_SUB_EXECS_TL.with(|c| c.set(c.get() + 1));
5873 let result = match execute(&self.view(), &entry.ops, &Params(&empty_params)) {
5874 Ok(r) => r,
5875 Err(e) => {
5876 // Keep the subscription alive; skip the diff for this commit.
5877 // Re-run errors are transient (e.g., planner change) and
5878 // self-heal when the next commit succeeds.
5879 eprintln!("[mushroomdb] subscribe_query re-run failed: {e}");
5880 return true;
5881 }
5882 };
5883 // Build new row map: serialized-key → row data.
5884 let new_row_map = Self::result_to_row_map(&result);
5885 // Removed rows: in prev but not in new.
5886 for (key, row) in &entry.prev_row_map {
5887 if !new_row_map.contains_key(key) {
5888 inner.push(DbEvent::QueryRowRemoved {
5889 columns: entry.columns.clone(),
5890 row: row.clone(),
5891 });
5892 }
5893 }
5894 // Added rows: in new but not in prev.
5895 for (key, row) in &new_row_map {
5896 if !entry.prev_row_map.contains_key(key) {
5897 inner.push(DbEvent::QueryRowAdded {
5898 columns: entry.columns.clone(),
5899 row: row.clone(),
5900 });
5901 }
5902 }
5903 entry.prev_row_map = new_row_map;
5904 true
5905 });
5906 self.query_subscriptions = query_subs;
5907 }
5908 }
5909
5910 /// Returns `true` if any live subscriber or view definition requires delta
5911 /// accumulation. Used to set `engine.emit_deltas` on subscribe/view DDL.
5912 fn needs_emit_deltas(&self) -> bool {
5913 !self.view_store.is_empty()
5914 || self
5915 .subscriptions
5916 .iter()
5917 .any(|e| e.inner.upgrade().is_some())
5918 }
5919
5920 /// Convert a WAL record into `DbEvent` write events with the given seq.
5921 fn write_events_from_record(
5922 rec: &WalRecord,
5923 seq: u64,
5924 intern: &Interner,
5925 ids: &IdMap,
5926 ) -> Vec<DbEvent> {
5927 match rec {
5928 WalRecord::InsertNode { label, key, .. } => vec![DbEvent::NodeInserted {
5929 label: label.clone(),
5930 key: key.clone(),
5931 commit_seq: seq,
5932 }],
5933 // *Id arms run after a successful apply, so resolution can only
5934 // fail on a programming error. Skip the event rather than emit a
5935 // fabricated "" that clients can't tell from a real empty value
5936 // (mirrors event_from_record returning None).
5937 WalRecord::InsertNodeId { label, key, .. } => intern
5938 .resolve(*label)
5939 .map(|label| DbEvent::NodeInserted {
5940 label: label.to_string(),
5941 key: key.clone(),
5942 commit_seq: seq,
5943 })
5944 .into_iter()
5945 .collect(),
5946 WalRecord::SetProp { key, field, .. } => vec![DbEvent::PropSet {
5947 key: key.clone(),
5948 field: field.clone(),
5949 commit_seq: seq,
5950 }],
5951 WalRecord::SetPropId { id, field, .. } => ids
5952 .key_of(*id)
5953 .zip(intern.resolve(*field))
5954 .map(|(key, field)| DbEvent::PropSet {
5955 key: key.to_string(),
5956 field: field.to_string(),
5957 commit_seq: seq,
5958 })
5959 .into_iter()
5960 .collect(),
5961 WalRecord::RemoveProp { key, field } => vec![DbEvent::PropRemoved {
5962 key: key.clone(),
5963 field: field.clone(),
5964 commit_seq: seq,
5965 }],
5966 WalRecord::InsertEdge {
5967 edge_type,
5968 src_key,
5969 dst_key,
5970 } => vec![DbEvent::EdgeInserted {
5971 edge_type: edge_type.clone(),
5972 src: src_key.clone(),
5973 dst: dst_key.clone(),
5974 commit_seq: seq,
5975 }],
5976 WalRecord::InsertEdgeId { etype, src, dst } => (|| {
5977 Some(DbEvent::EdgeInserted {
5978 edge_type: intern.resolve(*etype)?.to_string(),
5979 src: ids.key_of(*src)?.to_string(),
5980 dst: ids.key_of(*dst)?.to_string(),
5981 commit_seq: seq,
5982 })
5983 })()
5984 .into_iter()
5985 .collect(),
5986 WalRecord::DeleteEdge {
5987 edge_type,
5988 src_key,
5989 dst_key,
5990 } => vec![DbEvent::EdgeDeleted {
5991 edge_type: edge_type.clone(),
5992 src: src_key.clone(),
5993 dst: dst_key.clone(),
5994 commit_seq: seq,
5995 }],
5996 WalRecord::DeleteNode { key } => vec![DbEvent::NodeDeleted {
5997 key: key.clone(),
5998 commit_seq: seq,
5999 }],
6000 WalRecord::Batch(inner) => inner
6001 .iter()
6002 .flat_map(|r| Self::write_events_from_record(r, seq, intern, ids))
6003 .collect(),
6004 WalRecord::CreateRule { .. }
6005 | WalRecord::DeleteRule { .. }
6006 | WalRecord::RebuildRule { .. }
6007 | WalRecord::CreateView { .. }
6008 | WalRecord::DeleteView { .. }
6009 | WalRecord::EnableFulltext { .. }
6010 | WalRecord::DisableFulltext { .. }
6011 | WalRecord::EnableIndex { .. }
6012 | WalRecord::DisableIndex { .. }
6013 | WalRecord::Intern { .. }
6014 // History markers produce no DbEvent — the engine delta already
6015 // fired the EdgeFired/EdgeRetracted subscription events.
6016 | WalRecord::DerivedEdgeAdded { .. }
6017 | WalRecord::DerivedEdgeRetracted { .. }
6018 // A count is not an edge event: the pair it counts already fired one
6019 // when it was first inserted.
6020 | WalRecord::SetEdgeCount { .. }
6021 | WalRecord::RenameNode { .. } => vec![],
6022 }
6023 }
6024
6025 /// Subscribe to edge-fire and edge-retract events for one named rule.
6026 ///
6027 /// Returns `Err(GraphError::RuleNotFound)` if `rule_name` is not
6028 /// currently registered. Dropping the returned [`Subscription`] handle
6029 /// unregisters the subscriber — no further events are queued, no
6030 /// resources leak.
6031 pub fn subscribe_rule(&mut self, rule_name: &str) -> core_storage::Result<Subscription> {
6032 if self.read_only {
6033 return Err(core_storage::GraphError::ReadOnly);
6034 }
6035 if !self.engine.rules().any(|r| r.name == rule_name) {
6036 return Err(core_storage::GraphError::RuleNotFound {
6037 name: rule_name.to_string(),
6038 });
6039 }
6040 let inner = SubInner::new(self.sub_capacity());
6041 self.subscriptions.push(SubEntry {
6042 filter: SubFilter::Rule(rule_name.to_string()),
6043 inner: std::sync::Arc::downgrade(&inner),
6044 });
6045 self.engine.set_emit_deltas(true);
6046 Ok(Subscription(inner))
6047 }
6048
6049 /// Subscribe to edge-fire and edge-retract events for **all** rules.
6050 ///
6051 /// Returns `Err(GraphError::ReadOnly)` if called on an as-of instance —
6052 /// as-of instances never commit, so `distribute_events` never runs and the
6053 /// subscription would never deliver events.
6054 pub fn subscribe_all_rules(&mut self) -> core_storage::Result<Subscription> {
6055 if self.read_only {
6056 return Err(core_storage::GraphError::ReadOnly);
6057 }
6058 let inner = SubInner::new(self.sub_capacity());
6059 self.subscriptions.push(SubEntry {
6060 filter: SubFilter::AllRules,
6061 inner: std::sync::Arc::downgrade(&inner),
6062 });
6063 self.engine.set_emit_deltas(true);
6064 Ok(Subscription(inner))
6065 }
6066
6067 /// Subscribe to write events: node insert/delete, prop set/remove.
6068 ///
6069 /// Does not include edge-fire / edge-retract (rule-derived edge events).
6070 ///
6071 /// Returns `Err(GraphError::ReadOnly)` if called on an as-of instance —
6072 /// as-of instances never commit, so `distribute_events` never runs and the
6073 /// subscription would never deliver events.
6074 pub fn subscribe_writes(&mut self) -> core_storage::Result<Subscription> {
6075 if self.read_only {
6076 return Err(core_storage::GraphError::ReadOnly);
6077 }
6078 let inner = SubInner::new(self.sub_capacity());
6079 self.subscriptions.push(SubEntry {
6080 filter: SubFilter::Writes,
6081 inner: std::sync::Arc::downgrade(&inner),
6082 });
6083 self.engine.set_emit_deltas(true);
6084 Ok(Subscription(inner))
6085 }
6086
6087 /// Subscribe to incremental Cypher query results.
6088 ///
6089 /// Parses and plans `cypher`; rejects the query if the plan is not in the
6090 /// allowlisted subset (see [`core_query::cypher::is_subscribable`]):
6091 /// - `MATCH (n:Label) WHERE … RETURN … [LIMIT n]`
6092 /// - `MATCH (a)-[r:TYPE]->(b) RETURN … [LIMIT n]` (exactly one hop)
6093 ///
6094 /// SKIP is not supported — it shifts the result window on every commit,
6095 /// causing spurious Added/Removed churn for rows whose data never changed.
6096 /// Multi-hop Expand chains are not supported; each additional MATCH clause
6097 /// widens scope beyond the documented single-scan / single-hop subset.
6098 ///
6099 /// After each successful commit, the plan is **fully re-executed** and the
6100 /// result is diffed against the previous run. Added rows produce
6101 /// [`DbEvent::QueryRowAdded`]; removed rows produce
6102 /// [`DbEvent::QueryRowRemoved`].
6103 ///
6104 /// **Full re-run per commit; use LIMIT to bound execution cost.**
6105 /// The existing 1 M intermediate-row cap applies. Differential evaluation
6106 /// is roadmap / Phase 5.
6107 ///
6108 /// Returns `Err(GraphError::ReadOnly)` if called on an as-of instance —
6109 /// as-of instances never commit, so `distribute_events` never runs and the
6110 /// subscription would never deliver events.
6111 ///
6112 /// Returns `Err(GraphError::QueryError)` if the query fails to parse, plan,
6113 /// or if the plan shape is not in the allowlist.
6114 pub fn subscribe_query(&mut self, cypher: &str) -> Result<Subscription> {
6115 if self.read_only {
6116 return Err(GraphError::ReadOnly);
6117 }
6118 let tokens = lex(cypher).map_err(|e| GraphError::QueryError {
6119 detail: format!("lex: {e}"),
6120 })?;
6121 let ast = parse(&tokens).map_err(|e| GraphError::QueryError {
6122 detail: format!("parse: {e}"),
6123 })?;
6124 let ops = plan(&ast).map_err(|e| GraphError::QueryError {
6125 detail: format!("plan: {e}"),
6126 })?;
6127 if !is_subscribable(&ops) {
6128 return Err(GraphError::QueryError {
6129 detail: "subscribe_query only supports allowlisted plan shapes: \
6130 MATCH (n:Label) WHERE … RETURN … [LIMIT n] or \
6131 MATCH (a)-[r:TYPE]->(b) RETURN … [LIMIT n] (exactly one hop). \
6132 Not supported: multi-hop Expand chains, SKIP (creates \
6133 unstable offset windows), ORDER BY, DISTINCT, aggregates, \
6134 variable-length paths, OPTIONAL MATCH, WITH, UNWIND. \
6135 Use LIMIT to bound re-execution cost."
6136 .to_string(),
6137 });
6138 }
6139 // Execute once to capture initial state (initial rows are not emitted as
6140 // events — the subscriber learns the baseline via the first query call).
6141 let empty_params = BTreeMap::new();
6142 let initial = execute(&self.view(), &ops, &Params(&empty_params)).map_err(|e| {
6143 GraphError::QueryError {
6144 detail: format!("execute: {e}"),
6145 }
6146 })?;
6147 let columns = initial.columns().to_vec();
6148 let prev_row_map = Self::result_to_row_map(&initial);
6149 let inner = SubInner::new(self.sub_capacity());
6150 // Derive the scan-label sym for the commit-skip fast-path. Any Expand op
6151 // or unrecognized leading scan → None (always re-execute).
6152 let scan_label = extract_scan_label(&ops, Arc::make_mut(&mut self.syms));
6153 self.query_subscriptions.push(QuerySubEntry {
6154 ops,
6155 columns,
6156 prev_row_map,
6157 inner: std::sync::Arc::downgrade(&inner),
6158 scan_label,
6159 });
6160 Ok(Subscription(inner))
6161 }
6162
6163 /// Queue capacity used for new subscriptions.
6164 fn sub_capacity(&self) -> usize {
6165 self.sub_capacity
6166 }
6167
6168 /// Override per-subscriber queue capacity for subsequently created
6169 /// subscriptions on this db instance.
6170 ///
6171 /// Default is [`DEFAULT_SUB_CAPACITY`] (65,536 events). Use a smaller
6172 /// value in tests to exercise the [`DbEvent::Lagged`] path without
6173 /// generating tens of thousands of events.
6174 ///
6175 /// This is a test-support escape hatch. Calling it in production reduces
6176 /// subscriber reliability (more Lagged events). It is hidden from rustdoc
6177 /// to discourage accidental production use.
6178 #[doc(hidden)]
6179 pub fn set_sub_capacity(&mut self, capacity: usize) {
6180 self.sub_capacity = capacity;
6181 }
6182
6183 // -----------------------------------------------------------------------
6184
6185 /// Start an atomic batch.
6186 ///
6187 /// The returned [`BatchBuilder`] borrows `self` mutably until
6188 /// [`BatchBuilder::commit`]. Builder methods queue ops only — no
6189 /// validation, no WAL I/O. `commit` validates every queued op against
6190 /// live state plus preceding ops in this batch (duplicate key inside
6191 /// the batch is `Err`; an edge between two nodes created earlier in
6192 /// the batch is valid; `delete_node` then insert of the same key is a
6193 /// fresh identity). Validation never mutates the database. Any failure
6194 /// leaves WAL bytes and in-memory state identical to before `commit`.
6195 /// On success, one `WalRecord::Batch` frame is appended (one fsync)
6196 /// and each inner record is applied in order so rules fire per record.
6197 /// An empty batch, or a batch of only no-ops, writes zero WAL bytes.
6198 ///
6199 /// **Rule-window limitation:** batch validation cannot see edges that a
6200 /// rule created earlier in the *same* batch will derive at apply time, so
6201 /// a `delete_edge` / `insert_edge` in that window is silently no-oped
6202 /// where sequential calls would return `Err(RuleOwned)`. State integrity
6203 /// is unaffected (idempotent apply, provenance intact). Create rules in
6204 /// their own batch, or sequentially, when later ops may touch derived
6205 /// edges.
6206 pub fn batch(&mut self) -> BatchBuilder<'_, F> {
6207 BatchBuilder {
6208 db: self,
6209 ops: Vec::new(),
6210 }
6211 }
6212
6213 /// Closure-style atomic write batch.
6214 ///
6215 /// Equivalent to calling [`GraphDb::batch`], invoking `build` to queue ops,
6216 /// then committing. All ops queued inside `build` are validated in order and
6217 /// committed as a single `WalRecord::Batch` frame (one fsync). Rules fire
6218 /// once per inner record, in order, after commit — semantically identical to
6219 /// sequential single-op writes.
6220 ///
6221 /// **Error semantics — validate-then-apply.** `build` queues ops without
6222 /// touching the database. [`BatchBuilder::commit`] validates every op against
6223 /// live state plus earlier ops in this batch before writing anything. If op N
6224 /// fails validation (duplicate key, unknown key, rule-owned edge, …) the
6225 /// entire batch is rejected: no WAL bytes are written and no in-memory state
6226 /// changes. The database is identical to its state before `write_batch` was
6227 /// called.
6228 ///
6229 /// **Atomicity is crash-level, NOT isolation-level.** On replay after a crash,
6230 /// a partial (torn) `Batch` frame applies NONE of its ops — the frame is
6231 /// either fully applied or not at all. However, while applying a committed
6232 /// batch, concurrent readers may observe intermediate states as ops are applied
6233 /// sequentially in memory. There is no interactive transaction isolation in v1.
6234 /// This is documented as "crash-atomic write batches; no interactive
6235 /// transactions or read isolation."
6236 ///
6237 /// **Returns** `(nodes_inserted, edges_inserted)`. An empty or all-noop batch
6238 /// writes zero WAL bytes and returns `(0, 0)`.
6239 ///
6240 /// # Example
6241 ///
6242 /// ```rust,ignore
6243 /// let (nodes, edges) = db.write_batch(|b| {
6244 /// b.insert_node("Person", "alice", vec![("age".into(), Value::Int(30))]);
6245 /// b.insert_node("Person", "bob", vec![]);
6246 /// b.insert_edge("KNOWS", "alice", "bob");
6247 /// b.set_prop("alice", "role", Value::Str("admin".into()));
6248 /// b.delete_node("old_key");
6249 /// })?;
6250 /// // One fsync; on crash replay: all five ops land or none do.
6251 /// ```
6252 pub fn write_batch<C>(&mut self, build: C) -> Result<(usize, usize)>
6253 where
6254 C: FnOnce(&mut BatchBuilder<'_, F>),
6255 {
6256 let mut b = self.batch();
6257 build(&mut b);
6258 b.commit()
6259 }
6260
6261 /// Insert `rows` as nodes of `label`. One call is one atomic batch:
6262 /// auto-declared KeyMatch rules (if any) first, then the accepted node
6263 /// inserts, so incremental fire sees the new rules. Per-row key problems
6264 /// are collected in [`IngestReport::row_errors`] and skipped; a commit
6265 /// `Err` means nothing was applied.
6266 ///
6267 /// Auto-FK rule names are `auto_fk_<src_label_lowercase>_<field>` so
6268 /// distinct source labels sharing an FK field each get their own rule.
6269 pub fn ingest(
6270 &mut self,
6271 label: &str,
6272 rows: Vec<BTreeMap<String, Value>>,
6273 opts: &IngestOptions,
6274 ) -> Result<IngestReport> {
6275 self.ingest_with_edges(label, rows, opts, &[])
6276 }
6277
6278 /// [`ingest`] plus user edges in the **same** previewed WAL batch.
6279 /// A failing edge rejects the whole request; nothing is applied.
6280 pub fn ingest_with_edges(
6281 &mut self,
6282 label: &str,
6283 rows: Vec<BTreeMap<String, Value>>,
6284 opts: &IngestOptions,
6285 edges: &[(String, String, String)],
6286 ) -> Result<IngestReport> {
6287 crate::ingest::run(self, label, rows, opts, edges)
6288 }
6289
6290 /// Parse `json` as an array of objects and ingest via [`GraphDb::ingest`].
6291 ///
6292 /// JSON `null` fields are silently omitted (not stored, not a row error).
6293 /// Nested objects and arrays-of-objects are a per-row error (row skipped).
6294 /// Parse failures and a top-level value that is not an array of objects
6295 /// return [`GraphError::IngestError`].
6296 pub fn ingest_json(
6297 &mut self,
6298 label: &str,
6299 json: &str,
6300 opts: &IngestOptions,
6301 ) -> Result<IngestReport> {
6302 crate::ingest::run_json(self, label, json, opts)
6303 }
6304
6305 fn commit_logged_batch(
6306 &mut self,
6307 ops: Vec<BatchOp>,
6308 ingest: Option<(String, usize)>,
6309 // Two-source rule: write_batch_authz threads authz here directly (never
6310 // touches pending_write_authz); query_write_authz sets the field instead
6311 // and passes None. Only one source is non-None per call.
6312 param_authz: Option<WriteAuthz>,
6313 ) -> Result<BatchOutcome> {
6314 // Read-only guard: catches empty-batch calls before the early-return
6315 // that skips log_then_apply_with, ensuring all mutation entry points fail.
6316 if self.read_only {
6317 return Err(GraphError::ReadOnly);
6318 }
6319 // Ensure provenance is decoded before MutPreview accesses it
6320 // (note_delete_rule / is_rule_owned may call engine.provenance()).
6321 self.engine.ensure_provenance_loaded_mut();
6322
6323 // ── Authz pre-check ──────────────────────────────────────────────────
6324 // Evaluate the decision table per-op BEFORE MutPreview so that a denial
6325 // produces no WAL frame (all-or-nothing at the authz boundary extends
6326 // the existing validate-then-apply contract to role-scope checks).
6327 //
6328 // `batch_created` tracks key→label for nodes created by earlier ops in
6329 // THIS batch, so InsertEdgeUpsert can count same-batch placeholder nodes
6330 // as visible without needing to call `self.ids.get` on not-yet-committed
6331 // keys (they won't be there yet).
6332 //
6333 // Two-source rule: param_authz (write_batch_authz path) takes precedence;
6334 // fall back to self.pending_write_authz (query_write_authz/Cypher path).
6335 // Cloning the field copy avoids a simultaneous borrow of self.ids below.
6336 let authz_opt = param_authz.or_else(|| self.pending_write_authz.clone());
6337 if let Some(ref authz) = authz_opt {
6338 let mut batch_created: BTreeMap<String, String> = BTreeMap::new();
6339 for op in &ops {
6340 self.check_single_op_authz(authz, op, &batch_created)?;
6341 // Update batch_created after a passing authz check so that
6342 // subsequent ops in this batch see the nodes as "about to exist".
6343 match op {
6344 BatchOp::InsertNode { label, key, .. } => {
6345 // Only track genuinely new nodes (absent from the
6346 // snapshot at authz-check time). A pre-existing visible
6347 // key would be a DuplicateKey — not a real creation —
6348 // so MutPreview handles it. Letting it into batch_created
6349 // would allow a later SetProp to bypass update_labels
6350 // via the "batch-created → always updatable" ruling
6351 // (delete+recreate exploit, fix for I1 review round 2).
6352 //
6353 // Accepted edge: for a delete+recreate-with-different-
6354 // label batch, node_status resolves the pre-delete
6355 // (store) label for any subsequent update checks. This
6356 // grants no net-new capability — a role that can delete+
6357 // create can already place arbitrary props via
6358 // InsertNode's own props field.
6359 if self.ids.get(key.as_str()).is_none() {
6360 batch_created.insert(key.clone(), label.clone());
6361 }
6362 }
6363 BatchOp::InsertEdgeUpsert {
6364 placeholder_label,
6365 src_key,
6366 dst_key,
6367 ..
6368 } => {
6369 // Both endpoints will be created if not already in store.
6370 for ep_key in [src_key, dst_key] {
6371 if self.ids.get(ep_key.as_str()).is_none()
6372 && !batch_created.contains_key(ep_key.as_str())
6373 {
6374 batch_created.insert(ep_key.clone(), placeholder_label.clone());
6375 }
6376 }
6377 }
6378 _ => {}
6379 }
6380 }
6381 }
6382
6383 let mut outcome = BatchOutcome::default();
6384 let recs = {
6385 let mut preview = MutPreview::new(self);
6386 let mut recs = Vec::with_capacity(ops.len());
6387 // Which node row we are on, counted over the node-insert ops only.
6388 // A caller that queues its rows in order reads this straight back
6389 // as the index into its own list.
6390 let mut node_row = 0usize;
6391 // Every field name the store knows, which a `Replace` needs to work
6392 // out what it removes. Resolved on the first `Replace` in the frame
6393 // and reused, so N replaces read the field list once, not N times.
6394 let mut store_fields: Option<Vec<String>> = None;
6395 // Duplicate inserts this frame has to count, each paired with the
6396 // position in `recs` it belongs at. The count itself is named in the
6397 // dense rewrite and not here: a duplicate's endpoints and edge type
6398 // may all be created by earlier ops in this same frame, and nothing
6399 // in the frame has a dense id yet. See [`PlannedRec`].
6400 let mut deferred_counts: Vec<(usize, String, String, String)> = Vec::new();
6401 for op in ops {
6402 match op {
6403 BatchOp::InsertNode { label, key, props } => {
6404 node_row += 1;
6405 preview.check_insert_node(&key, &props)?;
6406 preview.note_insert_node(&label, &key, &props);
6407 recs.push(WalRecord::InsertNode { label, key, props });
6408 }
6409 BatchOp::InsertNodeOnConflict {
6410 label,
6411 key,
6412 props,
6413 on_conflict,
6414 } => {
6415 let row = node_row;
6416 node_row += 1;
6417 if !preview.has_key(&key) {
6418 // No conflict: an ordinary insert on any policy —
6419 // except that a supplied view-owned field is the
6420 // same mistake here as on a taken key, and gets the
6421 // same row error rather than a frame error. Without
6422 // this, one op answered one request two ways
6423 // depending on whether the store already had the
6424 // key (defect #19).
6425 if let Some(why) = preview.supplied_view_owned_prop(&key, &props) {
6426 outcome.row_errors.push((row, why));
6427 continue;
6428 }
6429 preview.note_insert_node(&label, &key, &props);
6430 recs.push(WalRecord::InsertNode { label, key, props });
6431 continue;
6432 }
6433 match on_conflict {
6434 OnConflict::Error => {
6435 return Err(GraphError::DuplicateKey { key });
6436 }
6437 OnConflict::Skip => outcome.skipped += 1,
6438 OnConflict::Replace => {
6439 if store_fields.is_none() {
6440 store_fields = Some(preview.db.props_view().field_names());
6441 }
6442 let fields = store_fields.as_deref().unwrap_or_default();
6443 match preview.plan_replace(&label, &key, &props, fields) {
6444 Ok((writes, kept_view_owned)) => {
6445 outcome.kept_view_owned += kept_view_owned;
6446 for (field, value) in writes {
6447 match value {
6448 Some(value) => {
6449 preview.note_set_prop(&key, &field, &value);
6450 recs.push(WalRecord::SetProp {
6451 key: key.clone(),
6452 field,
6453 value,
6454 });
6455 }
6456 None => {
6457 preview.note_remove_prop(&key, &field);
6458 recs.push(WalRecord::RemoveProp {
6459 key: key.clone(),
6460 field,
6461 });
6462 }
6463 }
6464 }
6465 outcome.replaced += 1;
6466 }
6467 Err(why) => outcome.row_errors.push((row, why)),
6468 }
6469 }
6470 }
6471 }
6472 BatchOp::InsertEdge {
6473 edge_type,
6474 src_key,
6475 dst_key,
6476 } => {
6477 if preview.prepare_insert_edge(&edge_type, &src_key, &dst_key)? {
6478 preview.note_insert_edge(&edge_type, &src_key, &dst_key);
6479 recs.push(WalRecord::InsertEdge {
6480 edge_type,
6481 src_key,
6482 dst_key,
6483 });
6484 } else if preview.db.multiplicity {
6485 // A duplicate inside a batch counts the way a
6486 // duplicate through `insert_edge` does: `ingest` and
6487 // Cypher `CREATE` reach this choke-point and not
6488 // that one, and a count only one entry point keeps
6489 // would be worse than no count at all.
6490 //
6491 // This is the one gate on discriminant 23 from the
6492 // batch path: a store that never opted in queues
6493 // nothing here and so writes no such record.
6494 deferred_counts.push((recs.len(), edge_type, src_key, dst_key));
6495 }
6496 }
6497 BatchOp::SetProp { key, field, value } => {
6498 if let Some(view_name) = preview.db.view_store.view_for_prop(&field) {
6499 return Err(GraphError::ViewPropReadOnly {
6500 view_name: view_name.to_string(),
6501 });
6502 }
6503 preview.check_live_key(&key)?;
6504 preview.note_set_prop(&key, &field, &value);
6505 recs.push(WalRecord::SetProp { key, field, value });
6506 }
6507 BatchOp::RemoveProp { key, field } => {
6508 if preview.prepare_remove_prop(&key, &field)? {
6509 preview.note_remove_prop(&key, &field);
6510 recs.push(WalRecord::RemoveProp { key, field });
6511 }
6512 }
6513 BatchOp::DeleteEdge {
6514 edge_type,
6515 src_key,
6516 dst_key,
6517 } => {
6518 if preview.prepare_delete_edge(&edge_type, &src_key, &dst_key)? {
6519 preview.note_delete_edge(&edge_type, &src_key, &dst_key);
6520 recs.push(WalRecord::DeleteEdge {
6521 edge_type,
6522 src_key,
6523 dst_key,
6524 });
6525 }
6526 }
6527 BatchOp::DeleteNode { key } => {
6528 preview.check_live_key(&key)?;
6529 preview.note_delete_node(&key);
6530 recs.push(WalRecord::DeleteNode { key });
6531 }
6532 BatchOp::CreateRule(def) => {
6533 preview.check_create_rule(&def)?;
6534 let def_bytes =
6535 bincode::serialize(&def).map_err(|e| GraphError::Corrupt {
6536 detail: format!("serialize rule: {e}"),
6537 })?;
6538 preview.note_create_rule(&def);
6539 recs.push(WalRecord::CreateRule { def_bytes });
6540 }
6541 BatchOp::DeleteRule { name } => {
6542 preview.check_delete_rule(&name)?;
6543 preview.note_delete_rule(&name);
6544 recs.push(WalRecord::DeleteRule { name });
6545 }
6546 BatchOp::RenameNode { old_key, new_key } => {
6547 preview.check_rename_node(&old_key, &new_key)?;
6548 preview.note_rename_node(&old_key, &new_key);
6549 recs.push(WalRecord::RenameNode { old_key, new_key });
6550 }
6551 BatchOp::InsertEdgeUpsert {
6552 edge_type,
6553 src_key,
6554 dst_key,
6555 placeholder_label,
6556 } => {
6557 // Auto-create any missing endpoints as plain InsertNode ops.
6558 // Rules fire and last-change is updated for each created node.
6559 for key in [&src_key, &dst_key] {
6560 if !preview.has_key(key) {
6561 // A placeholder endpoint carries no props, so
6562 // the view-owned check has nothing to refuse.
6563 preview.check_insert_node(key, &[])?;
6564 preview.note_insert_node(&placeholder_label, key, &[]);
6565 recs.push(WalRecord::InsertNode {
6566 label: placeholder_label.clone(),
6567 key: key.clone(),
6568 props: vec![],
6569 });
6570 }
6571 }
6572 if preview.prepare_insert_edge(&edge_type, &src_key, &dst_key)? {
6573 preview.note_insert_edge(&edge_type, &src_key, &dst_key);
6574 recs.push(WalRecord::InsertEdge {
6575 edge_type,
6576 src_key,
6577 dst_key,
6578 });
6579 } else if preview.db.multiplicity {
6580 // Same choke-point, same gate as `BatchOp::InsertEdge`
6581 // above: an upsert that finds the pair already there
6582 // is a duplicate insert and counts as one.
6583 deferred_counts.push((recs.len(), edge_type, src_key, dst_key));
6584 }
6585 }
6586 }
6587 }
6588 // Splice the deferred counts back into the positions they were
6589 // raised at, so a count still sits exactly where the duplicate did
6590 // — before any later op in the frame that deletes the pair.
6591 let mut planned: Vec<PlannedRec> =
6592 Vec::with_capacity(recs.len() + deferred_counts.len());
6593 let mut deferred = deferred_counts.into_iter().peekable();
6594 for (i, rec) in recs.into_iter().enumerate() {
6595 while deferred.peek().is_some_and(|(at, ..)| *at == i) {
6596 let (_, edge_type, src_key, dst_key) = deferred.next().expect("just peeked");
6597 planned.push(PlannedRec::DuplicateCount {
6598 edge_type,
6599 src_key,
6600 dst_key,
6601 });
6602 }
6603 planned.push(PlannedRec::Rec(rec));
6604 }
6605 for (_, edge_type, src_key, dst_key) in deferred {
6606 planned.push(PlannedRec::DuplicateCount {
6607 edge_type,
6608 src_key,
6609 dst_key,
6610 });
6611 }
6612 planned
6613 };
6614 // A frame that is nothing but skips or refused rows writes no WAL, but
6615 // it still has counts to report, so the early returns carry `outcome`
6616 // rather than zeros.
6617 if recs.is_empty() {
6618 return Ok(outcome);
6619 }
6620 // rewrite_wal_dense converts every InsertNode/InsertEdge into its
6621 // *Id form, so only the dense variants can appear in `recs` here.
6622 let recs = self.rewrite_wal_dense_planned(recs)?;
6623 // The rewrite can empty a non-empty batch: a `SET n.ns` naming the
6624 // namespace the node is already in is a no-op and is dropped there. An
6625 // empty `Batch` frame would still take a commit sequence and a WAL
6626 // record, so a batch that turns out to be nothing writes nothing.
6627 if recs.is_empty() {
6628 return Ok(outcome);
6629 }
6630 outcome.nodes_inserted = recs
6631 .iter()
6632 .filter(|r| matches!(r, WalRecord::InsertNodeId { .. }))
6633 .count();
6634 outcome.edges_inserted = recs
6635 .iter()
6636 .filter(|r| matches!(r, WalRecord::InsertEdgeId { .. }))
6637 .count();
6638 // Ingest / write_batch / query_write: one Batch frame, one fsync per call
6639 // under Strict. Pass self.fsync directly so Strict stays Strict —
6640 // wal_needs_sync(Strict, _) always returns true regardless of op count.
6641 // Mapping Strict → Batched (the prior bug) caused wal_needs_sync to
6642 // short-circuit on single-op batches and silently skip the fsync.
6643 // Batched fsyncs only for multi-op batches; Relaxed always skips.
6644 self.log_then_apply_with(WalRecord::Batch(recs), ingest, self.fsync)?;
6645 Ok(outcome)
6646 }
6647
6648 fn commit_batch(&mut self, ops: Vec<BatchOp>) -> Result<(usize, usize)> {
6649 self.commit_logged_batch(ops, None, None).map(inserted_pair)
6650 }
6651
6652 /// Commit one submission WITHOUT an fsync — for use inside `commit_group`
6653 /// and the group-commit drain thread, which do a single group fsync later.
6654 fn commit_batch_nosync(&mut self, ops: Vec<BatchOp>) -> Result<(usize, usize)> {
6655 // Restore fsync policy even on panic via a raw-pointer drop guard.
6656 // A panic here would poison the RwLock anyway, but the correct policy
6657 // must be in place if the guard is ever unwrapped.
6658 struct RestoreFsync(*mut FsyncPolicy, FsyncPolicy);
6659 impl Drop for RestoreFsync {
6660 fn drop(&mut self) {
6661 // SAFETY: the pointer is valid for the full duration of
6662 // commit_batch_nosync; the guard is dropped before the frame
6663 // returns, and GraphDb outlives this frame.
6664 unsafe {
6665 *self.0 = self.1;
6666 }
6667 }
6668 }
6669 let saved = self.fsync;
6670 // SAFETY: raw pointer into self; guard dropped within this frame.
6671 let _g = RestoreFsync(&mut self.fsync as *mut FsyncPolicy, saved);
6672 self.fsync = FsyncPolicy::Relaxed;
6673 self.commit_logged_batch(ops, None, None).map(inserted_pair)
6674 }
6675
6676 /// Commit multiple op-batches as a **group**: each submission gets its own
6677 /// WAL `Batch` frame, but there is exactly **one** `Fs::sync` for the whole
6678 /// group (under `Strict` / `Batched` policy; `Relaxed` skips all syncs).
6679 ///
6680 /// # Durability semantics
6681 ///
6682 /// A crash before the group fsync may lose **all** submissions in the group.
6683 /// A crash after the group fsync preserves all of them. No submission is
6684 /// ever torn: each WAL frame is either fully applied on replay or dropped
6685 /// in its entirety (CRC-protected frame boundaries).
6686 ///
6687 /// Events and subscription notifications fire per-submission immediately
6688 /// after apply, which may be before the group fsync. From a subscriber's
6689 /// perspective this is equivalent to the `Relaxed` durability window.
6690 /// Submitters using [`SharedDb::submit_batch`] only unblock after the group
6691 /// fsync, so from their perspective durability is fully guaranteed.
6692 ///
6693 /// # MVCC interplay
6694 ///
6695 /// Each submission records its own `CommitDelta`; the fold-every-K counter
6696 /// increments per submission (not per group), preserving existing reader
6697 /// snapshot semantics.
6698 ///
6699 /// # Returns
6700 ///
6701 /// One `Result<(nodes_inserted, edges_inserted)>` per input group element,
6702 /// in order. Failures are per-submission (validation errors); the group
6703 /// fsync error (if any) is returned as the second tuple element.
6704 pub fn commit_group(
6705 &mut self,
6706 groups: Vec<Vec<BatchOp>>,
6707 ) -> (Vec<Result<(usize, usize)>>, Option<GraphError>) {
6708 let mut results = Vec::with_capacity(groups.len());
6709 for ops in groups {
6710 results.push(self.commit_batch_nosync(ops));
6711 }
6712 let any_ok = results.iter().any(|r| r.is_ok());
6713 let sync_err = if self.fsync != FsyncPolicy::Relaxed && any_ok {
6714 self.fs
6715 .sync(core_storage::fs::FileId::Wal)
6716 .map_err(GraphError::Io)
6717 .err()
6718 } else {
6719 None
6720 };
6721 (results, sync_err)
6722 }
6723
6724 /// Like [`commit_group`] but skips the group fsync entirely.
6725 ///
6726 /// Used by the drain thread to apply submissions under the write lock and
6727 /// then perform the single fsync OUTSIDE the lock (via
6728 /// `core_storage::sync_wal_at`), reducing the write-lock hold time visible
6729 /// to concurrent readers.
6730 pub fn commit_group_nosync(
6731 &mut self,
6732 groups: Vec<Vec<BatchOp>>,
6733 ) -> Vec<Result<(usize, usize)>> {
6734 let mut results = Vec::with_capacity(groups.len());
6735 for ops in groups {
6736 results.push(self.commit_batch_nosync(ops));
6737 }
6738 results
6739 }
6740
6741 pub fn insert_node(
6742 &mut self,
6743 label: &str,
6744 key: &str,
6745 props: Vec<(String, Value)>,
6746 ) -> Result<()> {
6747 if self.read_only {
6748 return Err(GraphError::ReadOnly);
6749 }
6750 MutPreview::new(self).check_insert_node(key, &props)?;
6751 self.log_dense(vec![WalRecord::InsertNode {
6752 label: label.into(),
6753 key: key.into(),
6754 props,
6755 }])
6756 }
6757
6758 /// Insert a user edge. `Ok(true)` when the pair was new, `Ok(false)` when it
6759 /// was already there — the question is "was this pair new", and a duplicate
6760 /// does not make it so.
6761 ///
6762 /// On a store that called [`enable_multiplicity`](Self::enable_multiplicity)
6763 /// a duplicate is no longer a total no-op: it raises the pair's insert count
6764 /// (§5.13). Adjacency is still a set, so [`degree`](Self::degree) is
6765 /// unchanged and the return value is still `Ok(false)`; the count is visible
6766 /// only through [`degree_multiplicity`](Self::degree_multiplicity) and the
6767 /// reserved [`EDGE_COUNT_PROP`]. On every other store a duplicate writes
6768 /// nothing at all, as it always has.
6769 pub fn insert_edge(&mut self, edge_type: &str, src_key: &str, dst_key: &str) -> Result<bool> {
6770 if self.read_only {
6771 return Err(GraphError::ReadOnly);
6772 }
6773 if !MutPreview::new(self).prepare_insert_edge(edge_type, src_key, dst_key)? {
6774 // The pair exists. The only thing left to record is that it was
6775 // asked for again, and only where the store asked to be told.
6776 if let Some(rec) = self.edge_count_record(edge_type, src_key, dst_key) {
6777 self.log_then_apply(rec)?;
6778 }
6779 return Ok(false);
6780 }
6781 self.log_dense(vec![WalRecord::InsertEdge {
6782 edge_type: edge_type.into(),
6783 src_key: src_key.into(),
6784 dst_key: dst_key.into(),
6785 }])?;
6786 Ok(true)
6787 }
6788
6789 pub fn set_prop(&mut self, key: &str, field: &str, value: Value) -> Result<()> {
6790 if self.read_only {
6791 return Err(GraphError::ReadOnly);
6792 }
6793 if let Some(view_name) = self.view_store.view_for_prop(field) {
6794 return Err(GraphError::ViewPropReadOnly {
6795 view_name: view_name.to_string(),
6796 });
6797 }
6798 MutPreview::new(self).check_live_key(key)?;
6799 self.log_dense(vec![WalRecord::SetProp {
6800 key: key.into(),
6801 field: field.into(),
6802 value,
6803 }])
6804 }
6805
6806 /// Set several properties on one live node in a single WAL commit.
6807 ///
6808 /// Every per-property check [`set_prop`](Self::set_prop) runs — view-owned
6809 /// names, live key, the `ns` immutability rule and its type — is evaluated
6810 /// for the whole list before any record is logged. The first refusal
6811 /// returns and the node is unchanged. An empty list writes nothing.
6812 pub fn set_props(&mut self, key: &str, props: Vec<(String, Value)>) -> Result<()> {
6813 if self.read_only {
6814 return Err(GraphError::ReadOnly);
6815 }
6816 MutPreview::new(self).check_live_key(key)?;
6817 for (field, _) in &props {
6818 if let Some(view_name) = self.view_store.view_for_prop(field) {
6819 return Err(GraphError::ViewPropReadOnly {
6820 view_name: view_name.to_string(),
6821 });
6822 }
6823 }
6824 if props.is_empty() {
6825 return Ok(());
6826 }
6827 self.write_batch(|b| {
6828 for (field, value) in props {
6829 b.set_prop(key, &field, value);
6830 }
6831 })
6832 .map(|_| ())
6833 }
6834
6835 /// Remove a property. Returns `Ok(false)` (and does not log) if the field
6836 /// is already absent. Unknown or tombstoned keys are `Err(KeyNotFound)`.
6837 /// A field a view owns is `Err(ViewPropReadOnly)` — stated once, in
6838 /// [`MutPreview::prepare_remove_prop`], so that the batch ops reaching that
6839 /// same choke-point cannot miss it.
6840 pub fn remove_prop(&mut self, key: &str, field: &str) -> Result<bool> {
6841 if self.read_only {
6842 return Err(GraphError::ReadOnly);
6843 }
6844 if !MutPreview::new(self).prepare_remove_prop(key, field)? {
6845 return Ok(false);
6846 }
6847 self.log_then_apply(WalRecord::RemoveProp {
6848 key: key.into(),
6849 field: field.into(),
6850 })?;
6851 Ok(true)
6852 }
6853
6854 /// Delete a user edge. Returns `Ok(false)` (and does not log) if the edge
6855 /// is absent. Unknown keys are `Err(KeyNotFound)`. Rule-owned edges — in
6856 /// provenance, or a pair a live rule would derive — are `Err(RuleOwned)`
6857 /// (the rule would just put the edge back; delete or change the rule).
6858 pub fn delete_edge(&mut self, edge_type: &str, src_key: &str, dst_key: &str) -> Result<bool> {
6859 if self.read_only {
6860 return Err(GraphError::ReadOnly);
6861 }
6862 if !MutPreview::new(self).prepare_delete_edge(edge_type, src_key, dst_key)? {
6863 return Ok(false);
6864 }
6865 self.log_then_apply(WalRecord::DeleteEdge {
6866 edge_type: edge_type.into(),
6867 src_key: src_key.into(),
6868 dst_key: dst_key.into(),
6869 })?;
6870 Ok(true)
6871 }
6872
6873 /// Delete a live node. Unknown or already-tombstoned keys are
6874 /// `Err(KeyNotFound)` and are not logged. Validation runs before the WAL
6875 /// write; `apply` of a logged `DeleteNode` for an already-tombstoned key
6876 /// (crash window) is a clean no-op.
6877 ///
6878 /// Returns a [`DeleteReport`] with counts of manual and derived edges
6879 /// removed (computed from live state before the deletion is applied).
6880 pub fn delete_node(&mut self, key: &str) -> Result<DeleteReport> {
6881 if self.read_only {
6882 return Err(GraphError::ReadOnly);
6883 }
6884 // Provenance must be loaded before we query provenance_touching.
6885 self.engine.ensure_provenance_loaded_mut();
6886 let id = self
6887 .ids
6888 .get(key)
6889 .ok_or_else(|| GraphError::KeyNotFound { key: key.into() })?;
6890
6891 // Count edges before the delete is applied so we can report counts.
6892 let derived_set: BTreeSet<(u32, u32, u32)> = self
6893 .engine
6894 .provenance_touching(id)
6895 .map(|(_, etype, src, dst)| (etype, src, dst))
6896 .collect();
6897 let derived_edges = derived_set.len() as u64;
6898
6899 let mut total_topo = 0u64;
6900 let tv = self.topo_view();
6901 for et in tv.etypes() {
6902 total_topo += tv.neighbors(et, Direction::Out, id).len() as u64
6903 + tv.neighbors(et, Direction::In, id).len() as u64;
6904 }
6905 // For symmetric rules (e.g. Overlap), a→b and b→a are two separate directed
6906 // triples in both the topo scan (Out and In from id) and in provenance_touching.
6907 // The subtraction remains correct because both counts include both directions.
6908 let manual_edges = total_topo.saturating_sub(derived_edges);
6909
6910 self.log_then_apply(WalRecord::DeleteNode { key: key.into() })?;
6911 Ok(DeleteReport {
6912 manual_edges,
6913 derived_edges,
6914 })
6915 }
6916
6917 /// Rename a live node's key. The dense id (and therefore all edges,
6918 /// props, history, and last-change tracking) is unaffected.
6919 ///
6920 /// Returns `Err(KeyNotFound)` if `old` is not a live key.
6921 /// Returns `Err(DuplicateKey)` if `new` is already live.
6922 pub fn rename_node(&mut self, old: &str, new: &str) -> Result<()> {
6923 if self.read_only {
6924 return Err(GraphError::ReadOnly);
6925 }
6926 MutPreview::new(self).check_rename_node(old, new)?;
6927 self.log_then_apply(WalRecord::RenameNode {
6928 old_key: old.into(),
6929 new_key: new.into(),
6930 })
6931 }
6932
6933 /// Return the IVF drift counter for the dst-side candidate index of `rule`.
6934 /// `None` if the rule does not exist or is not approximate.
6935 ///
6936 /// The drift counter increments on IVF insert/remove after the last fit.
6937 /// When dst-side drift exceeds [`core_rules::IVF_DRIFT_REBUILD`], apply
6938 /// WAL-logs `RebuildRule` as a second commit (rebuild resets the counter).
6939 pub fn ivf_dst_drift(&self, rule: &str) -> Option<u64> {
6940 // SideIvfExport = (centroids, node→cluster, drift)
6941 self.engine
6942 .export_ivf_state()
6943 .remove(rule)
6944 .map(|(_src, dst)| dst.2)
6945 }
6946
6947 /// Validate and WAL-log a new rule, then backfill derived edges inside apply.
6948 /// Validation and duplicate-name check run before logging so invalid rules
6949 /// never enter the WAL.
6950 pub fn create_rule(&mut self, def: RuleDef) -> Result<()> {
6951 if self.read_only {
6952 return Err(GraphError::ReadOnly);
6953 }
6954 MutPreview::new(self).check_create_rule(&def)?;
6955 let def_bytes = bincode::serialize(&def).map_err(|e| GraphError::Corrupt {
6956 detail: format!("serialize rule: {e}"),
6957 })?;
6958 self.log_then_apply(WalRecord::CreateRule { def_bytes })
6959 }
6960
6961 /// Override this handle's HNSW build-slice size, or `None` to restore
6962 /// [`core_rules::HNSW_BUILD_BATCH`].
6963 ///
6964 /// Exposed for tests that need a small slice without a large corpus; not
6965 /// part of the stable surface.
6966 #[doc(hidden)]
6967 pub fn set_hnsw_build_batch(&mut self, batch: Option<usize>) {
6968 self.engine.set_hnsw_build_batch(batch);
6969 }
6970
6971 /// Rules whose vector index is still being built, in name order.
6972 ///
6973 /// The same list [`GraphDb::stats`] reports per rule in `building`.
6974 /// After a clean open this includes a build a snapshot cut short, so
6975 /// `serve`'s ticker can pump it without a write.
6976 pub fn builds_in_progress(&self) -> Vec<BuildProgress> {
6977 self.engine.builds_in_progress()
6978 }
6979
6980 /// Advance any vector index still building and backfill each rule that
6981 /// finishes. Returns what is still outstanding.
6982 ///
6983 /// A map lookup when nothing is pending, so it is cheap to call on a timer.
6984 /// One write lock and at most [`core_rules::HNSW_BUILD_BATCH`] vector
6985 /// inserts per pending rule per call, so a caller can drive a large build
6986 /// to completion without ever holding the lock for more than a slice.
6987 ///
6988 /// A rule that finishes here is backfilled through the same
6989 /// `WalRecord::RebuildRule` second commit that IVF drift already uses, so
6990 /// its derived edges are produced by [`GraphDb::rebuild_rule`]'s code path
6991 /// and appear all at once.
6992 ///
6993 /// Every ordinary write pumps one slice on its own (see the post-commit
6994 /// hook in `log_then_apply_with`), so this is for quiescent stores and for
6995 /// operators who want the build finished before traffic arrives.
6996 pub fn pump_index_build(&mut self) -> Result<Vec<BuildProgress>> {
6997 Ok(self.pump_index_build_reporting()?.1)
6998 }
6999
7000 /// [`GraphDb::pump_index_build`], also reporting the builds that **this**
7001 /// call finished, so a progress display can say so.
7002 ///
7003 /// A build can be registered and completed inside a single call — that is
7004 /// what a mid-build snapshot looks like on reopen, where the index scan
7005 /// finishes the graph and only the backfill is outstanding — and the
7006 /// outstanding list alone cannot show that anything happened.
7007 pub fn pump_index_build_reporting(
7008 &mut self,
7009 ) -> Result<(Vec<BuildProgress>, Vec<BuildProgress>)> {
7010 // A read-only handle cannot issue the `RebuildRule` a finished build
7011 // needs, so it would advance the index and then silently fail to
7012 // produce the edges. Refusing is the honest answer.
7013 if self.read_only {
7014 return Err(GraphError::ReadOnly);
7015 }
7016 let finished = self.pump_one_slice();
7017 for done in &finished {
7018 // The index is whole but the rule still owns no edges. A failed
7019 // second commit must leave the rule re-pumpable rather than
7020 // silently edge-less, so the error is surfaced here — unlike the
7021 // post-commit hook, this call is not riding someone else's commit.
7022 self.log_then_apply(WalRecord::RebuildRule {
7023 name: done.rule.clone(),
7024 })?;
7025 }
7026 Ok((finished, self.engine.builds_in_progress()))
7027 }
7028
7029 /// Run the deferred candidate-index build, if it is still owed, against the
7030 /// graph as it stands *now* — before the caller applies anything.
7031 ///
7032 /// A no-op bool test once the indexes are populated, which is after the
7033 /// first write of the handle's life, and for a store with no rules at all.
7034 fn populate_indexes_before_write(&mut self) {
7035 if !self.engine.needs_index_population() {
7036 return;
7037 }
7038 // The retained snapshot blobs arrive with the V8 base sections; without
7039 // them the scan would rebuild every graph the snapshot already holds.
7040 self.ensure_v8_base_sections_loaded();
7041 if !self.engine.needs_index_population() {
7042 return;
7043 }
7044 let mut eng = std::mem::take(&mut self.engine);
7045 {
7046 let gm = make_graph_mut(
7047 &self.ids,
7048 Arc::make_mut(&mut self.syms),
7049 &self.labels,
7050 build_props_view(&self.props, &self.base),
7051 Arc::make_mut(&mut self.topo),
7052 &self.base,
7053 Arc::make_mut(&mut self.edge_props),
7054 );
7055 eng.populate_indexes(&gm);
7056 }
7057 self.engine = eng;
7058 }
7059
7060 /// One slice of build work for every pending rule. Returns the rules whose
7061 /// index just became whole, which the caller must `RebuildRule`.
7062 ///
7063 /// Goes through the engine even with nothing pending when the indexes have
7064 /// not been populated yet: that call adopts the persisted graphs and, for
7065 /// an incomplete blob already registered at open, leaves the remainder to
7066 /// this slice rather than inserting it inline.
7067 fn pump_one_slice(&mut self) -> Vec<BuildProgress> {
7068 // The retained snapshot blobs — and the id count an interrupted build
7069 // is recognised against — arrive with the V8 base sections, which a
7070 // clean open reads lazily. Without this a freshly opened handle pumps
7071 // against empty retained state and concludes there is nothing to do,
7072 // which is precisely the store `build-index` exists for.
7073 self.ensure_v8_base_sections_loaded();
7074 let mut eng = std::mem::take(&mut self.engine);
7075 let finished = {
7076 let mut gm = make_graph_mut(
7077 &self.ids,
7078 Arc::make_mut(&mut self.syms),
7079 &self.labels,
7080 build_props_view(&self.props, &self.base),
7081 Arc::make_mut(&mut self.topo),
7082 &self.base,
7083 Arc::make_mut(&mut self.edge_props),
7084 );
7085 eng.pump_index_build(&mut gm)
7086 };
7087 self.engine = eng;
7088 finished
7089 }
7090
7091 /// Register a sliced build a snapshot cut short, from blobs with
7092 /// `complete == false`.
7093 ///
7094 /// Peeks the V8 mmap for incomplete entries without copying complete
7095 /// graphs. V5–V7 already hold the blobs in the engine from restore.
7096 fn register_outstanding_index_builds(&mut self) {
7097 if self.engine.indexes_populated() {
7098 return;
7099 }
7100 let extra = self.collect_incomplete_hnsw_blobs();
7101 let mut eng = std::mem::take(&mut self.engine);
7102 {
7103 let gm = make_graph_mut(
7104 &self.ids,
7105 Arc::make_mut(&mut self.syms),
7106 &self.labels,
7107 build_props_view(&self.props, &self.base),
7108 Arc::make_mut(&mut self.topo),
7109 &self.base,
7110 Arc::make_mut(&mut self.edge_props),
7111 );
7112 eng.register_incomplete_hnsw_builds(&extra, &gm);
7113 }
7114 self.engine = eng;
7115 }
7116
7117 /// Incomplete `(src, dst)` HNSW blobs from the V8 mmap, copied only when
7118 /// `complete` is false. Empty when there is no mmap base (V5–V7 uses the
7119 /// engine's retained map instead).
7120 fn collect_incomplete_hnsw_blobs(&self) -> BTreeMap<String, (Vec<u8>, Vec<u8>)> {
7121 let Some(base) = &self.base else {
7122 return BTreeMap::new();
7123 };
7124 let Ok(archived) = base.hnsw_section() else {
7125 return BTreeMap::new();
7126 };
7127 archived
7128 .rules
7129 .iter()
7130 .filter_map(|e| {
7131 let src = e.src_blob.as_slice();
7132 let dst = e.dst_blob.as_slice();
7133 if core_rules::hnsw::hnsw_blob_complete(src) == Some(false)
7134 || core_rules::hnsw::hnsw_blob_complete(dst) == Some(false)
7135 {
7136 Some((e.name.as_str().to_string(), (src.to_vec(), dst.to_vec())))
7137 } else {
7138 None
7139 }
7140 })
7141 .collect()
7142 }
7143
7144 /// WAL-log rule deletion. Returns RuleNotFound if the rule does not exist.
7145 pub fn delete_rule(&mut self, name: &str) -> Result<()> {
7146 if self.read_only {
7147 return Err(GraphError::ReadOnly);
7148 }
7149 MutPreview::new(self).check_delete_rule(name)?;
7150 self.log_then_apply(WalRecord::DeleteRule { name: name.into() })
7151 }
7152
7153 /// Return a snapshot of all registered rules.
7154 pub fn rules(&self) -> Vec<RuleDef> {
7155 self.engine.rules().cloned().collect()
7156 }
7157
7158 // -----------------------------------------------------------------------
7159 // Rule suggestion API
7160 // -----------------------------------------------------------------------
7161
7162 /// Profile the database and suggest linking rules with previewed edge counts.
7163 ///
7164 /// Uses the default seed ([`core_rules::SUGGEST_DEFAULT_SEED`]) for deterministic
7165 /// sampling. Suggestions are sorted by estimated edge count (descending).
7166 /// **NO auto-accept** — call [`GraphDb::create_rule`] explicitly to apply.
7167 pub fn suggest_rules(&self) -> Vec<core_rules::RuleSuggestion> {
7168 self.suggest_rules_seeded(core_rules::SUGGEST_DEFAULT_SEED)
7169 }
7170
7171 /// Like [`suggest_rules`] but with a caller-supplied RNG seed for
7172 /// reproducibility. Same seed + same data = identical output.
7173 pub fn suggest_rules_seeded(&self, seed: u64) -> Vec<core_rules::RuleSuggestion> {
7174 self.suggest_rules_with_config(&core_rules::suggest::SuggestConfig::default(), seed)
7175 .suggestions
7176 }
7177
7178 /// [`suggest_rules_seeded`] with a fully custom [`SuggestConfig`].
7179 ///
7180 /// Returns a [`core_rules::SuggestReport`] that includes both the candidate list
7181 /// and a `truncated` flag indicating whether the global budget fired before all
7182 /// candidates were evaluated.
7183 pub fn suggest_rules_with_config(
7184 &self,
7185 config: &core_rules::suggest::SuggestConfig,
7186 seed: u64,
7187 ) -> core_rules::SuggestReport {
7188 use std::collections::BTreeMap;
7189
7190 // Collect (node_id, key) pairs per label, skipping tombstoned nodes.
7191 let mut label_nodes: BTreeMap<String, Vec<(u32, String)>> = BTreeMap::new();
7192 for id in 0..self.ids.len() as u32 {
7193 let Some(key) = self.ids.key_of(id) else {
7194 continue;
7195 };
7196 let Some(&sym) = self.labels.get(id as usize) else {
7197 continue;
7198 };
7199 if sym == u32::MAX {
7200 continue; // tombstoned
7201 }
7202 let Some(label) = self.syms.resolve(sym) else {
7203 continue;
7204 };
7205 label_nodes
7206 .entry(label.to_string())
7207 .or_default()
7208 .push((id, key.to_string()));
7209 }
7210
7211 let existing = self.rules();
7212 let pv = build_props_view(&self.props, &self.base);
7213 let all_fields: Vec<String> = pv.field_names();
7214
7215 core_rules::suggest::suggest_rules(
7216 &label_nodes,
7217 &|id, field| pv.get(id, field).map(|vr| vr.into_value()),
7218 &all_fields,
7219 &existing,
7220 config,
7221 seed,
7222 )
7223 }
7224
7225 /// Recompute a rule's derived edges from scratch. WAL-logged so un-trip
7226 /// plus later mutations replay identically (rebuild is a pure function
7227 /// of state).
7228 ///
7229 /// Only exit from the tripped latch: if the full desired set fits the
7230 /// budget, it is applied completely and `tripped` clears; if it still
7231 /// exceeds the budget, provenance is left untouched and `tripped` stays
7232 /// true. Counts as a fire evaluation (see [`RuleStats::fires`]).
7233 /// Unknown rule → `RuleNotFound`, nothing logged.
7234 pub fn rebuild_rule(&mut self, name: &str) -> Result<()> {
7235 if self.read_only {
7236 return Err(GraphError::ReadOnly);
7237 }
7238 if !self.engine.rules().any(|r| r.name == name) {
7239 return Err(GraphError::RuleNotFound { name: name.into() });
7240 }
7241 self.log_then_apply(WalRecord::RebuildRule { name: name.into() })
7242 }
7243
7244 // -----------------------------------------------------------------------
7245 // Materialized view API
7246 // -----------------------------------------------------------------------
7247
7248 /// Register a new materialized property view, backfill its values for all
7249 /// existing nodes, and WAL-log the definition.
7250 ///
7251 /// # Errors
7252 /// - `ReadOnly`: called on an as-of instance.
7253 /// - `RuleInvalid`: name collision, view_prop collision, or invalid def.
7254 pub fn create_view(&mut self, def: ViewDef) -> Result<()> {
7255 if self.read_only {
7256 return Err(GraphError::ReadOnly);
7257 }
7258 // Pre-validate before WAL write.
7259 def.validate()
7260 .map_err(|e| GraphError::RuleInvalid { detail: e })?;
7261 if self.view_store.has_view(&def.name) {
7262 return Err(GraphError::RuleInvalid {
7263 detail: format!("view {:?} already exists", def.name),
7264 });
7265 }
7266 if let Some(existing) = self.view_store.view_for_prop(&def.view_prop) {
7267 return Err(GraphError::RuleInvalid {
7268 detail: format!(
7269 "view_prop {:?} is already used by view {:?}",
7270 def.view_prop, existing
7271 ),
7272 });
7273 }
7274 let def_bytes = bincode::serialize(&def).map_err(|e| GraphError::Corrupt {
7275 detail: format!("serialize view: {e}"),
7276 })?;
7277 // Enable delta accumulation before the view is registered so subsequent
7278 // incremental edge events reach view maintenance from this point onward.
7279 // (The backfill inside create_view reads topo directly; it does not rely
7280 // on pending deltas.)
7281 self.engine.set_emit_deltas(true);
7282 self.log_then_apply(WalRecord::CreateView { def_bytes })
7283 }
7284
7285 /// Remove a named view and delete its values from every node.
7286 ///
7287 /// # Errors
7288 /// - `ReadOnly`: called on an as-of instance.
7289 /// - `RuleNotFound`: view does not exist.
7290 pub fn delete_view(&mut self, name: &str) -> Result<()> {
7291 if self.read_only {
7292 return Err(GraphError::ReadOnly);
7293 }
7294 if !self.view_store.has_view(name) {
7295 return Err(GraphError::RuleNotFound { name: name.into() });
7296 }
7297 let result = self.log_then_apply(WalRecord::DeleteView { name: name.into() });
7298 // After deletion, disable accumulation if no listeners remain.
7299 if !self.needs_emit_deltas() {
7300 self.engine.set_emit_deltas(false);
7301 }
7302 result
7303 }
7304
7305 /// Snapshot of all registered view definitions.
7306 pub fn views(&self) -> Vec<ViewDef> {
7307 self.view_store.views().cloned().collect()
7308 }
7309
7310 // -----------------------------------------------------------------------
7311 // Full-text-lite API
7312 // -----------------------------------------------------------------------
7313
7314 /// Enable full-text indexing for all nodes of `label` on property `field`.
7315 ///
7316 /// After this call, every subsequent write to `(label, field)` is reflected
7317 /// in the index incrementally. Existing nodes are backfilled immediately.
7318 /// The declaration is persisted as a WAL record; the index itself is rebuilt
7319 /// from scratch on re-open (no snapshot format changes).
7320 ///
7321 /// # Errors
7322 /// - [`GraphError::ReadOnly`]: called on an as-of instance.
7323 /// - [`GraphError::RuleInvalid`]: `(label, field)` is already indexed.
7324 pub fn enable_fulltext(&mut self, label: &str, field: &str) -> Result<()> {
7325 if self.read_only {
7326 return Err(GraphError::ReadOnly);
7327 }
7328 if self.fulltext.is_enabled(label, field) {
7329 return Err(GraphError::RuleInvalid {
7330 detail: format!("full-text index for ({label:?}, {field:?}) already enabled"),
7331 });
7332 }
7333 self.log_then_apply(WalRecord::EnableFulltext {
7334 label: label.into(),
7335 field: field.into(),
7336 })
7337 }
7338
7339 /// Disable full-text indexing for `(label, field)` and drop its postings.
7340 ///
7341 /// # Errors
7342 /// - [`GraphError::ReadOnly`]: called on an as-of instance.
7343 /// - [`GraphError::RuleNotFound`]: `(label, field)` is not currently indexed.
7344 pub fn disable_fulltext(&mut self, label: &str, field: &str) -> Result<()> {
7345 if self.read_only {
7346 return Err(GraphError::ReadOnly);
7347 }
7348 if !self.fulltext.is_enabled(label, field) {
7349 return Err(GraphError::RuleNotFound {
7350 name: format!("fulltext({label},{field})"),
7351 });
7352 }
7353 self.log_then_apply(WalRecord::DisableFulltext {
7354 label: label.into(),
7355 field: field.into(),
7356 })
7357 }
7358
7359 /// Whether `(label, field)` is currently indexed for full-text search.
7360 pub fn is_fulltext_enabled(&self, label: &str, field: &str) -> bool {
7361 self.fulltext.is_enabled(label, field)
7362 }
7363
7364 /// Every `(label, field)` pair with a live full-text index, sorted.
7365 ///
7366 /// Note that [`GraphDb::search`] is keyed by field alone — a pair only
7367 /// declares which nodes are *indexed*, so callers that want to search
7368 /// everything indexed should query each distinct field once.
7369 pub fn fulltext_pairs(&self) -> Vec<(String, String)> {
7370 let mut v: Vec<(String, String)> = self.fulltext.enabled_pairs().cloned().collect();
7371 v.sort();
7372 v
7373 }
7374
7375 /// Enable an equality index for all nodes of `label` on scalar property
7376 /// `field`. Subsequent `WHERE n.field = value` lookups become O(matches)
7377 /// instead of an O(N_label) scan. Existing nodes are backfilled; the
7378 /// declaration persists via WAL and the postings rebuild on re-open.
7379 ///
7380 /// # Errors
7381 /// - [`GraphError::ReadOnly`]: called on an as-of instance.
7382 /// - [`GraphError::RuleInvalid`]: `(label, field)` is already indexed.
7383 pub fn enable_index(&mut self, label: &str, field: &str) -> Result<()> {
7384 if self.read_only {
7385 return Err(GraphError::ReadOnly);
7386 }
7387 if self.prop_index.is_enabled(label, field) {
7388 return Err(GraphError::RuleInvalid {
7389 detail: format!("property index for ({label:?}, {field:?}) already enabled"),
7390 });
7391 }
7392 self.log_then_apply(WalRecord::EnableIndex {
7393 label: label.into(),
7394 field: field.into(),
7395 })
7396 }
7397
7398 /// Disable the equality index for `(label, field)` and drop its postings.
7399 ///
7400 /// # Errors
7401 /// - [`GraphError::ReadOnly`]: called on an as-of instance.
7402 /// - [`GraphError::RuleNotFound`]: `(label, field)` is not currently indexed.
7403 pub fn disable_index(&mut self, label: &str, field: &str) -> Result<()> {
7404 if self.read_only {
7405 return Err(GraphError::ReadOnly);
7406 }
7407 if !self.prop_index.is_enabled(label, field) {
7408 return Err(GraphError::RuleNotFound {
7409 name: format!("index({label},{field})"),
7410 });
7411 }
7412 self.log_then_apply(WalRecord::DisableIndex {
7413 label: label.into(),
7414 field: field.into(),
7415 })
7416 }
7417
7418 /// Whether `(label, field)` currently has an equality index.
7419 pub fn is_index_enabled(&self, label: &str, field: &str) -> bool {
7420 self.prop_index.is_enabled(label, field)
7421 }
7422
7423 /// Start recording insert-count multiplicity on this store (§5.13).
7424 ///
7425 /// Adjacency stays a set and nothing about an existing read changes: a
7426 /// duplicate [`insert_edge`](Self::insert_edge) still returns `Ok(false)`
7427 /// and still leaves [`degree`](Self::degree) alone. What it gains is that
7428 /// the duplicate is *counted*, as the reserved edge property
7429 /// [`EDGE_COUNT_PROP`], readable through
7430 /// [`degree_multiplicity`](Self::degree_multiplicity).
7431 ///
7432 /// # This is a one-way step, and that is why it is a call
7433 ///
7434 /// The count is durable, so it is written to the WAL — as discriminant 23,
7435 /// which no release before v0.6.10 knows. A reader meeting an unknown WAL
7436 /// discriminant cannot know what the record would have changed, so it
7437 /// cannot degrade the way an unreadable index blob can. **After this call
7438 /// the store can no longer be read by an older binary, and there is no call
7439 /// that undoes it.** Gating the record behind this method is what keeps
7440 /// that step a decision an operator makes when they want the feature,
7441 /// rather than one everybody takes by upgrading.
7442 ///
7443 /// # It fails loudly, and that costs a snapshot
7444 ///
7445 /// An older binary does not refuse discriminant 23 — it truncates the WAL
7446 /// at it and, with `repair_wal`, persists the truncation. So this call also
7447 /// writes a **V10 snapshot**, a version no earlier release knows, and it
7448 /// writes it *first*: the snapshot is read before the WAL, so an older
7449 /// binary stops at `snapshot: unsupported version 10` with the WAL
7450 /// untouched. Taking the snapshot before appending the record is what makes
7451 /// the guard unconditional — the store is never, at any interruption point,
7452 /// carrying the record without the stamp that announces it.
7453 ///
7454 /// The snapshot keeps the WAL (`keep_wal: true`): opting in is not a
7455 /// compaction, and history reachable by [`open_at`](Self::open_at) stays
7456 /// reachable. On a large store the call therefore costs one full snapshot
7457 /// write.
7458 ///
7459 /// # What it costs a store that archives
7460 ///
7461 /// Writing `snapshot.bin` is also how the archive path decides whether the
7462 /// store may have a *genesis chain* — whether `open_at` can replay
7463 /// archive-resident commits from empty state. The rule is conservative: a
7464 /// snapshot that was already on disk might have been a truncating one, and
7465 /// once the handle that took it is gone this binary cannot tell. A
7466 /// `keep_wal` snapshot taken by **this** handle is the case where it can, so
7467 /// opting in and then archiving **in the same session** keeps the chain.
7468 ///
7469 /// Opting in, closing the store, and archiving in a *later* session does
7470 /// not — but that is the answer any store with a prior snapshot gets, not
7471 /// something this call causes. A store that wants the chain should take its
7472 /// first archive in the session that opted in.
7473 ///
7474 /// Calling it on a store that has already opted in writes nothing and
7475 /// returns `Ok(())`: an operator should not have to ask first.
7476 ///
7477 /// # This call is not atomic, and an `Err` does not undo it
7478 ///
7479 /// There is no rollback here, and there never was one. An `Err` means this
7480 /// handle stopped believing the store is opted in — `self.multiplicity` is
7481 /// reset, so this handle reports `false` from then on — and nothing more. It
7482 /// says nothing about what reached disk. Two reachable failures leave the
7483 /// opt-in standing:
7484 ///
7485 /// * **The declaration landed and only its fsync failed.** `log_then_apply`
7486 /// appends, then syncs; a failed barrier leaves `MULTIPLICITY_ENABLED`
7487 /// already in `wal.bin`. The next open replays it and the store is opted
7488 /// in. No archive is involved — this one predates the recovery below.
7489 /// * **The declaration never landed, but the V10 snapshot did, on a store
7490 /// that already had an archive.** The open-time recovery in
7491 /// `load_from_disk` reads V10-beside-an-archive as an interrupted archive
7492 /// sequence and opts the store in.
7493 ///
7494 /// So a failed call may leave the opt-in on disk immediately (the first
7495 /// case) or conjure it at the next open (the second), and nothing puts the
7496 /// store back out. Treat `Err` as "the outcome is unknown", not as "nothing
7497 /// happened".
7498 ///
7499 /// **This is safe, and the ordering is the reason.** The V10 stamp is
7500 /// written *before* the declaration, so every one of these intermediate
7501 /// states is one an older binary refuses by name rather than truncates at.
7502 /// The failure direction costs a refusal, never a commit. That ordering is
7503 /// the property worth protecting, not the atomicity this call never had.
7504 ///
7505 /// **To know where the store stands, ask the store.** Reopen it and call
7506 /// [`is_multiplicity_enabled`](Self::is_multiplicity_enabled); that is the
7507 /// only answer that accounts for what reached disk.
7508 ///
7509 /// The one case that really does leave the store opted out is a failure with
7510 /// no archive present and no record written: a stray V10 snapshot remains,
7511 /// costing an older reader a refusal it did not strictly need, and *that*
7512 /// store's next snapshot rewrites at V9.
7513 ///
7514 /// # Errors
7515 /// - [`GraphError::ReadOnly`]: called on an as-of instance.
7516 /// - Anything [`snapshot_with`](Self::snapshot_with) can return, and
7517 /// anything the WAL append or its fsync can return. See the atomicity
7518 /// section above for what the store is left holding.
7519 pub fn enable_multiplicity(&mut self) -> Result<()> {
7520 if self.read_only {
7521 return Err(GraphError::ReadOnly);
7522 }
7523 if self.multiplicity {
7524 return Ok(());
7525 }
7526 // The snapshot goes first, and the order is the guard.
7527 //
7528 // An older reader refuses a V10 snapshot by name and stops; it does not
7529 // refuse discriminant 23, it truncates the WAL at it. So the store must
7530 // never hold the record without the snapshot that announces it — not
7531 // even for the width of one fsync. Writing the snapshot before the
7532 // record makes the only reachable intermediate state "V10 snapshot, no
7533 // record", which is merely conservative: this binary reads it as a
7534 // store that has not opted in, and an older one refuses it.
7535 //
7536 // `keep_wal: true` because opting in is not a compaction: an operator
7537 // asking for multiplicity has not asked to lose the history `open_at`
7538 // can reach.
7539 self.multiplicity = true;
7540 let forced = self
7541 .snapshot_with(SnapshotOptions {
7542 keep_wal: true,
7543 ..SnapshotOptions::default()
7544 })
7545 .and_then(|()| self.log_then_apply(core_storage::wal::MULTIPLICITY_ENABLED));
7546 if forced.is_err() {
7547 // This handle stops believing it is opted in. That is all this line
7548 // does — it is not a rollback, and cannot be one: the declaration
7549 // may already be in `wal.bin` (the append succeeded and only the
7550 // fsync failed), and even when it is not, the V10 snapshot beside an
7551 // existing archive is enough for the open-time recovery to opt the
7552 // store in. See the "not atomic" section on this method.
7553 //
7554 // It fails in the safe direction either way: the V10 stamp reached
7555 // disk before anything a v0.6.9 reader would truncate at, so the
7556 // worst an interruption costs that reader is a refusal by name.
7557 self.multiplicity = false;
7558 }
7559 forced
7560 }
7561
7562 /// Whether this store records insert-count multiplicity.
7563 ///
7564 /// `false` on every store that has not called
7565 /// [`enable_multiplicity`](Self::enable_multiplicity) — which is every
7566 /// store that did not ask for it, including one upgraded from an earlier
7567 /// release.
7568 pub fn is_multiplicity_enabled(&self) -> bool {
7569 self.multiplicity
7570 }
7571
7572 /// How many times `(etype, src, dst)` has been inserted: the reserved
7573 /// `count` edge property, or 1 when it is absent.
7574 ///
7575 /// Answers 1 for a pair on a store that never opted in, which is the truth
7576 /// available there — the pair was inserted at least once, and the store
7577 /// kept no record of any second insert.
7578 fn edge_insert_count(&self, etype: u32, src: u32, dst: u32) -> u64 {
7579 match self.edge_props_view().get(etype, src, dst, EDGE_COUNT_PROP) {
7580 Some(Value::Int(n)) if n > 0 => n as u64,
7581 _ => 1,
7582 }
7583 }
7584
7585 /// The `SetEdgeCount` record a duplicate insert of `(edge_type, src_key,
7586 /// dst_key)` should log, or `None` when nothing should be written.
7587 ///
7588 /// `None` when the store has not opted in, so **no discriminant-23 record
7589 /// is written at all** — the gate the whole feature rests on.
7590 ///
7591 /// The other two `None`s are unreachable from the one caller. This is the
7592 /// single-mutation path, where `prepare_insert_edge` has already refused a
7593 /// missing endpoint and an existing pair's edge type is necessarily
7594 /// interned. A batch is the case where a pair's endpoints and type can all
7595 /// be created by the same frame, and it does not come through here: it
7596 /// queues a [`PlannedRec::DuplicateCount`] and names the count in the dense
7597 /// rewrite, which is the only pass that knows the frame's own ids.
7598 fn edge_count_record(
7599 &self,
7600 edge_type: &str,
7601 src_key: &str,
7602 dst_key: &str,
7603 ) -> Option<WalRecord> {
7604 if !self.multiplicity {
7605 return None;
7606 }
7607 let etype = self.syms.get(edge_type)?;
7608 let src = self.ids.get(src_key)?;
7609 let dst = self.ids.get(dst_key)?;
7610 Some(WalRecord::SetEdgeCount {
7611 etype,
7612 src,
7613 dst,
7614 count: self.edge_insert_count(etype, src, dst).saturating_add(1),
7615 })
7616 }
7617
7618 /// Search a full-text-indexed field.
7619 ///
7620 /// Returns `(node_key, match_count)` pairs sorted by match_count descending,
7621 /// ties broken by key (lexicographic). Tombstoned nodes are excluded.
7622 ///
7623 /// **Query syntax:**
7624 /// - Space-separated terms are AND'd: `"foo bar"` requires both.
7625 /// - `OR` between terms forms disjunction: `"foo OR bar"` matches either.
7626 /// - Trailing `*` on a term is a prefix match: `"rust*"` matches `rustlang`, `rusty`.
7627 /// - `AND` keyword is accepted explicitly and is the default.
7628 /// - Tokenization is unicode-alphanumeric (same as index time); case-insensitive.
7629 ///
7630 /// **Unindexed field:** returns `Ok(vec![])` if `field` is not indexed.
7631 /// Pin: this is the documented, tested, stable behavior for v1.
7632 ///
7633 /// **Memory / performance:** O(postings) lookup; no scan. The index is
7634 /// in-memory and proportional to total indexed text across all enabled fields.
7635 ///
7636 /// **v2 grammar:** supports `"phrase"`, `-negation`, `prefix*`, `OR`, `AND`.
7637 /// Results are BM25-scored (k1=1.2, b=0.75) and sorted by score descending,
7638 /// key ascending for deterministic tiebreaking.
7639 pub fn search(&self, field: &str, query: &str) -> Vec<(String, f64)> {
7640 // Resolve node_ids to keys (excluding tombstones) then re-sort by
7641 // (score DESC, key ASC) to give a deterministic, key-lexicographic
7642 // tiebreak. FulltextIndex::search sorts by (score DESC, node_id ASC)
7643 // which diverges from key order when nodes were not inserted in key-lex order.
7644 let mut results: Vec<(String, f64)> = self
7645 .fulltext
7646 .search(field, query, 0)
7647 .into_iter()
7648 .filter_map(|(id, score)| self.ids.key_of(id).map(|key| (key.to_string(), score)))
7649 .collect();
7650 results.sort_by(|a, b| {
7651 b.1.partial_cmp(&a.1)
7652 .unwrap_or(std::cmp::Ordering::Equal)
7653 .then(a.0.cmp(&b.0))
7654 });
7655 results
7656 }
7657
7658 /// [`search`](Self::search), stopping at the `k` best hits.
7659 ///
7660 /// Same ranking and the same deterministic tiebreak, but the index drops
7661 /// everything past `k` before any key is resolved, so a caller that wants
7662 /// the top few out of a field that matched thousands does not pay to
7663 /// materialise and re-sort the tail. `k == 0` means no limit, exactly as
7664 /// [`search`](Self::search) behaves.
7665 ///
7666 /// The BM25 scoring itself is not bounded by `k` — every candidate is
7667 /// scored either way — so this trims the resolve and the sort, not the
7668 /// search.
7669 pub fn search_top(&self, field: &str, query: &str, k: usize) -> Vec<(String, f64)> {
7670 // A tombstoned id resolves to nothing, so asking the index for exactly
7671 // `k` could return fewer. Over-fetching a little and truncating after
7672 // the filter keeps the count right without unbounding the call.
7673 let want = if k == 0 { 0 } else { k.saturating_mul(2) };
7674 let mut results: Vec<(String, f64)> = self
7675 .fulltext
7676 .search(field, query, want)
7677 .into_iter()
7678 .filter_map(|(id, score)| self.ids.key_of(id).map(|key| (key.to_string(), score)))
7679 .collect();
7680 results.sort_by(|a, b| {
7681 b.1.partial_cmp(&a.1)
7682 .unwrap_or(std::cmp::Ordering::Equal)
7683 .then(a.0.cmp(&b.0))
7684 });
7685 if k > 0 {
7686 results.truncate(k);
7687 }
7688 results
7689 }
7690
7691 /// Hybrid search: Reciprocal Rank Fusion (RRF) over fulltext + vector results.
7692 ///
7693 /// Takes up to `4*k` fulltext hits for `(text_field, query_text)` and up to
7694 /// `4*k` vector hits for `(vector_field, query_vec, min=0.0)`, then fuses
7695 /// them with RRF using a fixed constant of 60.
7696 ///
7697 /// ```text
7698 /// score(d) = Σ 1 / (60 + rank_i(d)) (rank 1-based per list)
7699 /// ```
7700 ///
7701 /// Returns the top `k` nodes by fused score, ties broken by node key
7702 /// ascending (deterministic).
7703 ///
7704 /// # Vector leg fallback
7705 ///
7706 /// When `query_vec` is empty the vector leg is skipped entirely and
7707 /// results are ranked by the text list alone through the same RRF path
7708 /// (each text result scores `1/(60 + rank)` from that single list).
7709 ///
7710 /// When `label` is `None`, the vector leg **always** returns empty results.
7711 /// Internally `label` is mapped to `""`, which does not match any rule-created
7712 /// HNSW index (all such indexes are keyed to a specific non-empty label), and
7713 /// the brute-force fallback finds no nodes with an empty label. The fused
7714 /// ranking is therefore text-only in this case.
7715 pub fn search_hybrid(
7716 &self,
7717 text_field: &str,
7718 query_text: &str,
7719 vector_field: &str,
7720 query_vec: &[f64],
7721 label: Option<&str>,
7722 k: usize,
7723 ) -> Vec<(String, f64)> {
7724 self.search_hybrid_inner(
7725 text_field,
7726 query_text,
7727 vector_field,
7728 query_vec,
7729 label,
7730 k,
7731 None,
7732 )
7733 }
7734
7735 /// [`search_hybrid`](Self::search_hybrid) with **each leg** filtered to the
7736 /// mask before the fusion.
7737 ///
7738 /// Filtering the fused list afterwards would quietly return fewer than `k`.
7739 /// Each leg over-fetches `4*k` candidates, so when the visible nodes rank
7740 /// below `4*k` hidden ones neither leg carries them into the fusion at all
7741 /// and the post-filter has nothing left to keep. Filtering first spends the
7742 /// `4*k` on **visible** hits, so a scoped call is as long as the corpus it
7743 /// can see allows.
7744 ///
7745 /// The ranks that enter RRF are therefore the ranks of the visible corpus,
7746 /// not the visible entries of the store-wide ranking. The constant stays 60
7747 /// and the tiebreak stays key-ascending.
7748 #[allow(clippy::too_many_arguments)]
7749 pub fn search_hybrid_scoped(
7750 &self,
7751 text_field: &str,
7752 query_text: &str,
7753 vector_field: &str,
7754 query_vec: &[f64],
7755 label: Option<&str>,
7756 k: usize,
7757 mask: &crate::mask::NodeMask,
7758 ) -> Vec<(String, f64)> {
7759 self.search_hybrid_inner(
7760 text_field,
7761 query_text,
7762 vector_field,
7763 query_vec,
7764 label,
7765 k,
7766 Some(mask),
7767 )
7768 }
7769
7770 /// The body shared by [`search_hybrid`](Self::search_hybrid) and
7771 /// [`search_hybrid_scoped`](Self::search_hybrid_scoped). `mask = None` is
7772 /// the unscoped contract unchanged: the filter below is then a no-op and
7773 /// the vector leg is the same unmasked call it has always been.
7774 #[allow(clippy::too_many_arguments)]
7775 fn search_hybrid_inner(
7776 &self,
7777 text_field: &str,
7778 query_text: &str,
7779 vector_field: &str,
7780 query_vec: &[f64],
7781 label: Option<&str>,
7782 k: usize,
7783 mask: Option<&crate::mask::NodeMask>,
7784 ) -> Vec<(String, f64)> {
7785 use std::collections::HashMap;
7786
7787 const RRF_K: f64 = 60.0;
7788 let pool = 4 * k;
7789
7790 // Accumulate per-node RRF scores.
7791 let mut scores: HashMap<String, f64> = HashMap::new();
7792
7793 // Text leg. The mask bites on the candidates, before `take(pool)`, so
7794 // the over-fetch is a budget of visible hits rather than one a hidden
7795 // prefix can exhaust.
7796 let text_hits = self.search(text_field, query_text);
7797 let visible_text = text_hits
7798 .into_iter()
7799 .filter(|(key, _count)| mask.is_none_or(|m| m.contains_node(self, key)));
7800 for (rank0, (key, _count)) in visible_text.take(pool).enumerate() {
7801 let rank = (rank0 + 1) as f64;
7802 *scores.entry(key).or_insert(0.0) += 1.0 / (RRF_K + rank);
7803 }
7804
7805 // Vector leg (skipped when query_vec is empty). The masked variant
7806 // applies the mask before its own k-truncation, for the same reason.
7807 if !query_vec.is_empty() {
7808 // `ExactnessCaller::Hybrid`: the leg is the same one
7809 // `find_similar_vector_masked` runs, but the advice its warning
7810 // gives has to fit *this* signature, which has no `exact`.
7811 let vec_hits = self
7812 .find_similar_vector_as(
7813 vector_field,
7814 label,
7815 query_vec,
7816 pool,
7817 0.0,
7818 mask,
7819 None,
7820 false,
7821 ExactnessCaller::Hybrid,
7822 )
7823 .expect("find_similar_vector_as is infallible without where_");
7824 for (rank0, (key, _sim)) in vec_hits.into_iter().enumerate() {
7825 let rank = (rank0 + 1) as f64;
7826 *scores.entry(key).or_insert(0.0) += 1.0 / (RRF_K + rank);
7827 }
7828 }
7829
7830 // Sort: score DESC, then key ASC for deterministic tie-breaking.
7831 let mut ranked: Vec<(String, f64)> = scores.into_iter().collect();
7832 ranked.sort_by(|a, b| {
7833 b.1.partial_cmp(&a.1)
7834 .unwrap_or(std::cmp::Ordering::Equal)
7835 .then(a.0.cmp(&b.0))
7836 });
7837 ranked.truncate(k);
7838 ranked
7839 }
7840
7841 /// For DST/testing: scratch BM25 search over live nodes without the index.
7842 /// Walks every live node, re-stems field tokens, computes corpus stats, and
7843 /// returns BM25-ranked results.
7844 ///
7845 /// The oracle: the ordered key list of `search(field, q)` must equal that of
7846 /// `scratch_search(field, q)` at every quiescent state.
7847 #[doc(hidden)]
7848 pub fn scratch_search(&self, field: &str, query: &str) -> Vec<(String, f64)> {
7849 use core_storage::fulltext::{parse_query, value_tokens_stemmed_with_positions};
7850 use std::collections::BTreeMap;
7851
7852 let groups = parse_query(query);
7853 if groups.is_empty() {
7854 return vec![];
7855 }
7856
7857 // --- Pass 1: collect all live indexed nodes with stemmed token data ---
7858 struct NodeData {
7859 key: String,
7860 /// stemmed_token → positions (sorted)
7861 tokens: BTreeMap<String, Vec<u32>>,
7862 dl: u32,
7863 }
7864
7865 let mut nodes: Vec<NodeData> = Vec::new();
7866 for id in 0..self.ids.len() as u32 {
7867 let Some(key) = self.ids.key_of(id) else {
7868 continue;
7869 };
7870 let Some(&sym) = self.labels.get(id as usize) else {
7871 continue;
7872 };
7873 if sym == u32::MAX {
7874 continue;
7875 }
7876 let label = match self.syms.resolve(sym) {
7877 Some(l) => l,
7878 None => continue,
7879 };
7880 if !self.fulltext.is_enabled(label, field) {
7881 continue;
7882 }
7883 let Some(value) = self.props_view().get(id, field).map(|vr| vr.into_value()) else {
7884 continue;
7885 };
7886 // Use value_tokens_stemmed_with_positions so list elements are
7887 // separated by POSITION_GAP — identical to the index path, which
7888 // prevents phrase queries from matching across element boundaries.
7889 let stemmed_with_pos = match &value {
7890 Value::Str(_) | Value::List(_) => value_tokens_stemmed_with_positions(&value),
7891 _ => continue,
7892 };
7893 let dl = stemmed_with_pos.len() as u32;
7894 let mut tok_map: BTreeMap<String, Vec<u32>> = BTreeMap::new();
7895 for (tok, pos) in stemmed_with_pos {
7896 tok_map.entry(tok).or_default().push(pos);
7897 }
7898 nodes.push(NodeData {
7899 key: key.to_string(),
7900 tokens: tok_map,
7901 dl,
7902 });
7903 }
7904
7905 if nodes.is_empty() {
7906 return vec![];
7907 }
7908
7909 // --- BM25 corpus stats ---
7910 let n = nodes.len() as f64;
7911 let avg_dl: f64 = nodes.iter().map(|nd| nd.dl as f64).sum::<f64>() / n;
7912 // df per stemmed token across all live indexed nodes.
7913 let mut df_map: BTreeMap<&str, f64> = BTreeMap::new();
7914 for nd in &nodes {
7915 for tok in nd.tokens.keys() {
7916 *df_map.entry(tok.as_str()).or_insert(0.0) += 1.0;
7917 }
7918 }
7919
7920 const K1: f64 = 1.2;
7921 const B: f64 = 0.75;
7922
7923 // --- Pass 2: score each node against each OR-group ---
7924 let mut results: Vec<(String, f64)> = Vec::new();
7925 for nd in &nodes {
7926 let dl = nd.dl as f64;
7927 let mut total_score = 0.0f64;
7928
7929 'group: for group in &groups {
7930 let mut group_score = 0.0f64;
7931
7932 for term in group {
7933 if term.negated {
7934 // Negated: if doc has this stemmed token → group fails.
7935 let present = if term.prefix {
7936 nd.tokens.keys().any(|t| t.starts_with(term.token.as_str()))
7937 } else {
7938 nd.tokens.contains_key(term.token.as_str())
7939 };
7940 if present {
7941 continue 'group;
7942 }
7943 continue;
7944 }
7945 if term.prefix {
7946 // Prefix: sum BM25 for all matching stemmed tokens.
7947 let mut prefix_matched = false;
7948 for (tok, positions) in &nd.tokens {
7949 if tok.starts_with(term.token.as_str()) {
7950 let tf = positions.len() as f64;
7951 let df = df_map.get(tok.as_str()).copied().unwrap_or(1.0);
7952 let idf = ((n - df + 0.5) / (df + 0.5) + 1.0).ln();
7953 let tf_norm =
7954 tf * (K1 + 1.0) / (tf + K1 * (1.0 - B + B * dl / avg_dl));
7955 group_score += idf * tf_norm;
7956 prefix_matched = true;
7957 }
7958 }
7959 if !prefix_matched {
7960 continue 'group;
7961 }
7962 } else {
7963 // term.token is already stemmed by parse_query; use directly.
7964 match nd.tokens.get(term.token.as_str()) {
7965 None => continue 'group,
7966 Some(positions) => {
7967 let tf = positions.len() as f64;
7968 let df = df_map.get(term.token.as_str()).copied().unwrap_or(1.0);
7969 let idf = ((n - df + 0.5) / (df + 0.5) + 1.0).ln();
7970 let tf_norm =
7971 tf * (K1 + 1.0) / (tf + K1 * (1.0 - B + B * dl / avg_dl));
7972 group_score += idf * tf_norm;
7973 }
7974 }
7975 }
7976 }
7977
7978 if group_score > 0.0 {
7979 total_score += group_score;
7980 }
7981 }
7982
7983 if total_score > 0.0 {
7984 results.push((nd.key.clone(), total_score));
7985 }
7986 }
7987
7988 results.sort_by(|a, b| {
7989 b.1.partial_cmp(&a.1)
7990 .unwrap_or(std::cmp::Ordering::Equal)
7991 .then(a.0.cmp(&b.0))
7992 });
7993 results
7994 }
7995
7996 /// Return the current view-maintained value of `view_prop` for node `key`.
7997 /// Equivalent to `get_prop` but documents that it reads a view-managed column.
7998 pub fn get_view_prop(&self, key: &str, view_prop: &str) -> Option<Value> {
7999 let id = self.ids.get(key)?;
8000 self.props_view()
8001 .get(id, view_prop)
8002 .map(|vr| vr.into_value())
8003 }
8004
8005 /// For testing / DST oracle: scratch recompute of a view value for one node.
8006 ///
8007 /// Returns `None` if the node does not exist, the view does not exist, or
8008 /// the view has no result for the node (e.g. Avg with no qualifying neighbors).
8009 #[doc(hidden)]
8010 pub fn scratch_view_value(&self, key: &str, view_name: &str) -> Option<Value> {
8011 let node = self.ids.get(key)?;
8012 let def = self.view_store.views().find(|v| v.name == view_name)?;
8013 // Use TopologyView so that NeighborAgg sees base + overlay edges
8014 // without materialising a temporary Topology (I1).
8015 let topo_view = self.topo_view();
8016 core_rules::views::compute_view_value(
8017 def,
8018 node,
8019 self.props_view(),
8020 &topo_view,
8021 &self.ids,
8022 &self.syms,
8023 &self.labels,
8024 )
8025 }
8026
8027 // -----------------------------------------------------------------------
8028 // Graph algorithm API
8029 // -----------------------------------------------------------------------
8030
8031 /// Run PageRank over the unified topology (manual + derived edges).
8032 ///
8033 /// Returns a [`PageRankReport`] with scores sorted descending (ties: key
8034 /// ascending). Set `config.edge_type` to restrict to one edge type.
8035 /// `config.converged` is `true` only when the power iteration converged
8036 /// within `config.max_iters` and within any time budget.
8037 pub fn pagerank(&self, config: &crate::algo::PageRankConfig) -> crate::algo::PageRankReport {
8038 let topo = build_topo_view(&self.topo, &self.base);
8039 let edge_props = self.edge_props_view();
8040 crate::algo::pagerank(
8041 &topo,
8042 &self.ids,
8043 &self.syms,
8044 &self.labels,
8045 &edge_props,
8046 config,
8047 )
8048 }
8049
8050 /// Weakly-connected components over the unified topology (treated as
8051 /// undirected regardless of how edges were inserted).
8052 ///
8053 /// Component IDs are the key of the smallest member in the component
8054 /// (deterministic). Result sorted by (component_id, key).
8055 pub fn connected_components(&self, config: &crate::algo::WccConfig) -> crate::algo::WccReport {
8056 let topo = build_topo_view(&self.topo, &self.base);
8057 let edge_props = self.edge_props_view();
8058 crate::algo::wcc(
8059 &topo,
8060 &self.ids,
8061 &self.syms,
8062 &self.labels,
8063 &edge_props,
8064 config,
8065 )
8066 }
8067
8068 /// Degree centrality for every live node.
8069 ///
8070 /// `direction`: `AlgoDir::Out` = out-degree, `AlgoDir::In` = in-degree,
8071 /// `AlgoDir::Both` = out + in (total directed degree).
8072 ///
8073 /// For one-shot ranking use this; for a live property updated on every
8074 /// write, create a Degree materialized view instead (see `docs/site/algorithms.md`).
8075 pub fn degree_centrality(
8076 &self,
8077 config: &crate::algo::DegreeConfig,
8078 ) -> crate::algo::DegreeReport {
8079 let topo = build_topo_view(&self.topo, &self.base);
8080 let edge_props = self.edge_props_view();
8081 crate::algo::degree_centrality(
8082 &topo,
8083 &self.ids,
8084 &self.syms,
8085 &self.labels,
8086 &edge_props,
8087 config,
8088 )
8089 }
8090
8091 /// Louvain community detection over the unified topology (undirected).
8092 ///
8093 /// See [`crate::algo::LouvainConfig`] for edge-type/weight/label
8094 /// restriction and [`crate::algo::CommunityReport`] for the shape of the
8095 /// result (communities sorted size-desc, then smallest member key asc).
8096 pub fn communities(&self, config: &crate::algo::LouvainConfig) -> crate::algo::CommunityReport {
8097 let topo = build_topo_view(&self.topo, &self.base);
8098 let edge_props = self.edge_props_view();
8099 crate::algo::louvain(
8100 &topo,
8101 &self.ids,
8102 &self.syms,
8103 &self.labels,
8104 &edge_props,
8105 config,
8106 )
8107 }
8108
8109 /// Write a vector of `(node_key, score)` pairs as `prop_name` on each node,
8110 /// atomically via a single write-batch (one WAL frame, one fsync).
8111 ///
8112 /// # Errors
8113 /// - [`GraphError::ReadOnly`]: called on an as-of instance.
8114 /// - [`GraphError::RuleInvalid`]: `prop_name` is managed by an existing view
8115 /// (collision check mirrors `create_view`).
8116 /// - [`GraphError::KeyNotFound`]: a key in `scores` does not exist as a live node.
8117 pub fn write_scores(&mut self, prop_name: &str, scores: &[(String, f64)]) -> Result<()> {
8118 if self.read_only {
8119 return Err(GraphError::ReadOnly);
8120 }
8121 // Collision check: refuse if prop_name is view-managed.
8122 if let Some(view_name) = self.view_store.view_for_prop(prop_name) {
8123 return Err(GraphError::RuleInvalid {
8124 detail: format!(
8125 "prop {:?} is managed by view {:?} and cannot be written as scores",
8126 prop_name, view_name
8127 ),
8128 });
8129 }
8130 // Refuse if prop_name is a view name itself (confusing namespace collision).
8131 if self.view_store.has_view(prop_name) {
8132 return Err(GraphError::RuleInvalid {
8133 detail: format!(
8134 "prop_name {:?} collides with an existing view name",
8135 prop_name
8136 ),
8137 });
8138 }
8139 // Write all scores in a single crash-atomic batch.
8140 self.write_batch(|b| {
8141 for (key, score) in scores {
8142 b.set_prop(key, prop_name, Value::Float(*score));
8143 }
8144 })?;
8145 Ok(())
8146 }
8147
8148 /// Return the value of `field` for the node with key `key`, or `None` if
8149 /// the node or field is absent. Reads through the overlay-over-base
8150 /// `ColumnsView`, materialising base values on demand (zero heap cost for
8151 /// overlay hits; one clone per base hit).
8152 pub fn get_prop(&self, key: &str, field: &str) -> Option<Value> {
8153 let id = self.ids.get(key)?;
8154 self.props_view().get(id, field).map(|vr| vr.into_value())
8155 }
8156
8157 pub fn has_node(&self, key: &str) -> bool {
8158 self.ids.get(key).is_some()
8159 }
8160
8161 /// Borrow the raw id map. Used by `NodeMask::from_keys` to resolve keys.
8162 pub(crate) fn ids(&self) -> &IdMap {
8163 &self.ids
8164 }
8165
8166 // -----------------------------------------------------------------------
8167 // Namespaces
8168 // -----------------------------------------------------------------------
8169
8170 /// The index `name` already has in `ns_names`, if any.
8171 fn ns_index_of(&self, name: &str) -> Option<u32> {
8172 self.ns_names
8173 .iter()
8174 .position(|n| n == name)
8175 .map(|i| i as u32)
8176 }
8177
8178 /// The index for `name`, appending it to `ns_names` when it is new.
8179 ///
8180 /// The table holds one entry per distinct namespace in the store — a
8181 /// tenant count, not a node count — so the linear scan is cheaper than a
8182 /// map and keeps `namespaces()` allocation-free of a second index.
8183 fn ns_index_for(&mut self, name: &str) -> u32 {
8184 match self.ns_index_of(name) {
8185 Some(i) => i,
8186 None => {
8187 self.ns_names.push(name.to_string());
8188 (self.ns_names.len() - 1) as u32
8189 }
8190 }
8191 }
8192
8193 /// The namespace name at `idx`, or [`NS_DEFAULT`] for an index this handle
8194 /// does not know (unreachable; the default is the narrowing answer).
8195 fn ns_name(&self, idx: u32) -> &str {
8196 self.ns_names
8197 .get(idx as usize)
8198 .map(String::as_str)
8199 .unwrap_or(NS_DEFAULT)
8200 }
8201
8202 /// The namespace index of dense node `id`, defaulting for an id with no
8203 /// entry (a node inserted before this handle rebuilt the array cannot
8204 /// exist: every insert path maintains it).
8205 fn node_ns_idx(&self, id: u32) -> u32 {
8206 self.node_ns
8207 .get(id as usize)
8208 .copied()
8209 .unwrap_or(NS_DEFAULT_IDX)
8210 }
8211
8212 /// File node `id` under namespace `name`, growing `node_ns` as `labels`
8213 /// grows. Called from `apply` for every node insert, live and replayed.
8214 fn set_node_ns(&mut self, id: u32, name: &str) {
8215 let idx = if name == NS_DEFAULT {
8216 NS_DEFAULT_IDX
8217 } else {
8218 self.ns_index_for(name)
8219 };
8220 if self.node_ns.len() <= id as usize {
8221 self.node_ns.resize(id as usize + 1, NS_DEFAULT_IDX);
8222 }
8223 self.node_ns[id as usize] = idx;
8224 }
8225
8226 /// Rebuild `node_ns` from the `ns` column — one pass, at the end of an
8227 /// open or a reload, after the snapshot is restored and the WAL replayed.
8228 ///
8229 /// A store with no `ns` column reads nothing: the column-name check fails
8230 /// and the vector is filled with one constant.
8231 fn rebuild_node_ns(&mut self) {
8232 let total = self.ids.len();
8233 self.ns_names.truncate(1);
8234 self.node_ns.clear();
8235 self.node_ns.resize(total, NS_DEFAULT_IDX);
8236 let has_ns_column = {
8237 let cv = self.props_view();
8238 cv.field_names().iter().any(|f| f == NS_PROP)
8239 };
8240 if !has_ns_column {
8241 return;
8242 }
8243 // Collected first so the props view is released before `ns_index_for`
8244 // takes `&mut self`.
8245 let named: Vec<(u32, String)> = {
8246 let cv = self.props_view();
8247 (0..total as u32)
8248 .filter_map(|id| match cv.get(id, NS_PROP).map(|vr| vr.into_value()) {
8249 Some(Value::Str(s)) if s != NS_DEFAULT => Some((id, s)),
8250 _ => None,
8251 })
8252 .collect()
8253 };
8254 for (id, name) in named {
8255 let idx = self.ns_index_for(&name);
8256 self.node_ns[id as usize] = idx;
8257 }
8258 }
8259
8260 /// Every namespace with at least one live node, in name order.
8261 ///
8262 /// `["default"]` on any store that has never named a namespace, including
8263 /// an empty one: a store is always at least its default namespace.
8264 pub fn namespaces(&self) -> Vec<String> {
8265 let mut out: BTreeSet<&str> = BTreeSet::new();
8266 out.insert(NS_DEFAULT);
8267 for (id, &idx) in self.node_ns.iter().enumerate() {
8268 if idx == NS_DEFAULT_IDX || !self.is_live_node(id as u32) {
8269 continue;
8270 }
8271 out.insert(self.ns_name(idx));
8272 }
8273 out.into_iter().map(str::to_string).collect()
8274 }
8275
8276 /// The namespace of `key`, or `None` when the key names no live node.
8277 pub fn namespace_of(&self, key: &str) -> Option<String> {
8278 let id = self.ids.get(key)?;
8279 if !self.is_live_node(id) {
8280 return None;
8281 }
8282 Some(self.ns_name(self.node_ns_idx(id)).to_string())
8283 }
8284
8285 /// Every live node in `namespace`, as a visibility mask.
8286 ///
8287 /// Built off `node_ns` on whichever handle this is, so on a temporal handle
8288 /// it is the namespace's membership at that commit. A name no node uses
8289 /// gives an empty mask — a namespace scope never widens.
8290 pub fn mask_for_namespace(&self, namespace: &str) -> crate::mask::NodeMask {
8291 let Some(idx) = self.ns_index_of(namespace) else {
8292 return crate::mask::NodeMask::from_ids(std::collections::HashSet::new());
8293 };
8294 let visible: std::collections::HashSet<u32> = (0..self.ids.len() as u32)
8295 .filter(|&id| self.node_ns_idx(id) == idx && self.is_live_node(id))
8296 .collect();
8297 crate::mask::NodeMask::from_ids(visible)
8298 }
8299
8300 /// Live-node test used by the namespace accessors: a deleted node keeps its
8301 /// dense id and its `node_ns` slot, and the label sentinel is what marks it
8302 /// gone — the same test `mask_for_role`'s label leg applies implicitly.
8303 fn is_live_node(&self, id: u32) -> bool {
8304 self.labels
8305 .get(id as usize)
8306 .is_some_and(|&sym| sym != u32::MAX)
8307 && self.ids.key_of(id).is_some()
8308 }
8309
8310 /// Per-namespace live node counts for [`Stats`], in name order.
8311 fn namespace_stats(&self) -> Vec<NamespaceStats> {
8312 let mut counts: BTreeMap<&str, usize> = BTreeMap::new();
8313 counts.insert(NS_DEFAULT, 0);
8314 for id in 0..self.ids.len() as u32 {
8315 if !self.is_live_node(id) {
8316 continue;
8317 }
8318 *counts
8319 .entry(self.ns_name(self.node_ns_idx(id)))
8320 .or_insert(0) += 1;
8321 }
8322 counts
8323 .into_iter()
8324 .filter(|&(name, n)| n > 0 || name == NS_DEFAULT)
8325 .map(|(name, nodes_live)| NamespaceStats {
8326 name: name.to_string(),
8327 nodes_live,
8328 })
8329 .collect()
8330 }
8331
8332 /// The namespace a create-class op would put its node in: the `ns` entry of
8333 /// the props it carries, normalised, with absent meaning [`NS_DEFAULT`].
8334 fn created_namespace<'a>(key: &str, props: &'a [(String, Value)]) -> Result<&'a str> {
8335 Ok(namespace_of_value(Self::sole_ns_entry(key, props)?))
8336 }
8337
8338 /// The one `ns` entry in a node's props, or `None` when it carries none.
8339 ///
8340 /// A props list naming `ns` twice is refused. Without that refusal the
8341 /// write path and the authorisation path can read the same list
8342 /// differently — one taking the first entry, the other the last — and
8343 /// `CREATE (n:L {ns: 'mine', ns: 'theirs'})` lands a node in a namespace
8344 /// the role was checked against the other of. One entry is the only shape
8345 /// where "the node's namespace" is a single fact, so it is the only shape
8346 /// accepted, and every reader of it agrees by construction.
8347 fn sole_ns_entry<'a>(key: &str, props: &'a [(String, Value)]) -> Result<Option<&'a Value>> {
8348 let mut found: Option<&'a Value> = None;
8349 for (field, value) in props {
8350 if field != NS_PROP {
8351 continue;
8352 }
8353 if found.is_some() {
8354 return Err(GraphError::RuleInvalid {
8355 detail: format!(
8356 "node {key}: {NS_PROP} is given more than once; a node has exactly \
8357 one namespace"
8358 ),
8359 });
8360 }
8361 found = Some(value);
8362 }
8363 Ok(found)
8364 }
8365
8366 /// The definition of the role a write authorisation names.
8367 ///
8368 /// `None` when `roles.json` was corrupt at open or the role has since been
8369 /// removed — neither can reach a write, because the authorisation carries a
8370 /// mask `mask_for_role` already resolved for that name.
8371 fn role_def_for(&self, role: &str) -> Option<&RoleDef> {
8372 self.roles.as_ref()?.iter().find(|r| r.name == role)
8373 }
8374
8375 /// Validate the `ns` entry of a node's props and drop an explicit default.
8376 ///
8377 /// Runs on the write path only (see `rewrite_wal_dense`), never on replay:
8378 /// a record that reached the WAL was already accepted here.
8379 fn normalise_insert_ns(
8380 key: &str,
8381 props: Vec<(String, Value)>,
8382 ) -> Result<(Vec<(String, Value)>, String)> {
8383 // One `ns` or none: this is where that is enforced, so every later
8384 // reader of the list — the authorisation gate, the two `apply` arms,
8385 // `node_ns` — is looking at a single entry and cannot disagree about
8386 // which one counts.
8387 Self::sole_ns_entry(key, &props)?;
8388 let mut name = NS_DEFAULT.to_string();
8389 let mut out = Vec::with_capacity(props.len());
8390 for (field, value) in props {
8391 if field != NS_PROP {
8392 out.push((field, value));
8393 continue;
8394 }
8395 let Value::Str(ref s) = value else {
8396 return Err(GraphError::RuleInvalid {
8397 detail: format!(
8398 "node {key}: {NS_PROP} must be a string naming a namespace, \
8399 got {value:?}"
8400 ),
8401 });
8402 };
8403 if !valid_namespace(s) {
8404 return Err(GraphError::RuleInvalid {
8405 detail: format!(
8406 "node {key}: {s:?} is not a valid namespace name — 1 to {NS_MAX_LEN} \
8407 characters of [A-Za-z0-9_.-]"
8408 ),
8409 });
8410 }
8411 name = s.clone();
8412 // An explicit default stores nothing, so a single-tenant store
8413 // never grows an `ns` column.
8414 if name != NS_DEFAULT {
8415 out.push((field, value));
8416 }
8417 }
8418 Ok((out, name))
8419 }
8420
8421 // -----------------------------------------------------------------------
8422 // RBAC role resolution
8423 // -----------------------------------------------------------------------
8424
8425 /// Parse `roles.json` bytes from `fs`.
8426 ///
8427 /// Return values:
8428 /// `Ok(Some(roles))` — file absent (returns `vec![]`) **or** file present
8429 /// and valid; in both cases `mask_for_role` uses the
8430 /// list normally (an absent file means no roles defined).
8431 /// `Ok(None)` — file present but corrupt or unrecognised version
8432 /// → poisoned state; `mask_for_role` returns `Err` for
8433 /// any role name until the file is fixed and the DB
8434 /// re-opened (or `apply_schema` is called to repair it).
8435 ///
8436 /// Note: `None` signals corruption, not absence — the opposite of what an
8437 /// optional "file missing" convention would suggest. The open path stores
8438 /// this result on `db.roles` directly.
8439 fn load_roles_from_fs(fs: &F) -> Result<Option<Vec<RoleDef>>> {
8440 let bytes = fs.read(FileId::Roles).map_err(GraphError::Io)?;
8441 if bytes.is_empty() {
8442 // Empty bytes means either the file is absent or zero-byte — both
8443 // are treated identically as "no roles defined". A zero-byte
8444 // roles.json does NOT widen access: an absent file and a zero-byte
8445 // file both resolve to an empty role list (sees nothing by default).
8446 return Ok(Some(vec![]));
8447 }
8448 match serde_json::from_slice::<RolesFile>(&bytes) {
8449 Ok(f) if matches!(f.version, 1..=4) => Ok(Some(f.roles)),
8450 // Corrupt or unrecognised version (>4): poison the roles state.
8451 // Never widen: a version this binary does not know may carry a
8452 // narrowing this binary would not apply.
8453 _ => Ok(None),
8454 }
8455 }
8456
8457 /// Resolve a role to a node-visibility mask against the current graph state.
8458 ///
8459 /// Returns `Err` when:
8460 /// - `roles.json` was present but corrupt at open (poisoned state), or
8461 /// - `role` does not match any defined role name.
8462 ///
8463 /// The mask union is: explicit `keys` (unknown keys silently ignored) plus
8464 /// all live nodes carrying any label in `labels` that also pass the role's
8465 /// [`visible_where`](crate::roles::RoleDef::visible_where) predicate, if it
8466 /// has one. Label resolution is live — new nodes of an allowed label are
8467 /// visible without re-applying the schema, and a property edited out of the
8468 /// predicate takes its node out of the mask on the next read. An empty
8469 /// union = empty mask = sees nothing.
8470 ///
8471 /// This is the one resolver every read path calls, live and as-of alike, so
8472 /// the predicate applies everywhere at once. On an as-of handle the role
8473 /// *definition* is the current one and the graph is the historical one: the
8474 /// predicate is evaluated against the property values at the commit being
8475 /// read.
8476 ///
8477 /// The result is memoised per `(role, commit_seq)`, so a scoped reader
8478 /// between two writes resolves the role once. See
8479 /// [`RoleMaskCache`](crate::mask::RoleMaskCache) for why that cannot go
8480 /// stale.
8481 pub fn mask_for_role(&self, role: &str) -> Result<crate::mask::NodeMask> {
8482 self.role_masks
8483 .get_or_build(role, self.commit_seq, || self.build_mask_for_role(role))
8484 .map(|m| (*m).clone())
8485 }
8486
8487 /// The mask an [`AsOfScope`] names, resolved against this handle.
8488 ///
8489 /// Shared by [`GraphDb::query_at_scoped`] and
8490 /// [`GraphDb::query_at_scoped_in_namespace`] so one scope resolves one way
8491 /// however the namespace leg is added.
8492 fn mask_at_scope(&self, scope: AsOfScope<'_>) -> Result<crate::mask::NodeMask> {
8493 // One resolver answers "what may this role see" — `mask_for_role` — and
8494 // it runs against this handle, so on a temporal one the answer is the
8495 // as-of one.
8496 Ok(match scope {
8497 AsOfScope::Role(role) => self.mask_for_role(role)?,
8498 AsOfScope::Keys(keys) => {
8499 crate::mask::NodeMask::from_keys(self, keys.iter().map(String::as_str))
8500 }
8501 AsOfScope::RoleAndKeys(role, keys) => {
8502 self.mask_for_role(role)?
8503 .intersect(&crate::mask::NodeMask::from_keys(
8504 self,
8505 keys.iter().map(String::as_str),
8506 ))
8507 }
8508 AsOfScope::Namespace(namespace) => self.mask_for_namespace(namespace),
8509 })
8510 }
8511
8512 /// Resolve `role` against the current graph, ignoring the memo.
8513 fn build_mask_for_role(&self, role: &str) -> Result<crate::mask::NodeMask> {
8514 let roles = self.roles.as_ref().ok_or_else(roles_poisoned)?;
8515 let def = roles
8516 .iter()
8517 .find(|r| r.name == role)
8518 .ok_or_else(|| GraphError::KeyNotFound {
8519 key: format!("role:{role}"),
8520 })?;
8521
8522 let mut visible = std::collections::HashSet::new();
8523
8524 // Key leg: resolve explicit keys to dense ids (unknown keys ignored).
8525 // An administrative grant, never narrowed by the predicate.
8526 for key in &def.keys {
8527 if let Some(id) = self.ids.get(key) {
8528 visible.insert(id);
8529 }
8530 }
8531
8532 // Label leg: live scan — iterate labels vec for matching symbol, and
8533 // when the role carries a predicate, test the property as well. The
8534 // property comes from the store's own merged view (overlay over the
8535 // mmap'd base), so an as-of handle reads the values of its own commit.
8536 let props = def.visible_where.as_ref().map(|_| self.props_view());
8537 for label_name in &def.labels {
8538 if let Some(sym) = self.syms.get(label_name) {
8539 for (i, &s) in self.labels.iter().enumerate() {
8540 if s != sym {
8541 continue;
8542 }
8543 let id = i as u32;
8544 match (&def.visible_where, &props) {
8545 (Some(pred), Some(view)) => {
8546 let value = view.get(id, &pred.field).map(|vr| vr.into_value());
8547 if pred.holds(value.as_ref()) {
8548 visible.insert(id);
8549 }
8550 }
8551 _ => {
8552 visible.insert(id);
8553 }
8554 }
8555 }
8556 }
8557 }
8558
8559 // Namespace leg: an intersection over the whole union, the key leg
8560 // included. A namespace is a tenancy boundary, so a key naming a node in
8561 // another tenant's namespace is not an administrative grant — and
8562 // `apply_schema` has already refused that role, so this only has to be
8563 // right about the node that moved into existence afterwards.
8564 if def.namespaces.is_some() {
8565 visible.retain(|&id| def.sees_namespace(self.ns_name(self.node_ns_idx(id))));
8566 }
8567
8568 Ok(crate::mask::NodeMask::from_ids(visible))
8569 }
8570
8571 /// Return the current list of role definitions.
8572 ///
8573 /// Returns an empty list when no roles are defined or when `roles.json`
8574 /// was corrupt at open (check [`mask_for_role`](Self::mask_for_role) for
8575 /// the fail-loud error in that case, or call
8576 /// [`roles_checked`](Self::roles_checked), which is this readout with that
8577 /// error in it).
8578 pub fn roles(&self) -> Vec<RoleDef> {
8579 self.roles.as_deref().unwrap_or(&[]).to_vec()
8580 }
8581
8582 /// The role definitions, or the poison error when `roles.json` was corrupt
8583 /// at open.
8584 ///
8585 /// [`roles`](Self::roles) answers `[]` both for a store that defines no
8586 /// roles and for one whose sidecar did not parse, and a caller validating a
8587 /// role name at boot cannot tell those apart. The wrong reading of the pair
8588 /// is the dangerous one: a store with no roles at all is an unrestricted
8589 /// store, so a poisoned file would read as "nothing is restricted here".
8590 ///
8591 /// This is the same answer, for the same cause, that
8592 /// [`mask_for_role`](Self::mask_for_role) gives on the first read.
8593 pub fn roles_checked(&self) -> Result<Vec<RoleDef>> {
8594 match self.roles.as_deref() {
8595 Some(roles) => Ok(roles.to_vec()),
8596 None => Err(roles_poisoned()),
8597 }
8598 }
8599
8600 // ── Role-scoped write authz ───────────────────────────────────────────────
8601
8602 /// Execute `ops` with optional role-scoped write authorization.
8603 ///
8604 /// - `None` → full authority, identical to [`write_batch`](Self::write_batch)
8605 /// (zero-cost bypass of all authz checks).
8606 /// - `Some(authz)` → the decision table is evaluated per-op BEFORE any WAL
8607 /// record is built. A denial returns an error with no WAL frame written
8608 /// (all-or-nothing at the authz boundary, then at the MutPreview boundary).
8609 ///
8610 /// See the plan's "authz decision table" section for the full semantics.
8611 pub fn write_batch_authz(
8612 &mut self,
8613 authz: Option<&WriteAuthz>,
8614 ops: Vec<BatchOp>,
8615 ) -> Result<(usize, usize)> {
8616 // Thread authz as a direct parameter — never touches pending_write_authz.
8617 self.commit_logged_batch(ops, None, authz.cloned())
8618 .map(inserted_pair)
8619 }
8620
8621 /// Execute a Cypher write statement with role-scoped write authorization.
8622 ///
8623 /// Resolves scope + mask from `self.roles` inside the call (same write-guard
8624 /// lifetime as execution, satisfying §5 lock discipline). The resolved
8625 /// `WriteAuthz` is stored as `pending_write_authz` for the duration of the
8626 /// call so that all inner `batch.commit()` calls are authz-checked.
8627 ///
8628 /// MERGE is handled specially: the MERGE scope precondition (§3.3) is
8629 /// checked in `exec_merge` BEFORE `has_node` to close the §6.2
8630 /// timing-oracle item (hidden ≡ absent for unscoped roles).
8631 ///
8632 /// Roles with `write: None` (v1 behavior) → `RoleWriteDenied` with
8633 /// "this endpoint is not permitted".
8634 pub fn query_write_authz(
8635 &mut self,
8636 role: &str,
8637 cypher: &str,
8638 params: &BTreeMap<String, Value>,
8639 ) -> Result<ResultSet> {
8640 // Resolve scope (fails fast if role has no write scope).
8641 // write:None → byte-identical v1 blanket-403 body (plan §v1-sidecar mandate).
8642 let scope =
8643 {
8644 let roles = self.roles.as_deref().ok_or_else(|| GraphError::Corrupt {
8645 detail: "roles.json was corrupt at open; re-open to restore role access".into(),
8646 })?;
8647 let def = roles.iter().find(|r| r.name == role).ok_or_else(|| {
8648 GraphError::KeyNotFound {
8649 key: format!("role:{role}"),
8650 }
8651 })?;
8652 def.write
8653 .clone()
8654 .ok_or_else(|| GraphError::RoleWriteDenied {
8655 reason: "role-bound token: writes are not permitted".into(),
8656 })?
8657 };
8658 // Resolve mask inside the call (same guard, §5 coherence).
8659 let mask = self.mask_for_role(role)?;
8660 self.pending_write_authz = Some(WriteAuthz {
8661 role: role.into(),
8662 scope,
8663 mask,
8664 });
8665 // RAII guard: always clears pending_write_authz on scope exit, including
8666 // on panic or early-return, mirroring the RestoreEmitDeltas precedent.
8667 struct ClearPendingAuthzOnDrop(*mut Option<WriteAuthz>);
8668 impl Drop for ClearPendingAuthzOnDrop {
8669 fn drop(&mut self) {
8670 // SAFETY: pointer into the owning GraphDb; guard is dropped
8671 // within this function's frame before it returns.
8672 unsafe { *self.0 = None };
8673 }
8674 }
8675 // SAFETY: raw pointer into self; guard dropped before this fn returns.
8676 let _authz_guard = ClearPendingAuthzOnDrop(&mut self.pending_write_authz as *mut _);
8677 let tokens = lex(cypher).map_err(|e| GraphError::QueryError {
8678 detail: format!("lex: {e}"),
8679 })?;
8680 let stmt = parse_write(&tokens).map_err(|e| GraphError::QueryError {
8681 detail: format!("parse: {e}"),
8682 })?;
8683 self.exec_write_stmt(stmt, params)
8684 }
8685
8686 /// Execute `ops` with optional role-scoped write authorization, suppressing
8687 /// fsync (for use inside the group-commit drain thread, which performs one
8688 /// group fsync after releasing the write lock).
8689 ///
8690 /// Identical to [`write_batch_authz`] except the fsync policy is temporarily
8691 /// forced to `Relaxed` for the duration of the call, matching the drain-thread
8692 /// contract established by [`commit_batch_nosync`].
8693 pub(crate) fn write_batch_authz_nosync(
8694 &mut self,
8695 authz: Option<&WriteAuthz>,
8696 ops: Vec<BatchOp>,
8697 ) -> Result<(usize, usize)> {
8698 let saved = self.fsync;
8699 struct RestoreFsync(*mut FsyncPolicy, FsyncPolicy);
8700 impl Drop for RestoreFsync {
8701 fn drop(&mut self) {
8702 // SAFETY: pointer into the owning GraphDb; guard is dropped
8703 // within the enclosing function's frame before it returns.
8704 unsafe { *self.0 = self.1 };
8705 }
8706 }
8707 // SAFETY: raw pointer into self; guard dropped before this fn returns.
8708 let _g = RestoreFsync(&mut self.fsync as *mut FsyncPolicy, saved);
8709 self.fsync = FsyncPolicy::Relaxed;
8710 self.commit_logged_batch(ops, None, authz.cloned())
8711 .map(inserted_pair)
8712 }
8713
8714 /// Execute a `/ingest` request with role-scoped write authorization.
8715 ///
8716 /// Resolves the role's `WriteScope` and `NodeMask` inside this call (same
8717 /// write-guard lifetime as the mutation, satisfying §5 lock discipline).
8718 /// Sets `pending_write_authz` for the duration of the call so that the
8719 /// `commit_ingest` → `commit_logged_batch` path picks up the authz context
8720 /// and evaluates the decision table per-op before any WAL write.
8721 ///
8722 /// §7.3: roles with empty `create_labels` will see every `InsertNode` op
8723 /// denied by the decision table with the appropriate §4.3 scope reason;
8724 /// no special HTTP-layer check is needed.
8725 ///
8726 /// Roles with `write: None` return `RoleWriteDenied` with
8727 /// "writes are not permitted" (byte-identical to v1 blanket 403).
8728 pub fn ingest_with_edges_authz(
8729 &mut self,
8730 role: &str,
8731 label: &str,
8732 rows: Vec<std::collections::BTreeMap<String, Value>>,
8733 opts: &crate::ingest::IngestOptions,
8734 edges: &[(String, String, String)],
8735 ) -> Result<crate::ingest::IngestReport> {
8736 // Resolve scope (fails fast if role has no write scope).
8737 // write:None → byte-identical v1 blanket-403 body (plan §v1-sidecar mandate).
8738 let scope =
8739 {
8740 let roles = self.roles.as_deref().ok_or_else(|| GraphError::Corrupt {
8741 detail: "roles.json was corrupt at open; re-open to restore role access".into(),
8742 })?;
8743 let def = roles.iter().find(|r| r.name == role).ok_or_else(|| {
8744 GraphError::KeyNotFound {
8745 key: format!("role:{role}"),
8746 }
8747 })?;
8748 def.write
8749 .clone()
8750 .ok_or_else(|| GraphError::RoleWriteDenied {
8751 reason: "role-bound token: writes are not permitted".into(),
8752 })?
8753 };
8754 let mask = self.mask_for_role(role)?;
8755 self.pending_write_authz = Some(WriteAuthz {
8756 role: role.into(),
8757 scope,
8758 mask,
8759 });
8760 // RAII guard: always clears pending_write_authz on scope exit, including
8761 // on panic or early-return, mirroring the RestoreEmitDeltas precedent.
8762 struct ClearPendingAuthzOnDrop(*mut Option<WriteAuthz>);
8763 impl Drop for ClearPendingAuthzOnDrop {
8764 fn drop(&mut self) {
8765 // SAFETY: pointer into the owning GraphDb; guard is dropped
8766 // within this function's frame before it returns.
8767 unsafe { *self.0 = None };
8768 }
8769 }
8770 // SAFETY: raw pointer into self; guard dropped before this fn returns.
8771 let _authz_guard = ClearPendingAuthzOnDrop(&mut self.pending_write_authz as *mut _);
8772 self.ingest_with_edges(label, rows, opts, edges)
8773 }
8774
8775 /// Evaluate the write-authz decision table for one `BatchOp`.
8776 ///
8777 /// Called by `commit_logged_batch` for each op when `pending_write_authz`
8778 /// is `Some`, BEFORE MutPreview. A denial returns an error immediately;
8779 /// the remaining ops are not evaluated and no WAL frame is written.
8780 ///
8781 /// `batch_created` carries the key→label pairs of nodes that earlier ops in
8782 /// THIS batch will create. Used by `InsertEdgeUpsert` to count same-batch
8783 /// placeholder nodes as visible (spec: "a placeholder endpoint the SAME
8784 /// batch creates counts as visible if its label passed the create-class gate").
8785 fn check_single_op_authz(
8786 &self,
8787 authz: &WriteAuthz,
8788 op: &BatchOp,
8789 batch_created: &BTreeMap<String, String>,
8790 ) -> Result<()> {
8791 // Helper: 3-way node status under the authz mask.
8792 //
8793 // Batch-created nodes (from earlier InsertNode in THIS batch) are treated
8794 // as Visible with their recorded label — their create gate already passed
8795 // and they are not yet in self.ids (not committed). This fixes the
8796 // MERGE+ON CREATE SET case where InsertNode + SetProp arrive together:
8797 // the SetProp must not see the node as Absent.
8798 let node_status = |key: &str| -> NodeAuthzStatus {
8799 if let Some(label) = batch_created.get(key) {
8800 return NodeAuthzStatus::Visible(label.clone());
8801 }
8802 match self.ids.get(key) {
8803 None => NodeAuthzStatus::Absent,
8804 Some(id) if !authz.mask.contains_id(id) => NodeAuthzStatus::Hidden,
8805 Some(id) => {
8806 let label = self
8807 .labels
8808 .get(id as usize)
8809 .and_then(|&sym| {
8810 if sym == u32::MAX {
8811 None
8812 } else {
8813 self.syms.resolve(sym).map(str::to_string)
8814 }
8815 })
8816 .unwrap_or_default();
8817 NodeAuthzStatus::Visible(label)
8818 }
8819 }
8820 };
8821
8822 // Helper: is an InsertEdgeUpsert endpoint visible?
8823 // A same-batch placeholder counts as visible if its label passed
8824 // the create-class gate (spec "upsert placeholder-counts-as-visible").
8825 let upsert_ep_visible = |ep_key: &str, placeholder_label: &str| -> bool {
8826 // In store and visible?
8827 if let Some(id) = self.ids.get(ep_key) {
8828 return authz.mask.contains_id(id);
8829 }
8830 // Created by an earlier op in this batch?
8831 if let Some(created_label) = batch_created.get(ep_key) {
8832 return authz.scope.create_labels.contains(created_label);
8833 }
8834 // Will be created by THIS InsertEdgeUpsert: placeholder_label
8835 // must pass the create-class gate.
8836 authz
8837 .scope
8838 .create_labels
8839 .contains(&placeholder_label.to_string())
8840 };
8841
8842 match op {
8843 // RenameNode / CreateRule / DeleteRule: defense-in-depth gate.
8844 // These ops are never routed to role-scoped paths by the HTTP layer,
8845 // but we 403 them here to close any future bypass route.
8846 //
8847 // InsertNodeOnConflict joins them: it is reachable only from the
8848 // embedded Python binding, which has no role token, and `Replace`
8849 // is a create and an update at once. Rather than split the decision
8850 // table for an op no role-scoped path constructs, refuse it — a
8851 // role-scoped caller writes through the ops that are already in the
8852 // table.
8853 BatchOp::RenameNode { .. }
8854 | BatchOp::CreateRule(_)
8855 | BatchOp::DeleteRule { .. }
8856 | BatchOp::InsertNodeOnConflict { .. } => {
8857 return Err(GraphError::RoleWriteDenied {
8858 reason: "role-bound token: this endpoint is not permitted".into(),
8859 });
8860 }
8861
8862 // ── CREATE-class: InsertNode ─────────────────────────────────────
8863 //
8864 // Decision table row 1 (scope-before-lookup): check label in
8865 // create_labels BEFORE any key lookup. This is the structural
8866 // closure of the §6.2 timing-oracle item — the denial fires even
8867 // when the store is EMPTY (see test_create_scope_denied_empty_store).
8868 BatchOp::InsertNode { label, key, props } => {
8869 if !authz.scope.create_labels.contains(label) {
8870 return Err(GraphError::RoleWriteDenied {
8871 reason: format!(
8872 "role-bound token: label '{}' not in write scope (create_labels)",
8873 label
8874 ),
8875 });
8876 }
8877 // A role bound to namespaces may only create inside them. The
8878 // never-widen rule is about what a write makes visible to *any*
8879 // party, not only to the writer: a node this role could never
8880 // read back is a write into somebody else's tenancy. Also a
8881 // scope check, so it runs before the key lookup — it discloses
8882 // nothing about the store. Covers Cypher `CREATE` and the node
8883 // `MERGE` creates, both of which arrive as this op.
8884 // Resolved before the role lookup so a props list naming `ns`
8885 // twice is refused for every role, scoped or not: it is the same
8886 // malformed write the seam refuses, and leaving it to the seam
8887 // would mean the gate had already read one of the two.
8888 let target = Self::created_namespace(key, props)?;
8889 if let Some(def) = self.role_def_for(&authz.role) {
8890 if !def.sees_namespace(target) {
8891 return Err(GraphError::RoleWriteDenied {
8892 reason: format!(
8893 "role-bound token: namespace '{target}' not in the role's \
8894 namespaces"
8895 ),
8896 });
8897 }
8898 }
8899 // Row 2/3: key lookup.
8900 match self.ids.get(key.as_str()) {
8901 Some(id) if authz.mask.contains_id(id) => {
8902 // Visible: DuplicateKey — let MutPreview handle this.
8903 }
8904 Some(_) => {
8905 // Hidden: indistinguishable from absent to the role.
8906 return Err(GraphError::RoleWriteDenied {
8907 reason: "role-bound token: target node not visible".into(),
8908 });
8909 }
8910 None => {
8911 // Absent: proceed (create).
8912 }
8913 }
8914 }
8915
8916 // ── UPDATE-class: SetProp, RemoveProp ────────────────────────────
8917 BatchOp::SetProp { key, .. } | BatchOp::RemoveProp { key, .. } => {
8918 if batch_created.contains_key(key.as_str()) {
8919 // Batch-created node: create gate already passed this batch.
8920 // Updating it in the same batch is always allowed, regardless
8921 // of update_labels (ruling §3.5: "writer just created it").
8922 } else {
8923 let label = match node_status(key) {
8924 NodeAuthzStatus::Visible(lbl) => lbl,
8925 _ => {
8926 return Err(GraphError::RoleWriteDenied {
8927 reason: "role-bound token: target node not visible".into(),
8928 });
8929 }
8930 };
8931 if !authz.scope.update_labels.contains(&label) {
8932 return Err(GraphError::RoleWriteDenied {
8933 reason: format!(
8934 "role-bound token: label '{}' not in write scope (update_labels)",
8935 label
8936 ),
8937 });
8938 }
8939 }
8940 }
8941
8942 // ── DELETE-class: DeleteNode ─────────────────────────────────────
8943 BatchOp::DeleteNode { key } => {
8944 let label = match node_status(key) {
8945 NodeAuthzStatus::Visible(lbl) => lbl,
8946 _ => {
8947 return Err(GraphError::RoleWriteDenied {
8948 reason: "role-bound token: target node not visible".into(),
8949 });
8950 }
8951 };
8952 if !authz.scope.delete_labels.contains(&label) {
8953 return Err(GraphError::RoleWriteDenied {
8954 reason: format!(
8955 "role-bound token: label '{}' not in write scope (delete_labels)",
8956 label
8957 ),
8958 });
8959 }
8960 }
8961
8962 // ── DELETE-class: DeleteEdge ─────────────────────────────────────
8963 //
8964 // Derived-edge rejection runs BEFORE the delete_edge_types scope
8965 // check (spec §3.5: "existing derived-edge rejection precedes
8966 // delete_edge_types check").
8967 BatchOp::DeleteEdge {
8968 edge_type,
8969 src_key,
8970 dst_key,
8971 } => {
8972 // Check provenance ownership BEFORE scope (spec §3.5 ordering).
8973 if let (Some(src_id), Some(dst_id), Some(et_sym)) = (
8974 self.ids.get(src_key.as_str()),
8975 self.ids.get(dst_key.as_str()),
8976 self.syms.get(edge_type.as_str()),
8977 ) {
8978 if self.engine.is_owned(et_sym, src_id, dst_id) {
8979 return Err(GraphError::RuleOwned {
8980 detail: format!(
8981 "edge {edge_type} {src_key}→{dst_key} is rule-owned; \
8982 delete or change the owning rule"
8983 ),
8984 });
8985 }
8986 // Also check would_derive via MutPreview (empty overlay, pre-batch).
8987 let preview = MutPreview::new(self);
8988 if preview.would_derive(edge_type, src_key, dst_key) {
8989 return Err(GraphError::RuleOwned {
8990 detail: format!(
8991 "edge {edge_type} {src_key}→{dst_key} is rule-owned; \
8992 delete or change the owning rule, or a live rule would \
8993 re-derive it"
8994 ),
8995 });
8996 }
8997 }
8998 // Scope check (AFTER derived-edge check, BEFORE endpoint visibility).
8999 if !authz.scope.delete_edge_types.contains(edge_type) {
9000 return Err(GraphError::RoleWriteDenied {
9001 reason: format!(
9002 "role-bound token: edge type '{}' not in write scope (delete_edge_types)",
9003 edge_type
9004 ),
9005 });
9006 }
9007 // Both endpoints must be visible.
9008 for ep_key in [src_key.as_str(), dst_key.as_str()] {
9009 match self.ids.get(ep_key) {
9010 None => {
9011 return Err(GraphError::RoleWriteDenied {
9012 reason: "role-bound token: edge endpoint not visible".into(),
9013 });
9014 }
9015 Some(id) if !authz.mask.contains_id(id) => {
9016 return Err(GraphError::RoleWriteDenied {
9017 reason: "role-bound token: edge endpoint not visible".into(),
9018 });
9019 }
9020 _ => {}
9021 }
9022 }
9023 }
9024
9025 // ── EDGE-CREATE: InsertEdge ──────────────────────────────────────
9026 //
9027 // Scope check BEFORE endpoint lookup (preserves timing symmetry).
9028 BatchOp::InsertEdge {
9029 edge_type,
9030 src_key,
9031 dst_key,
9032 } => {
9033 if !authz.scope.create_edge_types.contains(edge_type) {
9034 return Err(GraphError::RoleWriteDenied {
9035 reason: format!(
9036 "role-bound token: edge type '{}' not in write scope (create_edge_types)",
9037 edge_type
9038 ),
9039 });
9040 }
9041 // Both endpoints must be visible. A node created by an earlier
9042 // InsertNode in the same batch (tracked in batch_created) counts
9043 // as visible if its label passed the create-class gate.
9044 for ep_key in [src_key.as_str(), dst_key.as_str()] {
9045 if batch_created.contains_key(ep_key) {
9046 // Created earlier this batch — already scope-checked.
9047 continue;
9048 }
9049 match self.ids.get(ep_key) {
9050 None => {
9051 return Err(GraphError::RoleWriteDenied {
9052 reason: "role-bound token: edge endpoint not visible".into(),
9053 });
9054 }
9055 Some(id) if !authz.mask.contains_id(id) => {
9056 return Err(GraphError::RoleWriteDenied {
9057 reason: "role-bound token: edge endpoint not visible".into(),
9058 });
9059 }
9060 _ => {}
9061 }
9062 }
9063 }
9064
9065 // ── EDGE-CREATE: InsertEdgeUpsert ────────────────────────────────
9066 //
9067 // Scope check first; then endpoint visibility using same-batch
9068 // placeholder awareness (spec: "a placeholder endpoint the SAME
9069 // batch creates counts as visible if its label passed the
9070 // create-class gate").
9071 BatchOp::InsertEdgeUpsert {
9072 edge_type,
9073 src_key,
9074 dst_key,
9075 placeholder_label,
9076 } => {
9077 if !authz.scope.create_edge_types.contains(edge_type) {
9078 return Err(GraphError::RoleWriteDenied {
9079 reason: format!(
9080 "role-bound token: edge type '{}' not in write scope (create_edge_types)",
9081 edge_type
9082 ),
9083 });
9084 }
9085 // Check placeholder label against create_labels (create-class gate).
9086 // This ensures the auto-created endpoints are scope-allowed.
9087 for ep_key in [src_key.as_str(), dst_key.as_str()] {
9088 if !upsert_ep_visible(ep_key, placeholder_label) {
9089 return Err(GraphError::RoleWriteDenied {
9090 reason: "role-bound token: edge endpoint not visible".into(),
9091 });
9092 }
9093 }
9094 // A placeholder is created with no props, so it lands in the
9095 // default namespace. A role that cannot read `default` must not
9096 // create one there, for the same reason it may not create a node
9097 // there outright.
9098 //
9099 // The refusal is byte-identical to the hidden-endpoint one above,
9100 // and deliberately so: this arm fires only for an endpoint that
9101 // does **not** exist, and the one above only for an endpoint that
9102 // does. Two different strings would make the pair an existence
9103 // oracle — ask for an upsert and read off whether the key is
9104 // taken. Hidden ≡ absent is the rule everywhere else in this
9105 // table and it holds here too.
9106 if let Some(def) = self.role_def_for(&authz.role) {
9107 if !def.sees_namespace(NS_DEFAULT) {
9108 for ep_key in [src_key.as_str(), dst_key.as_str()] {
9109 if self.ids.get(ep_key).is_none() && !batch_created.contains_key(ep_key)
9110 {
9111 return Err(GraphError::RoleWriteDenied {
9112 reason: "role-bound token: edge endpoint not visible".into(),
9113 });
9114 }
9115 }
9116 }
9117 }
9118 }
9119 }
9120 Ok(())
9121 }
9122
9123 /// Write `roles` to `roles.json` atomically and update the in-memory list.
9124 ///
9125 /// Called by `apply_schema` when roles change. Never called on unchanged
9126 /// re-apply — this preserves byte-identical idempotency.
9127 pub(crate) fn commit_roles(&mut self, roles: Vec<RoleDef>) -> Result<()> {
9128 let file = RolesFile::new_versioned(roles.clone());
9129 let bytes = serde_json::to_vec(&file).map_err(|e| GraphError::Corrupt {
9130 detail: format!("roles serialization: {e}"),
9131 })?;
9132 self.fs
9133 .write_atomic(FileId::Roles, &bytes)
9134 .map_err(GraphError::Io)?;
9135 self.roles = Some(roles);
9136 // Rewriting the sidecar is not a commit, so `commit_seq` does not move
9137 // and a memoised mask would still match its version. Install a fresh
9138 // cache instead of clearing the shared one: a reader snapshot frozen
9139 // against the old definitions keeps the old `Arc` to itself and can
9140 // never publish an answer this handle would read back.
9141 self.role_masks = Arc::new(crate::mask::RoleMaskCache::new());
9142 // Refresh the MVCC frozen overlay so that reader() immediately sees the
9143 // updated role definitions without waiting for the next K-commit fold.
9144 self.fold_now();
9145 Ok(())
9146 }
9147
9148 fn view(&self) -> GraphView<'_> {
9149 GraphView {
9150 ids: &self.ids,
9151 syms: &self.syms,
9152 labels: &self.labels,
9153 props: self.props_view(),
9154 topo: self.topo_view(),
9155 edge_props: self.edge_props_view(),
9156 mask: None,
9157 prop_index: Some(&self.prop_index),
9158 }
9159 }
9160
9161 fn view_masked<'a>(&'a self, mask: &'a crate::mask::NodeMask) -> GraphView<'a> {
9162 GraphView {
9163 ids: &self.ids,
9164 syms: &self.syms,
9165 labels: &self.labels,
9166 props: self.props_view(),
9167 topo: self.topo_view(),
9168 edge_props: self.edge_props_view(),
9169 mask: Some(&mask.visible),
9170 prop_index: Some(&self.prop_index),
9171 }
9172 }
9173
9174 /// Execute a read-only Cypher query with a node visibility mask.
9175 ///
9176 /// Only nodes whose key is in `mask` are accessible: label scans, key
9177 /// lookups, and neighbor expansions all respect the mask. Edges where
9178 /// either endpoint is hidden are silently dropped.
9179 ///
9180 /// Returns `Err` with a "masked queries are read-only" message when
9181 /// `cypher` is a write statement (CREATE / MERGE / MATCH…SET / DELETE).
9182 pub fn query_masked(
9183 &self,
9184 cypher: &str,
9185 params: &std::collections::BTreeMap<String, Value>,
9186 mask: &crate::mask::NodeMask,
9187 ) -> Result<ResultSet> {
9188 // Reject write statements up front.
9189 let tokens = lex(cypher).map_err(|e| GraphError::QueryError {
9190 detail: format!("lex: {e}"),
9191 })?;
9192 if is_write_tokens(&tokens) {
9193 return Err(GraphError::MaskedReadOnly);
9194 }
9195 let union = parse_read(&tokens).map_err(|e| GraphError::QueryError {
9196 detail: format!("parse: {e}"),
9197 })?;
9198 // Each UNION part executes against the same masked view, so the mask
9199 // applies uniformly across the chain.
9200 execute_union(&self.view_masked(mask), &union, &Params(params)).map_err(|e| {
9201 GraphError::QueryError {
9202 detail: format!("execute: {e}"),
9203 }
9204 })
9205 }
9206
9207 pub fn node_ref(&self, key: &str) -> Option<NodeRef<'_, F>> {
9208 let id = self.ids.get(key)?;
9209 Some(NodeRef { db: self, id })
9210 }
9211
9212 /// BFS neighborhood expansion restricted to visible nodes in `mask`.
9213 ///
9214 /// Hidden nodes are never used as traversal intermediaries in either
9215 /// [`MaskMode::Omit`] or [`MaskMode::Stub`] — a visible node reachable
9216 /// only through a hidden node will not appear in results.
9217 ///
9218 /// In [`MaskMode::Stub`] mode, hidden nodes that are direct neighbours of
9219 /// a visited visible node are appended to the result as stub rows
9220 /// (`label` column is `null`, same key+depth columns as visible rows).
9221 /// They are NOT added to the BFS frontier.
9222 ///
9223 /// Returns `None` when `key` does not exist (caller should 404).
9224 ///
9225 /// **SECURITY**: role-token callers always pass an Omit-mode mask, so
9226 /// stub rows are never produced on the role path.
9227 pub fn neighborhood_masked(
9228 &self,
9229 key: &str,
9230 depth: u32,
9231 edge_types: Option<&[&str]>,
9232 dir: Dir,
9233 mask: &crate::mask::NodeMask,
9234 ) -> Option<ResultSet> {
9235 let start_id = self.ids.get(key)?;
9236 let view = self.view_masked(mask);
9237 let resolved: Option<Vec<u32>> = edge_types.map(|names| {
9238 names
9239 .iter()
9240 .filter_map(|name| view.syms.get(name))
9241 .collect()
9242 });
9243 let nb = neighborhood(&view, start_id, depth, resolved.as_deref(), dir);
9244 let mut rs = ResultSet::new(vec!["key".into(), "label".into(), "depth".into()]);
9245 // Collect visible BFS results (start_id at depth 0, BFS nodes after).
9246 let mut visited: Vec<(u32, u32)> = Vec::with_capacity(nb.nodes.len() + 1);
9247 visited.push((start_id, 0));
9248 for (nid, d) in &nb.nodes {
9249 let k = view.key_of(*nid);
9250 let label = view
9251 .label_of(*nid)
9252 .expect("real nodes always have a label; u32::MAX sentinel cannot occur");
9253 rs.push_row(vec![
9254 Some(Value::Str(k.to_string())),
9255 Some(Value::Str(label.to_string())),
9256 Some(Value::Int(*d as i64)),
9257 ]);
9258 visited.push((*nid, *d));
9259 }
9260 // Stub mode: add hidden direct neighbours of each visited node as stubs.
9261 // Hidden nodes are edge-endpoints only — they are not added to the BFS
9262 // frontier, so the BFS never expands through them.
9263 if mask.mode() == crate::mask::MaskMode::Stub {
9264 let raw_view = self.view();
9265 let mut seen: std::collections::HashSet<u32> =
9266 visited.iter().map(|(id, _)| *id).collect();
9267 for (node_id, node_depth) in &visited {
9268 if *node_depth >= depth {
9269 continue;
9270 }
9271 for e in expand(&raw_view, *node_id, resolved.as_deref(), dir) {
9272 let nbr = if e.src == *node_id { e.dst } else { e.src };
9273 if !mask.contains_id(nbr) && seen.insert(nbr) {
9274 if let Some(k) = self.ids.key_of(nbr) {
9275 rs.push_row(vec![
9276 Some(Value::Str(k.to_string())),
9277 None,
9278 Some(Value::Int((*node_depth + 1) as i64)),
9279 ]);
9280 }
9281 }
9282 }
9283 }
9284 }
9285 Some(rs)
9286 }
9287
9288 /// [`neighborhood_masked`](Self::neighborhood_masked) with the **subject
9289 /// check** a scoped caller needs: a start key the mask hides answers exactly
9290 /// as an absent one does.
9291 ///
9292 /// `neighborhood_masked` expands from any existing key, hidden or not,
9293 /// because a full-token caller supplying a client mask already knows which
9294 /// keys exist. A scoped caller does not, so telling it apart a hidden key
9295 /// from an absent one would be an existence oracle.
9296 ///
9297 /// Expansion itself is unchanged: hidden nodes are neither returned nor used
9298 /// as traversal intermediaries, so a visible node reachable only through a
9299 /// hidden one stays out of the result.
9300 ///
9301 /// Hidden or unknown `key` → [`GraphError::KeyNotFound`].
9302 pub fn neighborhood_scoped(
9303 &self,
9304 key: &str,
9305 depth: u32,
9306 edge_types: Option<&[&str]>,
9307 dir: Dir,
9308 mask: &crate::mask::NodeMask,
9309 ) -> Result<ResultSet> {
9310 if !mask.contains_node(self, key) {
9311 return Err(GraphError::KeyNotFound { key: key.into() });
9312 }
9313 self.neighborhood_masked(key, depth, edge_types, dir, mask)
9314 .ok_or_else(|| GraphError::KeyNotFound { key: key.into() })
9315 }
9316
9317 /// Live node's key, label, and columnar props. Unknown or tombstoned → `None`.
9318 pub fn node_info(&self, key: &str) -> Option<NodeInfo> {
9319 let n = self.node_ref(key)?;
9320 Some(NodeInfo {
9321 key: n.key().to_string(),
9322 label: n.label().to_string(),
9323 props: n.props(),
9324 })
9325 }
9326
9327 /// Look up a node with mask awareness.
9328 ///
9329 /// | Key state | Omit mode | Stub mode |
9330 /// |-------------------|-----------------|------------------------|
9331 /// | does not exist | `None` (→ 404) | `None` (→ 404) |
9332 /// | exists, visible | `Some(Visible)` | `Some(Visible)` |
9333 /// | exists, hidden | `None` (→ 404) | `Some(Restricted)` |
9334 ///
9335 /// **SECURITY**: only call from client-mask (full-token) paths.
9336 /// Role-token paths must use [`node_info`] after an explicit visibility check.
9337 pub fn node_info_masked(
9338 &self,
9339 key: &str,
9340 mask: &crate::mask::NodeMask,
9341 ) -> Option<MaskedNodeResult> {
9342 let id = self.ids.get(key)?;
9343 if mask.contains_id(id) {
9344 Some(MaskedNodeResult::Visible(self.node_info(key)?))
9345 } else {
9346 match mask.mode() {
9347 crate::mask::MaskMode::Stub => Some(MaskedNodeResult::Restricted),
9348 crate::mask::MaskMode::Omit => None,
9349 }
9350 }
9351 }
9352
9353 /// Get edges for `key` with mask-aware hidden-endpoint handling.
9354 ///
9355 /// - Omit mode: edges to hidden endpoints are excluded (same as role-path filtering).
9356 /// - Stub mode: edges to hidden endpoints are included; `src_restricted`/`dst_restricted`
9357 /// is `true` for each hidden endpoint.
9358 ///
9359 /// Unknown key → [`GraphError::KeyNotFound`].
9360 ///
9361 /// **SECURITY**: only call from client-mask (full-token) paths.
9362 pub fn node_edges_masked(
9363 &self,
9364 key: &str,
9365 mask: &crate::mask::NodeMask,
9366 ) -> Result<Vec<MaskedEdge>> {
9367 self.ensure_v8_base_sections_loaded();
9368 let id = self
9369 .ids
9370 .get(key)
9371 .ok_or_else(|| GraphError::KeyNotFound { key: key.into() })?;
9372 let derived: BTreeSet<(u32, u32, u32)> = self
9373 .engine
9374 .provenance_touching(id)
9375 .map(|(_rule, etype, src, dst)| (etype, src, dst))
9376 .collect();
9377 let mut edges = Vec::new();
9378 let tv = self.topo_view();
9379 for etype in tv.etypes() {
9380 // etype comes from the archived CSR (access_unchecked, no eager CRC).
9381 // A bit-flip in the large TOPOLOGY section can produce an etype id
9382 // that is not in the interner. Return Corrupt rather than panic.
9383 let edge_type = self
9384 .syms
9385 .resolve(etype)
9386 .ok_or_else(|| GraphError::Corrupt {
9387 detail: format!("v8: topology etype {etype} not in interner"),
9388 })?
9389 .to_string();
9390 for dir in [Direction::Out, Direction::In] {
9391 for &nbr in tv.neighbors(etype, dir, id).as_ref() {
9392 let nbr_restricted = !mask.contains_id(nbr);
9393 if nbr_restricted && mask.mode() == crate::mask::MaskMode::Omit {
9394 continue;
9395 }
9396 let nbr_key = self
9397 .ids
9398 .key_of(nbr)
9399 .ok_or_else(|| GraphError::Corrupt {
9400 detail: format!("topology id {nbr} has no key"),
9401 })?
9402 .to_string();
9403 let (src_id, dst_id, src_key, dst_key, src_restricted, dst_restricted) =
9404 match dir {
9405 Direction::Out => {
9406 (id, nbr, key.to_string(), nbr_key, false, nbr_restricted)
9407 }
9408 Direction::In => {
9409 (nbr, id, nbr_key, key.to_string(), nbr_restricted, false)
9410 }
9411 };
9412 edges.push(MaskedEdge {
9413 edge_type: edge_type.clone(),
9414 src_key,
9415 src_restricted,
9416 dst_key,
9417 dst_restricted,
9418 derived: derived.contains(&(etype, src_id, dst_id)),
9419 });
9420 }
9421 }
9422 }
9423 edges.sort_by(|a, b| {
9424 a.edge_type
9425 .cmp(&b.edge_type)
9426 .then(a.src_key.cmp(&b.src_key))
9427 .then(a.dst_key.cmp(&b.dst_key))
9428 });
9429 edges.dedup_by(|a, b| {
9430 a.edge_type == b.edge_type && a.src_key == b.src_key && a.dst_key == b.dst_key
9431 });
9432 Ok(edges)
9433 }
9434
9435 /// [`node_edges_masked`](Self::node_edges_masked) with the **subject check**
9436 /// a scoped caller needs, and a plain [`EdgeInfo`] list.
9437 ///
9438 /// `node_edges_masked` raises [`GraphError::KeyNotFound`] only when `key` is
9439 /// unknown; a key that exists but is hidden still yields its (filtered) edge
9440 /// list, which is correct for a full-token client mask and an existence
9441 /// oracle for a scoped one. Here a hidden subject answers exactly as an
9442 /// absent one does.
9443 ///
9444 /// Every edge naming a hidden endpoint is dropped, whatever the mask's
9445 /// [`MaskMode`](crate::mask::MaskMode): a scoped caller never sees a
9446 /// restricted stub, so there is nothing for it to render.
9447 ///
9448 /// Hidden or unknown `key` → [`GraphError::KeyNotFound`].
9449 pub fn node_edges_scoped(
9450 &self,
9451 key: &str,
9452 mask: &crate::mask::NodeMask,
9453 ) -> Result<Vec<EdgeInfo>> {
9454 if !mask.contains_node(self, key) {
9455 return Err(GraphError::KeyNotFound { key: key.into() });
9456 }
9457 Ok(self
9458 .node_edges_masked(key, mask)?
9459 .into_iter()
9460 .filter(|e| !e.src_restricted && !e.dst_restricted)
9461 .map(|e| EdgeInfo {
9462 edge_type: e.edge_type,
9463 src_key: e.src_key,
9464 dst_key: e.dst_key,
9465 derived: e.derived,
9466 })
9467 .collect())
9468 }
9469
9470 /// Every directed edge incident on `key`, both directions, every etype.
9471 ///
9472 /// Walk is `topology.etypes()` × `{Out, In}` × `neighbors()`. `derived` is
9473 /// membership in [`RuleEngine::provenance_touching`] (O(degree) via the
9474 /// Plan-8 `by_node` index). Sorted by `(edge_type, src_key, dst_key)`.
9475 /// Unknown key → [`GraphError::KeyNotFound`].
9476 pub fn node_edges(&self, key: &str) -> Result<Vec<EdgeInfo>> {
9477 self.ensure_v8_base_sections_loaded();
9478 let id = self
9479 .ids
9480 .get(key)
9481 .ok_or_else(|| GraphError::KeyNotFound { key: key.into() })?;
9482 let derived: BTreeSet<(u32, u32, u32)> = self
9483 .engine
9484 .provenance_touching(id)
9485 .map(|(_rule, etype, src, dst)| (etype, src, dst))
9486 .collect();
9487 let mut edges = Vec::new();
9488 let tv = self.topo_view();
9489 for etype in tv.etypes() {
9490 // Same guard as node_edges_masked: etype from unchecked-CRC CSR.
9491 let edge_type = self
9492 .syms
9493 .resolve(etype)
9494 .ok_or_else(|| GraphError::Corrupt {
9495 detail: format!("v8: topology etype {etype} not in interner"),
9496 })?
9497 .to_string();
9498 for dir in [Direction::Out, Direction::In] {
9499 for &nbr in tv.neighbors(etype, dir, id).as_ref() {
9500 let (src, dst, src_key, dst_key) = match dir {
9501 Direction::Out => (
9502 id,
9503 nbr,
9504 key.to_string(),
9505 self.ids
9506 .key_of(nbr)
9507 .ok_or_else(|| GraphError::Corrupt {
9508 detail: format!("topology id {nbr} has no key"),
9509 })?
9510 .to_string(),
9511 ),
9512 Direction::In => (
9513 nbr,
9514 id,
9515 self.ids
9516 .key_of(nbr)
9517 .ok_or_else(|| GraphError::Corrupt {
9518 detail: format!("topology id {nbr} has no key"),
9519 })?
9520 .to_string(),
9521 key.to_string(),
9522 ),
9523 };
9524 edges.push(EdgeInfo {
9525 edge_type: edge_type.clone(),
9526 src_key,
9527 dst_key,
9528 derived: derived.contains(&(etype, src, dst)),
9529 });
9530 }
9531 }
9532 }
9533 edges.sort_by(|a, b| {
9534 a.edge_type
9535 .cmp(&b.edge_type)
9536 .then(a.src_key.cmp(&b.src_key))
9537 .then(a.dst_key.cmp(&b.dst_key))
9538 });
9539 // Self-loops appear in both Out and In; sort makes the pair adjacent
9540 // (sort key matches PartialEq for this case) so one pass drops the dup.
9541 edges.dedup();
9542 Ok(edges)
9543 }
9544
9545 // ── Backup ────────────────────────────────────────────────────────────────
9546
9547 /// Copy this store to `dest` as a consistent, verified snapshot.
9548 ///
9549 /// Copies every durable file in the database directory — `snapshot.bin`,
9550 /// `wal.bin`, all `wal.<N>.archive` files, `wal.floor`, `wal.genesis`, and
9551 /// `roles.json` — into a freshly created `dest` directory using OS-level
9552 /// `copy` calls (no large in-process buffers).
9553 ///
9554 /// # Consistency guarantee
9555 ///
9556 /// The guarantee is **process-local**: the caller holds `&self`, which
9557 /// prevents any concurrent writer in the **same process** from modifying
9558 /// the files during the copy. Running `mushroomdb backup` against a
9559 /// directory that is **concurrently being written by another process** (e.g.
9560 /// `mushroomdb serve`) is **unsafe** — the copy can be torn. The post-copy
9561 /// `verified: true` result reduces but does not eliminate the risk of a
9562 /// silent corrupt backup (CRC catches many bit-flips; it cannot catch a
9563 /// consistent mid-write snapshot).
9564 ///
9565 /// **The safe path for a live-served store is `POST /backup` on the HTTP
9566 /// server.** That handler acquires the read lock on the shared database
9567 /// before calling this method, which is the correct cross-process
9568 /// synchronisation point because the server is the single process writing
9569 /// the files.
9570 ///
9571 /// After copying, opens the destination read-only and runs the CRC section
9572 /// verifier (`verify_snapshot`) to confirm byte-for-byte integrity.
9573 /// `BackupReport::verified` reflects whether both checks passed.
9574 ///
9575 /// Returns `Err` when `self` is not backed by a `RealFs` (e.g. `SimFs`).
9576 pub fn backup_to(&self, dest: &std::path::Path) -> Result<BackupReport> {
9577 // Derive source directory from snapshot_path (RealFs only).
9578 let src_dir = match self.fs.snapshot_path() {
9579 Some(p) => p.parent().map(|d| d.to_path_buf()).ok_or_else(|| {
9580 GraphError::Io(std::io::Error::other("snapshot has no parent dir"))
9581 })?,
9582 None => {
9583 return Err(GraphError::Io(std::io::Error::other(
9584 "backup_to requires a real filesystem (RealFs)",
9585 )))
9586 }
9587 };
9588
9589 std::fs::create_dir_all(dest)?;
9590
9591 let mut files: Vec<String> = Vec::new();
9592 let mut bytes: u64 = 0;
9593
9594 // Helper: copy src_dir/name → dest/name if the file exists.
9595 let mut try_copy = |name: &str| -> std::io::Result<()> {
9596 let src_path = src_dir.join(name);
9597 if src_path.exists() {
9598 let n = std::fs::copy(&src_path, dest.join(name))?;
9599 bytes += n;
9600 files.push(name.to_string());
9601 }
9602 Ok(())
9603 };
9604
9605 try_copy("snapshot.bin")?;
9606 try_copy("snapshot.bin.bak")?;
9607 try_copy("wal.bin")?;
9608 try_copy("wal.floor")?;
9609 try_copy("wal.genesis")?;
9610 try_copy("roles.json")?;
9611
9612 // Copy WAL archives.
9613 let archives = self.fs.list_archives()?;
9614 for n in &archives {
9615 let name = format!("wal.{n}.archive");
9616 let n_bytes = std::fs::copy(src_dir.join(&name), dest.join(&name))?;
9617 bytes += n_bytes;
9618 files.push(name);
9619 }
9620
9621 files.sort();
9622
9623 // Post-copy verification: open dest and run CRC checks.
9624 let snap_in_dest = dest.join("snapshot.bin").exists();
9625 let crc_ok = if snap_in_dest {
9626 crate::verify_snapshot(dest)
9627 .map(|results| results.iter().all(|(_, _, _, r)| r.is_ok()))
9628 .unwrap_or(false)
9629 } else {
9630 true // WAL-only store: nothing to CRC-check in snapshot
9631 };
9632 let opens_ok = GraphDb::<core_storage::fs::RealFs>::open(dest).is_ok();
9633 let verified = crc_ok && opens_ok;
9634
9635 Ok(BackupReport {
9636 files,
9637 bytes,
9638 verified,
9639 })
9640 }
9641
9642 // ── Export helpers ────────────────────────────────────────────────────────
9643
9644 /// All live nodes, sorted by key (deterministic).
9645 ///
9646 /// Reads base + WAL overlay. Tombstoned nodes are excluded.
9647 pub fn all_nodes_for_export(&self) -> Vec<NodeInfo> {
9648 self.ensure_v8_base_sections_loaded();
9649 let pv = self.props_view();
9650 let mut nodes = Vec::new();
9651 for id in 0..self.ids.len() as u32 {
9652 let Some(key) = self.ids.key_of(id) else {
9653 continue;
9654 };
9655 let Some(&sym) = self.labels.get(id as usize) else {
9656 continue;
9657 };
9658 if sym == u32::MAX {
9659 continue; // tombstoned
9660 }
9661 let Some(label) = self.syms.resolve(sym) else {
9662 continue;
9663 };
9664 let mut props = BTreeMap::new();
9665 for field in pv.field_names() {
9666 if let Some(vr) = pv.get(id, &field) {
9667 props.insert(field, vr.into_value());
9668 }
9669 }
9670 nodes.push(NodeInfo {
9671 key: key.to_string(),
9672 label: label.to_string(),
9673 props,
9674 });
9675 }
9676 nodes.sort_by(|a, b| a.key.cmp(&b.key));
9677 nodes
9678 }
9679
9680 /// All directed edges, sorted by `(edge_type, src, dst)`. Each edge appears once.
9681 ///
9682 /// Derived edges carry `derived: true` and the creating rule's name in `rule`.
9683 /// Manual edges carry `derived: false` and `rule: None`.
9684 /// `weight` is the creating rule's `weight_prop` value read off the edge
9685 /// (numeric only), mirroring the convention used by [`GraphDb::explain`]
9686 /// and [`GraphDb::weighted_edges`]. Deterministic across runs on the same
9687 /// store state.
9688 pub fn all_edges_for_export(&self) -> Vec<ExportEdge> {
9689 self.ensure_v8_base_sections_loaded();
9690
9691 // Build (etype_sym, src_id, dst_id) → rule_name for O(1) derivation lookup.
9692 let mut prov: HashMap<(u32, u32, u32), String> = HashMap::new();
9693 for (rule_name, triples) in self.engine.provenance() {
9694 for &(etype, src, dst) in triples {
9695 prov.insert((etype, src, dst), rule_name.clone());
9696 }
9697 }
9698
9699 // rule_name → weight_prop, for O(1) lookup per derived edge.
9700 let weight_props: HashMap<&str, Option<&str>> = self
9701 .engine
9702 .rules()
9703 .map(|r| (r.name.as_str(), r.weight_prop.as_deref()))
9704 .collect();
9705
9706 let tv = self.topo_view();
9707 let ep = self.edge_props_view();
9708 let mut edges = Vec::new();
9709
9710 for id in 0..self.ids.len() as u32 {
9711 let Some(key) = self.ids.key_of(id) else {
9712 continue;
9713 };
9714 let Some(&lsym) = self.labels.get(id as usize) else {
9715 continue;
9716 };
9717 if lsym == u32::MAX {
9718 continue; // tombstoned
9719 }
9720
9721 for etype_sym in tv.etypes() {
9722 // etype from archived CSR (access_unchecked, no eager CRC).
9723 // Skip edges whose etype is not in the interner; this can only
9724 // occur with a corrupt large TOPOLOGY section (bit-flip on an
9725 // etype field in the archived data). The function returns Vec,
9726 // not Result, so we continue rather than propagate.
9727 let Some(edge_type) = self.syms.resolve(etype_sym) else {
9728 continue;
9729 };
9730 let edge_type = edge_type.to_string();
9731 for &nbr in tv.neighbors(etype_sym, Direction::Out, id).as_ref() {
9732 let Some(dst_key) = self.ids.key_of(nbr) else {
9733 continue; // skip corrupt entries
9734 };
9735 let prov_key = (etype_sym, id, nbr);
9736 let rule = prov.get(&prov_key).cloned();
9737 let derived = rule.is_some();
9738 let weight = rule
9739 .as_deref()
9740 .and_then(|rn| weight_props.get(rn).copied().flatten())
9741 .and_then(|prop| match ep.get(etype_sym, id, nbr, prop) {
9742 Some(Value::Float(f)) => Some(f),
9743 Some(Value::Int(i)) => Some(i as f64),
9744 _ => None,
9745 });
9746 edges.push(ExportEdge {
9747 edge_type: edge_type.clone(),
9748 src: key.to_string(),
9749 dst: dst_key.to_string(),
9750 derived,
9751 rule,
9752 weight,
9753 });
9754 }
9755 }
9756 }
9757
9758 edges.sort_by(|a, b| {
9759 a.edge_type
9760 .cmp(&b.edge_type)
9761 .then(a.src.cmp(&b.src))
9762 .then(a.dst.cmp(&b.dst))
9763 });
9764 edges
9765 }
9766
9767 /// What each edge type *is*, without building one record per edge.
9768 ///
9769 /// [`all_edges_for_export`](Self::all_edges_for_export) answers the same
9770 /// question by materialising every edge — three `String`s apiece, a
9771 /// provenance `HashMap` over every derived edge, and a final sort. That is
9772 /// the right shape for an export, and the wrong one for a summary: on a
9773 /// store with 1.3 M derived edges it allocates hundreds of megabytes to
9774 /// produce nine lines. This walks the topology instead, summing neighbour
9775 /// slice lengths and collecting *label symbols* rather than label strings,
9776 /// so the per-edge cost is an integer add and a set insert on a set with
9777 /// as many members as the store has labels.
9778 ///
9779 /// The rule names come off the rule *definitions*, which each declare the
9780 /// `edge_type` they derive, so naming them costs one pass over the rules
9781 /// rather than one provenance lookup per edge. That is also why `rules`
9782 /// is a list: two rules may derive the same type — the association store
9783 /// derives `INDUSTRY_ALIGNMENT` from both a talent→company and a
9784 /// talent→job rule — and naming only one of them would be a half-truth.
9785 /// A type with no rules is one written by hand.
9786 ///
9787 /// `sample` is the first edge of the type in the store's own id order,
9788 /// which is insertion order: deterministic for a given store, and not the
9789 /// same as key order, which cannot be had without resolving a key per
9790 /// edge. Sorted by `edge_type`.
9791 pub fn edge_type_census(&self) -> Vec<EdgeTypeCensus> {
9792 self.ensure_v8_base_sections_loaded();
9793
9794 let mut rules_by_type: BTreeMap<&str, BTreeSet<&str>> = BTreeMap::new();
9795 for r in self.engine.rules() {
9796 rules_by_type
9797 .entry(r.edge_type.as_str())
9798 .or_default()
9799 .insert(r.name.as_str());
9800 }
9801
9802 let tv = self.topo_view();
9803 let node_count = self.ids.len() as u32;
9804 let mut out = Vec::new();
9805 for etype_sym in tv.etypes() {
9806 // An etype the interner cannot resolve means a corrupt TOPOLOGY
9807 // section; skip it rather than name it, as `all_edges_for_export`
9808 // does for the same reason.
9809 let Some(edge_type) = self.syms.resolve(etype_sym) else {
9810 continue;
9811 };
9812 let mut edges: u64 = 0;
9813 let mut src_syms: BTreeSet<u32> = BTreeSet::new();
9814 let mut dst_syms: BTreeSet<u32> = BTreeSet::new();
9815 let mut sample: Option<(u32, u32)> = None;
9816 for id in 0..node_count {
9817 let Some(&lsym) = self.labels.get(id as usize) else {
9818 continue;
9819 };
9820 if lsym == u32::MAX {
9821 continue; // tombstoned
9822 }
9823 let nbrs = tv.neighbors(etype_sym, Direction::Out, id);
9824 let nbrs = nbrs.as_ref();
9825 if nbrs.is_empty() {
9826 continue;
9827 }
9828 edges += nbrs.len() as u64;
9829 src_syms.insert(lsym);
9830 for &nbr in nbrs {
9831 if let Some(&dsym) = self.labels.get(nbr as usize) {
9832 if dsym != u32::MAX {
9833 dst_syms.insert(dsym);
9834 }
9835 }
9836 }
9837 if sample.is_none() {
9838 sample = Some((id, nbrs[0]));
9839 }
9840 }
9841 let resolve = |syms: &BTreeSet<u32>| -> Vec<String> {
9842 syms.iter()
9843 .filter_map(|&s| self.syms.resolve(s))
9844 .map(ToString::to_string)
9845 .collect()
9846 };
9847 out.push(EdgeTypeCensus {
9848 edge_type: edge_type.to_string(),
9849 edges,
9850 src_labels: resolve(&src_syms),
9851 dst_labels: resolve(&dst_syms),
9852 rules: rules_by_type
9853 .get(edge_type)
9854 .map(|rs| rs.iter().map(ToString::to_string).collect())
9855 .unwrap_or_default(),
9856 sample: sample.and_then(|(s, d)| {
9857 Some((
9858 self.ids.key_of(s)?.to_string(),
9859 self.ids.key_of(d)?.to_string(),
9860 ))
9861 }),
9862 });
9863 }
9864 out.sort_by(|a, b| a.edge_type.cmp(&b.edge_type));
9865 out
9866 }
9867
9868 /// All directed edges of `edge_type`, with the raw value of `weight_prop`
9869 /// on each edge when given.
9870 ///
9871 /// `weight` is `Some(f)` only when `weight_prop` is set and the edge
9872 /// carries that property with a numeric (`Int`/`Float`) value; otherwise
9873 /// `None` — callers that want a default weight (e.g. `1.0` for missing
9874 /// props) apply it themselves, matching the convention used internally
9875 /// by [`GraphDb::pagerank`], [`GraphDb::connected_components`],
9876 /// [`GraphDb::degree_centrality`], and [`GraphDb::communities`].
9877 ///
9878 /// Sorted by `(src, dst)` for determinism. Reads the unified topology
9879 /// (manual + rule-derived edges). An unknown `edge_type` returns an
9880 /// empty vec.
9881 pub fn weighted_edges(
9882 &self,
9883 edge_type: &str,
9884 weight_prop: Option<&str>,
9885 ) -> Vec<(String, String, Option<f64>)> {
9886 let Some(etype_sym) = self.syms.get(edge_type) else {
9887 return Vec::new();
9888 };
9889 let tv = self.topo_view();
9890 let ep = self.edge_props_view();
9891 let mut out = Vec::new();
9892 for id in 0..self.ids.len() as u32 {
9893 let Some(key) = self.ids.key_of(id) else {
9894 continue;
9895 };
9896 let Some(&sym) = self.labels.get(id as usize) else {
9897 continue;
9898 };
9899 if sym == u32::MAX {
9900 continue; // tombstoned
9901 }
9902 for &nbr in tv.neighbors(etype_sym, Direction::Out, id).as_ref() {
9903 let Some(dst_key) = self.ids.key_of(nbr) else {
9904 continue;
9905 };
9906 let weight = weight_prop.and_then(|prop| match ep.get(etype_sym, id, nbr, prop) {
9907 Some(Value::Float(f)) => Some(f),
9908 Some(Value::Int(i)) => Some(i as f64),
9909 _ => None,
9910 });
9911 out.push((key.to_string(), dst_key.to_string(), weight));
9912 }
9913 }
9914 out.sort_by(|a, b| a.0.cmp(&b.0).then(a.1.cmp(&b.1)));
9915 out
9916 }
9917
9918 pub fn nodes_with_label(&self, label: &str) -> Vec<NodeRef<'_, F>> {
9919 self.view()
9920 .nodes_with_label(label)
9921 .into_iter()
9922 .map(|id| NodeRef { db: self, id })
9923 .collect()
9924 }
9925
9926 pub fn find_nodes(&self, label: &str, filter: &Filter) -> Vec<NodeRef<'_, F>> {
9927 let view = self.view();
9928 view.nodes_with_label(label)
9929 .into_iter()
9930 .filter(|&id| {
9931 eval_filter(filter, &|field| {
9932 view.prop(id, field).map(|vr| vr.into_value())
9933 })
9934 })
9935 .map(|id| NodeRef { db: self, id })
9936 .collect()
9937 }
9938
9939 /// Returns `true` if any approximate (HNSW) VectorSimilar rule covers
9940 /// `field`. Use as a capability probe: when `true`, `find_similar_vector`
9941 /// with `label = None` will use the native ANN path rather than the O(n)
9942 /// brute-force scan.
9943 pub fn has_vector_rule(&self, field: &str) -> bool {
9944 self.engine.hnsw_has_rule(field)
9945 }
9946
9947 /// How many HNSW graphs this handle has built from scratch since it was
9948 /// opened (one per side of an approximate rule).
9949 ///
9950 /// An open that restored every graph from the snapshot reports `0`.
9951 /// Exposed for tests that assert the open path reuses the persisted index
9952 /// rather than rebuilding it; not part of the stable surface.
9953 #[doc(hidden)]
9954 pub fn hnsw_build_count(&self) -> u64 {
9955 self.engine.hnsw_build_count()
9956 }
9957
9958 /// How many rules this handle still holds a lazily-decoded HNSW graph for.
9959 ///
9960 /// Zero before the first ANN query on a clean open, and again once the
9961 /// live indexes own the graphs. See [`core_rules::RuleEngine::lazy_hnsw_len`].
9962 /// Exposed for tests that assert the lazy copies are released; not part of
9963 /// the stable surface.
9964 #[doc(hidden)]
9965 pub fn lazy_hnsw_len(&self) -> usize {
9966 self.engine.lazy_hnsw_len()
9967 }
9968
9969 /// Find nodes whose `field` vector is most similar to `q` (cosine
9970 /// similarity), returning up to `k` results with similarity ≥ `min`,
9971 /// sorted descending.
9972 ///
9973 /// When `label` is `None` the search spans all labels (via
9974 /// `hnsw_search_any_dst` or a full brute-force scan); when `label` is
9975 /// `Some(lbl)` it restricts to nodes with that label.
9976 ///
9977 /// Uses the HNSW index when one is available (fast path); otherwise falls
9978 /// back to an O(n) brute-force scan.
9979 ///
9980 /// **The index supplies candidates, never scores.** Its own distances are
9981 /// `f32` (accurate to ~1e-6, so an exact duplicate scores 0.9999999), so
9982 /// every candidate is re-scored from the `f64` property vectors by
9983 /// [`exact_vector_similarity`] before `min`, the ordering and the reported
9984 /// score are decided. `k + VECTOR_RESCORE_MARGIN` candidates are fetched so
9985 /// the re-ordering cannot drop a true top-`k` member; see that constant for
9986 /// the rule. The score a caller receives is therefore the same number the
9987 /// brute-force path would have produced, to `f64` precision, and `min = 1.0`
9988 /// finds an exact duplicate.
9989 pub fn find_similar_vector(
9990 &self,
9991 field: &str,
9992 label: Option<&str>,
9993 q: &[f64],
9994 k: usize,
9995 min: f64,
9996 ) -> Vec<(String, f64)> {
9997 self.find_similar_vector_filtered(field, label, q, k, min, None, None, false)
9998 .expect("find_similar_vector_filtered is infallible without where_")
9999 }
10000
10001 /// Like [`find_similar_vector`] but restricts results to nodes visible in
10002 /// `mask`. Hidden nodes never appear in results; the mask is applied
10003 /// **before** k-truncation so a caller still receives up to `k` visible
10004 /// hits.
10005 ///
10006 /// # HNSW path (widening beam)
10007 ///
10008 /// When an HNSW index covers the request, the beam starts at an over-fetch
10009 /// of `k × n / |visible|` (plus the rescore margin) when the mask's
10010 /// selectivity is known from the index length, otherwise at `k` plus that
10011 /// margin. If fewer than `k` visible candidates remain after the mask and
10012 /// `min` filter, the beam doubles — the same ×2 loop exact `VectorSimilar`
10013 /// rules use, capped at `ef_max()` (`EF_MAX` = 4,096). Reaching the cap,
10014 /// or a beam that comes back short of its own width, falls through to the
10015 /// exhaustive masked scan rather than returning a short result.
10016 ///
10017 /// Every surviving candidate is re-scored from the `f64` property vectors,
10018 /// exactly as [`find_similar_vector`] does and for the same reason.
10019 ///
10020 /// # Brute-force path
10021 ///
10022 /// When no HNSW index covers the request, or the beam cannot admit `k`
10023 /// hits, the function builds a masked [`GraphView`] so that `nodes_all` /
10024 /// `nodes_with_label` return only visible nodes, guaranteeing exact `k`
10025 /// results (or all visible nodes if fewer than `k` exist).
10026 pub fn find_similar_vector_masked(
10027 &self,
10028 field: &str,
10029 label: Option<&str>,
10030 q: &[f64],
10031 k: usize,
10032 min: f64,
10033 mask: &crate::mask::NodeMask,
10034 ) -> Vec<(String, f64)> {
10035 self.find_similar_vector_filtered(field, label, q, k, min, Some(mask), None, false)
10036 .expect("find_similar_vector_filtered is infallible without where_")
10037 }
10038
10039 /// Exact or ANN kNN with optional key-list `mask` and property `where_`.
10040 ///
10041 /// `where_` present and failing [`PropPredicate::validate_named`] `"where"`
10042 /// → `QueryError`. `exact=true` or `where_=Some` skip HNSW and GEMM-brute
10043 /// the candidate set (`label ∩ mask ∩ holds(where)`). `mask` alone still
10044 /// uses HNSW when an index covers the field.
10045 #[allow(clippy::too_many_arguments)]
10046 pub fn find_similar_vector_filtered(
10047 &self,
10048 field: &str,
10049 label: Option<&str>,
10050 q: &[f64],
10051 k: usize,
10052 min: f64,
10053 mask: Option<&crate::mask::NodeMask>,
10054 where_: Option<&PropPredicate>,
10055 exact: bool,
10056 ) -> Result<Vec<(String, f64)>> {
10057 self.find_similar_vector_as(
10058 field,
10059 label,
10060 q,
10061 k,
10062 min,
10063 mask,
10064 where_,
10065 exact,
10066 ExactnessCaller::Vector,
10067 )
10068 }
10069
10070 /// [`find_similar_vector_filtered`](Self::find_similar_vector_filtered)
10071 /// with the caller shape named, so the exactness warning can advise the
10072 /// signature that actually reached it. Everything else is identical.
10073 #[allow(clippy::too_many_arguments)]
10074 fn find_similar_vector_as(
10075 &self,
10076 field: &str,
10077 label: Option<&str>,
10078 q: &[f64],
10079 k: usize,
10080 min: f64,
10081 mask: Option<&crate::mask::NodeMask>,
10082 where_: Option<&PropPredicate>,
10083 exact: bool,
10084 caller: ExactnessCaller,
10085 ) -> Result<Vec<(String, f64)>> {
10086 if let Some(pred) = where_ {
10087 pred.validate_named("where")
10088 .map_err(|detail| GraphError::QueryError { detail })?;
10089 }
10090
10091 // Ensure any HNSW blobs retained from the snapshot are deserialized
10092 // before the first ANN query on a clean-open (no-WAL) path. The
10093 // section read has to come first: on a clean open nothing else has
10094 // called it, so without it `retained_hnsw_blobs` is empty,
10095 // `ensure_hnsw_loaded` caches an empty map in its `OnceLock`, and every
10096 // approximate query on the handle runs brute force — correct results,
10097 // silently off the index. Both calls are idempotent and cheap once hot.
10098 self.ensure_v8_base_sections_loaded();
10099 self.engine.ensure_hnsw_loaded();
10100 let norm: f64 = q.iter().map(|x| x * x).sum::<f64>().sqrt();
10101 if norm == 0.0 {
10102 return Ok(vec![]);
10103 }
10104 if let Some(m) = mask {
10105 if k == 0 || m.is_empty() {
10106 return Ok(vec![]);
10107 }
10108 }
10109 let q_unit: Vec<f64> = q.iter().map(|x| x / norm).collect();
10110
10111 // `where` implies exact: a predicate must not ride a silent ANN.
10112 let skip_hnsw = exact || where_.is_some();
10113 if !skip_hnsw {
10114 if let Some(mask) = mask {
10115 if let Some(out) =
10116 self.find_similar_hnsw_masked(field, label, &q_unit, k, min, mask, caller)
10117 {
10118 return Ok(out);
10119 }
10120 } else if let Some(out) = self.find_similar_hnsw(field, label, &q_unit, k, min) {
10121 return Ok(out);
10122 }
10123 }
10124
10125 let view = match mask {
10126 Some(m) => self.view_masked(m),
10127 None => self.view(),
10128 };
10129 let candidate_ids = Self::vector_candidates(&view, label, where_);
10130 Ok(self.brute_vector_hits(&view, candidate_ids, field, &q_unit, k, min))
10131 }
10132
10133 /// Unmasked HNSW path. `None` when no populated index covers the request.
10134 fn find_similar_hnsw(
10135 &self,
10136 field: &str,
10137 label: Option<&str>,
10138 q_unit: &[f64],
10139 k: usize,
10140 min: f64,
10141 ) -> Option<Vec<(String, f64)>> {
10142 // Try HNSW fast path.
10143 // `None` label searches across all VectorSimilar rules covering `field`
10144 // (merging their results); `Some(lbl)` restricts to rules whose
10145 // dst_label matches. Returns `None` when no populated HNSW index
10146 // covers the request — the O(n) brute-force fallback handles that case.
10147 let over_k = k.saturating_add(VECTOR_RESCORE_MARGIN);
10148 let hits = match label {
10149 Some(lbl) => self.engine.hnsw_search_dst(field, lbl, q_unit, over_k)?,
10150 None => self.engine.hnsw_search_any_dst(field, q_unit, over_k)?,
10151 };
10152 // Candidates only: the index's `f32` similarity is discarded and
10153 // each hit is re-scored against the `f64` vectors.
10154 let view = self.view();
10155 let mut out: Vec<(String, f64)> = hits
10156 .into_iter()
10157 .filter_map(|(id, _)| {
10158 let sim = exact_vector_similarity(&view, id, field, q_unit)?;
10159 if sim < min {
10160 return None;
10161 }
10162 Some((self.ids.key_of(id)?.to_string(), sim))
10163 })
10164 .collect();
10165 out.sort_by(|a, b| {
10166 b.1.partial_cmp(&a.1)
10167 .unwrap_or(std::cmp::Ordering::Equal)
10168 .then_with(|| a.0.cmp(&b.0))
10169 });
10170 out.truncate(k);
10171 Some(out)
10172 }
10173
10174 /// Masked HNSW widening beam. `None` when no index covers the request or
10175 /// the beam cannot admit `k` visible hits (caller falls through to brute).
10176 #[allow(clippy::too_many_arguments)]
10177 fn find_similar_hnsw_masked(
10178 &self,
10179 field: &str,
10180 label: Option<&str>,
10181 q_unit: &[f64],
10182 k: usize,
10183 min: f64,
10184 mask: &crate::mask::NodeMask,
10185 caller: ExactnessCaller,
10186 ) -> Option<Vec<(String, f64)>> {
10187 let index_len = match label {
10188 Some(lbl) => self.engine.hnsw_dst_len(field, lbl, q_unit.len()),
10189 None => self.engine.hnsw_any_dst_len(field, q_unit.len()),
10190 };
10191 let n = index_len?;
10192 // The `?` above is the coverage test: past it, an index exists and this
10193 // masked, non-exact call is about to ride it.
10194 self.note_ambiguous_exactness(field, label, caller);
10195 // Same ceiling the exact-rule widening loop in `hnsw_candidates`
10196 // consults — including the `with_ef_max` test hook.
10197 let cap = ef_max();
10198 let visible = mask.len();
10199 let mut ef = k.saturating_add(VECTOR_RESCORE_MARGIN);
10200 if visible > 0 && n > 0 {
10201 let over = k
10202 .saturating_mul(n)
10203 .div_ceil(visible)
10204 .saturating_add(VECTOR_RESCORE_MARGIN);
10205 ef = ef.max(over);
10206 }
10207 loop {
10208 let hits = match label {
10209 Some(lbl) => self
10210 .engine
10211 .hnsw_search_dst_with_ef(field, lbl, q_unit, ef, ef),
10212 None => self
10213 .engine
10214 .hnsw_search_any_dst_with_ef(field, q_unit, ef, ef),
10215 };
10216 let hits = hits?;
10217 let full = hits.len() == ef;
10218 let mut out = self.score_masked_hnsw_hits(&hits, field, q_unit, min, mask);
10219 if out.len() >= k {
10220 out.truncate(k);
10221 return Some(out);
10222 }
10223 // Short of its width (frontier exhausted) or at the ceiling:
10224 // a wider beam reaches nothing new, so the scan answers.
10225 if !full || ef >= cap {
10226 return None;
10227 }
10228 ef = ef.saturating_mul(2);
10229 }
10230 }
10231
10232 /// Say once, per `(field, label)` index and caller shape, that a masked
10233 /// search is answering approximately.
10234 ///
10235 /// A mask narrows *which nodes may be returned*. It does not choose a
10236 /// kernel — `exact=true` and a `where=` predicate do, and nothing else
10237 /// does. A caller who needed exact answers, passed `mask=` alone, and read
10238 /// the mask as a promise of exhaustiveness gets a correct-looking
10239 /// approximate answer and no signal at all; that is a silent wrong answer,
10240 /// and it has cost an integration team real time.
10241 ///
10242 /// The fix is a question, not a behaviour change. Making a mask imply
10243 /// `exact` would turn every existing masked caller's ANN into an O(n) GEMM
10244 /// without asking them, which is a worse trade than the ambiguity.
10245 ///
10246 /// Printed once per index for the reason the dimension-mismatch skip in
10247 /// `core_rules::hnsw` is: a line on every call is a line callers learn to
10248 /// scroll past.
10249 ///
10250 /// `caller` decides the advice. The same leg is reached from two signatures
10251 /// and only one of them has an `exact` argument to pass; see
10252 /// [`ExactnessCaller`].
10253 fn note_ambiguous_exactness(&self, field: &str, label: Option<&str>, caller: ExactnessCaller) {
10254 let entry = (field.to_string(), label.unwrap_or("").to_string(), caller);
10255 let first = match self.warned_ambiguous_exactness.lock() {
10256 Ok(mut seen) => seen.insert(entry),
10257 Err(poisoned) => poisoned.into_inner().insert(entry),
10258 };
10259 if !first {
10260 return;
10261 }
10262 AMBIGUOUS_EXACTNESS_WARNS.with(|c| c.set(c.get().saturating_add(1)));
10263 let which = match label {
10264 Some(lbl) => format!(" (label `{lbl}`)"),
10265 None => String::new(),
10266 };
10267 let subject = caller.subject();
10268 let advice = caller.advice();
10269 let line = format!(
10270 "mushroomdb: {subject} on field `{field}`{which} is answering \
10271 approximately. A mask narrows which nodes may be returned; it does not \
10272 change which kernel runs, and an index covers this field. For an exact \
10273 answer over the same visible candidate set, {advice} Further masked \
10274 searches of this shape on this index are silent."
10275 );
10276 AMBIGUOUS_EXACTNESS_LAST.with(|c| *c.borrow_mut() = Some(line.clone()));
10277 eprintln!("{line}");
10278 }
10279
10280 /// `label ∩ mask ∩ holds(where)`. Index fast path when `label` is `Some`
10281 /// and `(label, where.field)` is enabled; otherwise scan with `visible()`.
10282 fn vector_candidates(
10283 view: &GraphView<'_>,
10284 label: Option<&str>,
10285 where_: Option<&PropPredicate>,
10286 ) -> Vec<u32> {
10287 if let (Some(lbl), Some(pred)) = (label, where_) {
10288 let indexed = view
10289 .prop_index
10290 .is_some_and(|idx| idx.is_enabled(lbl, &pred.field));
10291 if indexed {
10292 match (&pred.eq, &pred.in_) {
10293 (Some(eq), None) => {
10294 if let Some(ids) = view.nodes_with_prop(lbl, &pred.field, eq) {
10295 return ids;
10296 }
10297 }
10298 (None, Some(allowed)) => {
10299 let mut seen = HashSet::new();
10300 let mut out = Vec::new();
10301 for v in allowed {
10302 if let Some(ids) = view.nodes_with_prop(lbl, &pred.field, v) {
10303 for id in ids {
10304 if seen.insert(id) {
10305 out.push(id);
10306 }
10307 }
10308 }
10309 }
10310 return out;
10311 }
10312 _ => {}
10313 }
10314 }
10315 }
10316
10317 let mut ids: Vec<u32> = match label {
10318 Some(lbl) => view
10319 .nodes_with_label(lbl)
10320 .into_iter()
10321 .filter(|&id| view.visible(id))
10322 .collect(),
10323 None => view.nodes_all(),
10324 };
10325 if let Some(pred) = where_ {
10326 ids.retain(|&id| match view.prop(id, &pred.field) {
10327 None => pred.holds(None),
10328 Some(vr) => pred.holds(Some(vr.as_value())),
10329 });
10330 }
10331 ids
10332 }
10333
10334 /// Exact brute kNN: pack candidates at `q_unit`'s dim, GEMV, keep
10335 /// `score >= min`, sort `(sim desc, key asc)`, truncate to `k`.
10336 fn brute_vector_hits(
10337 &self,
10338 view: &GraphView<'_>,
10339 candidate_ids: impl IntoIterator<Item = u32>,
10340 field: &str,
10341 q_unit: &[f64],
10342 k: usize,
10343 min: f64,
10344 ) -> Vec<(String, f64)> {
10345 let rows: Vec<(u32, std::borrow::Cow<'_, [f64]>)> = candidate_ids
10346 .into_iter()
10347 .filter_map(|id| crate::exact_knn::vector_f64(view, id, field).map(|v| (id, v)))
10348 .collect();
10349 let packed =
10350 crate::exact_knn::pack(rows.iter().map(|(id, v)| (*id, v.as_ref())), q_unit.len());
10351 let scores = crate::exact_knn::gemv(&packed, q_unit);
10352 let mut scored: Vec<(String, f64)> = packed
10353 .ids
10354 .iter()
10355 .zip(scores.iter())
10356 .filter_map(|(&id, &sim)| {
10357 if sim < min {
10358 return None;
10359 }
10360 let key = self.ids.key_of(id)?.to_string();
10361 Some((key, sim))
10362 })
10363 .collect();
10364 scored.sort_by(|a, b| {
10365 b.1.partial_cmp(&a.1)
10366 .unwrap_or(std::cmp::Ordering::Equal)
10367 .then_with(|| a.0.cmp(&b.0))
10368 });
10369 scored.truncate(k);
10370 scored
10371 }
10372
10373 /// Exact cosine top-k for each key in `keys`, scored only against `keys`.
10374 ///
10375 /// `min` is cosine similarity in [-1, 1], inclusive (`score >= min`), the
10376 /// same unit and inequality as `find_similar_vector`. Self-matches are
10377 /// excluded. Unknown keys, keys with no `field`, zero-norm or wrong-dim
10378 /// embeddings are omitted as both query and candidate. Duplicate keys are
10379 /// collapsed, first-seen order. Empty `keys` → empty `Ok(vec![])`. Never
10380 /// uses HNSW. `n > PAIRWISE_MAX_N` → `QueryError`.
10381 #[allow(clippy::type_complexity)]
10382 pub fn pairwise_similar(
10383 &self,
10384 keys: &[&str],
10385 field: &str,
10386 k: usize,
10387 min: f64,
10388 ) -> Result<Vec<(String, Vec<(String, f64)>)>> {
10389 let mut seen = HashSet::new();
10390 let mut unique_ids = Vec::new();
10391 for key in keys {
10392 let Some(id) = self.ids.get(key) else {
10393 continue;
10394 };
10395 if seen.insert(id) {
10396 unique_ids.push(id);
10397 }
10398 }
10399 let max_n = crate::exact_knn::pairwise_max_n();
10400 if unique_ids.len() > max_n {
10401 return Err(GraphError::QueryError {
10402 detail: format!(
10403 "pairwise_similar: n={} exceeds PAIRWISE_MAX_N ({max_n})",
10404 unique_ids.len()
10405 ),
10406 });
10407 }
10408 if unique_ids.is_empty() {
10409 return Ok(Vec::new());
10410 }
10411
10412 let view = self.view();
10413 let mut rows: Vec<(u32, std::borrow::Cow<'_, [f64]>)> = Vec::new();
10414 let mut counts: HashMap<usize, usize> = HashMap::new();
10415 for id in unique_ids {
10416 let Some(v) = crate::exact_knn::vector_f64(&view, id, field) else {
10417 continue;
10418 };
10419 let norm: f64 = v.iter().map(|x| x * x).sum::<f64>().sqrt();
10420 if norm == 0.0 {
10421 continue;
10422 }
10423 *counts.entry(v.len()).or_default() += 1;
10424 rows.push((id, v));
10425 }
10426 if rows.is_empty() {
10427 return Ok(Vec::new());
10428 }
10429 let dim = counts
10430 .into_iter()
10431 .max_by_key(|&(d, c)| (c, d))
10432 .map(|(d, _)| d)
10433 .expect("rows non-empty");
10434 let packed = crate::exact_knn::pack(rows.iter().map(|(id, v)| (*id, v.as_ref())), dim);
10435 let n = packed.ids.len();
10436 if n == 0 {
10437 return Ok(Vec::new());
10438 }
10439 let src_keys: Vec<String> = packed
10440 .ids
10441 .iter()
10442 .map(|&id| self.ids.key_of(id).unwrap_or("").to_string())
10443 .collect();
10444
10445 let mut out = Vec::with_capacity(n);
10446 if n <= crate::exact_knn::pairwise_gram_max() {
10447 let sims = crate::exact_knn::gram(&packed);
10448 for i in 0..n {
10449 out.push(Self::topk_from_row(
10450 &src_keys,
10451 i,
10452 &sims[i * n..(i + 1) * n],
10453 k,
10454 min,
10455 ));
10456 }
10457 } else {
10458 for i in 0..n {
10459 let row = &packed.data[i * packed.dim..(i + 1) * packed.dim];
10460 let scores = crate::exact_knn::gemv(&packed, row);
10461 out.push(Self::topk_from_row(&src_keys, i, &scores, k, min));
10462 }
10463 }
10464 Ok(out)
10465 }
10466
10467 /// [`pairwise_similar`](Self::pairwise_similar) over the keys the mask
10468 /// admits — intersected **before** the matmul, never filtered after it.
10469 ///
10470 /// A hidden vector packed into the Gram is a row every visible key is
10471 /// scored against. It can take a visible neighbour's place in the top-`k`,
10472 /// and because the packed dimension is a majority vote over the candidate
10473 /// rows it can decide whether a visible pair is scored at all. Dropping
10474 /// hidden names from the finished answer leaves both effects standing, so
10475 /// the intersection happens first and the answer is byte-for-byte the one
10476 /// `pairwise_similar` gives for the visible keys alone.
10477 ///
10478 /// The caps therefore measure the **post-filter** count: a key set over
10479 /// [`PAIRWISE_MAX_N`](crate::PAIRWISE_MAX_N) unscoped can come under it
10480 /// scoped and succeed, because the work the cap refuses is work this call
10481 /// no longer does. A filtered count still over the cap is still refused.
10482 #[allow(clippy::type_complexity)]
10483 pub fn pairwise_similar_scoped(
10484 &self,
10485 keys: &[&str],
10486 field: &str,
10487 k: usize,
10488 min: f64,
10489 mask: &crate::mask::NodeMask,
10490 ) -> Result<Vec<(String, Vec<(String, f64)>)>> {
10491 let visible: Vec<&str> = keys
10492 .iter()
10493 .copied()
10494 .filter(|key| mask.contains_node(self, key))
10495 .collect();
10496 self.pairwise_similar(&visible, field, k, min)
10497 }
10498
10499 /// Neighbours of packed row `i`: drop self, keep `score >= min`, sort
10500 /// `(sim desc, key asc)`, truncate to `k`. Packed srcs with no survivors
10501 /// still appear as `(src, [])`.
10502 fn topk_from_row(
10503 src_keys: &[String],
10504 i: usize,
10505 scores: &[f64],
10506 k: usize,
10507 min: f64,
10508 ) -> (String, Vec<(String, f64)>) {
10509 let mut neigh: Vec<(String, f64)> = scores
10510 .iter()
10511 .enumerate()
10512 .filter_map(|(j, &sim)| {
10513 if i == j || sim < min {
10514 return None;
10515 }
10516 Some((src_keys[j].clone(), sim))
10517 })
10518 .collect();
10519 neigh.sort_by(|a, b| {
10520 b.1.partial_cmp(&a.1)
10521 .unwrap_or(std::cmp::Ordering::Equal)
10522 .then_with(|| a.0.cmp(&b.0))
10523 });
10524 neigh.truncate(k);
10525 (src_keys[i].clone(), neigh)
10526 }
10527
10528 /// Re-score HNSW candidates from the `f64` vectors, drop hidden / below-`min`
10529 /// hits, order by score then key. The index's own `f32` similarity is discarded.
10530 fn score_masked_hnsw_hits(
10531 &self,
10532 hits: &[(u32, f64)],
10533 field: &str,
10534 q_unit: &[f64],
10535 min: f64,
10536 mask: &crate::mask::NodeMask,
10537 ) -> Vec<(String, f64)> {
10538 let view = self.view_masked(mask);
10539 let mut out: Vec<(String, f64)> = hits
10540 .iter()
10541 .copied()
10542 .filter(|&(id, _)| mask.contains_id(id))
10543 .filter_map(|(id, _)| {
10544 let sim = exact_vector_similarity(&view, id, field, q_unit)?;
10545 if sim < min {
10546 return None;
10547 }
10548 Some((self.ids.key_of(id)?.to_string(), sim))
10549 })
10550 .collect();
10551 out.sort_by(|a, b| {
10552 b.1.partial_cmp(&a.1)
10553 .unwrap_or(std::cmp::Ordering::Equal)
10554 .then_with(|| a.0.cmp(&b.0))
10555 });
10556 out
10557 }
10558
10559 /// Read a single property from an edge.
10560 ///
10561 /// Returns `None` when the edge does not exist, the field is absent, or any
10562 /// of the string keys cannot be resolved to interned ids. Only edge props
10563 /// written by rules (weight fields) are accessible without a `set_edge_prop`
10564 /// binding; topology-only edges (no props set) return `None` for every field.
10565 pub fn get_edge_prop(
10566 &self,
10567 edge_type: &str,
10568 src_key: &str,
10569 dst_key: &str,
10570 field: &str,
10571 ) -> Option<Value> {
10572 let etype = self.syms.get(edge_type)?;
10573 let src = self.ids.get(src_key)?;
10574 let dst = self.ids.get(dst_key)?;
10575 self.edge_props_view().get(etype, src, dst, field)
10576 }
10577
10578 /// Lex → parse → plan → execute `cypher` over a read-only view.
10579 /// Every pipeline `Err(String)` becomes `GraphError::QueryError` with a
10580 /// stage prefix (`lex:` / `parse:` / `plan:` / `execute:`).
10581 pub fn query(&self, cypher: &str, params: &BTreeMap<String, Value>) -> Result<ResultSet> {
10582 let tokens = lex(cypher).map_err(|e| GraphError::QueryError {
10583 detail: format!("lex: {e}"),
10584 })?;
10585 let union = parse_read(&tokens).map_err(|e| GraphError::QueryError {
10586 detail: format!("parse: {e}"),
10587 })?;
10588 let t0 = std::time::Instant::now();
10589 let result = execute_union(&self.view(), &union, &Params(params)).map_err(|e| {
10590 GraphError::QueryError {
10591 detail: format!("execute: {e}"),
10592 }
10593 });
10594 let elapsed_ms = t0.elapsed().as_millis() as u64;
10595 let threshold = self.slow_query_threshold_ms;
10596 if threshold > 0 && elapsed_ms >= threshold {
10597 eprintln!("[mushroomdb] slow query ({elapsed_ms}ms): {cypher}");
10598 let entry = SlowQueryEntry {
10599 ms: elapsed_ms,
10600 query: cypher.to_string(),
10601 at_commit: self.commit_seq,
10602 };
10603 if let Ok(mut log) = self.slow_queries.lock() {
10604 if log.entries.len() == SLOW_QUERY_RING_CAP {
10605 log.entries.pop_front();
10606 }
10607 log.entries.push_back(entry);
10608 log.total += 1;
10609 }
10610 }
10611 result
10612 }
10613
10614 /// Convenience entry-point that accepts a slice of `(name, value)` pairs
10615 /// instead of a pre-built `BTreeMap`. Equivalent to building the map and
10616 /// calling [`GraphDb::query`].
10617 pub fn query_with_params(&self, cypher: &str, params: &[(&str, Value)]) -> Result<ResultSet> {
10618 let map: BTreeMap<String, Value> = params
10619 .iter()
10620 .map(|(k, v)| (k.to_string(), v.clone()))
10621 .collect();
10622 self.query(cypher, &map)
10623 }
10624
10625 /// Execute a Cypher write statement (CREATE / MATCH…SET / MATCH…DELETE / MERGE).
10626 ///
10627 /// All mutations flow through the same `insert_node` / `set_prop` /
10628 /// `delete_edge` / `insert_edge` path as the Rust API so the rule engine
10629 /// fires and the WAL captures everything with one fsync per statement.
10630 ///
10631 /// Returns a one-row [`ResultSet`] with columns `created`, `properties_set`,
10632 /// and `deleted` matching the write-result contract.
10633 ///
10634 /// **Mutation routing**: mutations are collected into a single
10635 /// [`BatchBuilder`] and committed atomically (one WAL `Batch` frame, one
10636 /// fsync). The MATCH phase for SET/DELETE uses a read-only `execute` call
10637 /// over `self.view()` — the borrow is dropped before the batch is opened.
10638 ///
10639 /// **Limitations (v1)**:
10640 /// - SET RHS must be a literal, `$param`, or arithmetic; bare property copy → named error.
10641 /// - `DETACH DELETE n` → calls `delete_node` for each matched node (removes all edges).
10642 /// - Bare `DELETE n` → error if n has any incident edges; succeeds for isolated nodes.
10643 /// - MERGE supports `ON CREATE SET` / `ON MATCH SET` in the same write batch.
10644 /// - Deleting a derived edge → named error "cannot delete derived edge".
10645 pub fn query_write(
10646 &mut self,
10647 cypher: &str,
10648 params: &BTreeMap<String, Value>,
10649 ) -> Result<ResultSet> {
10650 let tokens = lex(cypher).map_err(|e| GraphError::QueryError {
10651 detail: format!("lex: {e}"),
10652 })?;
10653 let stmt = parse_write(&tokens).map_err(|e| GraphError::QueryError {
10654 detail: format!("parse: {e}"),
10655 })?;
10656 self.exec_write_stmt(stmt, params)
10657 }
10658
10659 fn exec_write_stmt(
10660 &mut self,
10661 stmt: WriteStatement,
10662 params: &BTreeMap<String, Value>,
10663 ) -> Result<ResultSet> {
10664 match stmt {
10665 WriteStatement::Create(s) => self.exec_create(s, params),
10666 WriteStatement::MatchSet(s) => self.exec_match_set(s, params),
10667 WriteStatement::MatchDelete(s) => self.exec_match_delete(s, params),
10668 WriteStatement::MatchDeleteNode(s) => self.exec_match_delete_node(s, params),
10669 WriteStatement::Merge(s) => self.exec_merge(s, params),
10670 }
10671 }
10672
10673 fn exec_create(
10674 &mut self,
10675 stmt: core_query::cypher::CreateStmt,
10676 params: &BTreeMap<String, Value>,
10677 ) -> Result<ResultSet> {
10678 // Extract the node key from props: require a string-valued `id` field.
10679 let mut var_to_key: BTreeMap<String, String> = BTreeMap::new();
10680 for node in &stmt.nodes {
10681 let var = node.var.as_deref().unwrap_or("_cn0");
10682 let key = node
10683 .props
10684 .iter()
10685 .find(|(f, _)| f == "id")
10686 .and_then(|(_, v)| {
10687 if let Value::Str(s) = v {
10688 Some(s.clone())
10689 } else {
10690 None
10691 }
10692 })
10693 .ok_or_else(|| GraphError::QueryError {
10694 detail: format!(
10695 "CREATE node ({}:{}) requires a string 'id' property",
10696 var, node.label
10697 ),
10698 })?;
10699 var_to_key.insert(var.to_string(), key);
10700 }
10701
10702 let mut batch = self.batch();
10703 let mut created: usize = 0;
10704 for node in &stmt.nodes {
10705 let var = node.var.as_deref().unwrap_or("_cn0");
10706 let key = &var_to_key[var];
10707 batch.insert_node(&node.label, key, node.props.clone());
10708 created += 1;
10709 }
10710 for edge in &stmt.edges {
10711 let src_key = var_to_key
10712 .get(&edge.src_var)
10713 .ok_or_else(|| GraphError::QueryError {
10714 detail: format!("CREATE edge src variable '{}' is not bound", edge.src_var),
10715 })?;
10716 let dst_key = var_to_key
10717 .get(&edge.dst_var)
10718 .ok_or_else(|| GraphError::QueryError {
10719 detail: format!("CREATE edge dst variable '{}' is not bound", edge.dst_var),
10720 })?;
10721 batch.insert_edge(&edge.etype, src_key, dst_key);
10722 }
10723 batch.commit()?;
10724
10725 // Optional RETURN clause: project created bindings as a read result.
10726 if let Some(returns) = stmt.returns {
10727 // Each created node is looked up by its key via a separate MATCH pattern.
10728 // Multiple single-node patterns cross-join to produce 1 output row with
10729 // all variables bound (each pattern returns exactly 1 row).
10730 let patterns: Vec<Pattern> = stmt
10731 .nodes
10732 .iter()
10733 .map(|node| {
10734 let var = node.var.as_deref().unwrap_or("_cn0");
10735 let key = var_to_key[var].clone();
10736 Pattern {
10737 start: NodePat {
10738 var: Some(var.to_string()),
10739 label: Some(node.label.clone()),
10740 props: vec![("id".to_string(), Operand::Lit(Value::Str(key)))],
10741 },
10742 chain: vec![],
10743 shortest: false,
10744 }
10745 })
10746 .collect();
10747 let q = Query {
10748 matches: patterns,
10749 optional_clauses: vec![],
10750 where_expr: None,
10751 unwinds: vec![],
10752 post_unwind_where: None,
10753 stages: vec![],
10754 returns,
10755 distinct: false,
10756 order_by: vec![],
10757 skip: None,
10758 limit: None,
10759 };
10760 let ops = plan(&q).map_err(|e| GraphError::QueryError {
10761 detail: format!("plan: {e}"),
10762 })?;
10763 return execute(&self.view(), &ops, &Params(params)).map_err(|e| {
10764 GraphError::QueryError {
10765 detail: format!("execute: {e}"),
10766 }
10767 });
10768 }
10769
10770 let mut rs = write_result_set();
10771 rs.push_row(vec![
10772 Some(Value::Int(created as i64)),
10773 Some(Value::Int(0)),
10774 Some(Value::Int(0)),
10775 ]);
10776 Ok(rs)
10777 }
10778
10779 fn exec_match_set(
10780 &mut self,
10781 stmt: core_query::cypher::MatchSetStmt,
10782 params: &BTreeMap<String, Value>,
10783 ) -> Result<ResultSet> {
10784 let project_returns = stmt.returns.clone();
10785 // Collect unique node vars targeted by SET clauses, plus RETURN bindings
10786 // so the post-write projection can look them up by key.
10787 let mut set_vars: Vec<String> = Vec::new();
10788 for s in &stmt.sets {
10789 if !set_vars.contains(&s.var) {
10790 set_vars.push(s.var.clone());
10791 }
10792 }
10793 let rel_vars = pattern_rel_vars(&stmt.matches);
10794 // `count` is the engine's, on an edge: it is the insert-count §5.13
10795 // maintains, and a `SET` that overwrote it would make the number mean
10796 // whatever the last writer said rather than how many times the pair was
10797 // inserted. Refused by name here, before the match runs, so the caller
10798 // is told what is actually wrong instead of meeting the executor's
10799 // generic "did not resolve to a node key" — and so the answer does not
10800 // depend on whether the pattern happened to match a row. The same name
10801 // on a *node* is an ordinary property and is untouched.
10802 for s in &stmt.sets {
10803 if s.field == EDGE_COUNT_PROP && rel_vars.iter().any(|r| r == &s.var) {
10804 return Err(GraphError::QueryError {
10805 detail: format!(
10806 "cannot SET {}.{EDGE_COUNT_PROP}: `{EDGE_COUNT_PROP}` is a reserved edge \
10807 property holding the pair's insert count",
10808 s.var
10809 ),
10810 });
10811 }
10812 }
10813 let mut lookup_vars = set_vars.clone();
10814 for v in pattern_node_vars(&stmt.matches) {
10815 add_var(&mut lookup_vars, &v);
10816 }
10817 if let Some(ref returns) = project_returns {
10818 for v in ret_node_vars(returns) {
10819 if !rel_vars.iter().any(|r| r == &v) {
10820 add_var(&mut lookup_vars, &v);
10821 }
10822 }
10823 }
10824
10825 // Synthesize a read query: MATCH … WHERE … RETURN <lookup_vars>, <set_values…>
10826 // SET values are projected as ScalarExpr items so that arithmetic expressions
10827 // (e.g. `SET n.score = n.score * 1.5`) are evaluated in the matched-row context.
10828 let mut set_returns: Vec<RetItem> = lookup_vars
10829 .iter()
10830 .map(|v| RetItem {
10831 value: RetVal::Var(v.clone()),
10832 alias: None,
10833 })
10834 .collect();
10835 // One computed column per SET clause; alias is `__sv_<i>`.
10836 let set_val_cols: Vec<String> = stmt
10837 .sets
10838 .iter()
10839 .enumerate()
10840 .map(|(i, _)| format!("__sv_{i}"))
10841 .collect();
10842 for (sc, col) in stmt.sets.iter().zip(&set_val_cols) {
10843 set_returns.push(RetItem {
10844 value: RetVal::ScalarExpr(sc.value.clone()),
10845 alias: Some(col.clone()),
10846 });
10847 }
10848 // Capture relationship types while r is bound; SET does not change them.
10849 for r in &rel_vars {
10850 set_returns.push(RetItem {
10851 value: RetVal::FuncCall {
10852 name: "type".into(),
10853 args: vec![Operand::Var(r.clone())],
10854 },
10855 alias: Some(rel_type_alias(r)),
10856 });
10857 }
10858
10859 let read_q = Query {
10860 matches: stmt.matches.clone(),
10861 optional_clauses: vec![],
10862 where_expr: stmt.where_expr.clone(),
10863 unwinds: vec![],
10864 post_unwind_where: None,
10865 stages: vec![],
10866 returns: set_returns,
10867 distinct: false,
10868 order_by: vec![],
10869 skip: None,
10870 limit: None,
10871 };
10872 let ops = plan(&read_q).map_err(|e| GraphError::QueryError {
10873 detail: format!("plan: {e}"),
10874 })?;
10875 // MATCH phase is read-only; borrow ends before batch opens.
10876 //
10877 // When a role-scoped write is in flight, run the MATCH read through
10878 // view_masked so hidden nodes are invisible → hidden ≡ absent ≡
10879 // zero-rows (no SetProp ops generated, no existence-oracle 403).
10880 // Full-authority writes (pending_write_authz=None) keep view().
10881 let match_rs = {
10882 let mask_opt = self.pending_write_authz.as_ref().map(|a| a.mask.clone());
10883 if let Some(ref mask) = mask_opt {
10884 execute(&self.view_masked(mask), &ops, &Params(params))
10885 } else {
10886 execute(&self.view(), &ops, &Params(params))
10887 }
10888 }
10889 .map_err(|e| GraphError::QueryError {
10890 detail: format!("execute: {e}"),
10891 })?;
10892
10893 // Collect (key, field, value) for each matched row × each SET clause.
10894 let mut set_ops: Vec<(String, String, Value)> = Vec::new();
10895 for row_i in 0..match_rs.len() {
10896 for (sc, col) in stmt.sets.iter().zip(&set_val_cols) {
10897 let key = match match_rs.get(row_i, &sc.var) {
10898 Some(Value::Str(k)) => k.clone(),
10899 _ => {
10900 return Err(GraphError::QueryError {
10901 detail: format!(
10902 "SET variable '{}' did not resolve to a node key",
10903 sc.var
10904 ),
10905 })
10906 }
10907 };
10908 // The SET value was already evaluated by the executor.
10909 let value = match match_rs.get(row_i, col) {
10910 Some(v) => v.clone(),
10911 None => {
10912 return Err(GraphError::QueryError {
10913 detail: format!(
10914 "SET value for {}.{} evaluated to null",
10915 sc.var, sc.field
10916 ),
10917 })
10918 }
10919 };
10920 set_ops.push((key, sc.field.clone(), value));
10921 }
10922 }
10923
10924 // Apply as one atomic batch.
10925 let props_set = set_ops.len();
10926 let mut batch = self.batch();
10927 for (key, field, value) in set_ops {
10928 batch.set_prop(&key, &field, value);
10929 }
10930 batch.commit()?;
10931
10932 if let Some(returns) = project_returns {
10933 return project_set_return_rows(self, &rel_vars, &match_rs, &returns, params);
10934 }
10935
10936 let mut rs = write_result_set();
10937 rs.push_row(vec![
10938 Some(Value::Int(0)),
10939 Some(Value::Int(props_set as i64)),
10940 Some(Value::Int(0)),
10941 ]);
10942 Ok(rs)
10943 }
10944
10945 fn exec_match_delete(
10946 &mut self,
10947 stmt: core_query::cypher::MatchDeleteStmt,
10948 params: &BTreeMap<String, Value>,
10949 ) -> Result<ResultSet> {
10950 // Collect unique node vars needed to identify edge endpoints.
10951 let mut node_vars: Vec<String> = Vec::new();
10952 for ed in &stmt.deletes {
10953 if !node_vars.contains(&ed.src_var) {
10954 node_vars.push(ed.src_var.clone());
10955 }
10956 if !node_vars.contains(&ed.dst_var) {
10957 node_vars.push(ed.dst_var.clone());
10958 }
10959 }
10960
10961 // Synthesize read query.
10962 let returns: Vec<RetItem> = node_vars
10963 .iter()
10964 .map(|v| RetItem {
10965 value: RetVal::Var(v.clone()),
10966 alias: None,
10967 })
10968 .collect();
10969 let read_q = Query {
10970 matches: stmt.matches,
10971 optional_clauses: vec![],
10972 where_expr: stmt.where_expr,
10973 unwinds: vec![],
10974 post_unwind_where: None,
10975 stages: vec![],
10976 returns,
10977 distinct: false,
10978 order_by: vec![],
10979 skip: None,
10980 limit: None,
10981 };
10982 let ops = plan(&read_q).map_err(|e| GraphError::QueryError {
10983 detail: format!("plan: {e}"),
10984 })?;
10985 // Role-scoped writes: mask the MATCH read phase so hidden nodes are
10986 // invisible → hidden ≡ absent ≡ zero-rows (spec §3.1, hidden ≡ absent).
10987 let match_rs = {
10988 let mask_opt = self.pending_write_authz.as_ref().map(|a| a.mask.clone());
10989 if let Some(ref mask) = mask_opt {
10990 execute(&self.view_masked(mask), &ops, &Params(params))
10991 } else {
10992 execute(&self.view(), &ops, &Params(params))
10993 }
10994 }
10995 .map_err(|e| GraphError::QueryError {
10996 detail: format!("execute: {e}"),
10997 })?;
10998
10999 // Collect (etype, src_key, dst_key) for each row × each delete target.
11000 let mut del_ops: Vec<(String, String, String)> = Vec::new();
11001 for row_i in 0..match_rs.len() {
11002 for ed in &stmt.deletes {
11003 let src_key = match match_rs.get(row_i, &ed.src_var) {
11004 Some(Value::Str(k)) => k.clone(),
11005 _ => {
11006 return Err(GraphError::QueryError {
11007 detail: format!(
11008 "DELETE src variable '{}' did not resolve to a node key",
11009 ed.src_var
11010 ),
11011 })
11012 }
11013 };
11014 let dst_key = match match_rs.get(row_i, &ed.dst_var) {
11015 Some(Value::Str(k)) => k.clone(),
11016 _ => {
11017 return Err(GraphError::QueryError {
11018 detail: format!(
11019 "DELETE dst variable '{}' did not resolve to a node key",
11020 ed.dst_var
11021 ),
11022 })
11023 }
11024 };
11025 del_ops.push((ed.etype.clone(), src_key, dst_key));
11026 }
11027 }
11028
11029 // Apply as one atomic batch.
11030 let deleted = del_ops.len();
11031 let mut batch = self.batch();
11032 for (etype, src_key, dst_key) in del_ops {
11033 batch.delete_edge(&etype, &src_key, &dst_key);
11034 }
11035 batch.commit().map_err(|e| match e {
11036 GraphError::RuleOwned { .. } => GraphError::QueryError {
11037 detail: "cannot delete derived edge; retract via the rule or change the property"
11038 .to_string(),
11039 },
11040 other => other,
11041 })?;
11042
11043 let mut rs = write_result_set();
11044 rs.push_row(vec![
11045 Some(Value::Int(0)),
11046 Some(Value::Int(0)),
11047 Some(Value::Int(deleted as i64)),
11048 ]);
11049 Ok(rs)
11050 }
11051
11052 /// Execute `MATCH … [DETACH] DELETE <node_var> [, …]`.
11053 ///
11054 /// Collects the matching node keys via an ephemeral read query, then calls
11055 /// `delete_node` on each one. When `stmt.detach` is `false` (bare DELETE)
11056 /// the executor first checks that the node has no incident edges; if any
11057 /// remain it returns a named error matching openCypher semantics.
11058 fn exec_match_delete_node(
11059 &mut self,
11060 stmt: MatchDeleteNodeStmt,
11061 params: &BTreeMap<String, Value>,
11062 ) -> Result<ResultSet> {
11063 // Build a read query returning only the node keys we need.
11064 let returns: Vec<RetItem> = stmt
11065 .node_vars
11066 .iter()
11067 .map(|v| RetItem {
11068 value: RetVal::Var(v.clone()),
11069 alias: None,
11070 })
11071 .collect();
11072 let read_q = Query {
11073 matches: stmt.matches,
11074 optional_clauses: vec![],
11075 where_expr: stmt.where_expr,
11076 unwinds: vec![],
11077 post_unwind_where: None,
11078 stages: vec![],
11079 returns,
11080 distinct: false,
11081 order_by: vec![],
11082 skip: None,
11083 limit: None,
11084 };
11085 let ops = plan(&read_q).map_err(|e| GraphError::QueryError {
11086 detail: format!("plan: {e}"),
11087 })?;
11088 // Role-scoped writes: mask the MATCH read phase so hidden nodes are
11089 // invisible → hidden ≡ absent ≡ zero-rows (spec §3.1, hidden ≡ absent).
11090 let match_rs = {
11091 let mask_opt = self.pending_write_authz.as_ref().map(|a| a.mask.clone());
11092 if let Some(ref mask) = mask_opt {
11093 execute(&self.view_masked(mask), &ops, &Params(params))
11094 } else {
11095 execute(&self.view(), &ops, &Params(params))
11096 }
11097 }
11098 .map_err(|e| GraphError::QueryError {
11099 detail: format!("execute: {e}"),
11100 })?;
11101
11102 // Collect unique node keys to delete (deduplicate across rows × vars).
11103 let mut keys: Vec<String> = Vec::new();
11104 for row_i in 0..match_rs.len() {
11105 for var in &stmt.node_vars {
11106 if let Some(Value::Str(k)) = match_rs.get(row_i, var) {
11107 if !keys.contains(k) {
11108 keys.push(k.clone());
11109 }
11110 }
11111 }
11112 }
11113
11114 if !stmt.detach {
11115 // openCypher bare DELETE: error if any matched node has incident edges.
11116 for key in &keys {
11117 if let Some(id) = self.ids.get(key) {
11118 let tv = self.topo_view();
11119 let has_edges = tv.etypes().any(|et| {
11120 !tv.neighbors(et, Direction::Out, id).is_empty()
11121 || !tv.neighbors(et, Direction::In, id).is_empty()
11122 });
11123 if has_edges {
11124 return Err(GraphError::QueryError {
11125 detail: format!(
11126 "Cannot delete node `{key}` because it still has incident edges. \
11127 Use DETACH DELETE to remove the node and all its edges."
11128 ),
11129 });
11130 }
11131 }
11132 }
11133 }
11134
11135 let mut nodes_deleted = 0i64;
11136 let mut edges_deleted = 0i64;
11137 for key in keys {
11138 match self.delete_node(&key) {
11139 Ok(report) => {
11140 nodes_deleted += 1;
11141 edges_deleted += (report.manual_edges + report.derived_edges) as i64;
11142 }
11143 Err(GraphError::KeyNotFound { .. }) => {
11144 // Node may have been deleted by an earlier iteration (e.g., via
11145 // multiple MATCH rows for the same node). Safe to skip.
11146 }
11147 Err(e) => return Err(e),
11148 }
11149 }
11150
11151 let mut rs = write_result_set();
11152 rs.push_row(vec![
11153 Some(Value::Int(0)),
11154 Some(Value::Int(0)),
11155 Some(Value::Int(nodes_deleted + edges_deleted)),
11156 ]);
11157 Ok(rs)
11158 }
11159
11160 /// Props the MERGE create arm inserts: the identifying key, plus `ns` when
11161 /// the pattern named one, or the executing role's sole namespace when it
11162 /// did not. A role bound to two or more namespaces cannot choose, and is
11163 /// refused with [`MERGE_CREATE_NEEDS_ONE_NAMESPACE`]. The authorizer still
11164 /// refuses a named `ns` the role cannot write.
11165 fn merge_create_props(
11166 &self,
11167 key_field: &str,
11168 key_value: &Value,
11169 named_ns: Option<&Value>,
11170 ) -> Result<Vec<(String, Value)>> {
11171 let mut props = vec![(key_field.to_string(), key_value.clone())];
11172 if let Some(ns) = named_ns {
11173 props.push((NS_PROP.to_string(), ns.clone()));
11174 return Ok(props);
11175 }
11176 if let Some(ns) = self.merge_create_stamp_ns()? {
11177 props.push((NS_PROP.to_string(), Value::Str(ns)));
11178 }
11179 Ok(props)
11180 }
11181
11182 /// The namespace a role-scoped MERGE create stamps when the pattern does
11183 /// not name `ns`. `None` = unscoped / full authority, so the node lands in
11184 /// `default`.
11185 fn merge_create_stamp_ns(&self) -> Result<Option<String>> {
11186 let Some(authz) = self.pending_write_authz.as_ref() else {
11187 return Ok(None);
11188 };
11189 let Some(def) = self.role_def_for(&authz.role) else {
11190 return Ok(None);
11191 };
11192 match def.namespaces.as_deref() {
11193 Some([only]) => Ok(Some(only.clone())),
11194 Some(_) => Err(GraphError::RoleWriteDenied {
11195 reason: MERGE_CREATE_NEEDS_ONE_NAMESPACE.to_string(),
11196 }),
11197 None => Ok(None),
11198 }
11199 }
11200
11201 fn exec_merge(
11202 &mut self,
11203 stmt: core_query::cypher::MergeStmt,
11204 params: &BTreeMap<String, Value>,
11205 ) -> Result<ResultSet> {
11206 // MERGE: check if a node with the given key already exists.
11207 let key = match &stmt.key_value {
11208 Value::Str(s) => s.clone(),
11209 _ => {
11210 return Err(GraphError::QueryError {
11211 detail: format!(
11212 "MERGE key value must be a string (got {:?})",
11213 stmt.key_value
11214 ),
11215 })
11216 }
11217 };
11218
11219 if let Some(var) = stmt.var.as_deref() {
11220 for sc in stmt.on_create.iter().chain(&stmt.on_match) {
11221 if sc.var != var {
11222 return Err(GraphError::QueryError {
11223 detail: format!(
11224 "SET variable '{}' does not match MERGE variable '{var}'",
11225 sc.var
11226 ),
11227 });
11228 }
11229 }
11230 }
11231
11232 // ── MERGE authz pre-check (when role-scoped) ─────────────────────────
11233 //
11234 // MERGE scope precondition: check create OR update scope for the
11235 // declared label BEFORE calling `has_node` (timing-oracle closure,
11236 // spec §6.2 "MERGE visibility oracle" item: hidden ≡ absent for
11237 // unscoped roles — the scope denial fires without touching the key store).
11238 //
11239 // Clone to avoid holding a borrow on `self.pending_write_authz` while
11240 // also calling `self.ids.get(key)`.
11241 let merge_existed: bool = if let Some(authz) = self.pending_write_authz.clone() {
11242 let has_create = authz.scope.create_labels.contains(&stmt.label);
11243 let has_update = authz.scope.update_labels.contains(&stmt.label);
11244 if !has_create && !has_update {
11245 // Scope-before-lookup: 403 without has_node call (timing oracle
11246 // closure — see test_merge_unscoped_no_key_lookup).
11247 return Err(GraphError::RoleWriteDenied {
11248 reason: format!(
11249 "role-bound token: label '{}' not in write scope (create_labels)",
11250 stmt.label
11251 ),
11252 });
11253 }
11254 // Key lookup under mask.
11255 match self.ids.get(key.as_str()) {
11256 Some(id) if authz.mask.contains_id(id) => {
11257 // Visible: must have update scope to proceed to match arm.
11258 if !has_update {
11259 return Err(GraphError::RoleWriteDenied {
11260 reason: format!(
11261 "role-bound token: label '{}' not in write scope (update_labels)",
11262 stmt.label
11263 ),
11264 });
11265 }
11266 true // existed = true → match arm
11267 }
11268 Some(_) => {
11269 // Hidden: same error as absent to the role (spec §3.1/§3.3).
11270 return Err(GraphError::RoleWriteDenied {
11271 reason: "role-bound token: target node not visible".into(),
11272 });
11273 }
11274 None => {
11275 // Absent: must have create scope to proceed to the create arm.
11276 //
11277 // Update-only roles (create_labels empty, update_labels set):
11278 // return the SAME "not visible" error as the hidden-key branch
11279 // so hidden ≡ absent — no distinguishing oracle (spec §6.1
11280 // "confirm existence of hidden nodes: No").
11281 //
11282 // Create-scoped roles (has_create=true): absent → create arm
11283 // as before. The accepted structural key-existence disclosure
11284 // (§THREAT-MODEL) applies only when the role holds create scope.
11285 if !has_create {
11286 return Err(GraphError::RoleWriteDenied {
11287 reason: "role-bound token: target node not visible".into(),
11288 });
11289 }
11290 false // existed = false → create arm
11291 }
11292 }
11293 } else {
11294 // Full authority: use the existing non-masked has_node check.
11295 self.has_node(&key)
11296 };
11297
11298 let existed = merge_existed;
11299 let create_props = if existed {
11300 None
11301 } else {
11302 Some(self.merge_create_props(&stmt.key_field, &stmt.key_value, stmt.ns.as_ref())?)
11303 };
11304 let mut created = 0i64;
11305 if create_props.is_some() || !stmt.on_match.is_empty() {
11306 let mut batch = self.batch();
11307 if let Some(props) = create_props {
11308 batch.insert_node(&stmt.label, &key, props);
11309 for sc in &stmt.on_create {
11310 let value = resolve_merge_set_value(&sc.value, params)?;
11311 batch.set_prop(&key, &sc.field, value);
11312 }
11313 created = 1;
11314 } else {
11315 for sc in &stmt.on_match {
11316 let value = resolve_merge_set_value(&sc.value, params)?;
11317 batch.set_prop(&key, &sc.field, value);
11318 }
11319 }
11320 batch.commit()?;
11321 }
11322
11323 // Refresh the role mask so the just-created node is visible to this
11324 // statement's RETURN (read-after-write). Safe: create_labels ⊆ read labels
11325 // (apply_schema subset rule), so the new node's label is already in the
11326 // role's read scope — this never widens beyond the role's declared labels.
11327 if !existed {
11328 if let Some(role) = self.pending_write_authz.as_ref().map(|a| a.role.clone()) {
11329 let new_mask = self.mask_for_role(&role)?;
11330 if let Some(a) = self.pending_write_authz.as_mut() {
11331 a.mask = new_mask;
11332 }
11333 }
11334 }
11335
11336 // Optional RETURN clause: project the node (created or matched) as a read result.
11337 if let Some(returns) = stmt.returns {
11338 let var = stmt.var.as_deref().unwrap_or("_mn0");
11339 let q = Query {
11340 matches: vec![Pattern {
11341 start: NodePat {
11342 var: Some(var.to_string()),
11343 label: Some(stmt.label.clone()),
11344 props: vec![("id".to_string(), Operand::Lit(stmt.key_value.clone()))],
11345 },
11346 chain: vec![],
11347 shortest: false,
11348 }],
11349 optional_clauses: vec![],
11350 where_expr: None,
11351 unwinds: vec![],
11352 post_unwind_where: None,
11353 stages: vec![],
11354 returns,
11355 distinct: false,
11356 order_by: vec![],
11357 skip: None,
11358 limit: None,
11359 };
11360 let ops = plan(&q).map_err(|e| GraphError::QueryError {
11361 detail: format!("plan: {e}"),
11362 })?;
11363 // Use view_masked when a role-scoped write is in flight so the
11364 // post-merge projection is consistent with the masked read phase.
11365 let mask_opt = self.pending_write_authz.as_ref().map(|a| a.mask.clone());
11366 return (if let Some(ref mask) = mask_opt {
11367 execute(&self.view_masked(mask), &ops, &Params(params))
11368 } else {
11369 execute(&self.view(), &ops, &Params(params))
11370 })
11371 .map_err(|e| GraphError::QueryError {
11372 detail: format!("execute: {e}"),
11373 });
11374 }
11375
11376 let mut rs = write_result_set();
11377 rs.push_row(vec![
11378 Some(Value::Int(created)),
11379 Some(Value::Int(0)),
11380 Some(Value::Int(0)),
11381 ]);
11382 Ok(rs)
11383 }
11384
11385 /// Return all rule-owned edges between `key_a` and `key_b` (either direction),
11386 /// annotated with rule name, edge type, direction, and weight.
11387 /// Results are sorted by (rule, edge_type).
11388 /// Returns `Err(KeyNotFound)` if either key is unknown.
11389 pub fn explain(&self, key_a: &str, key_b: &str) -> Result<Vec<Explanation>> {
11390 self.ensure_v8_base_sections_loaded();
11391 let id_a = self
11392 .ids
11393 .get(key_a)
11394 .ok_or_else(|| GraphError::KeyNotFound { key: key_a.into() })?;
11395 let id_b = self
11396 .ids
11397 .get(key_b)
11398 .ok_or_else(|| GraphError::KeyNotFound { key: key_b.into() })?;
11399
11400 let mut results = Vec::new();
11401
11402 // Walk the smaller incident set so explain is O(min(deg(a), deg(b)))
11403 // rather than O(total provenance).
11404 let scan = if self.engine.provenance_touching_len(id_a)
11405 <= self.engine.provenance_touching_len(id_b)
11406 {
11407 id_a
11408 } else {
11409 id_b
11410 };
11411 for (rule_name, etype, src, dst) in self.engine.provenance_touching(scan) {
11412 if !((src == id_a && dst == id_b) || (src == id_b && dst == id_a)) {
11413 continue;
11414 }
11415 let Some(rule_def) = self.engine.rules().find(|r| r.name == rule_name) else {
11416 continue;
11417 };
11418 let edge_type = match self.syms.resolve(etype) {
11419 Some(s) => s.to_string(),
11420 None => continue,
11421 };
11422 // Provenance (src, dst) ids come from the archived PROVENANCE section
11423 // (large, no eager CRC). A corrupt section can produce ids that are
11424 // out of range; return Corrupt rather than panic.
11425 let src_key = self
11426 .ids
11427 .key_of(src)
11428 .ok_or_else(|| GraphError::Corrupt {
11429 detail: format!("v8: provenance src id {src} not in id table"),
11430 })?
11431 .to_string();
11432 let dst_key = self
11433 .ids
11434 .key_of(dst)
11435 .ok_or_else(|| GraphError::Corrupt {
11436 detail: format!("v8: provenance dst id {dst} not in id table"),
11437 })?
11438 .to_string();
11439 let stored = rule_def.weight_prop.as_deref().and_then(|prop| {
11440 self.edge_props_view()
11441 .get(etype, src, dst, prop)
11442 .and_then(|v| {
11443 if let Value::Float(f) = v {
11444 Some(f)
11445 } else {
11446 None
11447 }
11448 })
11449 });
11450 // Rules that store no weight (KeyMatch/FieldEqual defaults, auto-FK)
11451 // still have a score: recompute it from the predicate so explain
11452 // never reports "no score" for an edge the engine scored. Via-hop
11453 // rules score over their via set, not over (src, dst), so leave
11454 // those None rather than report a number the rule did not produce.
11455 let weight = stored.or_else(|| {
11456 if rule_def.via_edge.is_some() {
11457 return None;
11458 }
11459 let props_view = build_props_view(&self.props, &self.base);
11460 let src_get = |field: &str| props_view.get(src, field).map(|vr| vr.into_value());
11461 let dst_get = |field: &str| props_view.get(dst, field).map(|vr| vr.into_value());
11462 let src_view = NodeView {
11463 key: &src_key,
11464 props: &src_get,
11465 };
11466 let dst_view = NodeView {
11467 key: &dst_key,
11468 props: &dst_get,
11469 };
11470 evaluate(&rule_def.predicate, &src_view, &dst_view)
11471 });
11472 results.push(Explanation {
11473 rule: rule_name.to_string(),
11474 edge_type,
11475 src_key,
11476 dst_key,
11477 weight,
11478 predicate: PredicateSummary {
11479 approximate: rule_def.approximate,
11480 ..PredicateSummary::from(&rule_def.predicate)
11481 },
11482 via_edge: rule_def.via_edge.clone(),
11483 });
11484 }
11485
11486 results.sort_by(|a, b| a.rule.cmp(&b.rule).then(a.edge_type.cmp(&b.edge_type)));
11487 Ok(results)
11488 }
11489
11490 /// [`explain`](Self::explain) with the scoped read contract (§5.3): both
11491 /// endpoints are subject-checked, and any explanation whose evidence runs
11492 /// through a hidden node is **dropped entirely, not redacted**.
11493 ///
11494 /// A plain two-node rule's evidence is the pair itself, so once both
11495 /// subjects are visible there is nothing left to hide. A **via-hop** rule is
11496 /// different: it fires `src → dst` because some node carrying `via_label`
11497 /// sits between them, and [`Explanation`] carries the hop's edge *type*
11498 /// (`via_edge`) and never the hop's key. There is no field to blank, so a
11499 /// redacted explanation would still say "these two are linked through
11500 /// something you cannot see" — which discloses that the something exists.
11501 /// The explanation is therefore kept only when at least one **visible** via
11502 /// node satisfies the rule on its own.
11503 ///
11504 /// The weight is the **visible corpus's** number, not the store's: a via-hop
11505 /// rule stores the max over every via it hopped through, so the stored value
11506 /// can be a score only a hidden via produced. It is recomputed over the
11507 /// visible vias alone.
11508 ///
11509 /// Hidden or unknown `key_a` or `key_b` → [`GraphError::KeyNotFound`].
11510 pub fn explain_scoped(
11511 &self,
11512 key_a: &str,
11513 key_b: &str,
11514 mask: &crate::mask::NodeMask,
11515 ) -> Result<Vec<Explanation>> {
11516 for key in [key_a, key_b] {
11517 if !mask.contains_node(self, key) {
11518 return Err(GraphError::KeyNotFound { key: key.into() });
11519 }
11520 }
11521 Ok(self
11522 .explain(key_a, key_b)?
11523 .into_iter()
11524 .filter_map(|e| self.scoped_explanation(e, mask))
11525 .collect())
11526 }
11527
11528 /// `e` as a caller limited to `mask` may have it, or `None` when it must be
11529 /// dropped entirely.
11530 ///
11531 /// Every non-via-hop explanation passes through untouched: its only nodes
11532 /// are the two subjects, which [`explain_scoped`](Self::explain_scoped) has
11533 /// already checked, and its weight is scored over that pair alone.
11534 ///
11535 /// A via-hop explanation is kept only when some via node the caller may see
11536 /// satisfies the rule on its own — and then its weight is recomputed as the
11537 /// max over exactly those vias. The engine writes the max over **all** of
11538 /// them (`core-rules::engine`, `best = prev.max(score)`), so passing the
11539 /// stored number through would let a hidden node set a figure the caller
11540 /// reads: the same disclosure dropping the explanation exists to prevent.
11541 ///
11542 /// A rule that stores no weight still reports none. The recomputed score is
11543 /// a sanitised version of a number `explain` already returned, never a new
11544 /// one — a scoped read must not say more than the unscoped read it narrows.
11545 fn scoped_explanation(
11546 &self,
11547 e: Explanation,
11548 mask: &crate::mask::NodeMask,
11549 ) -> Option<Explanation> {
11550 let Some(via_edge) = e.via_edge.clone() else {
11551 return Some(e);
11552 };
11553 let Some(rule_def) = self.engine.rules().find(|r| r.name == e.rule) else {
11554 // The rule is gone but its provenance is not; nothing can vouch for
11555 // the hop, so nothing is shown.
11556 return None;
11557 };
11558 let Some(via_label) = rule_def.via_label.as_deref() else {
11559 return Some(e);
11560 };
11561 let (Some(src), Some(dst)) = (self.ids.get(&e.src_key), self.ids.get(&e.dst_key)) else {
11562 return None;
11563 };
11564 let (Some(via_etype), Some(via_sym)) = (self.syms.get(&via_edge), self.syms.get(via_label))
11565 else {
11566 return None;
11567 };
11568 let via_dir = rule_def.via_dir.unwrap_or(Direction::Out);
11569 let props_view = build_props_view(&self.props, &self.base);
11570 let dst_get = |field: &str| props_view.get(dst, field).map(|vr| vr.into_value());
11571 let dst_view = NodeView {
11572 key: &e.dst_key,
11573 props: &dst_get,
11574 };
11575 // The rule's own namespace test, the one the engine applies to each via
11576 // candidate (`core-rules::engine::rule_sees_node`). Without it a
11577 // visible, out-of-namespace via — right label, satisfying predicate —
11578 // vouches for a hop the engine never made, and an explanation whose real
11579 // evidence is a hidden in-namespace node is kept.
11580 let rule_sees = |id: u32| match rule_def.namespace.as_deref() {
11581 None => true,
11582 Some(ns) => {
11583 let value = props_view.get(id, NS_PROP).map(|vr| vr.into_value());
11584 namespace_of_value(value.as_ref()) == ns
11585 }
11586 };
11587 let best = self
11588 .topo_view()
11589 .neighbors(via_etype, via_dir, src)
11590 .iter()
11591 .copied()
11592 .filter_map(|via| {
11593 if !mask.contains_id(via) {
11594 return None;
11595 }
11596 if self.labels.get(via as usize).copied() != Some(via_sym) {
11597 return None;
11598 }
11599 if !rule_sees(via) {
11600 return None;
11601 }
11602 let via_key = self.ids.key_of(via)?;
11603 let via_get = |field: &str| props_view.get(via, field).map(|vr| vr.into_value());
11604 let via_view = NodeView {
11605 key: via_key,
11606 props: &via_get,
11607 };
11608 evaluate(&rule_def.predicate, &via_view, &dst_view)
11609 })
11610 .fold(None::<f64>, |best, score| {
11611 Some(match best {
11612 None => score,
11613 Some(prev) => prev.max(score),
11614 })
11615 })?;
11616 let weight = e.weight.map(|_| best);
11617 Some(Explanation { weight, ..e })
11618 }
11619
11620 pub fn neighbors(&self, key: &str, edge_type: &str, dir: Direction) -> Result<Vec<String>> {
11621 let id = self
11622 .ids
11623 .get(key)
11624 .ok_or_else(|| GraphError::KeyNotFound { key: key.into() })?;
11625 let Some(sym) = self.syms.get(edge_type) else {
11626 return Ok(Vec::new());
11627 };
11628 self.topo_view()
11629 .neighbors(sym, dir, id)
11630 .iter()
11631 .map(|&n| {
11632 self.ids
11633 .key_of(n)
11634 .map(|k| k.to_string())
11635 .ok_or_else(|| GraphError::Corrupt {
11636 detail: format!("topology id {n} has no key"),
11637 })
11638 })
11639 .collect::<Result<Vec<_>>>()
11640 }
11641
11642 /// Unique directed degree of `key`. Unknown key → [`GraphError::KeyNotFound`].
11643 /// Unknown `edge_type` → 0. [`crate::algo::AlgoDir::Both`] is out + in (sum).
11644 pub fn degree(
11645 &self,
11646 key: &str,
11647 edge_type: Option<&str>,
11648 direction: crate::algo::AlgoDir,
11649 ) -> Result<u64> {
11650 let id = self
11651 .ids
11652 .get(key)
11653 .ok_or_else(|| GraphError::KeyNotFound { key: key.into() })?;
11654 let topo = self.topo_view();
11655 Ok(Self::unique_directed_degree(
11656 &topo, &self.syms, id, edge_type, direction,
11657 ))
11658 }
11659
11660 /// [`degree`](Self::degree) summing each pair's **insert count** instead of
11661 /// counting each pair once (§5.13).
11662 ///
11663 /// The unique degree asks how many neighbours there are; this asks how many
11664 /// times they were inserted. A pair with no recorded count contributes 1,
11665 /// so on a store that never called
11666 /// [`enable_multiplicity`](Self::enable_multiplicity) this returns exactly
11667 /// what [`degree`](Self::degree) returns rather than erroring — the
11668 /// distinction is a readout preference, not a demand the store cannot meet.
11669 ///
11670 /// `AlgoDir::Both` still sums out + in, so a pair visible on both sides
11671 /// still contributes twice: multiplicity changes what a pair is worth, never
11672 /// how a direction is counted.
11673 pub fn degree_multiplicity(
11674 &self,
11675 key: &str,
11676 edge_type: Option<&str>,
11677 direction: crate::algo::AlgoDir,
11678 ) -> Result<u64> {
11679 let id = self
11680 .ids
11681 .get(key)
11682 .ok_or_else(|| GraphError::KeyNotFound { key: key.into() })?;
11683 Ok(Self::multiplicity_directed_degree(
11684 &self.topo_view(),
11685 &self.edge_props_view(),
11686 &self.syms,
11687 id,
11688 edge_type,
11689 direction,
11690 None,
11691 ))
11692 }
11693
11694 /// [`degree_multiplicity`](Self::degree_multiplicity) under a scope.
11695 ///
11696 /// The sum covers **visible pairs only**. A hidden neighbour's inserts stay
11697 /// out of it for the reason
11698 /// [`degree_scoped`](Self::degree_scoped) documents, and more sharply: an
11699 /// unscoped multiplicity count discloses not only that a hidden neighbour
11700 /// exists but how often it was written.
11701 ///
11702 /// Hidden or unknown `key` → [`GraphError::KeyNotFound`].
11703 pub fn degree_scoped_multiplicity(
11704 &self,
11705 key: &str,
11706 edge_type: Option<&str>,
11707 direction: crate::algo::AlgoDir,
11708 mask: &crate::mask::NodeMask,
11709 ) -> Result<u64> {
11710 let id = self
11711 .ids
11712 .get(key)
11713 .filter(|&id| mask.contains_id(id))
11714 .ok_or_else(|| GraphError::KeyNotFound { key: key.into() })?;
11715 Ok(Self::multiplicity_directed_degree(
11716 &self.topo_view(),
11717 &self.edge_props_view(),
11718 &self.syms,
11719 id,
11720 edge_type,
11721 direction,
11722 Some(mask),
11723 ))
11724 }
11725
11726 /// [`degree`](Self::degree) counting **only neighbours the mask admits**.
11727 ///
11728 /// The filter is a correctness requirement, not an optimisation: an
11729 /// unfiltered count discloses the existence of a hidden neighbour to a
11730 /// caller who cannot see it, which is the same leak
11731 /// [`node_edges_scoped`](Self::node_edges_scoped) exists to prevent —
11732 /// reached by arithmetic instead of by name.
11733 ///
11734 /// Hidden or unknown `key` → [`GraphError::KeyNotFound`]. Unknown
11735 /// `edge_type` is still 0, as it is unscoped.
11736 pub fn degree_scoped(
11737 &self,
11738 key: &str,
11739 edge_type: Option<&str>,
11740 direction: crate::algo::AlgoDir,
11741 mask: &crate::mask::NodeMask,
11742 ) -> Result<u64> {
11743 let id = self
11744 .ids
11745 .get(key)
11746 .filter(|&id| mask.contains_id(id))
11747 .ok_or_else(|| GraphError::KeyNotFound { key: key.into() })?;
11748 let topo = self.topo_view();
11749 Ok(Self::visible_directed_degree(
11750 &topo, &self.syms, id, edge_type, direction, mask,
11751 ))
11752 }
11753
11754 /// Unique directed degree for a subset or a label scan.
11755 ///
11756 /// Unknown keys in `keys` are omitted (mask-like). `keys = Some(&[])` →
11757 /// empty `Ok(vec![])`. `limit` is applied after sorting degree desc, key
11758 /// asc, and only when `Some`. Invalid `where_` → `QueryError`.
11759 #[allow(clippy::too_many_arguments)]
11760 pub fn degrees(
11761 &self,
11762 keys: Option<&[String]>,
11763 label: Option<&str>,
11764 where_: Option<&PropPredicate>,
11765 edge_type: Option<&str>,
11766 direction: crate::algo::AlgoDir,
11767 limit: Option<usize>,
11768 ) -> Result<Vec<(String, u64)>> {
11769 self.degrees_inner(
11770 keys, label, where_, edge_type, direction, limit, None, false,
11771 )
11772 }
11773
11774 /// [`degrees`](Self::degrees) reporting each row's **insert-count** sum
11775 /// instead of its unique neighbour count, as
11776 /// [`degree_multiplicity`](Self::degree_multiplicity) does for one key.
11777 ///
11778 /// The sort is still degree descending, key ascending — over the counts this
11779 /// reading produces — and `limit` still applies after it.
11780 #[allow(clippy::too_many_arguments)]
11781 pub fn degrees_multiplicity(
11782 &self,
11783 keys: Option<&[String]>,
11784 label: Option<&str>,
11785 where_: Option<&PropPredicate>,
11786 edge_type: Option<&str>,
11787 direction: crate::algo::AlgoDir,
11788 limit: Option<usize>,
11789 ) -> Result<Vec<(String, u64)>> {
11790 self.degrees_inner(keys, label, where_, edge_type, direction, limit, None, true)
11791 }
11792
11793 /// [`degrees_scoped`](Self::degrees_scoped) reporting insert counts.
11794 ///
11795 /// Both filters apply: a hidden key stays out of the result, and every
11796 /// row's sum covers its **visible** pairs only.
11797 #[allow(clippy::too_many_arguments)]
11798 pub fn degrees_scoped_multiplicity(
11799 &self,
11800 keys: Option<&[String]>,
11801 label: Option<&str>,
11802 where_: Option<&PropPredicate>,
11803 edge_type: Option<&str>,
11804 direction: crate::algo::AlgoDir,
11805 limit: Option<usize>,
11806 mask: &crate::mask::NodeMask,
11807 ) -> Result<Vec<(String, u64)>> {
11808 self.degrees_inner(
11809 keys,
11810 label,
11811 where_,
11812 edge_type,
11813 direction,
11814 limit,
11815 Some(mask),
11816 true,
11817 )
11818 }
11819
11820 /// [`degrees`](Self::degrees) with the scope applied on both sides: a hidden
11821 /// key is omitted from the input — whether it arrived in `keys` or came out
11822 /// of the `label`/`where_` scan — and every row's count is the count of its
11823 /// **visible** neighbours, for the reason
11824 /// [`degree_scoped`](Self::degree_scoped) documents.
11825 ///
11826 /// Unlike `degree_scoped`, a hidden key here is not
11827 /// [`GraphError::KeyNotFound`]: `degrees` already drops unknown keys
11828 /// silently, so hidden and absent stay one answer by staying out of the
11829 /// result. `limit` still applies after the sort, and so counts visible rows.
11830 #[allow(clippy::too_many_arguments)]
11831 pub fn degrees_scoped(
11832 &self,
11833 keys: Option<&[String]>,
11834 label: Option<&str>,
11835 where_: Option<&PropPredicate>,
11836 edge_type: Option<&str>,
11837 direction: crate::algo::AlgoDir,
11838 limit: Option<usize>,
11839 mask: &crate::mask::NodeMask,
11840 ) -> Result<Vec<(String, u64)>> {
11841 self.degrees_inner(
11842 keys,
11843 label,
11844 where_,
11845 edge_type,
11846 direction,
11847 limit,
11848 Some(mask),
11849 false,
11850 )
11851 }
11852
11853 /// The body shared by [`degrees`](Self::degrees) and
11854 /// [`degrees_scoped`](Self::degrees_scoped). `mask = None` is the unscoped
11855 /// contract unchanged.
11856 #[allow(clippy::too_many_arguments)]
11857 fn degrees_inner(
11858 &self,
11859 keys: Option<&[String]>,
11860 label: Option<&str>,
11861 where_: Option<&PropPredicate>,
11862 edge_type: Option<&str>,
11863 direction: crate::algo::AlgoDir,
11864 limit: Option<usize>,
11865 mask: Option<&crate::mask::NodeMask>,
11866 multiplicity: bool,
11867 ) -> Result<Vec<(String, u64)>> {
11868 if let Some(pred) = where_ {
11869 pred.validate_named("where")
11870 .map_err(|detail| GraphError::QueryError { detail })?;
11871 }
11872 if matches!(keys, Some(ks) if ks.is_empty()) {
11873 return Ok(Vec::new());
11874 }
11875 let view = self.view();
11876 let ids: Vec<u32> = match keys {
11877 Some(ks) => {
11878 let mut seen = HashSet::new();
11879 let mut out = Vec::new();
11880 for k in ks {
11881 let Some(id) = view.ids.get(k) else {
11882 continue;
11883 };
11884 if !seen.insert(id) {
11885 continue;
11886 }
11887 if let Some(pred) = where_ {
11888 let holds = match view.prop(id, &pred.field) {
11889 None => pred.holds(None),
11890 Some(vr) => pred.holds(Some(vr.as_value())),
11891 };
11892 if !holds {
11893 continue;
11894 }
11895 }
11896 out.push(id);
11897 }
11898 out
11899 }
11900 None => Self::vector_candidates(&view, label, where_),
11901 };
11902 let mut out: Vec<(String, u64)> = ids
11903 .into_iter()
11904 // A hidden candidate leaves as quietly as an unknown key does.
11905 .filter(|&id| mask.is_none_or(|m| m.contains_id(id)))
11906 .filter_map(|id| {
11907 let key = self.ids.key_of(id)?.to_string();
11908 let deg = match (multiplicity, mask) {
11909 // The same `view` the unique arms read, so the per-row
11910 // rebuild F9 measured is gone and all three arms agree on
11911 // the state they are reading.
11912 (true, m) => Self::multiplicity_directed_degree(
11913 &view.topo,
11914 &view.edge_props,
11915 view.syms,
11916 id,
11917 edge_type,
11918 direction,
11919 m,
11920 ),
11921 (false, Some(m)) => Self::visible_directed_degree(
11922 &view.topo, view.syms, id, edge_type, direction, m,
11923 ),
11924 (false, None) => Self::unique_directed_degree(
11925 &view.topo, view.syms, id, edge_type, direction,
11926 ),
11927 };
11928 Some((key, deg))
11929 })
11930 .collect();
11931 out.sort_by(|a, b| b.1.cmp(&a.1).then_with(|| a.0.cmp(&b.0)));
11932 if let Some(lim) = limit {
11933 out.truncate(lim);
11934 }
11935 Ok(out)
11936 }
11937
11938 /// Unique neighbour count for `id` across `edge_type` (or all types) and
11939 /// `direction`. Unknown `edge_type` → 0. `Both` sums out + in.
11940 fn unique_directed_degree(
11941 topo: &TopologyView<'_>,
11942 syms: &Interner,
11943 id: u32,
11944 edge_type: Option<&str>,
11945 direction: crate::algo::AlgoDir,
11946 ) -> u64 {
11947 let dirs: &[Direction] = match direction {
11948 crate::algo::AlgoDir::Out => &[Direction::Out],
11949 crate::algo::AlgoDir::In => &[Direction::In],
11950 crate::algo::AlgoDir::Both => &[Direction::Out, Direction::In],
11951 };
11952 match edge_type {
11953 Some(name) => {
11954 let Some(et) = syms.get(name) else {
11955 return 0;
11956 };
11957 dirs.iter().map(|&d| topo.degree(et, d, id) as u64).sum()
11958 }
11959 None => topo
11960 .etypes()
11961 .map(|et| {
11962 dirs.iter()
11963 .map(|&d| topo.degree(et, d, id) as u64)
11964 .sum::<u64>()
11965 })
11966 .sum(),
11967 }
11968 }
11969
11970 /// [`unique_directed_degree`](Self::unique_directed_degree) counting only
11971 /// neighbours `mask` admits.
11972 ///
11973 /// Same shape, one substitution: `topo.degree` is a length, so it cannot be
11974 /// filtered; the neighbour list it measures can. `Both` still sums out + in,
11975 /// so a node visible on both sides still counts twice — the filter changes
11976 /// which neighbours are counted, never how a degree is defined.
11977 fn visible_directed_degree(
11978 topo: &TopologyView<'_>,
11979 syms: &Interner,
11980 id: u32,
11981 edge_type: Option<&str>,
11982 direction: crate::algo::AlgoDir,
11983 mask: &crate::mask::NodeMask,
11984 ) -> u64 {
11985 let dirs: &[Direction] = match direction {
11986 crate::algo::AlgoDir::Out => &[Direction::Out],
11987 crate::algo::AlgoDir::In => &[Direction::In],
11988 crate::algo::AlgoDir::Both => &[Direction::Out, Direction::In],
11989 };
11990 let visible = |et: u32| -> u64 {
11991 dirs.iter()
11992 .map(|&d| {
11993 topo.neighbors(et, d, id)
11994 .iter()
11995 .filter(|&&n| mask.contains_id(n))
11996 .count() as u64
11997 })
11998 .sum()
11999 };
12000 match edge_type {
12001 Some(name) => syms.get(name).map_or(0, visible),
12002 None => topo.etypes().map(visible).sum(),
12003 }
12004 }
12005
12006 /// Sum of the insert counts of `id`'s pairs (§5.13), over `edge_type` (or
12007 /// all types) and `direction`, restricted to what `mask` admits when one is
12008 /// given.
12009 ///
12010 /// The same neighbour lists the unique reading measures, with each entry
12011 /// worth its pair's count rather than worth 1 — so the filter decides which
12012 /// pairs are in the sum and the count decides what each contributes. A
12013 /// direction decides which way round the pair is addressed: an `In`
12014 /// neighbour `n` of `id` is the pair `(et, n, id)`.
12015 ///
12016 /// Takes its views as parameters, exactly as the unique helpers do, because
12017 /// it is called once per row from a label scan. `edge_props_view()` reaches
12018 /// into the mmap'd base's rkyv section on every call, so building the two
12019 /// views inside made an N-row `degrees(multiplicity=True)` do N section
12020 /// accesses where the unique reading does one: worth 2.57 ms of 16.68 ms
12021 /// over 20 000 rows, about 0.13 us per row (defect #29,
12022 /// `tests/f9_bench.rs`). Most of that call's cost is the per-neighbour
12023 /// count lookup and is inherent, so this is a hoist, not a rescue.
12024 ///
12025 /// The views are exactly `self.view()`'s own `topo` and `edge_props`, so a
12026 /// caller that already has a view passes its halves and reads the same
12027 /// state it reads everything else from.
12028 fn multiplicity_directed_degree(
12029 topo: &TopologyView<'_>,
12030 edge_props: &EdgePropsView<'_>,
12031 syms: &Interner,
12032 id: u32,
12033 edge_type: Option<&str>,
12034 direction: crate::algo::AlgoDir,
12035 mask: Option<&crate::mask::NodeMask>,
12036 ) -> u64 {
12037 let dirs: &[Direction] = match direction {
12038 crate::algo::AlgoDir::Out => &[Direction::Out],
12039 crate::algo::AlgoDir::In => &[Direction::In],
12040 crate::algo::AlgoDir::Both => &[Direction::Out, Direction::In],
12041 };
12042 let count_of = |et: u32, src: u32, dst: u32| -> u64 {
12043 match edge_props.get(et, src, dst, EDGE_COUNT_PROP) {
12044 Some(Value::Int(n)) if n > 0 => n as u64,
12045 _ => 1,
12046 }
12047 };
12048 let per_etype = |et: u32| -> u64 {
12049 dirs.iter()
12050 .map(|&d| {
12051 topo.neighbors(et, d, id)
12052 .iter()
12053 .filter(|&&n| mask.is_none_or(|m| m.contains_id(n)))
12054 .map(|&n| match d {
12055 Direction::Out => count_of(et, id, n),
12056 Direction::In => count_of(et, n, id),
12057 })
12058 .sum::<u64>()
12059 })
12060 .sum()
12061 };
12062 match edge_type {
12063 Some(name) => syms.get(name).map_or(0, per_etype),
12064 None => topo.etypes().map(per_etype).sum(),
12065 }
12066 }
12067
12068 /// Return the last-change commit sequence for `key`, or `None` if the node
12069 /// does not exist or has never been mutated since the last V5-V7 snapshot
12070 /// (horizon-bounded for legacy stores).
12071 ///
12072 /// The returned sequence is a monotonically increasing counter that starts
12073 /// at 1 for the first commit after `open` and increments with every
12074 /// successful write. WAL replay at open also assigns sequences (1..N for N
12075 /// replayed frames), so sequences are consistent across snapshot+WAL cycles.
12076 ///
12077 /// For V5-V7 stores opened without a V8 snapshot, nodes that were present
12078 /// in the snapshot but not touched by any WAL frame will return `None`
12079 /// (horizon-bounded: CAS against such nodes is only safe after the first
12080 /// V8 snapshot or after the node is next mutated).
12081 pub fn last_changed(&self, key: &str) -> Option<u64> {
12082 let id = self.ids.get(key)?;
12083 self.last_change.get(&id).copied()
12084 }
12085
12086 /// Which loaded store this handle is.
12087 ///
12088 /// Paired with [`commit_seq`](GraphDb::commit_seq) it identifies a graph
12089 /// state outright, which `commit_seq` alone does not: two stores of the same
12090 /// age share a sequence, and a reload can return to one. Memos of dense
12091 /// node ids that the handle does not own are stamped with both; see
12092 /// [`StoreStamp`](crate::mask::StoreStamp).
12093 pub(crate) fn store_id(&self) -> crate::mask::StoreId {
12094 self.store_id
12095 }
12096
12097 /// The current commit sequence (number of successful commits since open,
12098 /// including WAL replay frames). Useful for recording a baseline before
12099 /// a read-modify-write cycle.
12100 pub fn commit_seq(&self) -> u64 {
12101 self.commit_seq
12102 }
12103
12104 /// Check that all `preconds` are satisfied against the current db state.
12105 /// Returns `Err(GraphError::CasConflict)` on the first failing precondition.
12106 pub(crate) fn check_preconditions(&self, preconds: &[Precondition]) -> Result<()> {
12107 for precond in preconds {
12108 match precond {
12109 Precondition::NodeUnchangedSince { key, expected } => {
12110 // Missing entry means the node predates the WAL window or
12111 // does not exist; treat as 0 (before any commit).
12112 let actual = self.last_changed(key).unwrap_or_default();
12113 if actual != *expected {
12114 return Err(GraphError::CasConflict {
12115 key: key.clone(),
12116 expected: *expected,
12117 actual,
12118 });
12119 }
12120 }
12121 Precondition::NodeAbsent { key } => {
12122 // Node must not exist (not live).
12123 if self.ids.get(key).is_some() {
12124 let actual = self.last_changed(key).unwrap_or(0);
12125 return Err(GraphError::CasConflict {
12126 key: key.clone(),
12127 expected: u64::MAX,
12128 actual,
12129 });
12130 }
12131 }
12132 }
12133 }
12134 Ok(())
12135 }
12136
12137 /// Apply a batch of mutations with compare-and-set preconditions.
12138 ///
12139 /// All preconditions are checked atomically before any operation is applied.
12140 /// If any precondition fails, the entire batch is rejected with
12141 /// [`GraphError::CasConflict`] and no WAL frame is written.
12142 ///
12143 /// # Returns
12144 /// `(nodes_inserted, edges_inserted)` on success, same as [`write_batch`].
12145 ///
12146 /// # Errors
12147 /// - [`GraphError::CasConflict`] if any precondition is not satisfied.
12148 /// - Any error that [`write_batch`] would return for the ops themselves.
12149 pub fn write_batch_cas(
12150 &mut self,
12151 preconds: Vec<Precondition>,
12152 ops: Vec<BatchOp>,
12153 ) -> Result<(usize, usize)> {
12154 self.check_preconditions(&preconds)?;
12155 self.commit_logged_batch(ops, None, None).map(inserted_pair)
12156 }
12157
12158 /// Update the per-node last-change map for a WAL record at commit `seq`.
12159 ///
12160 /// Called after a successful apply to record which nodes were touched.
12161 /// For replay, called with the WAL-frame's replayed seq.
12162 ///
12163 /// Touch definition (see [`Precondition`] doc):
12164 /// - InsertNode / InsertNodeId / SetProp / SetPropId / RemoveProp → the node.
12165 /// - InsertEdge / InsertEdgeId / DeleteEdge → both src and dst.
12166 /// - DeleteNode → node tombstoned; last_changed() returns None so no update needed.
12167 /// - DerivedEdge markers, Intern, rule/view records → no-ops.
12168 /// - Batch → recurse into inner records.
12169 fn update_last_change_from_rec(&mut self, rec: &WalRecord, seq: u64) {
12170 match rec {
12171 WalRecord::InsertNode { key, .. }
12172 | WalRecord::SetProp { key, .. }
12173 | WalRecord::RemoveProp { key, .. } => {
12174 if let Some(id) = self.ids.get(key) {
12175 self.last_change.insert(id, seq);
12176 }
12177 }
12178 WalRecord::InsertNodeId { key, .. } => {
12179 if let Some(id) = self.ids.get(key) {
12180 self.last_change.insert(id, seq);
12181 }
12182 }
12183 WalRecord::SetPropId { id, .. } => {
12184 self.last_change.insert(*id, seq);
12185 }
12186 WalRecord::InsertEdge {
12187 src_key, dst_key, ..
12188 }
12189 | WalRecord::DeleteEdge {
12190 src_key, dst_key, ..
12191 } => {
12192 if let Some(src_id) = self.ids.get(src_key) {
12193 self.last_change.insert(src_id, seq);
12194 }
12195 if let Some(dst_id) = self.ids.get(dst_key) {
12196 self.last_change.insert(dst_id, seq);
12197 }
12198 }
12199 WalRecord::InsertEdgeId { src, dst, .. } => {
12200 self.last_change.insert(*src, seq);
12201 self.last_change.insert(*dst, seq);
12202 }
12203 // A count record touches the pair, so it touches both endpoints —
12204 // the same reading `InsertEdgeId` gets, because a duplicate insert
12205 // that raises the count *is* a mutation of that pair. The opt-in
12206 // declaration touches nothing.
12207 WalRecord::SetEdgeCount { src, dst, .. } if !rec.is_multiplicity_decl() => {
12208 self.last_change.insert(*src, seq);
12209 self.last_change.insert(*dst, seq);
12210 }
12211 WalRecord::SetEdgeCount { .. } => {}
12212 // DeleteNode: node is tombstoned; last_changed(key) returns None for
12213 // deleted keys (ids.get() returns None post-tombstone), so no update needed.
12214 // History markers: state no-ops; the underlying mutation already
12215 // touched the relevant nodes' last_change entries.
12216 WalRecord::DeleteNode { .. }
12217 | WalRecord::DerivedEdgeAdded { .. }
12218 | WalRecord::DerivedEdgeRetracted { .. }
12219 | WalRecord::Intern { .. }
12220 | WalRecord::CreateRule { .. }
12221 | WalRecord::DeleteRule { .. }
12222 | WalRecord::RebuildRule { .. }
12223 | WalRecord::CreateView { .. }
12224 | WalRecord::DeleteView { .. }
12225 | WalRecord::EnableFulltext { .. }
12226 | WalRecord::DisableFulltext { .. }
12227 | WalRecord::EnableIndex { .. }
12228 | WalRecord::DisableIndex { .. } => {}
12229 // RenameNode: node id is stable; update last_change via the new key.
12230 // Called after apply(), so ids already reflects new_key.
12231 WalRecord::RenameNode { new_key, .. } => {
12232 if let Some(id) = self.ids.get(new_key) {
12233 self.last_change.insert(id, seq);
12234 }
12235 }
12236 WalRecord::Batch(inner) => {
12237 for inner_rec in inner {
12238 self.update_last_change_from_rec(inner_rec, seq);
12239 }
12240 }
12241 }
12242 }
12243
12244 pub fn node_count(&self) -> usize {
12245 self.ids.len()
12246 }
12247
12248 /// Configure archive retention: keep the `N` newest WAL archives at each
12249 /// [`snapshot_with`] call when `archive_wal: true`.
12250 ///
12251 /// `Some(N)` where N > 0 → prune oldest archives keeping the newest N.
12252 /// `Some(0)` or `None` → unlimited (no pruning).
12253 ///
12254 /// Pruning only ever happens inside [`snapshot_with`]; this method only
12255 /// stores the policy. Archives below the retention limit are deleted
12256 /// oldest-first. The horizon floor is updated so that
12257 /// [`was_linked`] / history APIs return `CommitOutOfRange` for commits
12258 /// in pruned archives rather than silently returning wrong data.
12259 pub fn set_wal_archive_retention(&mut self, keep: Option<u32>) {
12260 self.wal_archive_retention = keep;
12261 }
12262
12263 /// Delete any WAL archives that are fully below the current horizon floor.
12264 ///
12265 /// Orphaned archives arise when the floor is written first during retention
12266 /// pruning and then a crash interrupts the archive-delete sequence. The
12267 /// opening cleanup ensures no subsequent read path sees stale data.
12268 ///
12269 /// Under the monotonic naming scheme, the archive name N equals the
12270 /// cumulative end-frame index of the archive in global commit space (i.e.
12271 /// the archive covers global frames `[prev_n, N)`). An archive is
12272 /// fully orphaned when `N <= wal_horizon_floor`: all of its frames fall
12273 /// below the floor and have already been counted in it.
12274 fn cleanup_orphaned_archives(&mut self) -> Result<()> {
12275 if self.wal_horizon_floor == 0 {
12276 // Floor at 0 means no pruning has ever occurred; nothing to clean.
12277 return Ok(());
12278 }
12279 let archive_ns = self.fs.list_archives()?;
12280 for n in archive_ns {
12281 if n <= self.wal_horizon_floor {
12282 // Archive N ends at global frame N; all its frames are below
12283 // the floor (floor already accounts for them) → orphaned.
12284 self.fs.delete_archive(n).map_err(GraphError::Io)?;
12285 } else {
12286 // Archives are sorted ascending; first one above floor stops scan.
12287 break;
12288 }
12289 }
12290 Ok(())
12291 }
12292
12293 /// Collect all WAL frames from surviving archives (oldest-first) then the
12294 /// live WAL into one flat list, and return the total along with the number
12295 /// of archive frames at the front of the list.
12296 ///
12297 /// Commit indices into the returned list are LOCAL (0 = first frame of
12298 /// oldest surviving archive). To obtain the GLOBAL index add
12299 /// `self.wal_horizon_floor`.
12300 /// How many frames the surviving archives hold, without materialising them.
12301 ///
12302 /// The same count `all_frames` puts at the front of its list. Used to seed
12303 /// [`wal_frames_written`](GraphDb::wal_frames_written) at open without
12304 /// decoding the live WAL a second time; free on a store with no archives,
12305 /// which is most of them.
12306 fn archive_frame_count(&self) -> Result<u64> {
12307 let mut n = 0u64;
12308 for a in self.fs.list_archives()? {
12309 let bytes = self.fs.read_archive(a)?;
12310 let (frames, _) = decode_all(&bytes);
12311 n += frames.len() as u64;
12312 }
12313 Ok(n)
12314 }
12315
12316 fn all_frames(&self) -> Result<(Vec<WalRecord>, u64)> {
12317 let archive_ns = self.fs.list_archives()?;
12318 let mut all: Vec<WalRecord> = Vec::new();
12319 for n in archive_ns {
12320 let bytes = self.fs.read_archive(n)?;
12321 let (frames, _) = decode_all(&bytes);
12322 all.extend(frames);
12323 }
12324 let archive_count = all.len() as u64;
12325 let live_bytes = self.fs.read(FileId::Wal)?;
12326 let (live_frames, _) = decode_all(&live_bytes);
12327 all.extend(live_frames);
12328 Ok((all, archive_count))
12329 }
12330
12331 /// Return the total number of committed WAL frames visible in the current
12332 /// horizon window, including frames in surviving WAL archives.
12333 ///
12334 /// This is the exclusive upper bound for valid `at_commit` indices in
12335 /// `was_linked`. Valid indices are `wal_horizon_floor()..wal_total_commits()`.
12336 ///
12337 /// Returns the horizon floor when all surviving history is empty.
12338 pub fn wal_total_commits(&self) -> Result<u64> {
12339 let (frames, _) = self.all_frames()?;
12340 Ok(self.wal_horizon_floor + frames.len() as u64)
12341 }
12342
12343 /// The global frame index of the first commit reachable through surviving
12344 /// archives (0 when no archives have been pruned).
12345 pub fn wal_horizon_floor(&self) -> u64 {
12346 self.wal_horizon_floor
12347 }
12348
12349 /// Return the per-node change history for `key` by scanning the on-disk WAL.
12350 ///
12351 /// ## Horizon
12352 ///
12353 /// History reaches back only to the last WAL-truncating snapshot, exactly like `open_at`.
12354 /// Snapshots written with `keep_wal: true` preserve deeper history. This is the honest,
12355 /// zero-cost contract; a durable history log is out of scope.
12356 ///
12357 /// ## Derived edges
12358 ///
12359 /// Rule-created (derived) edges are **not** in the WAL and therefore do not appear in
12360 /// history. Only edges written directly by the application are recorded.
12361 ///
12362 /// ## Deleted nodes
12363 ///
12364 /// For nodes that have been deleted, dense-id records (SetPropId, InsertEdgeId) that
12365 /// predate the deletion may not resolve (the id is tombstoned in the live map). The
12366 /// string-keyed `DeleteNode` record still matches and produces a `NodeDeleted` entry.
12367 /// Prop/edge history of a deleted node may therefore be partially unresolvable.
12368 ///
12369 /// ## Dense-id edge entries and tombstoned partners
12370 ///
12371 /// Edge entries from dense-id WAL records (`InsertEdgeId`) are omitted when the partner
12372 /// endpoint's dense id is tombstoned. As a result, a live node's history can contain an
12373 /// `EdgeRemoved` (string-keyed, always resolves) without a corresponding `EdgeAdded`.
12374 /// Build commit-bounded alias intervals for `queried_key`.
12375 ///
12376 /// Returns a list of `(key, valid_from_inclusive, valid_until_exclusive)` tuples.
12377 /// A record written under `key` at commit `c` matches the queried identity iff
12378 /// `c >= valid_from && (valid_until.is_none() || c < valid_until)`.
12379 ///
12380 /// Each alias entry carries both a lower and an upper bound so that key-reuse
12381 /// after a rename is handled correctly: if "a" is renamed to "b" at commit 5,
12382 /// then a NEW node is created as "a" at commit 7 and renamed to "c" at commit 10,
12383 /// querying "c" must NOT surface identity-1's events (commits 0–4 under "a");
12384 /// only identity-2's events (commits 7–9 under "a") are in scope.
12385 ///
12386 /// Only **forward aliasing**: querying the *new* key surfaces events written
12387 /// under the *old* key. The reverse direction is not supported.
12388 fn build_key_alias_intervals(
12389 &self,
12390 frames: &[core_storage::wal::WalRecord],
12391 queried_key: &str,
12392 ) -> Vec<(String, u64, Option<u64>)> {
12393 use core_storage::wal::WalRecord;
12394
12395 // Pre-pass: build reverse_rename and key_starts maps.
12396 let mut reverse_rename: HashMap<String, (String, u64)> = HashMap::new();
12397 let mut key_starts: HashMap<String, Vec<u64>> = HashMap::new();
12398
12399 for (local_i, frame) in frames.iter().enumerate() {
12400 let commit = self.wal_horizon_floor + local_i as u64;
12401 let records: &[WalRecord] = match frame {
12402 WalRecord::Batch(inner) => inner.as_slice(),
12403 single => std::slice::from_ref(single),
12404 };
12405 for rec in records {
12406 match rec {
12407 WalRecord::InsertNode { key, .. } | WalRecord::InsertNodeId { key, .. } => {
12408 key_starts.entry(key.clone()).or_default().push(commit);
12409 }
12410 WalRecord::RenameNode { old_key, new_key } => {
12411 // new_key came into existence at this commit.
12412 key_starts.entry(new_key.clone()).or_default().push(commit);
12413 // Record the reverse rename: new_key was introduced by renaming old_key.
12414 reverse_rename.insert(new_key.clone(), (old_key.clone(), commit));
12415 }
12416 _ => {}
12417 }
12418 }
12419 }
12420
12421 // Build alias intervals by following the reverse rename chain.
12422 let mut result: Vec<(String, u64, Option<u64>)> = Vec::new();
12423 let mut current_key = queried_key.to_string();
12424 let mut current_valid_until: Option<u64> = None;
12425
12426 loop {
12427 // valid_from: the most recent commit where current_key was assigned to this
12428 // identity. For aliases (valid_until = Some(vu)), find the last start event
12429 // for the key strictly before vu — this is where the alias's occupancy by
12430 // this identity began, correctly excluding prior identities that reused the key.
12431 let valid_from = if let Some(vu) = current_valid_until {
12432 key_starts
12433 .get(¤t_key)
12434 .and_then(|starts| starts.iter().rev().find(|&&s| s < vu).copied())
12435 .unwrap_or(self.wal_horizon_floor)
12436 } else {
12437 // Queried key — no upper bound; may have been introduced at any commit.
12438 self.wal_horizon_floor
12439 };
12440
12441 result.push((current_key.clone(), valid_from, current_valid_until));
12442
12443 match reverse_rename.get(¤t_key) {
12444 Some((old_key, rename_commit)) => {
12445 current_valid_until = Some(*rename_commit);
12446 current_key = old_key.clone();
12447 }
12448 None => break,
12449 }
12450 }
12451
12452 result
12453 }
12454
12455 /// Returns true if `record_key` matches any alias interval that covers `commit`.
12456 fn aliases_match(
12457 intervals: &[(String, u64, Option<u64>)],
12458 record_key: &str,
12459 commit: u64,
12460 ) -> bool {
12461 intervals
12462 .iter()
12463 .any(|(k, vf, vu)| k == record_key && commit >= *vf && vu.is_none_or(|u| commit < u))
12464 }
12465
12466 /// Return the change history of node `key` by scanning the on-disk WAL.
12467 ///
12468 /// ## Horizon
12469 ///
12470 /// History reaches back only as far as the retained WAL. The returned
12471 /// [`HistoryResult`](crate::history::HistoryResult) carries `total_commits`
12472 /// (the exclusive upper bound for valid commit indices) and `horizon` (the
12473 /// oldest commit still reachable). When `horizon > 0`, older events were
12474 /// pruned and are not in `items`.
12475 pub fn node_history(
12476 &self,
12477 key: &str,
12478 ) -> Result<crate::history::HistoryResult<crate::history::HistoryEntry>> {
12479 use crate::history::{HistoryChange, HistoryEntry, HistoryResult};
12480 use core_storage::wal::WalRecord;
12481
12482 let (frames, _) = self.all_frames()?;
12483 let total_commits = self.wal_horizon_floor + frames.len() as u64;
12484
12485 // Resolve commit-bounded alias intervals for `key` (handles renames in the WAL).
12486 let alias_intervals = self.build_key_alias_intervals(&frames, key);
12487
12488 let mut out: Vec<HistoryEntry> = Vec::new();
12489
12490 for (local_i, frame) in frames.iter().enumerate() {
12491 let commit = self.wal_horizon_floor + local_i as u64;
12492 // Collect the inner records to process — Batch is one commit, single records are one commit.
12493 let records: &[WalRecord] = match frame {
12494 WalRecord::Batch(inner) => inner.as_slice(),
12495 single => std::slice::from_ref(single),
12496 };
12497
12498 for rec in records {
12499 let change = match rec {
12500 WalRecord::InsertNode { label, key: k, .. }
12501 if Self::aliases_match(&alias_intervals, k, commit) =>
12502 {
12503 Some(HistoryChange::NodeInserted {
12504 label: label.clone(),
12505 })
12506 }
12507 WalRecord::InsertNodeId { label, key: k, .. }
12508 if Self::aliases_match(&alias_intervals, k, commit) =>
12509 {
12510 let label_str = match self.syms.resolve(*label) {
12511 Some(s) => s.to_string(),
12512 None => continue,
12513 };
12514 Some(HistoryChange::NodeInserted { label: label_str })
12515 }
12516 WalRecord::SetProp {
12517 key: k,
12518 field,
12519 value,
12520 } if Self::aliases_match(&alias_intervals, k, commit) => {
12521 Some(HistoryChange::PropSet {
12522 field: field.clone(),
12523 value: value.clone(),
12524 })
12525 }
12526 WalRecord::SetPropId { id, field, value } => {
12527 // Use key_of_historical (not key_of) so a node's prop_set
12528 // events remain visible after the node is later deleted:
12529 // key_of returns None for a tombstoned id, which would
12530 // silently drop every PropSet between insert and delete.
12531 // Mirrors the InsertEdgeId arm below and edge_history's
12532 // own id-keyed arms.
12533 match self.ids.key_of_historical(*id) {
12534 // key_of_historical returns the last-known (possibly
12535 // post-rename, possibly post-delete) key; compare to queried key.
12536 Some(resolved) if resolved == key => {
12537 let field_str = match self.syms.resolve(*field) {
12538 Some(s) => s.to_string(),
12539 None => continue,
12540 };
12541 Some(HistoryChange::PropSet {
12542 field: field_str,
12543 value: value.clone(),
12544 })
12545 }
12546 _ => None,
12547 }
12548 }
12549 WalRecord::RemoveProp { key: k, field }
12550 if Self::aliases_match(&alias_intervals, k, commit) =>
12551 {
12552 Some(HistoryChange::PropRemoved {
12553 field: field.clone(),
12554 })
12555 }
12556 WalRecord::InsertEdge {
12557 edge_type,
12558 src_key,
12559 dst_key,
12560 } => {
12561 if Self::aliases_match(&alias_intervals, src_key, commit) {
12562 Some(HistoryChange::EdgeAdded {
12563 edge_type: edge_type.clone(),
12564 other: dst_key.clone(),
12565 outgoing: true,
12566 })
12567 } else if Self::aliases_match(&alias_intervals, dst_key, commit) {
12568 Some(HistoryChange::EdgeAdded {
12569 edge_type: edge_type.clone(),
12570 other: src_key.clone(),
12571 outgoing: false,
12572 })
12573 } else {
12574 None
12575 }
12576 }
12577 WalRecord::InsertEdgeId { etype, src, dst } => {
12578 let etype_str = match self.syms.resolve(*etype) {
12579 Some(s) => s.to_string(),
12580 None => continue,
12581 };
12582 // key_of_historical (not key_of): an edge added before
12583 // either endpoint was later deleted must still resolve —
12584 // see the SetPropId arm above and edge_history's
12585 // InsertEdgeId arm, which use the same lookup for the
12586 // same reason.
12587 let src_key = self.ids.key_of_historical(*src);
12588 let dst_key = self.ids.key_of_historical(*dst);
12589 if src_key == Some(key) {
12590 let other = match dst_key {
12591 Some(s) => s.to_string(),
12592 None => continue,
12593 };
12594 Some(HistoryChange::EdgeAdded {
12595 edge_type: etype_str,
12596 other,
12597 outgoing: true,
12598 })
12599 } else if dst_key == Some(key) {
12600 let other = match src_key {
12601 Some(s) => s.to_string(),
12602 None => continue,
12603 };
12604 Some(HistoryChange::EdgeAdded {
12605 edge_type: etype_str,
12606 other,
12607 outgoing: false,
12608 })
12609 } else {
12610 None
12611 }
12612 }
12613 WalRecord::DeleteEdge {
12614 edge_type,
12615 src_key,
12616 dst_key,
12617 } => {
12618 if Self::aliases_match(&alias_intervals, src_key, commit) {
12619 Some(HistoryChange::EdgeRemoved {
12620 edge_type: edge_type.clone(),
12621 other: dst_key.clone(),
12622 outgoing: true,
12623 })
12624 } else if Self::aliases_match(&alias_intervals, dst_key, commit) {
12625 Some(HistoryChange::EdgeRemoved {
12626 edge_type: edge_type.clone(),
12627 other: src_key.clone(),
12628 outgoing: false,
12629 })
12630 } else {
12631 None
12632 }
12633 }
12634 WalRecord::DeleteNode { key: k }
12635 if Self::aliases_match(&alias_intervals, k, commit) =>
12636 {
12637 Some(HistoryChange::NodeDeleted)
12638 }
12639 // Skip: rule/view/fulltext/intern metadata; Batch wrapper handled above.
12640 _ => None,
12641 };
12642
12643 if let Some(change) = change {
12644 out.push(HistoryEntry { commit, change });
12645 }
12646 }
12647 }
12648
12649 Ok(HistoryResult {
12650 items: out,
12651 total_commits,
12652 horizon: self.wal_horizon_floor,
12653 })
12654 }
12655
12656 /// Return the per-edge change history between nodes `a` and `b` by scanning
12657 /// the on-disk WAL.
12658 ///
12659 /// ## Horizon
12660 ///
12661 /// History reaches back only to the last WAL-truncating snapshot, exactly
12662 /// like `node_history` and `open_at`. The returned [`HistoryResult`] carries
12663 /// `total_commits` (= number of WAL frames), which is the exclusive upper
12664 /// bound for valid commit indices.
12665 ///
12666 /// ## Derived edges
12667 ///
12668 /// Rule-derived edges appear via `DerivedEdgeAdded` / `DerivedEdgeRetracted`
12669 /// WAL markers written by `log_then_apply_with` after each rule-firing
12670 /// mutation. The `rule` field of those events carries the rule name.
12671 ///
12672 /// ## DeleteNode
12673 ///
12674 /// When a node is deleted, its manual incident edges are swept inline without
12675 /// individual `DeleteEdge` WAL records. `edge_history` detects `DeleteNode`
12676 /// events for either endpoint and synthesises `Retracted(rule:None)` events
12677 /// for each manual edge that was active at that point. Derived edges active at
12678 /// the time of deletion are handled by the `DerivedEdgeRetracted` marker that
12679 /// the engine appends immediately after the `DeleteNode` record; those events
12680 /// carry correct rule attribution and are emitted by the marker arm, not the
12681 /// synthetic sweep.
12682 ///
12683 /// ## Masks
12684 ///
12685 /// Like `node_history`, this method has no mask parameter and returns WAL
12686 /// history regardless of any role mask. For masked history semantics, apply
12687 /// the mask at the caller level.
12688 pub fn edge_history(
12689 &self,
12690 a: &str,
12691 b: &str,
12692 ) -> Result<crate::history::HistoryResult<crate::history::EdgeHistoryEvent>> {
12693 use crate::history::{EdgeEvent, EdgeHistoryEvent, HistoryResult};
12694 use core_storage::wal::WalRecord;
12695
12696 let (frames, _) = self.all_frames()?;
12697 let total_commits = self.wal_horizon_floor + frames.len() as u64;
12698
12699 // Resolve all historical names for a and b (handles RenameNode in the WAL).
12700 // Intervals are commit-bounded so recycled keys don't contaminate histories.
12701 let alias_a = self.build_key_alias_intervals(&frames, a);
12702 let alias_b = self.build_key_alias_intervals(&frames, b);
12703
12704 // Active edges between a and b tracked as (edge_type, src_key, dst_key, is_derived).
12705 // The is_derived flag is used by the DeleteNode sweep: manual edges are
12706 // swept with a synthetic Retracted(rule:None); derived edges are skipped
12707 // because the engine writes a DerivedEdgeRetracted marker immediately after
12708 // the DeleteNode record, which carries the correct rule attribution.
12709 let mut active: Vec<(String, String, String, bool)> = Vec::new();
12710 let mut out: Vec<EdgeHistoryEvent> = Vec::new();
12711
12712 for (local_i, frame) in frames.iter().enumerate() {
12713 let commit = self.wal_horizon_floor + local_i as u64;
12714 let records: &[WalRecord] = match frame {
12715 WalRecord::Batch(inner) => inner.as_slice(),
12716 single => std::slice::from_ref(single),
12717 };
12718
12719 for rec in records {
12720 match rec {
12721 WalRecord::InsertEdge {
12722 edge_type,
12723 src_key,
12724 dst_key,
12725 } => {
12726 let is_ab = Self::aliases_match(&alias_a, src_key, commit)
12727 && Self::aliases_match(&alias_b, dst_key, commit);
12728 let is_ba = Self::aliases_match(&alias_b, src_key, commit)
12729 && Self::aliases_match(&alias_a, dst_key, commit);
12730 if is_ab || is_ba {
12731 active.push((
12732 edge_type.clone(),
12733 src_key.clone(),
12734 dst_key.clone(),
12735 false,
12736 ));
12737 out.push(EdgeHistoryEvent {
12738 edge_type: edge_type.clone(),
12739 commit,
12740 event: EdgeEvent::Added,
12741 rule: None,
12742 });
12743 }
12744 }
12745 WalRecord::InsertEdgeId { etype, src, dst } => {
12746 let etype_str = match self.syms.resolve(*etype) {
12747 Some(s) => s.to_string(),
12748 None => continue,
12749 };
12750 // Use key_of_historical so tombstoned nodes (deleted
12751 // later in the WAL) still resolve during the scan.
12752 let src_key = self.ids.key_of_historical(*src);
12753 let dst_key = self.ids.key_of_historical(*dst);
12754 let is_ab = src_key == Some(a) && dst_key == Some(b);
12755 let is_ba = src_key == Some(b) && dst_key == Some(a);
12756 if is_ab || is_ba {
12757 let src_str = src_key.unwrap().to_string();
12758 let dst_str = dst_key.unwrap().to_string();
12759 active.push((etype_str.clone(), src_str, dst_str, false));
12760 out.push(EdgeHistoryEvent {
12761 edge_type: etype_str,
12762 commit,
12763 event: EdgeEvent::Added,
12764 rule: None,
12765 });
12766 }
12767 }
12768 WalRecord::DeleteEdge {
12769 edge_type,
12770 src_key,
12771 dst_key,
12772 } => {
12773 let is_ab = Self::aliases_match(&alias_a, src_key, commit)
12774 && Self::aliases_match(&alias_b, dst_key, commit);
12775 let is_ba = Self::aliases_match(&alias_b, src_key, commit)
12776 && Self::aliases_match(&alias_a, dst_key, commit);
12777 if is_ab || is_ba {
12778 // Remove the first matching active entry (flag ignored).
12779 if let Some(pos) = active.iter().position(|(et, s, d, _)| {
12780 et == edge_type && s == src_key && d == dst_key
12781 }) {
12782 active.remove(pos);
12783 }
12784 out.push(EdgeHistoryEvent {
12785 edge_type: edge_type.clone(),
12786 commit,
12787 event: EdgeEvent::Retracted,
12788 rule: None,
12789 });
12790 }
12791 }
12792 WalRecord::DeleteNode { key: k }
12793 if Self::aliases_match(&alias_a, k, commit)
12794 || Self::aliases_match(&alias_b, k, commit) =>
12795 {
12796 // Sweep: implicitly retract only MANUAL active edges.
12797 // Derived active edges are skipped here because the rule
12798 // engine appends a DerivedEdgeRetracted marker immediately
12799 // after this DeleteNode record; that marker produces the
12800 // single correctly-attributed Retracted event. Derived
12801 // entries are dropped from `active` (the marker arm's
12802 // idempotent retain finds nothing to remove).
12803 for (et, _, _, is_derived) in active.drain(..) {
12804 if !is_derived {
12805 out.push(EdgeHistoryEvent {
12806 edge_type: et,
12807 commit,
12808 event: EdgeEvent::Retracted,
12809 rule: None,
12810 });
12811 }
12812 // Derived: drop silently; marker carries the Retracted event.
12813 }
12814 }
12815 WalRecord::DerivedEdgeAdded {
12816 rule,
12817 edge_type: et,
12818 src_key,
12819 dst_key,
12820 } => {
12821 let is_ab = Self::aliases_match(&alias_a, src_key, commit)
12822 && Self::aliases_match(&alias_b, dst_key, commit);
12823 let is_ba = Self::aliases_match(&alias_b, src_key, commit)
12824 && Self::aliases_match(&alias_a, dst_key, commit);
12825 if is_ab || is_ba {
12826 active.push((et.clone(), src_key.clone(), dst_key.clone(), true));
12827 out.push(EdgeHistoryEvent {
12828 edge_type: et.clone(),
12829 commit,
12830 event: EdgeEvent::Added,
12831 rule: Some(rule.clone()),
12832 });
12833 }
12834 }
12835 WalRecord::DerivedEdgeRetracted {
12836 rule,
12837 edge_type: et,
12838 src_key,
12839 dst_key,
12840 } => {
12841 let is_ab = Self::aliases_match(&alias_a, src_key, commit)
12842 && Self::aliases_match(&alias_b, dst_key, commit);
12843 let is_ba = Self::aliases_match(&alias_b, src_key, commit)
12844 && Self::aliases_match(&alias_a, dst_key, commit);
12845 if is_ab || is_ba {
12846 // Push unconditionally: a derived edge whose Added marker
12847 // predates the history horizon has no `active` entry, but
12848 // the retraction is still a real in-window event.
12849 // Remove from active idempotently if present.
12850 active.retain(|(aet, s, d, _)| {
12851 !(aet == et && s == src_key && d == dst_key)
12852 });
12853 out.push(EdgeHistoryEvent {
12854 edge_type: et.clone(),
12855 commit,
12856 event: EdgeEvent::Retracted,
12857 rule: Some(rule.clone()),
12858 });
12859 }
12860 }
12861 // All other records (InsertNode, SetProp, CreateRule, etc.)
12862 // do not affect edges between a and b.
12863 _ => {}
12864 }
12865 }
12866 }
12867
12868 Ok(HistoryResult {
12869 items: out,
12870 total_commits,
12871 horizon: self.wal_horizon_floor,
12872 })
12873 }
12874
12875 /// Return `true` iff an edge of `edge_type` existed between `a` and `b`
12876 /// (in either direction) at the WAL commit `at_commit`.
12877 ///
12878 /// ## Horizon
12879 ///
12880 /// Valid commit indices are `0..total_commits` where `total_commits` is the
12881 /// number of WAL frames. An `at_commit >= total_commits` is outside the
12882 /// visible horizon and returns [`GraphError::CommitOutOfRange`].
12883 ///
12884 /// ## Derived edges
12885 ///
12886 /// Rule-derived edges are tracked via `DerivedEdgeAdded` / `DerivedEdgeRetracted`
12887 /// WAL markers appended at firing time (Task 1). `was_linked` reads these markers
12888 /// and therefore includes derived edges in its point-in-time evaluation,
12889 /// matching `edge_history`'s fidelity.
12890 pub fn was_linked(&self, a: &str, b: &str, edge_type: &str, at_commit: u64) -> Result<bool> {
12891 use core_storage::wal::WalRecord;
12892
12893 let (frames, _) = self.all_frames()?;
12894 let total_commits = self.wal_horizon_floor + frames.len() as u64;
12895
12896 // Horizon floor: commits in pruned archives are unreachable.
12897 if at_commit < self.wal_horizon_floor {
12898 return Err(GraphError::CommitOutOfRange {
12899 commit: at_commit,
12900 total: total_commits,
12901 floor: self.wal_horizon_floor,
12902 });
12903 }
12904 if at_commit >= total_commits {
12905 return Err(GraphError::CommitOutOfRange {
12906 commit: at_commit,
12907 total: total_commits,
12908 floor: self.wal_horizon_floor,
12909 });
12910 }
12911
12912 // Resolve all historical names for a and b (handles RenameNode in the WAL).
12913 // Intervals are commit-bounded so recycled keys don't contaminate point-in-time reads.
12914 let alias_a = self.build_key_alias_intervals(&frames, a);
12915 let alias_b = self.build_key_alias_intervals(&frames, b);
12916
12917 // Local index into surviving frames (0 = first frame of oldest archive).
12918 let local_commit = at_commit - self.wal_horizon_floor;
12919
12920 // Replay local frames 0..=local_commit, tracking active edges.
12921 let mut active: BTreeSet<(String, String, String)> = BTreeSet::new();
12922
12923 for (local_i, frame) in frames.iter().enumerate().take((local_commit + 1) as usize) {
12924 let commit = self.wal_horizon_floor + local_i as u64;
12925 let records: &[WalRecord] = match frame {
12926 WalRecord::Batch(inner) => inner.as_slice(),
12927 single => std::slice::from_ref(single),
12928 };
12929
12930 for rec in records {
12931 match rec {
12932 WalRecord::InsertEdge {
12933 edge_type: et,
12934 src_key,
12935 dst_key,
12936 } => {
12937 let is_ab = Self::aliases_match(&alias_a, src_key, commit)
12938 && Self::aliases_match(&alias_b, dst_key, commit);
12939 let is_ba = Self::aliases_match(&alias_b, src_key, commit)
12940 && Self::aliases_match(&alias_a, dst_key, commit);
12941 if is_ab || is_ba {
12942 active.insert((et.clone(), src_key.clone(), dst_key.clone()));
12943 }
12944 }
12945 WalRecord::InsertEdgeId { etype, src, dst } => {
12946 let etype_str = match self.syms.resolve(*etype) {
12947 Some(s) => s.to_string(),
12948 None => continue,
12949 };
12950 // Use key_of_historical so tombstoned nodes resolve.
12951 let src_key = self.ids.key_of_historical(*src);
12952 let dst_key = self.ids.key_of_historical(*dst);
12953 let is_ab = src_key == Some(a) && dst_key == Some(b);
12954 let is_ba = src_key == Some(b) && dst_key == Some(a);
12955 if is_ab || is_ba {
12956 active.insert((
12957 etype_str,
12958 src_key.unwrap().to_string(),
12959 dst_key.unwrap().to_string(),
12960 ));
12961 }
12962 }
12963 WalRecord::DeleteEdge {
12964 edge_type: et,
12965 src_key,
12966 dst_key,
12967 } => {
12968 let is_ab = Self::aliases_match(&alias_a, src_key, commit)
12969 && Self::aliases_match(&alias_b, dst_key, commit);
12970 let is_ba = Self::aliases_match(&alias_b, src_key, commit)
12971 && Self::aliases_match(&alias_a, dst_key, commit);
12972 if is_ab || is_ba {
12973 active.remove(&(et.clone(), src_key.clone(), dst_key.clone()));
12974 }
12975 }
12976 WalRecord::DeleteNode { key: k }
12977 if Self::aliases_match(&alias_a, k, commit)
12978 || Self::aliases_match(&alias_b, k, commit) =>
12979 {
12980 // All edges touching the deleted node are gone.
12981 active.retain(|(_, s, d)| s != k && d != k);
12982 }
12983 WalRecord::DerivedEdgeAdded {
12984 edge_type: et,
12985 src_key,
12986 dst_key,
12987 ..
12988 } => {
12989 let is_ab = Self::aliases_match(&alias_a, src_key, commit)
12990 && Self::aliases_match(&alias_b, dst_key, commit);
12991 let is_ba = Self::aliases_match(&alias_b, src_key, commit)
12992 && Self::aliases_match(&alias_a, dst_key, commit);
12993 if is_ab || is_ba {
12994 active.insert((et.clone(), src_key.clone(), dst_key.clone()));
12995 }
12996 }
12997 WalRecord::DerivedEdgeRetracted {
12998 edge_type: et,
12999 src_key,
13000 dst_key,
13001 ..
13002 } => {
13003 let is_ab = Self::aliases_match(&alias_a, src_key, commit)
13004 && Self::aliases_match(&alias_b, dst_key, commit);
13005 let is_ba = Self::aliases_match(&alias_b, src_key, commit)
13006 && Self::aliases_match(&alias_a, dst_key, commit);
13007 if is_ab || is_ba {
13008 active.remove(&(et.clone(), src_key.clone(), dst_key.clone()));
13009 }
13010 }
13011 _ => {}
13012 }
13013 }
13014 }
13015
13016 Ok(active.iter().any(|(et, _, _)| et == edge_type))
13017 }
13018
13019 /// Every edge incident to `key` — either endpoint — that existed at WAL
13020 /// commit `commit`, from ONE scan of the WAL.
13021 ///
13022 /// This is the bulk form of [`was_linked`](GraphDb::was_linked): answering
13023 /// "what did K's relationships look like at commit C" with one call instead
13024 /// of one [`edge_history`](GraphDb::edge_history) per candidate partner.
13025 /// The two agree edge for edge.
13026 ///
13027 /// Results are sorted by `(edge_type, src_key, dst_key)`.
13028 ///
13029 /// ## Horizon
13030 ///
13031 /// Valid commit indices are `wal_horizon_floor()..wal_total_commits()`;
13032 /// anything outside is [`GraphError::CommitOutOfRange`], exactly like
13033 /// `was_linked`. An unknown key is not an error — it simply had no edges.
13034 ///
13035 /// ## Derived edges
13036 ///
13037 /// `DerivedEdgeAdded` / `DerivedEdgeRetracted` markers carry rule
13038 /// attribution, so a rule-owned edge comes back with `derived: true` and
13039 /// `rule: Some(name)`.
13040 ///
13041 /// ## Renames
13042 ///
13043 /// `key` is matched through the same commit-bounded alias intervals
13044 /// `edge_history` uses, so querying a node's *current* key surfaces edges
13045 /// written under an earlier name. Endpoint keys in the result are reported
13046 /// under the name the node carries today, so they can be fed straight back
13047 /// into `node_info`, `explain` or another `edges_at`.
13048 ///
13049 /// ## Masks
13050 ///
13051 /// Like `edge_history` and `node_history`, this reads the WAL regardless of
13052 /// any role mask. Apply masking at the caller level.
13053 pub fn edges_at(&self, key: &str, commit: u64) -> Result<Vec<EdgeAt>> {
13054 use core_storage::wal::WalRecord;
13055
13056 let (frames, _) = self.all_frames()?;
13057 let total_commits = self.wal_horizon_floor + frames.len() as u64;
13058
13059 // Horizon floor: commits in pruned archives are unreachable.
13060 if commit < self.wal_horizon_floor || commit >= total_commits {
13061 return Err(GraphError::CommitOutOfRange {
13062 commit,
13063 total: total_commits,
13064 floor: self.wal_horizon_floor,
13065 });
13066 }
13067
13068 // Commit-bounded historical names of `key` (handles RenameNode).
13069 let alias = self.build_key_alias_intervals(&frames, key);
13070
13071 // Forward rename chain, for reporting endpoints under their current
13072 // names: old key → [(commit, new key)] in ascending commit order.
13073 // Built over the whole WAL, not just the prefix up to `commit`, because
13074 // a rename after `commit` still changes what the node is called today.
13075 let mut renames: HashMap<String, Vec<(u64, String)>> = HashMap::new();
13076 for (local_i, frame) in frames.iter().enumerate() {
13077 let c = self.wal_horizon_floor + local_i as u64;
13078 let records: &[WalRecord] = match frame {
13079 WalRecord::Batch(inner) => inner.as_slice(),
13080 single => std::slice::from_ref(single),
13081 };
13082 for rec in records {
13083 if let WalRecord::RenameNode { old_key, new_key } = rec {
13084 renames
13085 .entry(old_key.clone())
13086 .or_default()
13087 .push((c, new_key.clone()));
13088 }
13089 }
13090 }
13091
13092 // The name a node written as `k` at commit `from` carries today.
13093 // Follows the first rename at or after `from`, then keeps going. The
13094 // iteration cap bounds a rename cycle inside a single batch.
13095 let canon = |k: &str, from: u64| -> String {
13096 if renames.is_empty() {
13097 return k.to_string();
13098 }
13099 let mut cur = k.to_string();
13100 let mut at = from;
13101 for _ in 0..64 {
13102 match renames
13103 .get(&cur)
13104 .and_then(|v| v.iter().find(|(c, _)| *c >= at))
13105 {
13106 Some((c, new)) => {
13107 at = *c;
13108 cur = new.clone();
13109 }
13110 None => break,
13111 }
13112 }
13113 cur
13114 };
13115
13116 let local_commit = commit - self.wal_horizon_floor;
13117 // (edge_type, src_key, dst_key) → (derived, rule)
13118 let mut active: BTreeMap<(String, String, String), (bool, Option<String>)> =
13119 BTreeMap::new();
13120
13121 for (local_i, frame) in frames.iter().enumerate().take((local_commit + 1) as usize) {
13122 let c = self.wal_horizon_floor + local_i as u64;
13123 let records: &[WalRecord] = match frame {
13124 WalRecord::Batch(inner) => inner.as_slice(),
13125 single => std::slice::from_ref(single),
13126 };
13127
13128 for rec in records {
13129 match rec {
13130 WalRecord::InsertEdge {
13131 edge_type,
13132 src_key,
13133 dst_key,
13134 } => {
13135 if Self::aliases_match(&alias, src_key, c)
13136 || Self::aliases_match(&alias, dst_key, c)
13137 {
13138 active.insert(
13139 (edge_type.clone(), canon(src_key, c), canon(dst_key, c)),
13140 (false, None),
13141 );
13142 }
13143 }
13144 WalRecord::InsertEdgeId { etype, src, dst } => {
13145 let Some(etype_str) = self.syms.resolve(*etype) else {
13146 continue;
13147 };
13148 // `key_of_historical` resolves tombstoned ids too, and
13149 // already returns the node's current key — no rename
13150 // canonicalisation needed on this arm.
13151 let (Some(src_key), Some(dst_key)) = (
13152 self.ids.key_of_historical(*src),
13153 self.ids.key_of_historical(*dst),
13154 ) else {
13155 continue;
13156 };
13157 if src_key == key || dst_key == key {
13158 active.insert(
13159 (
13160 etype_str.to_string(),
13161 src_key.to_string(),
13162 dst_key.to_string(),
13163 ),
13164 (false, None),
13165 );
13166 }
13167 }
13168 WalRecord::DeleteEdge {
13169 edge_type,
13170 src_key,
13171 dst_key,
13172 } => {
13173 if Self::aliases_match(&alias, src_key, c)
13174 || Self::aliases_match(&alias, dst_key, c)
13175 {
13176 active.remove(&(
13177 edge_type.clone(),
13178 canon(src_key, c),
13179 canon(dst_key, c),
13180 ));
13181 }
13182 }
13183 WalRecord::DeleteNode { key: k } => {
13184 if active.is_empty() {
13185 continue;
13186 }
13187 if Self::aliases_match(&alias, k, c) {
13188 // Our node is gone; every incident edge goes with it.
13189 active.clear();
13190 } else {
13191 // A partner is gone; its edges to us go with it.
13192 let ck = canon(k, c);
13193 active.retain(|(_, s, d), _| *s != ck && *d != ck);
13194 }
13195 }
13196 WalRecord::DerivedEdgeAdded {
13197 rule,
13198 edge_type,
13199 src_key,
13200 dst_key,
13201 } => {
13202 if Self::aliases_match(&alias, src_key, c)
13203 || Self::aliases_match(&alias, dst_key, c)
13204 {
13205 active.insert(
13206 (edge_type.clone(), canon(src_key, c), canon(dst_key, c)),
13207 (true, Some(rule.clone())),
13208 );
13209 }
13210 }
13211 WalRecord::DerivedEdgeRetracted {
13212 edge_type,
13213 src_key,
13214 dst_key,
13215 ..
13216 } => {
13217 if Self::aliases_match(&alias, src_key, c)
13218 || Self::aliases_match(&alias, dst_key, c)
13219 {
13220 active.remove(&(
13221 edge_type.clone(),
13222 canon(src_key, c),
13223 canon(dst_key, c),
13224 ));
13225 }
13226 }
13227 // InsertNode, SetProp, CreateRule, … do not move edges.
13228 _ => {}
13229 }
13230 }
13231 }
13232
13233 // BTreeMap iteration is already (edge_type, src, dst) order.
13234 Ok(active
13235 .into_iter()
13236 .map(|((edge_type, src_key, dst_key), (derived, rule))| EdgeAt {
13237 edge_type,
13238 src_key,
13239 dst_key,
13240 derived,
13241 rule,
13242 })
13243 .collect())
13244 }
13245
13246 /// The derived edges that would be retracted and derived if `key.field`
13247 /// were set to `value` — computed WITHOUT writing anything.
13248 ///
13249 /// Nothing is committed and nothing on `self` is mutated: the rule engine's
13250 /// provenance, its candidate indexes, the topology and the property columns
13251 /// are all cloned first, the change is applied to the clone, and the real
13252 /// per-node re-derivation (`RuleEngine::on_node_changed` — the same call
13253 /// `set_prop` makes during apply) runs against it. The derived-edge deltas
13254 /// it emits are the answer, so rule semantics — predicates, top-k,
13255 /// via-hops, chaining, weights — are the engine's, not a re-implementation.
13256 ///
13257 /// Works on a read-only handle.
13258 ///
13259 /// **While a rule's vector index is still building** (`RuleStats::building`)
13260 /// the clone carries no pending-build state, so this reports the edges that
13261 /// rule would derive — which the live store will not derive until its
13262 /// backfill runs. Right about the end state, early about the timing.
13263 ///
13264 /// Returns `Err(KeyNotFound)` for an unknown or tombstoned key and
13265 /// `Err(ViewPropReadOnly)` for a field a view owns — matching
13266 /// [`set_prop`](GraphDb::set_prop)'s validation. A change with no effect
13267 /// (the node already holds `value`, or no rule watches `field`) returns
13268 /// empty lists.
13269 ///
13270 /// ## Cost
13271 ///
13272 /// One clone of the property columns, the topology overlay, the symbol
13273 /// interner, the edge properties and the provenance map, plus one candidate
13274 /// re-index (O(nodes × rules)). That is much cheaper than copying the store
13275 /// directory, but it is not free — this is an interactive "what if", not a
13276 /// hot path.
13277 pub fn what_if_set_prop(&self, key: &str, field: &str, value: Value) -> Result<WhatIf> {
13278 // The engine's provenance, HNSW and IVF state live in the mmap'd base
13279 // until something asks for them. On a store opened cold from a snapshot
13280 // this is the first ask, and without it the clone below starts from an
13281 // empty provenance map: nothing to retract, so `lost` comes back empty.
13282 self.ensure_v8_base_sections_loaded();
13283
13284 let empty = WhatIf {
13285 lost: Vec::new(),
13286 gained: Vec::new(),
13287 };
13288
13289 if let Some(view_name) = self.view_store.view_for_prop(field) {
13290 return Err(GraphError::ViewPropReadOnly {
13291 view_name: view_name.to_string(),
13292 });
13293 }
13294 MutPreview::new(self).check_live_key(key)?;
13295 let id = self
13296 .ids
13297 .get(key)
13298 .ok_or_else(|| GraphError::KeyNotFound { key: key.into() })?;
13299
13300 let rules: Vec<RuleDef> = self.engine.rules().cloned().collect();
13301 if rules.is_empty() {
13302 return Ok(empty);
13303 }
13304
13305 // No rule watches this field → no derivation can change.
13306 if !rules.iter().any(|r| r.watched_fields().contains(field)) {
13307 return Ok(empty);
13308 }
13309
13310 let old_value = build_props_view(&self.props, &self.base)
13311 .get(id, field)
13312 .map(|vr| vr.into_value());
13313 if old_value.as_ref() == Some(&value) {
13314 return Ok(empty);
13315 }
13316
13317 // --- Clone every piece of state the re-derivation writes to. ---
13318 // `what_if` mutates its own throwaway copies and writes nothing, so it
13319 // takes owned clones rather than sharing the `Arc`s. Deref-clone, not
13320 // `Arc::clone`: sharing here would make `make_mut` copy on first touch
13321 // anyway, and an owned local keeps the rest of this function unchanged.
13322 let mut props = (*self.props).clone();
13323 let mut topo = (*self.topo).clone();
13324 let mut syms = (*self.syms).clone();
13325 let mut edge_props = (*self.edge_props).clone();
13326
13327 let mut tripped: BTreeMap<String, bool> = BTreeMap::new();
13328 let mut fires: BTreeMap<String, u64> = BTreeMap::new();
13329 for r in &rules {
13330 tripped.insert(r.name.clone(), self.engine.is_tripped(&r.name));
13331 fires.insert(r.name.clone(), self.engine.fire_count(&r.name));
13332 }
13333 // `provenance()` decodes retained snapshot bytes on first use; the
13334 // engine clone needs the real map, not an empty one.
13335 let provenance = self.engine.provenance().clone();
13336 let mut engine = core_rules::RuleEngine::from_persist(rules, provenance, tripped, fires);
13337
13338 // Build the candidate indexes from the state BEFORE the change, exactly
13339 // as apply() sees them: `on_node_changed` withdraws the node under its
13340 // old value and refiles it under the new one, so the index must not
13341 // already reflect the change.
13342 engine.reindex_all_load_state(
13343 &self.ids,
13344 &syms,
13345 &self.labels,
13346 build_props_view(&self.props, &self.base),
13347 self.engine.export_ivf_state(),
13348 self.engine.export_hnsw_state_passthrough(),
13349 );
13350 engine.set_emit_deltas(true);
13351
13352 // --- Apply the hypothetical change and re-derive. ---
13353 props.set(id, field, value);
13354 {
13355 let mut gm = make_graph_mut(
13356 &self.ids,
13357 &mut syms,
13358 &self.labels,
13359 build_props_view(&props, &self.base),
13360 &mut topo,
13361 &self.base,
13362 &mut edge_props,
13363 );
13364 engine.on_node_changed(id, Some((field, old_value)), &mut gm);
13365 }
13366
13367 let mut lost: BTreeSet<EdgeAt> = BTreeSet::new();
13368 let mut gained: BTreeSet<EdgeAt> = BTreeSet::new();
13369 for d in engine.drain_deltas() {
13370 let edge = EdgeAt {
13371 edge_type: d.edge_type,
13372 src_key: d.src_key,
13373 dst_key: d.dst_key,
13374 derived: true,
13375 rule: Some(d.rule),
13376 };
13377 if d.fired {
13378 gained.insert(edge);
13379 } else {
13380 lost.insert(edge);
13381 }
13382 }
13383 // An edge retracted and re-derived within the same re-derivation (top-k
13384 // churn) is not a change the caller would see.
13385 let churn: Vec<EdgeAt> = lost.intersection(&gained).cloned().collect();
13386 for e in churn {
13387 lost.remove(&e);
13388 gained.remove(&e);
13389 }
13390
13391 Ok(WhatIf {
13392 lost: lost.into_iter().collect(),
13393 gained: gained.into_iter().collect(),
13394 })
13395 }
13396
13397 pub fn edge_count(&self) -> u64 {
13398 self.topo_view().edge_count()
13399 }
13400
13401 /// Live/tombstone/edge counts plus per-rule provenance size, trip latch,
13402 /// and fire counter (includes rebuild evaluations). Rules are sorted by name.
13403 pub fn stats(&self) -> Stats {
13404 self.ensure_v8_base_sections_loaded();
13405 let building = self.engine.builds_in_progress();
13406 let rules: Vec<RuleStats> = self
13407 .engine
13408 .rules()
13409 .map(|r| RuleStats {
13410 name: r.name.clone(),
13411 edges: self
13412 .engine
13413 .provenance()
13414 .get(&r.name)
13415 .map(|s| s.len() as u64)
13416 .unwrap_or(0),
13417 tripped: self.engine.is_tripped(&r.name),
13418 fires: self.engine.fire_count(&r.name),
13419 approximate: r.approximate,
13420 building: building.iter().find(|b| b.rule == r.name).cloned(),
13421 })
13422 .collect();
13423 Stats {
13424 nodes_live: self.ids.live_len(),
13425 nodes_tombstoned: self.ids.len() - self.ids.live_len(),
13426 edges: self.topo_view().edge_count(),
13427 rules,
13428 chain_truncations: self.engine.chain_truncations(),
13429 history_floor: self.wal_horizon_floor,
13430 namespaces: self.namespace_stats(),
13431 }
13432 }
13433
13434 /// On-disk size of the WAL file in bytes.
13435 ///
13436 /// Reads file metadata without loading WAL contents. Returns `Err` for
13437 /// in-memory (`SimFs`) databases where no WAL file exists on disk.
13438 pub fn wal_size_bytes(&self) -> std::io::Result<u64> {
13439 let path = self.fs.wal_path().ok_or_else(|| {
13440 std::io::Error::new(
13441 std::io::ErrorKind::Unsupported,
13442 "wal_path not available for this Fs implementation",
13443 )
13444 })?;
13445 Ok(std::fs::metadata(path)?.len())
13446 }
13447
13448 /// Set the slow-query threshold. Queries whose execution time equals or
13449 /// exceeds `ms` milliseconds are logged. Pass `0` to disable.
13450 ///
13451 /// Use this setter in tests — the environment variable
13452 /// `MUSHROOMDB_SLOW_QUERY_MS` is process-global and races parallel test
13453 /// threads.
13454 pub fn set_slow_query_threshold_ms(&mut self, ms: u64) {
13455 self.slow_query_threshold_ms = ms;
13456 }
13457
13458 /// Snapshot of the slow-query ring buffer and lifetime counter.
13459 pub fn slow_query_snapshot(&self) -> SlowQuerySnapshot {
13460 let log = self.slow_queries.lock().unwrap_or_else(|e| e.into_inner());
13461 SlowQuerySnapshot {
13462 threshold_ms: self.slow_query_threshold_ms,
13463 count: log.total,
13464 last: log.entries.iter().cloned().collect(),
13465 }
13466 }
13467
13468 /// Instant the database was opened. Used by consumers (e.g. `/metrics`)
13469 /// to compute uptime.
13470 pub fn started_at(&self) -> std::time::Instant {
13471 self.started_at
13472 }
13473
13474 /// The on-disk snapshot version a store that has opted in to nothing
13475 /// writes — the **floor**, not the whole answer.
13476 ///
13477 /// It is not "the version this binary writes", and it is not "the version
13478 /// this binary reads". Since v0.6.10 this binary writes 9 **or** 10
13479 /// depending on the store — [`snapshot::version_for`] decides, and a store
13480 /// that has called [`enable_multiplicity`](Self::enable_multiplicity)
13481 /// writes 10 — and it reads 5 through 10. A caller comparing a store's
13482 /// stamp against this value must use `>=`, not `==`, or it will report an
13483 /// opted-in store as needing a migration *down*; `cli::run_migrate` is the
13484 /// worked example.
13485 ///
13486 /// The name is kept for compatibility: it is public API reachable from the
13487 /// CLI and from any embedder, and respelling it would break them for a
13488 /// doc-level clarification.
13489 ///
13490 /// [`snapshot::version_for`]: core_storage::snapshot::version_for
13491 pub fn format_version() -> u16 {
13492 core_storage::snapshot::VERSION
13493 }
13494
13495 /// Test-support: total bytes appended (SimFs only usage).
13496 pub fn fs_total_appended(&self) -> usize
13497 where
13498 F: FsIntrospect,
13499 {
13500 self.fs.total_appended()
13501 }
13502
13503 /// Test-support: successful `Fs::sync` calls (SimFs / counting fs).
13504 pub fn fs_sync_count(&self) -> usize
13505 where
13506 F: FsIntrospect,
13507 {
13508 self.fs.sync_count()
13509 }
13510
13511 /// Consume the db, returning its fs (for crash simulation).
13512 pub fn into_fs(self) -> F {
13513 self.fs
13514 }
13515
13516 pub fn snapshot(&mut self) -> Result<()> {
13517 self.snapshot_with(SnapshotOptions::default())
13518 }
13519
13520 /// Snapshot with explicit options.
13521 ///
13522 /// # `keep_wal`
13523 ///
13524 /// When `keep_wal` is `false` (the default, same as [`snapshot`]):
13525 /// - The WAL is replaced with a minimal baseline containing one
13526 /// `EnableFulltext` record per active declaration. All pre-snapshot
13527 /// history is discarded; `open_at` can only reach post-snapshot commits.
13528 ///
13529 /// When `keep_wal` is `true`:
13530 /// - The WAL is left intact. All pre-snapshot commits remain reachable
13531 /// via `open_at`. The existing WAL already contains the original
13532 /// `EnableFulltext` records, so no baseline re-write is needed; the
13533 /// recovery guards in `apply()` silently skip any duplicate records on
13534 /// replay.
13535 /// - Crash window: a crash after the snapshot write but before the next
13536 /// WAL write leaves the full pre-snapshot WAL intact. On reopen the
13537 /// snapshot is loaded and the WAL replayed idempotently over it — safe
13538 /// because every `apply()` arm is idempotent when replayed over an
13539 /// already-current snapshot.
13540 pub fn snapshot_with(&mut self, opts: SnapshotOptions) -> Result<()> {
13541 if self.read_only {
13542 return Err(GraphError::ReadOnly);
13543 }
13544 // A snapshot rewrites `wal.bin` through a tmp+rename, so a peer that is
13545 // appending ends up holding a descriptor on an unlinked inode and loses
13546 // commits it believes durable. Snapshotting therefore requires the
13547 // cross-process write lock, exactly as appending does. Unlike the WAL
13548 // append path this does not go through `log_then_apply_with`, so both
13549 // guards are repeated here.
13550 if self.degraded {
13551 return Err(GraphError::Io(std::io::Error::other(
13552 "database degraded after group-commit fsync failure; reopen required",
13553 )));
13554 }
13555 if self.lock_denied {
13556 return Err(GraphError::Busy { holder: None });
13557 }
13558 // Capture whether snapshot.bin already existed BEFORE this snapshot write.
13559 // Used by the archive path's conservative genesis-chain check: if a prior
13560 // snapshot exists but wal.truncated does not, we cannot distinguish a
13561 // legacy store (may have been truncated in an older code version) from a
13562 // new store that only used keep_wal=true. Conservative: refuse genesis in
13563 // both cases. Must be sampled here, before the snapshot write below.
13564 //
13565 // `snapshot_preserved_history` is the one case where the answer is not a
13566 // guess: a snapshot *this handle* took, on a store that had none when it
13567 // opened, and that kept the WAL. The proxy defers to it, because
13568 // otherwise `enable_multiplicity` — whose forced snapshot is exactly
13569 // that — would permanently disqualify the store from a genesis chain it
13570 // is fully entitled to (defect #23).
13571 let had_prior_snapshot = self.fs.snapshot_path().map(|p| p.exists()).unwrap_or(false)
13572 && !self.snapshot_preserved_history;
13573 // Which version this store writes. V9 unless it has opted in to
13574 // multiplicity, in which case V10 — the stamp that makes a reader which
13575 // does not know WAL discriminant 23 refuse the open instead of
13576 // truncating the WAL at the first such frame. The container is
13577 // identical either way; only these two header bytes move.
13578 let snapshot_version = core_storage::snapshot::version_for(self.multiplicity);
13579 self.ensure_v8_base_sections_loaded();
13580 // Ensure provenance is decoded before to_persist() clones it.
13581 self.engine.ensure_provenance_loaded_mut();
13582 let (rule_defs_typed, provenance, rule_tripped, rule_fires) = self.engine.to_persist();
13583 let rule_defs = rule_defs_typed
13584 .iter()
13585 .map(|r| bincode::serialize(r).expect("RuleDef serialize cannot fail"))
13586 .collect();
13587 // Collect HNSW state and IVF state. When indexes are not yet
13588 // populated (clean open, no mutation since open), pass the retained
13589 // raw bytes through directly so that migrate/snapshot does not
13590 // silently discard fitted approximate-rule indexes.
13591 let hnsw_state = self.engine.export_hnsw_state_passthrough();
13592 let ivf_bytes = if !self.engine.indexes_populated() {
13593 // Pass retained IVF bytes through unchanged (no re-encode).
13594 self.engine.retained_ivf_bytes_clone().unwrap_or_default()
13595 } else {
13596 // Indexes live: encode from current state.
13597 let raw_ivf = self.engine.export_ivf_state();
13598 let ivf_state_map: BTreeMap<String, core_storage::snapshot::PerRuleIvfState> = raw_ivf
13599 .into_iter()
13600 .map(|(name, ((sc, sa, sd), (dc, da, dd)))| {
13601 (
13602 name,
13603 core_storage::snapshot::PerRuleIvfState {
13604 src: core_storage::snapshot::SideIvfState {
13605 centroids: sc,
13606 clusters: sa,
13607 drift: sd,
13608 },
13609 dst: core_storage::snapshot::SideIvfState {
13610 centroids: dc,
13611 clusters: da,
13612 drift: dd,
13613 },
13614 },
13615 )
13616 })
13617 .collect();
13618 if ivf_state_map.is_empty() {
13619 Vec::new()
13620 } else {
13621 bincode::serialize(&ivf_state_map).expect("IVF state serialize cannot fail")
13622 }
13623 };
13624 let view_defs: Vec<Vec<u8>> = self
13625 .view_store
13626 .views()
13627 .map(|v| bincode::serialize(v).expect("ViewDef serialize cannot fail"))
13628 .collect();
13629 if self.base.is_some() {
13630 // V8 merge-snapshot path: encode base+overlay into a new V8 snapshot,
13631 // write it atomically, remap it as the new base, then clear the overlay.
13632 let meta = V8Meta {
13633 labels: (*self.labels).clone(),
13634 edge_props: (*self.edge_props).clone(),
13635 rule_defs,
13636 provenance,
13637 rule_tripped,
13638 rule_fires,
13639 ivf_bytes,
13640 view_defs,
13641 wal_truncated: !opts.keep_wal,
13642 hnsw: hnsw_state,
13643 last_change: self.last_change.clone(),
13644 };
13645 let mut buf: Vec<u8> = Vec::new();
13646 {
13647 // Clone the Arc so the old base stays alive while we encode.
13648 // The borrow of archived_csr (into old_base's mmap) is released
13649 // at the end of this block, before we replace self.base.
13650 let old_base = self.base.clone().expect("is_some checked above");
13651 let archived_csr = old_base.topology().map_err(|e| GraphError::Corrupt {
13652 detail: format!("v8 snapshot: topology section: {e:?}"),
13653 })?;
13654 let archived_cols = old_base.columns().map_err(|e| GraphError::Corrupt {
13655 detail: format!("v8 snapshot: columns section: {e:?}"),
13656 })?;
13657 // `None` when the base predates V9 — the migration path: its
13658 // string columns still carry their own tables and this snapshot
13659 // is the rewrite that collapses them into section 12.
13660 let archived_strings =
13661 old_base
13662 .string_table()
13663 .transpose()
13664 .map_err(|e| GraphError::Corrupt {
13665 detail: format!("v8 snapshot: strings section: {e:?}"),
13666 })?;
13667 let archived_edge_props =
13668 old_base
13669 .edge_props_section()
13670 .map_err(|e| GraphError::Corrupt {
13671 detail: format!("v8 snapshot: edge_props section: {e:?}"),
13672 })?;
13673 let edge_props_raw =
13674 old_base
13675 .edge_props_raw_bytes()
13676 .map_err(|e| GraphError::Corrupt {
13677 detail: format!("v8 snapshot: edge_props raw bytes: {e:?}"),
13678 })?;
13679 let prov_raw =
13680 old_base
13681 .provenance_raw_bytes()
13682 .map_err(|e| GraphError::Corrupt {
13683 detail: format!("v8 snapshot: provenance raw bytes: {e:?}"),
13684 })?;
13685 encode_v8(
13686 Some(archived_csr),
13687 Some(archived_cols),
13688 archived_strings,
13689 Some((archived_edge_props, edge_props_raw)),
13690 Some(prov_raw),
13691 &self.topo,
13692 &self.props,
13693 &self.ids,
13694 &self.syms,
13695 &meta,
13696 &mut buf,
13697 )?;
13698 }
13699 core_storage::snapshot::stamp_container_version(&mut buf, snapshot_version)?;
13700 self.fs.write_atomic(FileId::Snapshot, &buf)?;
13701 // Remap the freshly-written snapshot as the new base.
13702 // C2: use file mmap on RealFs; fall back to from_bytes on SimFs.
13703 let new_base = if let Some(snap_path) = self.fs.snapshot_path() {
13704 core_storage::v8::MappedBase::map(&snap_path)
13705 } else {
13706 core_storage::v8::MappedBase::from_bytes(buf)
13707 }
13708 .map_err(|e| GraphError::Corrupt {
13709 detail: format!("v8 snapshot: remap new base: {e:?}"),
13710 })?;
13711 self.base = Some(Arc::new(new_base));
13712 // Clear the overlay and prop tombstones — all data is now in the new base.
13713 self.topo = Arc::new(Topology::new());
13714 self.props = Arc::new(core_storage::columns::ColumnStore::new());
13715 } else {
13716 // Legacy path (V5–V7 stores without a V8 base).
13717 //
13718 // Memory-diet path: build V8Meta directly from &self — no SnapshotState
13719 // clone and no encode_v8_from_state intermediate clones. The big
13720 // structures (self.topo, self.props) are borrowed, not cloned.
13721 // self.edge_props is moved (not cloned) because we immediately clear it
13722 // when we remap the new V8 snapshot as self.base (see below).
13723 //
13724 // Eliminates from peak RSS vs. the old SnapshotState path:
13725 // • self.topo.clone() (~topology HashMap footprint)
13726 // • self.props.clone() (~column-store footprint)
13727 // • encode_v8_from_state V8Meta secondary clones (labels, edge_props, …)
13728 let meta = V8Meta {
13729 labels: (*self.labels).clone(),
13730 wal_truncated: !opts.keep_wal,
13731 // Move edge_props out so the large overlay is freed when meta
13732 // drops at end of this block (self.edge_props is now empty; reads
13733 // after base assignment go through the mmap'd base section).
13734 edge_props: std::mem::take(Arc::make_mut(&mut self.edge_props)),
13735 rule_defs,
13736 provenance,
13737 rule_tripped,
13738 rule_fires,
13739 ivf_bytes,
13740 view_defs,
13741 hnsw: hnsw_state,
13742 last_change: self.last_change.clone(),
13743 };
13744 let mut buf = Vec::new();
13745 encode_v8(
13746 None,
13747 None,
13748 None,
13749 None,
13750 None,
13751 &self.topo,
13752 &self.props,
13753 &self.ids,
13754 &self.syms,
13755 &meta,
13756 &mut buf,
13757 )?;
13758 // meta (and the moved edge_props inside it) is no longer needed;
13759 // drop it before the write to keep the peak window narrow.
13760 drop(meta);
13761 core_storage::snapshot::stamp_container_version(&mut buf, snapshot_version)?;
13762 self.fs.write_atomic(FileId::Snapshot, &buf)?;
13763 // Remap the freshly-written V8 snapshot as self.base.
13764 // On RealFs: drop the encode buffer before mmap to recover ~1.9 GiB.
13765 // On SimFs (tests): pass buf to from_bytes.
13766 let new_base = if let Some(snap_path) = self.fs.snapshot_path() {
13767 drop(buf);
13768 core_storage::v8::MappedBase::map(&snap_path)
13769 } else {
13770 core_storage::v8::MappedBase::from_bytes(buf)
13771 }
13772 .map_err(|e| GraphError::Corrupt {
13773 detail: format!("v8 snapshot: remap new base (legacy path): {e:?}"),
13774 })?;
13775 self.base = Some(Arc::new(new_base));
13776 // Free the large heap-allocated decoded state — all data is now in the
13777 // mmap'd base. Mirrors the V8 merge-snapshot path (see above).
13778 // self.edge_props was already moved into meta and is effectively empty.
13779 self.topo = Arc::new(Topology::new());
13780 self.props = Arc::new(core_storage::columns::ColumnStore::new());
13781 }
13782
13783 if opts.archive_wal {
13784 // History-preserving snapshot (Task 4):
13785 // 1. Snapshot already written above (write_atomic → fsynced).
13786 // 2. Rename WAL → wal.<commit_seq>.archive (atomic, same fs).
13787 // Crash window B: crash here leaves archive present, WAL
13788 // absent. Reopen: snapshot loaded (full state), no WAL
13789 // replay. Archive is NOT replayed into live state — it is
13790 // pre-snapshot by construction. Safe.
13791 // 3. Optionally write genesis marker (first archive only, no
13792 // prior WAL truncation).
13793 // 4. Prune old archives (retention), update horizon floor.
13794 // Pruning invalidates the genesis chain; delete marker.
13795 // 5. Write new minimal baseline WAL (write_atomic).
13796 // Crash window C: crash here leaves new archive plus no live
13797 // WAL. Same as window B — handled above.
13798 //
13799 // Sample existing archives BEFORE the rename so we can detect
13800 // whether this is the first archive.
13801 let existing_archives = self.fs.list_archives()?;
13802 let is_first_archive = existing_archives.is_empty();
13803
13804 // Compute a globally-monotonic archive name: the name equals the
13805 // cumulative end-frame index of the archive in global commit space.
13806 //
13807 // Using `commit_seq` directly is UNSOUND across sessions: on reopen
13808 // commit_seq is seeded from max(last_change), which underestimates
13809 // the WAL depth when trailing commits (e.g. insert_edge) do not
13810 // update last_change. A session-2 archive could then receive a name
13811 // ≤ the session-1 archive, causing incorrect sort order or collision.
13812 //
13813 // Instead: read and decode the live WAL here (before the rename) to
13814 // get its exact frame count, then add it to the last known global
13815 // end-frame index (the name of the most recent existing archive, or
13816 // wal_horizon_floor if no archives exist). This is O(WAL size) but
13817 // snapshot is already serialising the full graph state, so the cost
13818 // is dominated.
13819 let live_wal_bytes_for_name = self.fs.read(FileId::Wal)?;
13820 let (live_frames_for_name, _) = decode_all(&live_wal_bytes_for_name);
13821 let archive_n = existing_archives
13822 .last()
13823 .copied()
13824 .unwrap_or(self.wal_horizon_floor)
13825 + live_frames_for_name.len() as u64;
13826 self.fs.archive_wal(archive_n)?;
13827
13828 // The replacement WAL goes in **immediately**, with no fallible call
13829 // between it and the rename above.
13830 //
13831 // The rename is what removes the store's live declarations — the
13832 // multiplicity opt-in, and every `EnableFulltext` / `EnableIndex` —
13833 // and this write is what puts them back. Every call that used to sit
13834 // in between (the genesis marker, the retention sweep's reads, the
13835 // floor write, the archive deletes) was a `?` that could leave the
13836 // store with neither, so a single transient `Err` was enough to lose
13837 // a declaration that no rebuild can recover (defect #22).
13838 //
13839 // Ordering alone cannot close the crash window between two
13840 // filesystem calls; for the multiplicity declaration the V10 stamp
13841 // does that on the open path. What ordering does close is the much
13842 // wider window in which an ordinary I/O error did it — and that half
13843 // covers all three declarations, not just the one with a stamp.
13844 let mut baseline_wal: Vec<u8> = Vec::new();
13845 // The multiplicity opt-in is a declaration like the two below it,
13846 // and it is re-emitted for the same reason: truncation must not
13847 // silently opt the store back out and stop counting.
13848 if self.multiplicity {
13849 baseline_wal
13850 .extend_from_slice(&encode_record(&core_storage::wal::MULTIPLICITY_ENABLED));
13851 }
13852 for (label, field) in self.fulltext.enabled_pairs() {
13853 let rec = WalRecord::EnableFulltext {
13854 label: label.clone(),
13855 field: field.clone(),
13856 };
13857 baseline_wal.extend_from_slice(&encode_record(&rec));
13858 }
13859 for (label, field) in self.prop_index.enabled_pairs() {
13860 let rec = WalRecord::EnableIndex {
13861 label: label.clone(),
13862 field: field.clone(),
13863 };
13864 baseline_wal.extend_from_slice(&encode_record(&rec));
13865 }
13866 self.fs.write_atomic(FileId::Wal, &baseline_wal)?;
13867
13868 // Genesis marker: written once when the first archive is taken
13869 // from a store that has never undergone a WAL-truncating snapshot.
13870 // When present, `open_at` may replay archive-resident commits from
13871 // empty state (the archive chain covers from global index 0).
13872 //
13873 // Two conditions must ALL hold:
13874 // 1. This is the first archive (existing_archives was empty).
13875 // 2. No snapshot.bin existed before this operation (had_prior_snapshot=false).
13876 // A WAL-truncating snapshot (keep_wal=false) always writes snapshot.bin
13877 // before truncating the WAL, so if any prior truncating snapshot was taken
13878 // — even in a previous session — snapshot.bin is present and this condition
13879 // is false. This subsumes the cross-session truncation case without
13880 // requiring a separate wal.truncated sidecar file.
13881 // For legacy stores (snapshot.bin written by an older code version that
13882 // may have truncated the WAL), the same conservative refusal applies:
13883 // we cannot prove the chain is complete, so we refuse genesis (cost =
13884 // no as-of-through-archives; never silent wrong data).
13885 // The one exception is a snapshot this handle took itself, on a store
13886 // that had none when it opened, with the WAL kept: there the answer is
13887 // known rather than guessed, and `snapshot_preserved_history` says so.
13888 // Without that exception `enable_multiplicity`'s forced keep_wal
13889 // snapshot would disqualify the store forever (defect #23).
13890 // On SimFs (snapshot_path() == None) had_prior_snapshot is always false,
13891 // so SimFs always passes this check.
13892 if is_first_archive && !had_prior_snapshot {
13893 self.fs.write_genesis_marker()?;
13894 self.archive_genesis_chain = true;
13895 }
13896
13897 // Retention pruning: keep newest `keep` archives; delete oldest.
13898 // Pruning is the ONLY deletion site for archives.
13899 //
13900 // Crash-safety ordering (C1 fix):
13901 // 1. Count frames in surplus archives (reads only — no mutation).
13902 // 2. Advance and PERSIST the horizon floor FIRST via write-then-
13903 // rename (atomic). A crash after this point leaves orphaned
13904 // archives on disk, but the floor is correct. The opening
13905 // cleanup sweep (`cleanup_orphaned_archives`) removes them on
13906 // the next open, so the store is always safe to reopen.
13907 // 3. Delete the genesis marker (floor > 0 already blocks open_at
13908 // via the conjunctive gate; marker cleanup is belt-and-suspenders).
13909 // 4. Delete surplus archives. A crash between any two deletes
13910 // leaves the floor committed and orphaned archives cleaned at
13911 // next open — never a stale floor with a missing archive prefix.
13912 if let Some(keep) = self.wal_archive_retention {
13913 if keep > 0 {
13914 let archives = self.fs.list_archives()?;
13915 // archives is sorted ascending (oldest first)
13916 if archives.len() as u32 > keep {
13917 let surplus = archives.len() - keep as usize;
13918 // Step 1: count pruned frames (reads, no mutation).
13919 let mut pruned_frames = 0u64;
13920 for &n in &archives[..surplus] {
13921 let bytes = self.fs.read_archive(n)?;
13922 let (frames, _) = decode_all(&bytes);
13923 pruned_frames += frames.len() as u64;
13924 }
13925 // Step 2: advance and persist floor FIRST.
13926 self.wal_horizon_floor += pruned_frames;
13927 self.fs.write_horizon_floor(self.wal_horizon_floor)?;
13928 // The time map must not outlive the commits it
13929 // describes: an entry below the new floor would resolve
13930 // a date to a commit the engine can no longer replay,
13931 // which is worse than having no entry at all.
13932 self.commit_times.truncate_below(self.wal_horizon_floor);
13933 self.rewrite_commit_times();
13934 // Step 3: delete genesis marker (floor > 0 already
13935 // blocks open_at; this is belt-and-suspenders cleanup).
13936 if pruned_frames > 0 && self.archive_genesis_chain {
13937 self.fs.delete_genesis_marker()?;
13938 self.archive_genesis_chain = false;
13939 }
13940 // Step 4: delete surplus archives. Crash here →
13941 // orphaned archives; cleaned at next open.
13942 for &n in &archives[..surplus] {
13943 self.fs.delete_archive(n)?;
13944 }
13945 }
13946 }
13947 }
13948 } else if opts.keep_wal {
13949 // keep_wal=true: WAL is left untouched. The existing WAL already
13950 // contains the EnableFulltext records from the original enable calls;
13951 // replay is idempotent (guards in apply() skip already-live entries).
13952 // No baseline re-write is needed or safe here — the full WAL history
13953 // must remain intact for open_at to reach pre-snapshot commits.
13954 } else {
13955 // keep_wal=false (default): truncate by replacing the WAL with a
13956 // minimal baseline of one EnableFulltext record per active pair.
13957 //
13958 // Crash-ordering: write_atomic is atomic.
13959 // • Crash before snapshot write → WAL unchanged. Safe.
13960 // • Crash after snapshot write but before this WAL write → full
13961 // pre-snapshot WAL still present; open_with replays idempotently.
13962 // • Crash after both writes → normal post-snapshot state.
13963 //
13964 // Genesis chain: a WAL-truncating snapshot breaks the archive chain
13965 // for any archives taken AFTER this point (their WAL slices would
13966 // not start at genesis). Delete any existing genesis marker so that
13967 // open_at refuses archive-resident commits. Future sessions are
13968 // covered by had_prior_snapshot: snapshot.bin written here persists
13969 // across sessions and prevents a later archiving session from
13970 // incorrectly claiming a complete genesis chain.
13971 if self.archive_genesis_chain {
13972 self.fs.delete_genesis_marker()?;
13973 self.archive_genesis_chain = false;
13974 }
13975 // And this handle can no longer prove the WAL is whole: it is about
13976 // to truncate it itself. Same-session archives after this point get
13977 // the conservative answer, exactly as cross-session ones do.
13978 self.snapshot_preserved_history = false;
13979 let mut baseline_wal: Vec<u8> = Vec::new();
13980 // The multiplicity opt-in is a declaration like the two below it,
13981 // and it is re-emitted for the same reason: truncation must not
13982 // silently opt the store back out and stop counting.
13983 if self.multiplicity {
13984 baseline_wal
13985 .extend_from_slice(&encode_record(&core_storage::wal::MULTIPLICITY_ENABLED));
13986 }
13987 for (label, field) in self.fulltext.enabled_pairs() {
13988 let rec = WalRecord::EnableFulltext {
13989 label: label.clone(),
13990 field: field.clone(),
13991 };
13992 baseline_wal.extend_from_slice(&encode_record(&rec));
13993 }
13994 for (label, field) in self.prop_index.enabled_pairs() {
13995 let rec = WalRecord::EnableIndex {
13996 label: label.clone(),
13997 field: field.clone(),
13998 };
13999 baseline_wal.extend_from_slice(&encode_record(&rec));
14000 }
14001 self.fs.write_atomic(FileId::Wal, &baseline_wal)?;
14002 // This branch discards history rather than archiving it: the frames
14003 // the map describes are gone and the replacement WAL renumbers from
14004 // the floor, so every surviving entry now names a different frame.
14005 // Keeping them would resolve a date onto an unrelated commit — and
14006 // the stamps written after this point would sit far above the
14007 // store's own frame count, which is how the date surface came to
14008 // refuse commits the index path served perfectly well.
14009 //
14010 // A store that cannot answer a date says so by name
14011 // (`NoRecordedTime`). That is the honest state after discarding the
14012 // history the dates addressed.
14013 self.commit_times = core_storage::commit_times::CommitTimes::default();
14014 self.rewrite_commit_times();
14015 }
14016 // After snapshot the overlay may have changed (V8 merge path clears
14017 // self.topo and self.props). Refresh the MVCC fold so future readers
14018 // see the post-snapshot state rather than stale overlay data.
14019 self.fold_now();
14020 // We wrote the snapshot and (unless keep_wal) replaced the WAL, so both
14021 // markers this handle uses to detect other processes' work must be
14022 // re-taken from disk. Skipping this would make our own snapshot look
14023 // like a peer's on the next staleness check and force a needless
14024 // reload.
14025 self.wal_consumed = self.fs.wal_len().map_err(GraphError::Io)?;
14026 // A snapshot can replace the live WAL with a baseline, which renumbers
14027 // every frame after it. Re-derive the frame cursor from what the store
14028 // now actually holds rather than carrying the pre-snapshot count
14029 // forward — `wal_total_commits` is the same sequence the history
14030 // surfaces index, and the snapshot has already paid a far larger cost
14031 // than one decode.
14032 self.wal_frames_written = self.wal_total_commits()?;
14033 self.snapshot_ident = self.fs.snapshot_ident().map_err(GraphError::Io)?;
14034 Ok(())
14035 }
14036}
14037
14038/// What a batch node insert does when its key is already taken.
14039///
14040/// A mirror rebuild writes a frame onto a store that already has content, so
14041/// "the key exists" is a routine answer rather than a failure. The decision is
14042/// made during the batch's existing validate pass, from one id-map lookup per
14043/// row, so the frame stays atomic and re-ingest stays O(n).
14044#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)]
14045pub enum OnConflict {
14046 /// Refuse the whole frame with [`GraphError::DuplicateKey`]. The default,
14047 /// and the only behaviour before v0.6.10.
14048 #[default]
14049 Error,
14050 /// Leave the stored node exactly as it is — properties, label and edges —
14051 /// and count it in [`BatchOutcome::skipped`].
14052 Skip,
14053 /// Keep the key and make the node's properties **exactly** the supplied
14054 /// props: supplied fields are set, fields absent from the supplied props
14055 /// are removed. A supplied label that differs from the stored one, and a
14056 /// supplied `ns` that would move the node, are row errors — relabelling is
14057 /// [`GraphDb::rename_node`], not a side effect of a rebuild.
14058 ///
14059 /// Two properties are outside "exactly", both because they are not the
14060 /// caller's to supply:
14061 ///
14062 /// - `ns` is immutable, so an omitted `ns` leaves the node where it is
14063 /// rather than moving it to `default`;
14064 /// - a property a **view** owns is kept, not removed. Supplying one is a
14065 /// row error, so omitting it cannot be a request to delete it, and
14066 /// refusing the row instead would make `Replace` impossible for the whole
14067 /// population a view has written to. Each such field kept is counted in
14068 /// [`BatchOutcome::kept_view_owned`] — the row is still `replaced` and
14069 /// still raises no row error, so that count is the only signal a caller
14070 /// gets that the stored node carries a field its frame did not describe.
14071 Replace,
14072}
14073
14074/// What one [`OnConflict::Replace`] row resolves to.
14075///
14076/// The property writes that make the node exactly the supplied props —
14077/// `Some(value)` is a set, `None` a removal — paired with how many view-owned
14078/// fields the row kept instead of removing, which is the one way the result is
14079/// not exactly the supplied props. See [`MutPreview::plan_replace`].
14080type ReplacePlan = (Vec<(String, Option<Value>)>, usize);
14081
14082/// What one committed batch did.
14083///
14084/// [`BatchBuilder::commit`] returns the first two fields as a tuple; the rest
14085/// exist for [`OnConflict`] and are always zero / empty without it.
14086#[derive(Clone, Debug, Default, PartialEq, Eq)]
14087pub struct BatchOutcome {
14088 /// Node records actually written.
14089 pub nodes_inserted: usize,
14090 /// Edge records actually written. A duplicate edge is a silent no-op under
14091 /// every policy — adjacency is a set — and is not counted.
14092 pub edges_inserted: usize,
14093 /// Rows whose key was taken and whose policy was [`OnConflict::Skip`].
14094 pub skipped: usize,
14095 /// Rows whose key was taken and whose policy was [`OnConflict::Replace`].
14096 pub replaced: usize,
14097 /// View-owned properties an [`OnConflict::Replace`] row **kept** although
14098 /// the caller did not supply them — counted per field, so one row that
14099 /// keeps two contributes two.
14100 ///
14101 /// This is the one respect in which `Replace` does not make a node's props
14102 /// exactly the supplied ones (see [`OnConflict::Replace`]). Those rows
14103 /// still count in `replaced` and still raise no `row_errors`, because
14104 /// nothing went wrong: a view's property is not the caller's to supply or
14105 /// to remove. A mirror rebuild that needs its copy to be byte-exact reads
14106 /// this to learn that the store kept fields its frame did not describe.
14107 pub kept_view_owned: usize,
14108 /// `(row, why)` for rows an [`OnConflict::Replace`] refused. `row` counts
14109 /// node-insert ops in this batch from zero, which for a caller that queues
14110 /// its nodes in order is the index of the offending node. The rest of the
14111 /// frame still commits; the refused row changes nothing.
14112 pub row_errors: Vec<(usize, String)>,
14113}
14114
14115/// The `(nodes_inserted, edges_inserted)` pair every pre-0.6.10 commit entry
14116/// point returns. Keeps those signatures unchanged now that the validate pass
14117/// produces a [`BatchOutcome`].
14118fn inserted_pair(outcome: BatchOutcome) -> (usize, usize) {
14119 (outcome.nodes_inserted, outcome.edges_inserted)
14120}
14121
14122/// One entry of a frame the validate pass has decided on, before
14123/// [`GraphDb::rewrite_wal_dense_planned`] turns it into dense-id records.
14124///
14125/// Almost every entry is already a finished [`WalRecord`]. The exception is a
14126/// duplicate edge insert: its count names a dense triple, and on the batch path
14127/// the endpoints and the edge type may all be created by earlier records in the
14128/// *same* frame, so no id for them exists until the dense rewrite allocates it.
14129/// Carrying the keys this far and resolving them there is what lets the count
14130/// survive the shape a mirror rebuild writes (defect #24).
14131enum PlannedRec {
14132 Rec(WalRecord),
14133 DuplicateCount {
14134 edge_type: String,
14135 src_key: String,
14136 dst_key: String,
14137 },
14138}
14139
14140/// Queued mutation for a [`BatchBuilder`] or [`GraphDb::commit_group`].
14141///
14142/// The `submit_batch` / `commit_group` APIs accept `Vec<BatchOp>` so that
14143/// callers can build a set of mutations without holding `&mut GraphDb` and
14144/// hand them off to the group-committing writer for durable, batched I/O.
14145pub enum BatchOp {
14146 InsertNode {
14147 label: String,
14148 key: String,
14149 props: Vec<(String, Value)>,
14150 },
14151 InsertEdge {
14152 edge_type: String,
14153 src_key: String,
14154 dst_key: String,
14155 },
14156 SetProp {
14157 key: String,
14158 field: String,
14159 value: Value,
14160 },
14161 RemoveProp {
14162 key: String,
14163 field: String,
14164 },
14165 DeleteEdge {
14166 edge_type: String,
14167 src_key: String,
14168 dst_key: String,
14169 },
14170 DeleteNode {
14171 key: String,
14172 },
14173 CreateRule(RuleDef),
14174 DeleteRule {
14175 name: String,
14176 },
14177 /// Rename a node's key. Validated: old must exist, new must not.
14178 RenameNode {
14179 old_key: String,
14180 new_key: String,
14181 },
14182 /// Insert an edge, auto-creating any missing endpoint as a plain node with
14183 /// `placeholder_label` and no props. Rules fire and last-change is updated
14184 /// for each created endpoint (normal InsertNode semantics in the batch frame).
14185 InsertEdgeUpsert {
14186 edge_type: String,
14187 src_key: String,
14188 dst_key: String,
14189 placeholder_label: String,
14190 },
14191 /// Insert `key`, or — when the key is already taken — do what `on_conflict`
14192 /// says. Queued by [`BatchBuilder::insert_node_on_conflict`]; `Error`
14193 /// queues a plain [`BatchOp::InsertNode`] instead, so this variant only
14194 /// ever carries `Skip` or `Replace`.
14195 InsertNodeOnConflict {
14196 label: String,
14197 key: String,
14198 props: Vec<(String, Value)>,
14199 on_conflict: OnConflict,
14200 },
14201}
14202
14203/// Three-way node visibility status used by `check_single_op_authz`.
14204enum NodeAuthzStatus {
14205 /// Node exists in the store and is in the role's read mask.
14206 Visible(String), // carries the node's label
14207 /// Node exists in the store but is NOT in the role's read mask.
14208 Hidden,
14209 /// Node does not exist in the store.
14210 Absent,
14211}
14212
14213/// Overlay of ops already accepted earlier in the same batch. Never written
14214/// back to the database — validation only.
14215#[derive(Default)]
14216struct Overlay {
14217 extra_keys: BTreeSet<String>,
14218 /// Label of each node inserted earlier in this batch. The store does not
14219 /// have these keys yet, so `label_of` cannot answer for them, and
14220 /// `OnConflict::Replace` has to compare labels.
14221 extra_labels: BTreeMap<String, String>,
14222 deleted_keys: BTreeSet<String>,
14223 extra_props: BTreeMap<(String, String), Value>,
14224 removed_props: BTreeSet<(String, String)>,
14225 extra_edges: BTreeSet<(String, String, String)>,
14226 deleted_edges: BTreeSet<(String, String, String)>,
14227 extra_rules: BTreeSet<String>,
14228 deleted_rules: BTreeSet<String>,
14229 /// `rule name → (via_edge, edge_type)` for every via-hop rule accepted
14230 /// earlier in this batch. Feeds the rule-chain cycle check, which otherwise
14231 /// sees only the rules already committed to the engine. Keyed by name so a
14232 /// later `DeleteRule` in the same batch drops the arc with the rule.
14233 extra_rule_arcs: BTreeMap<String, (String, String)>,
14234}
14235
14236/// Read-only view of live db state plus a batch overlay. Shared by single-op
14237/// public methods (empty overlay) and `commit_batch`.
14238struct MutPreview<'a, F: Fs> {
14239 db: &'a GraphDb<F>,
14240 overlay: Overlay,
14241}
14242
14243/// Shortest path from `start` to `target` following `arcs` (`from → to`), or
14244/// `None` if `target` is unreachable.
14245///
14246/// Used for rule-chain cycle detection, where an arc is "a rule hops over
14247/// `from` and writes `to`". Breadth-first over BTree-ordered adjacency, so the
14248/// reported path is stable for a given rule set, and iterative so a pathological
14249/// rule graph cannot overflow the stack.
14250fn find_cycle_through(arcs: &[(String, String)], start: &str, target: &str) -> Option<Vec<String>> {
14251 let mut adj: BTreeMap<&str, BTreeSet<&str>> = BTreeMap::new();
14252 for (from, to) in arcs {
14253 adj.entry(from.as_str()).or_default().insert(to.as_str());
14254 }
14255 let mut parent: BTreeMap<&str, &str> = BTreeMap::new();
14256 let mut visited: BTreeSet<&str> = BTreeSet::new();
14257 let mut queue: std::collections::VecDeque<&str> = std::collections::VecDeque::new();
14258 visited.insert(start);
14259 queue.push_back(start);
14260 while let Some(node) = queue.pop_front() {
14261 if node == target {
14262 let mut path = vec![node.to_string()];
14263 let mut cur = node;
14264 while let Some(&p) = parent.get(cur) {
14265 path.push(p.to_string());
14266 cur = p;
14267 }
14268 path.reverse();
14269 return Some(path);
14270 }
14271 for &next in adj.get(node).into_iter().flatten() {
14272 if visited.insert(next) {
14273 parent.insert(next, node);
14274 queue.push_back(next);
14275 }
14276 }
14277 }
14278 None
14279}
14280
14281impl<'a, F: Fs> MutPreview<'a, F> {
14282 fn new(db: &'a GraphDb<F>) -> Self {
14283 Self {
14284 db,
14285 overlay: Overlay::default(),
14286 }
14287 }
14288
14289 fn has_key(&self, key: &str) -> bool {
14290 if self.overlay.extra_keys.contains(key) {
14291 return true;
14292 }
14293 if self.overlay.deleted_keys.contains(key) {
14294 return false;
14295 }
14296 self.db.ids.get(key).is_some()
14297 }
14298
14299 fn has_prop(&self, key: &str, field: &str) -> bool {
14300 if !self.has_key(key) {
14301 return false;
14302 }
14303 let k = (key.to_string(), field.to_string());
14304 if self.overlay.removed_props.contains(&k) {
14305 return false;
14306 }
14307 if self.overlay.extra_props.contains_key(&k) {
14308 return true;
14309 }
14310 // Fresh identity (first insert in this batch, or delete+reinsert):
14311 // ignore props still sitting on the soon-to-be-tombstoned slot.
14312 if self.overlay.extra_keys.contains(key) {
14313 return false;
14314 }
14315 self.db.get_prop(key, field).is_some()
14316 }
14317
14318 fn has_edge(&self, edge_type: &str, src_key: &str, dst_key: &str) -> bool {
14319 let k = (
14320 edge_type.to_string(),
14321 src_key.to_string(),
14322 dst_key.to_string(),
14323 );
14324 if self.overlay.deleted_edges.contains(&k) {
14325 return false;
14326 }
14327 if self.overlay.extra_edges.contains(&k) {
14328 return true;
14329 }
14330 // A key created in this batch (including reinsert) has no db edges.
14331 if self.overlay.extra_keys.contains(src_key) || self.overlay.extra_keys.contains(dst_key) {
14332 return false;
14333 }
14334 if self.overlay.deleted_keys.contains(src_key)
14335 || self.overlay.deleted_keys.contains(dst_key)
14336 {
14337 return false;
14338 }
14339 let Some(src) = self.db.ids.get(src_key) else {
14340 return false;
14341 };
14342 let Some(dst) = self.db.ids.get(dst_key) else {
14343 return false;
14344 };
14345 let Some(sym) = self.db.syms.get(edge_type) else {
14346 return false;
14347 };
14348 self.db
14349 .topo_view()
14350 .neighbors(sym, Direction::Out, src)
14351 .binary_search(&dst)
14352 .is_ok()
14353 }
14354
14355 fn has_rule(&self, name: &str) -> bool {
14356 if self.overlay.extra_rules.contains(name) {
14357 return true;
14358 }
14359 if self.overlay.deleted_rules.contains(name) {
14360 return false;
14361 }
14362 self.db.engine.rules().any(|r| r.name == name)
14363 }
14364
14365 fn is_rule_owned(&self, edge_type: &str, src_key: &str, dst_key: &str) -> bool {
14366 if self.overlay.extra_keys.contains(src_key) || self.overlay.extra_keys.contains(dst_key) {
14367 return false;
14368 }
14369 if self.overlay.deleted_keys.contains(src_key)
14370 || self.overlay.deleted_keys.contains(dst_key)
14371 {
14372 return false;
14373 }
14374 let Some(src) = self.db.ids.get(src_key) else {
14375 return false;
14376 };
14377 let Some(dst) = self.db.ids.get(dst_key) else {
14378 return false;
14379 };
14380 let Some(et) = self.db.syms.get(edge_type) else {
14381 return false;
14382 };
14383 // extra_rules is deliberately not consulted: a CreateRule earlier in
14384 // this batch has not fired, so it contributes no provenance. That is
14385 // the documented rule-window gap (see GraphDb::batch).
14386 if self.overlay.deleted_rules.is_empty() {
14387 return self.db.engine.is_owned(et, src, dst);
14388 }
14389 for (rule, triples) in self.db.engine.provenance() {
14390 if self.overlay.deleted_rules.contains(rule) {
14391 continue;
14392 }
14393 if triples.contains(&(et, src, dst)) {
14394 return true;
14395 }
14396 }
14397 false
14398 }
14399
14400 /// The refusals a node creation makes, in the order it makes them.
14401 ///
14402 /// A view owns its property, and creating a node that carries one is a
14403 /// write to it exactly as `set_prop` is — so it is refused here, at the one
14404 /// choke-point `GraphDb::insert_node`, `BatchOp::InsertNode` and the
14405 /// no-conflict arm of `BatchOp::InsertNodeOnConflict` all pass through.
14406 ///
14407 /// Leaving creation exempt was not harmless. The value was stored and
14408 /// served: a created node the view has no reason to revisit keeps the
14409 /// caller's number for the life of the handle, and the backfill at the next
14410 /// open overwrites it — so the store answered `deg = 777` before a restart
14411 /// and `deg = 0` after, for a property every other surface calls read-only.
14412 /// It also split one op two ways: supplying a view-owned field under
14413 /// `OnConflict::Replace` was already a row error on a taken key while the
14414 /// same field on a fresh key was accepted.
14415 ///
14416 /// Checked before the key, like [`MutPreview::prepare_remove_prop`], so the
14417 /// answer does not depend on whether the key exists.
14418 fn check_insert_node(&self, key: &str, props: &[(String, Value)]) -> Result<()> {
14419 for (field, _) in props {
14420 if let Some(view_name) = self.db.view_store.view_for_prop(field) {
14421 return Err(GraphError::ViewPropReadOnly {
14422 view_name: view_name.to_string(),
14423 });
14424 }
14425 }
14426 if self.has_key(key) {
14427 Err(GraphError::DuplicateKey { key: key.into() })
14428 } else {
14429 Ok(())
14430 }
14431 }
14432
14433 fn check_live_key(&self, key: &str) -> Result<()> {
14434 if self.has_key(key) {
14435 Ok(())
14436 } else {
14437 Err(GraphError::KeyNotFound { key: key.into() })
14438 }
14439 }
14440
14441 fn prepare_insert_edge(&self, edge_type: &str, src_key: &str, dst_key: &str) -> Result<bool> {
14442 for k in [src_key, dst_key] {
14443 if !self.has_key(k) {
14444 return Err(GraphError::KeyNotFound { key: k.into() });
14445 }
14446 }
14447 if self.is_rule_owned(edge_type, src_key, dst_key) {
14448 return Err(GraphError::RuleOwned {
14449 detail: format!("edge {edge_type} {src_key}→{dst_key} is rule-owned"),
14450 });
14451 }
14452 // A user-written edge stays inside one namespace. Derived edges do not
14453 // come through here — the engine adds them directly — and the rule
14454 // scoping check is what keeps those pure.
14455 let src_ns = self.namespace_in_batch(src_key);
14456 let dst_ns = self.namespace_in_batch(dst_key);
14457 if src_ns != dst_ns {
14458 return Err(GraphError::CrossNamespace {
14459 src: src_key.to_string(),
14460 src_ns,
14461 dst: dst_key.to_string(),
14462 dst_ns,
14463 });
14464 }
14465 Ok(!self.has_edge(edge_type, src_key, dst_key))
14466 }
14467
14468 fn prepare_remove_prop(&self, key: &str, field: &str) -> Result<bool> {
14469 // A view owns its property, and the refusal has to live here rather
14470 // than on `GraphDb::remove_prop`: `BatchOp::RemoveProp` never meets
14471 // that one, and it is what the HTTP `DELETE /node/{key}/prop/{field}`
14472 // route, `Batch::remove_prop` and the CLI all submit. This is the one
14473 // choke-point every removal passes, exactly as it is for `ns` below.
14474 // Checked before the key, so the answer does not depend on whether the
14475 // key exists — which is also what `GraphDb::remove_prop` answered when
14476 // it carried the only copy of this guard.
14477 if let Some(view_name) = self.db.view_store.view_for_prop(field) {
14478 return Err(GraphError::ViewPropReadOnly {
14479 view_name: view_name.to_string(),
14480 });
14481 }
14482 self.check_live_key(key)?;
14483 // Removing `ns` is changing the namespace — to `default`, the namespace
14484 // an absent property names. It goes through this one choke-point and NOT
14485 // through `rewrite_wal_dense` (a `RemoveProp` needs no dense rewrite), so
14486 // the immutability rule has to be stated here as well. Without it the
14487 // node silently lands in `default` on the next open: the cross-namespace
14488 // edge guard is defeated and a default-bound role reads a tenant's node.
14489 if field == NS_PROP {
14490 let from = self.namespace_in_batch(key);
14491 if from != NS_DEFAULT {
14492 return Err(GraphError::NamespaceImmutable {
14493 key: key.to_string(),
14494 from,
14495 to: NS_DEFAULT.to_string(),
14496 });
14497 }
14498 // Already in `default`: the removal changes no namespace. It is the
14499 // no-op `set_prop` to the current namespace is, not an error.
14500 return Ok(false);
14501 }
14502 Ok(self.has_prop(key, field))
14503 }
14504
14505 fn prepare_delete_edge(&self, edge_type: &str, src_key: &str, dst_key: &str) -> Result<bool> {
14506 for k in [src_key, dst_key] {
14507 if !self.has_key(k) {
14508 return Err(GraphError::KeyNotFound { key: k.into() });
14509 }
14510 }
14511 // Provenance-owned OR a live rule would derive this pair. User-first
14512 // edges that a later rule matches are not in `owned`, but deleting
14513 // them would leave a hole `rebuild_rule` immediately fills.
14514 if self.is_rule_owned(edge_type, src_key, dst_key) {
14515 return Err(GraphError::RuleOwned {
14516 detail: format!(
14517 "edge {edge_type} {src_key}→{dst_key} is rule-owned; \
14518 delete or change the owning rule"
14519 ),
14520 });
14521 }
14522 if self.would_derive(edge_type, src_key, dst_key) {
14523 return Err(GraphError::RuleOwned {
14524 detail: format!(
14525 "edge {edge_type} {src_key}→{dst_key} is rule-owned; \
14526 delete or change the owning rule, or a live rule would re-derive it"
14527 ),
14528 });
14529 }
14530 Ok(self.has_edge(edge_type, src_key, dst_key))
14531 }
14532
14533 /// True if any live rule (minus overlay-deleted names) would derive
14534 /// `(edge_type, src, dst)` from current overlay-visible props/labels.
14535 /// CreateRule names in `extra_rules` are ignored — same documented
14536 /// same-batch rule-window as [`Self::is_rule_owned`].
14537 fn would_derive(&self, edge_type: &str, src_key: &str, dst_key: &str) -> bool {
14538 if src_key == dst_key {
14539 return false;
14540 }
14541 let Some(src_label) = self.label_of(src_key) else {
14542 return false;
14543 };
14544 let Some(dst_label) = self.label_of(dst_key) else {
14545 return false;
14546 };
14547 for rule in self.db.engine.rules() {
14548 if self.overlay.deleted_rules.contains(&rule.name) {
14549 continue;
14550 }
14551 if rule.edge_type != edge_type {
14552 continue;
14553 }
14554 if rule.src_label != src_label || rule.dst_label != dst_label {
14555 continue;
14556 }
14557 let src_props = |f: &str| self.prop_value(src_key, f);
14558 let dst_props = |f: &str| self.prop_value(dst_key, f);
14559 let src_view = NodeView {
14560 key: src_key,
14561 props: &src_props,
14562 };
14563 let dst_view = NodeView {
14564 key: dst_key,
14565 props: &dst_props,
14566 };
14567 if evaluate(&rule.predicate, &src_view, &dst_view).is_some() {
14568 return true;
14569 }
14570 }
14571 false
14572 }
14573
14574 fn label_of(&self, key: &str) -> Option<String> {
14575 if self.overlay.deleted_keys.contains(key) {
14576 return None;
14577 }
14578 // Fresh identities created in this batch have no stored label in the
14579 // overlay; they cannot be provenance-owned yet either.
14580 let id = self.db.ids.get(key)?;
14581 let sym = self.db.labels.get(id as usize).copied()?;
14582 if sym == u32::MAX {
14583 return None;
14584 }
14585 self.db.syms.resolve(sym).map(str::to_string)
14586 }
14587
14588 /// The label `key` carries as this batch sees it — including a node
14589 /// inserted earlier in the same batch, which the store does not have yet.
14590 fn label_in_batch(&self, key: &str) -> Option<String> {
14591 if self.overlay.deleted_keys.contains(key) {
14592 return None;
14593 }
14594 if let Some(label) = self.overlay.extra_labels.get(key) {
14595 return Some(label.clone());
14596 }
14597 self.label_of(key)
14598 }
14599
14600 /// The property writes that make `key`'s props exactly `props`, or why the
14601 /// row is refused.
14602 ///
14603 /// `Some(value)` is a set and `None` is a removal. `store_fields` is every
14604 /// field name the store knows, hoisted by the caller so a frame of N
14605 /// replaces reads the field list once rather than N times.
14606 ///
14607 /// The second half of the pair is how many view-owned fields this row kept
14608 /// rather than removed — the one part of "exactly the supplied props" that
14609 /// does not hold, and the caller's only signal that it did not.
14610 ///
14611 /// The refusals are row errors, not frame errors: a mirror rebuild should
14612 /// learn which of its rows disagree with the store without losing the rows
14613 /// that agree.
14614 fn plan_replace(
14615 &self,
14616 label: &str,
14617 key: &str,
14618 props: &[(String, Value)],
14619 store_fields: &[String],
14620 ) -> std::result::Result<ReplacePlan, String> {
14621 // A different label is a relabel, and a rebuild does not relabel: that
14622 // is `rename_node` or an explicit write, never a side effect here.
14623 let stored = self.label_in_batch(key).unwrap_or_default();
14624 if stored != label {
14625 return Err(format!(
14626 "node {key}: on_conflict=\"replace\" will not relabel {stored:?} to {label:?}; \
14627 relabelling is rename_node or an explicit write"
14628 ));
14629 }
14630 // `ns` is immutable. Replace removes what the supplied props omit, so
14631 // an omitted `ns` is a move to `default` exactly as a different `ns` is
14632 // a move to that one; both are the same refusal.
14633 let from = self.namespace_in_batch(key);
14634 let to = match props.iter().find(|(field, _)| field == NS_PROP) {
14635 Some((_, Value::Str(ns))) => ns.clone(),
14636 Some((_, value)) => {
14637 return Err(format!(
14638 "node {key}: {NS_PROP} must be a string naming a namespace, got {value:?}"
14639 ));
14640 }
14641 None => NS_DEFAULT.to_string(),
14642 };
14643 if to != from {
14644 return Err(format!(
14645 "node {key}: {NS_PROP} is immutable; on_conflict=\"replace\" cannot move it \
14646 from {from:?} to {to:?}"
14647 ));
14648 }
14649
14650 if let Some(why) = self.supplied_view_owned_prop(key, props) {
14651 return Err(why);
14652 }
14653
14654 let supplied: BTreeSet<&str> = props.iter().map(|(field, _)| field.as_str()).collect();
14655 let mut writes = Vec::new();
14656 for (field, value) in props {
14657 // `ns` names the namespace the node is already in, so the write is
14658 // the no-op the dense-rewrite seam would drop anyway.
14659 if field == NS_PROP {
14660 continue;
14661 }
14662 // Already exactly this value: a rebuild of an unchanged row should
14663 // cost no WAL record.
14664 if self.prop_value(key, field).as_ref() == Some(value) {
14665 continue;
14666 }
14667 writes.push((field.clone(), Some(value.clone())));
14668 }
14669 // Everything the node still carries that the supplied props do not.
14670 // `ns` is never removed: it is immutable, and the check above has
14671 // already established the node stays where it is.
14672 let overlay_fields = self
14673 .overlay
14674 .extra_props
14675 .keys()
14676 .filter(|(k, _)| k == key)
14677 .map(|(_, field)| field.as_str());
14678 //
14679 // A view-owned field is filtered out rather than refused. It is not the
14680 // caller's to supply (supplying one is still the row error above) and
14681 // so it is not part of what "exactly the supplied ones" ranges over:
14682 // omitting it is not a request to delete it. Refusing here instead
14683 // would make `replace` impossible for every node a view has written to
14684 // — which on a store carrying a view is the whole population a mirror
14685 // rebuild has to cover.
14686 let omitted: BTreeSet<&str> = store_fields
14687 .iter()
14688 .map(String::as_str)
14689 .chain(overlay_fields)
14690 .filter(|field| {
14691 *field != NS_PROP && !supplied.contains(field) && self.has_prop(key, field)
14692 })
14693 .collect();
14694 // The view-owned half is kept, and counted: the row still commits and
14695 // still reports no error, so without this number a mirror rebuild is
14696 // told it got exactly what it asked for when it did not (defect #18).
14697 let (stale, kept): (Vec<&str>, Vec<&str>) = omitted
14698 .into_iter()
14699 .partition(|field| self.db.view_store.view_for_prop(field).is_none());
14700 writes.extend(stale.into_iter().map(|field| (field.to_string(), None)));
14701 Ok((writes, kept.len()))
14702 }
14703
14704 /// The row error a supplied view-owned field earns, or `None`.
14705 ///
14706 /// Shared by [`MutPreview::plan_replace`] and the no-conflict arm of
14707 /// `BatchOp::InsertNodeOnConflict` so that one op answers a supplied
14708 /// view-owned field the same way whether or not the key was already taken.
14709 fn supplied_view_owned_prop(&self, key: &str, props: &[(String, Value)]) -> Option<String> {
14710 props.iter().find_map(|(field, _)| {
14711 self.db.view_store.view_for_prop(field).map(|view_name| {
14712 format!(
14713 "node {key}: property {field:?} is owned by view {view_name:?} and is \
14714 read-only"
14715 )
14716 })
14717 })
14718 }
14719
14720 /// The namespace `key` is in as this batch sees it — including a node
14721 /// inserted earlier in the same batch, which the store does not have yet.
14722 fn namespace_in_batch(&self, key: &str) -> String {
14723 namespace_of_value(self.prop_value(key, NS_PROP).as_ref()).to_string()
14724 }
14725
14726 fn prop_value(&self, key: &str, field: &str) -> Option<Value> {
14727 if !self.has_key(key) {
14728 return None;
14729 }
14730 let k = (key.to_string(), field.to_string());
14731 if self.overlay.removed_props.contains(&k) {
14732 return None;
14733 }
14734 if let Some(v) = self.overlay.extra_props.get(&k) {
14735 return Some(v.clone());
14736 }
14737 if self.overlay.extra_keys.contains(key) {
14738 return None;
14739 }
14740 self.db.get_prop(key, field)
14741 }
14742
14743 fn check_create_rule(&self, def: &RuleDef) -> Result<()> {
14744 def.validate()
14745 .map_err(|e| GraphError::RuleInvalid { detail: e })?;
14746 if self.has_rule(&def.name) {
14747 return Err(GraphError::RuleInvalid {
14748 detail: format!("rule {:?} already exists", def.name),
14749 });
14750 }
14751 // Rule-chain cycle rejection. Derived edges feed via-hop rules, so a
14752 // rule set forms a graph whose arcs are "hops over `via_edge`, writes
14753 // `edge_type`". A cycle in that graph is a rule set that would re-fire
14754 // itself forever; the engine's depth cap would silently truncate it
14755 // instead, leaving an arbitrary partial result. Reject it here, the one
14756 // place that sees the whole rule set.
14757 //
14758 // Rules accepted earlier in the same batch count too: the overlay
14759 // carries their arcs, so a cycle cannot be assembled one op at a time.
14760 if let Some(via) = def.via_edge.as_deref() {
14761 if via == def.edge_type {
14762 return Err(GraphError::RuleInvalid {
14763 detail: format!("rule chain cycle: {} -> {}", via, def.edge_type),
14764 });
14765 }
14766 let mut arcs: Vec<(String, String)> = self
14767 .db
14768 .engine
14769 .rules()
14770 .filter(|r| !self.overlay.deleted_rules.contains(&r.name))
14771 .filter_map(|r| r.via_edge.clone().map(|v| (v, r.edge_type.clone())))
14772 .collect();
14773 arcs.extend(self.overlay.extra_rule_arcs.values().cloned());
14774 arcs.push((via.to_string(), def.edge_type.clone()));
14775 if let Some(path) = find_cycle_through(&arcs, &def.edge_type, via) {
14776 return Err(GraphError::RuleInvalid {
14777 detail: format!("rule chain cycle: {} -> {}", via, path.join(" -> ")),
14778 });
14779 }
14780 }
14781 Ok(())
14782 }
14783
14784 fn check_delete_rule(&self, name: &str) -> Result<()> {
14785 if self.has_rule(name) {
14786 Ok(())
14787 } else {
14788 Err(GraphError::RuleNotFound { name: name.into() })
14789 }
14790 }
14791
14792 fn note_insert_node(&mut self, label: &str, key: &str, props: &[(String, Value)]) {
14793 self.overlay.deleted_keys.remove(key);
14794 self.overlay.extra_keys.insert(key.to_string());
14795 self.overlay
14796 .extra_labels
14797 .insert(key.to_string(), label.to_string());
14798 self.overlay.extra_props.retain(|(k, _), _| k != key);
14799 self.overlay.removed_props.retain(|(k, _)| k != key);
14800 for (field, value) in props {
14801 self.overlay
14802 .extra_props
14803 .insert((key.to_string(), field.clone()), value.clone());
14804 }
14805 }
14806
14807 fn note_insert_edge(&mut self, edge_type: &str, src_key: &str, dst_key: &str) {
14808 let k = (
14809 edge_type.to_string(),
14810 src_key.to_string(),
14811 dst_key.to_string(),
14812 );
14813 self.overlay.deleted_edges.remove(&k);
14814 self.overlay.extra_edges.insert(k);
14815 }
14816
14817 fn note_set_prop(&mut self, key: &str, field: &str, value: &Value) {
14818 let k = (key.to_string(), field.to_string());
14819 self.overlay.removed_props.remove(&k);
14820 self.overlay.extra_props.insert(k, value.clone());
14821 }
14822
14823 fn note_remove_prop(&mut self, key: &str, field: &str) {
14824 let k = (key.to_string(), field.to_string());
14825 self.overlay.extra_props.remove(&k);
14826 self.overlay.removed_props.insert(k);
14827 }
14828
14829 fn note_delete_edge(&mut self, edge_type: &str, src_key: &str, dst_key: &str) {
14830 let k = (
14831 edge_type.to_string(),
14832 src_key.to_string(),
14833 dst_key.to_string(),
14834 );
14835 self.overlay.extra_edges.remove(&k);
14836 self.overlay.deleted_edges.insert(k);
14837 }
14838
14839 fn note_delete_node(&mut self, key: &str) {
14840 self.overlay.extra_keys.remove(key);
14841 self.overlay.deleted_keys.insert(key.to_string());
14842 self.overlay.extra_props.retain(|(k, _), _| k != key);
14843 self.overlay.removed_props.retain(|(k, _)| k != key);
14844 self.overlay
14845 .extra_edges
14846 .retain(|(_, s, d)| s != key && d != key);
14847 self.overlay
14848 .deleted_edges
14849 .retain(|(_, s, d)| s != key && d != key);
14850 }
14851
14852 fn note_create_rule(&mut self, def: &RuleDef) {
14853 self.overlay.deleted_rules.remove(&def.name);
14854 self.overlay.extra_rules.insert(def.name.clone());
14855 // Rules accepted earlier in this batch are not in the engine yet, so
14856 // the cycle check would not see their arcs. Keep the arc, not just the
14857 // name, so a batch cannot smuggle in a cycle one op at a time.
14858 if let Some(via) = def.via_edge.clone() {
14859 self.overlay
14860 .extra_rule_arcs
14861 .insert(def.name.clone(), (via, def.edge_type.clone()));
14862 }
14863 }
14864
14865 fn check_rename_node(&self, old: &str, new: &str) -> Result<()> {
14866 if !self.has_key(old) {
14867 return Err(GraphError::KeyNotFound { key: old.into() });
14868 }
14869 if self.has_key(new) {
14870 return Err(GraphError::DuplicateKey { key: new.into() });
14871 }
14872 Ok(())
14873 }
14874
14875 fn note_rename_node(&mut self, old: &str, new: &str) {
14876 // Mark old as deleted so subsequent batch ops cannot reference it.
14877 self.overlay.extra_keys.remove(old);
14878 self.overlay.deleted_keys.insert(old.to_string());
14879 // Mark new as extra so subsequent batch ops can reference it.
14880 self.overlay.deleted_keys.remove(new);
14881 self.overlay.extra_keys.insert(new.to_string());
14882 // Migrate any overlay props from old key to new key.
14883 let new_str = new.to_string();
14884 let transferred: Vec<((String, String), Value)> = self
14885 .overlay
14886 .extra_props
14887 .iter()
14888 .filter(|((k, _), _)| k.as_str() == old)
14889 .map(|((_, f), v)| ((new_str.clone(), f.clone()), v.clone()))
14890 .collect();
14891 self.overlay
14892 .extra_props
14893 .retain(|(k, _), _| k.as_str() != old);
14894 for (k, v) in transferred {
14895 self.overlay.extra_props.insert(k, v);
14896 }
14897 // Migrate removed_props.
14898 let transferred_removed: Vec<(String, String)> = self
14899 .overlay
14900 .removed_props
14901 .iter()
14902 .filter(|(k, _)| k.as_str() == old)
14903 .map(|(_, f)| (new_str.clone(), f.clone()))
14904 .collect();
14905 self.overlay
14906 .removed_props
14907 .retain(|(k, _)| k.as_str() != old);
14908 for k in transferred_removed {
14909 self.overlay.removed_props.insert(k);
14910 }
14911 }
14912
14913 fn note_delete_rule(&mut self, name: &str) {
14914 self.overlay.extra_rules.remove(name);
14915 // Drop its chain arc too: a rule created and then deleted in the same
14916 // batch must not make a later, legal rule look like a cycle.
14917 self.overlay.extra_rule_arcs.remove(name);
14918 self.overlay.deleted_rules.insert(name.to_string());
14919 // Treat the deleted rule's current provenance as gone so a later
14920 // delete_edge of those triples is a no-op (matches sequential).
14921 if let Some(triples) = self.db.engine.provenance().get(name) {
14922 for &(et, s, d) in triples {
14923 let Some(etype) = self.db.syms.resolve(et) else {
14924 continue;
14925 };
14926 let Some(src) = self.db.ids.key_of(s) else {
14927 continue;
14928 };
14929 let Some(dst) = self.db.ids.key_of(d) else {
14930 continue;
14931 };
14932 let k = (etype.to_string(), src.to_string(), dst.to_string());
14933 self.overlay.extra_edges.remove(&k);
14934 self.overlay.deleted_edges.insert(k);
14935 }
14936 }
14937 }
14938}
14939
14940/// Collects mutations and commits them as one WAL `Batch` frame.
14941///
14942/// Holds `&mut GraphDb` for its lifetime. Queue with the same method names
14943/// as [`GraphDb`]; call [`commit`](Self::commit) to validate, log, and apply.
14944/// See [`GraphDb::batch`] for validation and atomicity rules.
14945pub struct BatchBuilder<'a, F: Fs> {
14946 db: &'a mut GraphDb<F>,
14947 ops: Vec<BatchOp>,
14948}
14949
14950impl<'a, F: Fs> BatchBuilder<'a, F> {
14951 pub fn insert_node(
14952 &mut self,
14953 label: &str,
14954 key: &str,
14955 props: Vec<(String, Value)>,
14956 ) -> &mut Self {
14957 self.ops.push(BatchOp::InsertNode {
14958 label: label.into(),
14959 key: key.into(),
14960 props,
14961 });
14962 self
14963 }
14964
14965 /// Queue a node insert whose answer to a taken key is `on_conflict`.
14966 ///
14967 /// [`OnConflict::Error`] queues exactly the op [`insert_node`](Self::insert_node)
14968 /// does, so the default path is unchanged.
14969 pub fn insert_node_on_conflict(
14970 &mut self,
14971 label: &str,
14972 key: &str,
14973 props: Vec<(String, Value)>,
14974 on_conflict: OnConflict,
14975 ) -> &mut Self {
14976 self.ops.push(match on_conflict {
14977 OnConflict::Error => BatchOp::InsertNode {
14978 label: label.into(),
14979 key: key.into(),
14980 props,
14981 },
14982 on_conflict => BatchOp::InsertNodeOnConflict {
14983 label: label.into(),
14984 key: key.into(),
14985 props,
14986 on_conflict,
14987 },
14988 });
14989 self
14990 }
14991
14992 pub fn insert_edge(&mut self, edge_type: &str, src_key: &str, dst_key: &str) -> &mut Self {
14993 self.ops.push(BatchOp::InsertEdge {
14994 edge_type: edge_type.into(),
14995 src_key: src_key.into(),
14996 dst_key: dst_key.into(),
14997 });
14998 self
14999 }
15000
15001 pub fn set_prop(&mut self, key: &str, field: &str, value: Value) -> &mut Self {
15002 self.ops.push(BatchOp::SetProp {
15003 key: key.into(),
15004 field: field.into(),
15005 value,
15006 });
15007 self
15008 }
15009
15010 pub fn remove_prop(&mut self, key: &str, field: &str) -> &mut Self {
15011 self.ops.push(BatchOp::RemoveProp {
15012 key: key.into(),
15013 field: field.into(),
15014 });
15015 self
15016 }
15017
15018 pub fn delete_edge(&mut self, edge_type: &str, src_key: &str, dst_key: &str) -> &mut Self {
15019 self.ops.push(BatchOp::DeleteEdge {
15020 edge_type: edge_type.into(),
15021 src_key: src_key.into(),
15022 dst_key: dst_key.into(),
15023 });
15024 self
15025 }
15026
15027 pub fn delete_node(&mut self, key: &str) -> &mut Self {
15028 self.ops.push(BatchOp::DeleteNode { key: key.into() });
15029 self
15030 }
15031
15032 pub fn create_rule(&mut self, def: RuleDef) -> &mut Self {
15033 self.ops.push(BatchOp::CreateRule(def));
15034 self
15035 }
15036
15037 pub fn delete_rule(&mut self, name: &str) -> &mut Self {
15038 self.ops.push(BatchOp::DeleteRule { name: name.into() });
15039 self
15040 }
15041
15042 /// Queue a node-rename in this batch.
15043 ///
15044 /// Validation (old exists, new not taken) runs at commit time.
15045 pub fn rename_node(&mut self, old_key: &str, new_key: &str) -> &mut Self {
15046 self.ops.push(BatchOp::RenameNode {
15047 old_key: old_key.into(),
15048 new_key: new_key.into(),
15049 });
15050 self
15051 }
15052
15053 /// Queue an edge insert with endpoint auto-creation.
15054 ///
15055 /// Any missing endpoint is created as a plain node `{key, label:
15056 /// placeholder_label, no props}` inside this batch frame. Rules fire and
15057 /// last-change is updated for each auto-created node.
15058 pub fn insert_edge_upsert(
15059 &mut self,
15060 edge_type: &str,
15061 src_key: &str,
15062 dst_key: &str,
15063 placeholder_label: &str,
15064 ) -> &mut Self {
15065 self.ops.push(BatchOp::InsertEdgeUpsert {
15066 edge_type: edge_type.into(),
15067 src_key: src_key.into(),
15068 dst_key: dst_key.into(),
15069 placeholder_label: placeholder_label.into(),
15070 });
15071 self
15072 }
15073
15074 /// Validate every queued op, then log one `Batch` frame and apply.
15075 /// Empty / all-noop batches return `Ok(())` without writing the WAL.
15076 /// A second `commit()` after a successful one is an empty-batch no-op
15077 /// (queued ops were taken).
15078 /// Takes `&mut self` so it chains after the queue methods (`b.insert_node(..).commit()`)
15079 /// and also works as `let mut b = db.batch(); b.insert_node(..); b.commit()`.
15080 ///
15081 /// **Rule-window limitation:** batch validation cannot see edges that a
15082 /// rule created earlier in the *same* batch will derive at apply time, so
15083 /// a `delete_edge` / `insert_edge` in that window is silently no-oped
15084 /// where sequential calls would return `Err(RuleOwned)`. State integrity
15085 /// is unaffected (idempotent apply, provenance intact). Create rules in
15086 /// their own batch, or sequentially, when later ops may touch derived
15087 /// edges.
15088 /// Validate every queued op and commit atomically.
15089 ///
15090 /// Returns `(nodes_inserted, edges_inserted)` — the counts of node and edge
15091 /// WAL records actually written (duplicate edges are silent no-ops and are
15092 /// NOT counted). Both are 0 when the batch is empty or all-noop.
15093 pub fn commit(&mut self) -> Result<(usize, usize)> {
15094 let ops = std::mem::take(&mut self.ops);
15095 self.db.commit_batch(ops)
15096 }
15097
15098 /// [`commit`](Self::commit) with the full [`BatchOutcome`] — the counts a
15099 /// caller needs when its rows carry an [`OnConflict`] policy.
15100 pub fn commit_outcome(&mut self) -> Result<BatchOutcome> {
15101 let ops = std::mem::take(&mut self.ops);
15102 self.db.commit_logged_batch(ops, None, None)
15103 }
15104
15105 /// Same as [`commit`](Self::commit) but tail the inner events with
15106 /// [`MutationEvent::Ingested`] instead of [`MutationEvent::BatchApplied`].
15107 pub(crate) fn commit_ingest(&mut self, label: &str, inserted: usize) -> Result<(usize, usize)> {
15108 let ops = std::mem::take(&mut self.ops);
15109 self.db
15110 .commit_logged_batch(ops, Some((label.to_string(), inserted)), None)
15111 .map(inserted_pair)
15112 }
15113}
15114
15115pub struct NodeRef<'a, F: Fs> {
15116 db: &'a GraphDb<F>,
15117 id: u32,
15118}
15119
15120impl<'a, F: Fs> NodeRef<'a, F> {
15121 pub fn key(&self) -> &str {
15122 self.db.ids.key_of(self.id).expect("dense ids")
15123 }
15124
15125 pub fn label(&self) -> &str {
15126 let sym = self
15127 .db
15128 .labels
15129 .get(self.id as usize)
15130 .copied()
15131 .filter(|&s| s != u32::MAX)
15132 .expect("real nodes always have a label; u32::MAX sentinel cannot occur");
15133 self.db.syms.resolve(sym).expect("interned label symbol")
15134 }
15135
15136 pub fn prop(&self, field: &str) -> Option<Value> {
15137 self.db
15138 .props_view()
15139 .get(self.id, field)
15140 .map(|vr| vr.into_value())
15141 }
15142
15143 /// All stored fields for this node, sorted by field name.
15144 ///
15145 /// Reads from the full base+overlay view so that props stored only in the
15146 /// V8 snapshot base (i.e. before any post-snapshot WAL writes) are visible.
15147 pub fn props(&self) -> BTreeMap<String, Value> {
15148 let mut out = BTreeMap::new();
15149 let pv = self.db.props_view();
15150 for field in pv.field_names() {
15151 if let Some(vr) = pv.get(self.id, &field) {
15152 out.insert(field, vr.into_value());
15153 }
15154 }
15155 out
15156 }
15157
15158 /// depth-N BFS as a ResultSet: columns ["key","label","depth"], BFS order.
15159 pub fn neighborhood(&self, depth: u32, edge_types: Option<&[&str]>, dir: Dir) -> ResultSet {
15160 let view = self.db.view();
15161 let resolved: Option<Vec<u32>> = edge_types.map(|names| {
15162 names
15163 .iter()
15164 .filter_map(|name| view.syms.get(name))
15165 .collect()
15166 });
15167 let nb = neighborhood(&view, self.id, depth, resolved.as_deref(), dir);
15168 let mut rs = ResultSet::new(vec!["key".into(), "label".into(), "depth".into()]);
15169 for (nid, d) in nb.nodes {
15170 let key = view.key_of(nid);
15171 let label = view
15172 .label_of(nid)
15173 .expect("real nodes always have a label; u32::MAX sentinel cannot occur");
15174 rs.push_row(vec![
15175 Some(Value::Str(key.to_string())),
15176 Some(Value::Str(label.to_string())),
15177 Some(Value::Int(d as i64)),
15178 ]);
15179 }
15180 rs
15181 }
15182
15183 /// 1-hop, Both directions: edge-type name → sorted unique neighbor keys.
15184 pub fn grouped_by_edge_type(&self) -> BTreeMap<String, Vec<String>> {
15185 let view = self.db.view();
15186 let mut groups: BTreeMap<String, BTreeSet<String>> = BTreeMap::new();
15187 for e in expand(&view, self.id, None, Dir::Both) {
15188 // Skip edges with unknown etypes (only possible from corrupt large
15189 // TOPOLOGY section; function returns BTreeMap not Result).
15190 let Some(etype) = view.syms.resolve(e.etype) else {
15191 continue;
15192 };
15193 let etype = etype.to_string();
15194 let nbr = if e.src == self.id { e.dst } else { e.src };
15195 groups
15196 .entry(etype)
15197 .or_default()
15198 .insert(view.key_of(nbr).to_string());
15199 }
15200 groups
15201 .into_iter()
15202 .map(|(k, v)| (k, v.into_iter().collect()))
15203 .collect()
15204 }
15205}
15206
15207#[cfg(test)]
15208mod tests {
15209 use super::*;
15210 use core_rules::Predicate;
15211
15212 fn tmp_dir(name: &str) -> std::path::PathBuf {
15213 let d =
15214 std::env::temp_dir().join(format!("graphdb-db-unit-{}-{}", name, std::process::id()));
15215 let _ = std::fs::remove_dir_all(&d);
15216 d
15217 }
15218
15219 fn fk_rule() -> RuleDef {
15220 RuleDef {
15221 name: "works_at".into(),
15222 src_label: "Person".into(),
15223 dst_label: "Org".into(),
15224 predicate: Predicate::KeyMatch {
15225 field: "org_id".into(),
15226 },
15227 edge_type: "WORKS_AT".into(),
15228 weight_prop: None,
15229 max_edges: None,
15230 approximate: false,
15231 via_label: None,
15232 via_edge: None,
15233 via_dir: None,
15234 namespace: None,
15235 }
15236 }
15237
15238 /// Regression guard for the no-views delta-copy fast path.
15239 ///
15240 /// When no views are defined, `pending_deltas_since().to_vec()` must never
15241 /// be called — even during a large CreateRule backfill. The DELTA_COPY_COUNT
15242 /// thread-local is incremented inside every `if !view_store.is_empty()` block;
15243 /// a count of 0 after the entire sequence proves the guard fires correctly.
15244 #[test]
15245 fn no_delta_copy_when_no_views() {
15246 DELTA_COPY_COUNT.with(|c| c.set(0));
15247 let dir = tmp_dir("no-delta-copy");
15248 {
15249 let mut db = GraphDb::open(&dir).unwrap();
15250 // Insert 50 Org + 50 Person nodes with FK links.
15251 for i in 0..50u32 {
15252 db.insert_node("Org", &format!("o{i}"), vec![]).unwrap();
15253 }
15254 for i in 0..50u32 {
15255 db.insert_node(
15256 "Person",
15257 &format!("p{i}"),
15258 vec![("org_id".into(), Value::Str(format!("o{i}")))],
15259 )
15260 .unwrap();
15261 }
15262 // CreateRule backfill should NOT invoke to_vec() when no views are defined.
15263 db.create_rule(fk_rule()).unwrap();
15264
15265 // Counter must stay 0 — no views, no copies.
15266 let copies = DELTA_COPY_COUNT.with(|c| c.get());
15267 assert_eq!(
15268 copies, 0,
15269 "pending_deltas_since().to_vec() called despite no views"
15270 );
15271
15272 // Derived edges must still be correct (the guard skips only the
15273 // empty delta propagation loop, not the rule application itself).
15274 let nbrs = db.neighbors("p0", "WORKS_AT", Direction::Out).unwrap();
15275 assert_eq!(
15276 nbrs,
15277 vec!["o0"],
15278 "rule must derive edges even with no views"
15279 );
15280 }
15281 let _ = std::fs::remove_dir_all(&dir);
15282 }
15283
15284 /// Gating regression: subscribe AFTER a backfill must see no stale events.
15285 /// subscribe BEFORE a backfill must see every edge-fire event.
15286 #[test]
15287 fn subscribe_after_backfill_no_stale_events() {
15288 let dir = tmp_dir("sub-after-backfill");
15289 {
15290 let mut db = GraphDb::open(&dir).unwrap();
15291 for i in 0..10u32 {
15292 db.insert_node("Org", &format!("o{i}"), vec![]).unwrap();
15293 db.insert_node(
15294 "Person",
15295 &format!("p{i}"),
15296 vec![("org_id".into(), Value::Str(format!("o{i}")))],
15297 )
15298 .unwrap();
15299 }
15300 // Create rule BEFORE subscribing — emit_deltas is false during backfill.
15301 db.create_rule(fk_rule()).unwrap();
15302
15303 // Subscribe AFTER the backfill — queue must be empty (no stale events).
15304 let sub = db.subscribe_all_rules().unwrap();
15305 // No events should have queued for the prior backfill.
15306 assert!(
15307 sub.try_recv().is_none(),
15308 "subscribe after backfill must see no stale events"
15309 );
15310
15311 // Inserting a new node now should fire an event (emit_deltas is now true).
15312 db.insert_node("Org", "o_new", vec![]).unwrap();
15313 db.insert_node(
15314 "Person",
15315 "p_new",
15316 vec![("org_id".into(), Value::Str("o_new".into()))],
15317 )
15318 .unwrap();
15319 let ev = sub.recv_timeout(std::time::Duration::from_millis(200));
15320 assert!(
15321 ev.is_some(),
15322 "edge-fire event must arrive after subscribe (emit_deltas=true)"
15323 );
15324 }
15325 let _ = std::fs::remove_dir_all(&dir);
15326 }
15327
15328 /// Gating regression: subscribe BEFORE a backfill → events flow.
15329 #[test]
15330 fn subscribe_before_backfill_events_flow() {
15331 let dir = tmp_dir("sub-before-backfill");
15332 {
15333 let mut db = GraphDb::open(&dir).unwrap();
15334 // Subscribe FIRST — emit_deltas becomes true.
15335 let sub = db.subscribe_all_rules().unwrap();
15336
15337 for i in 0..5u32 {
15338 db.insert_node("Org", &format!("o{i}"), vec![]).unwrap();
15339 db.insert_node(
15340 "Person",
15341 &format!("p{i}"),
15342 vec![("org_id".into(), Value::Str(format!("o{i}")))],
15343 )
15344 .unwrap();
15345 }
15346 // Backfill fires with emit_deltas=true → events queued.
15347 db.create_rule(fk_rule()).unwrap();
15348
15349 // Should receive at least one edge-fired event from the backfill.
15350 let mut received = 0usize;
15351 while sub.try_recv().is_some() {
15352 received += 1;
15353 }
15354 assert!(
15355 received > 0,
15356 "subscribe before backfill must receive edge-fire events (got 0)"
15357 );
15358 }
15359 let _ = std::fs::remove_dir_all(&dir);
15360 }
15361
15362 /// Companion: when a view IS defined, the delta path fires and view values update.
15363 #[test]
15364 fn delta_copy_fires_when_view_exists() {
15365 use core_rules::ViewSource;
15366 DELTA_COPY_COUNT.with(|c| c.set(0));
15367 let dir = tmp_dir("delta-copy-with-view");
15368 {
15369 let mut db = GraphDb::open(&dir).unwrap();
15370 db.insert_node("Org", "o1", vec![]).unwrap();
15371 db.insert_node(
15372 "Person",
15373 "p1",
15374 vec![("org_id".into(), Value::Str("o1".into()))],
15375 )
15376 .unwrap();
15377 // Declare a Degree view so is_empty() returns false.
15378 db.create_view(ViewDef {
15379 name: "degree_out".into(),
15380 label: "Person".into(),
15381 view_prop: "degree_out".into(),
15382 source: ViewSource::Degree {
15383 edge_type: "WORKS_AT".into(),
15384 direction: Direction::Out,
15385 },
15386 })
15387 .unwrap();
15388 db.create_rule(fk_rule()).unwrap();
15389
15390 // At least one delta copy should have happened (CreateRule backfill).
15391 let copies = DELTA_COPY_COUNT.with(|c| c.get());
15392 assert!(
15393 copies > 0,
15394 "expected delta copy to fire when a view is defined"
15395 );
15396
15397 // View value should be computed: p1 has one WORKS_AT out-edge.
15398 let info = db.node_info("p1").unwrap();
15399 let degree = info.props.get("degree_out");
15400 assert!(
15401 degree.is_some(),
15402 "view prop should be written to node props"
15403 );
15404 }
15405 let _ = std::fs::remove_dir_all(&dir);
15406 }
15407
15408 /// Regression: `open_at_with` must call `rebuild_all` after WAL replay so
15409 /// derived-edge-driven view values reflect the as-of state rather than just
15410 /// the initial backfill written at `CreateView` time.
15411 ///
15412 /// Base WAL frames (indices 0..=5 before history markers):
15413 /// 0: insert Org "o1"
15414 /// 1: create_view "employee_count" (Degree / WORKS_AT / In) on Org
15415 /// 2: create_rule fk_rule (WORKS_AT, Person→Org via org_id)
15416 /// 3: insert Person "p1" → rule fires WORKS_AT p1→o1 (degree = 1) ← mid
15417 /// 4: insert Person "p2" → rule fires WORKS_AT p2→o1 (degree = 2)
15418 /// 5: insert Person "p3" → rule fires WORKS_AT p3→o1 (degree = 3) ← latest
15419 ///
15420 /// Each rule-fire also appends a DerivedEdgeAdded history-marker frame (state
15421 /// no-op), so the total commit count is higher than the base frame count.
15422 /// The "latest" open_at commit is computed dynamically via `wal_commit_count_at`.
15423 ///
15424 /// Without `rebuild_all`, the as-of instance's "emp" view stays at the
15425 /// initial backfill value (0) instead of reflecting the replayed derived edges.
15426 #[test]
15427 fn open_at_derived_edge_view_values_correct() {
15428 use core_rules::ViewSource;
15429 let dir = tmp_dir("open-at-view-rebuild");
15430 {
15431 let mut db = GraphDb::open(&dir).unwrap();
15432 // frame 0
15433 db.insert_node("Org", "o1", vec![]).unwrap();
15434 // frame 1: create view — initial backfill sees 0 derived edges (none fired yet)
15435 db.create_view(ViewDef {
15436 name: "employee_count".into(),
15437 label: "Org".into(),
15438 view_prop: "emp".into(),
15439 source: ViewSource::Degree {
15440 edge_type: "WORKS_AT".into(),
15441 direction: Direction::In,
15442 },
15443 })
15444 .unwrap();
15445 // frame 2: create rule — no Persons yet; backfill is a no-op
15446 db.create_rule(fk_rule()).unwrap();
15447 // frame 3: p1 — rule fires WORKS_AT p1→o1; degree = 1
15448 db.insert_node(
15449 "Person",
15450 "p1",
15451 vec![("org_id".into(), Value::Str("o1".into()))],
15452 )
15453 .unwrap();
15454 // frame 4: p2 — degree = 2
15455 db.insert_node(
15456 "Person",
15457 "p2",
15458 vec![("org_id".into(), Value::Str("o1".into()))],
15459 )
15460 .unwrap();
15461 // frame 5: p3 — degree = 3
15462 db.insert_node(
15463 "Person",
15464 "p3",
15465 vec![("org_id".into(), Value::Str("o1".into()))],
15466 )
15467 .unwrap();
15468 // Sanity: normal open sees degree = 3.
15469 assert_eq!(
15470 db.get_view_prop("o1", "emp"),
15471 Some(Value::Int(3)),
15472 "normal db must show degree 3 after 3 derived edges"
15473 );
15474 } // WAL flushed
15475
15476 // Re-open normally to get the authoritative reference value.
15477 let normal_db = GraphDb::open(&dir).unwrap();
15478 let normal_emp = normal_db.get_view_prop("o1", "emp");
15479 assert_eq!(
15480 normal_emp,
15481 Some(Value::Int(3)),
15482 "re-opened normal db must show degree 3"
15483 );
15484
15485 // Latest as-of (last WAL commit): must match the normal open.
15486 // History-marker frames are appended after each rule-fire, so the total
15487 // commit count is computed dynamically rather than hardcoded.
15488 let total = crate::wal_commit_count_at(&dir).unwrap();
15489 let aof_latest = GraphDb::open_at(&dir, total - 1).unwrap();
15490 assert_eq!(
15491 aof_latest.get_view_prop("o1", "emp"),
15492 normal_emp,
15493 "open_at latest: derived-edge view must equal normal open (rebuild_all required)"
15494 );
15495
15496 // Mid-history as-of (commit 3 = p1 insert Batch frame): only p1; degree = 1.
15497 // The DerivedEdgeAdded marker for p1 is at frame 4 (state no-op on replay),
15498 // so replaying 0..=3 correctly re-derives only the p1→o1 edge.
15499 let aof_mid = GraphDb::open_at(&dir, 3).unwrap();
15500 assert_eq!(
15501 aof_mid.get_view_prop("o1", "emp"),
15502 Some(Value::Int(1)),
15503 "open_at mid-history: only p1 exists at frame 3, degree must be 1"
15504 );
15505
15506 let _ = std::fs::remove_dir_all(&dir);
15507 }
15508
15509 /// Pin: subscribe_* on an as-of instance must return Err(ReadOnly) —
15510 /// as-of instances never commit, so distribute_events never runs and any
15511 /// subscription would wait forever.
15512 #[test]
15513 fn subscribe_on_as_of_returns_read_only_error() {
15514 let dir = tmp_dir("sub-as-of-read-only");
15515 {
15516 let mut db = GraphDb::open(&dir).unwrap();
15517 db.insert_node("Org", "o1", vec![]).unwrap();
15518 db.create_rule(fk_rule()).unwrap();
15519 }
15520 let mut aof = GraphDb::open_at(&dir, 0).unwrap();
15521
15522 assert!(
15523 matches!(
15524 aof.subscribe_all_rules(),
15525 Err(core_storage::GraphError::ReadOnly)
15526 ),
15527 "subscribe_all_rules on as-of must return ReadOnly"
15528 );
15529 assert!(
15530 matches!(
15531 aof.subscribe_writes(),
15532 Err(core_storage::GraphError::ReadOnly)
15533 ),
15534 "subscribe_writes on as-of must return ReadOnly"
15535 );
15536 assert!(
15537 matches!(
15538 aof.subscribe_rule("works_at"),
15539 Err(core_storage::GraphError::ReadOnly)
15540 ),
15541 "subscribe_rule on as-of must return ReadOnly"
15542 );
15543 let _ = std::fs::remove_dir_all(&dir);
15544 }
15545
15546 /// Regression: a failed dense WAL rewrite must not leave speculative
15547 /// interns in `syms`. If it does, the next successful mutation logs an
15548 /// `Intern` record with an inflated id; replay (which never saw the
15549 /// orphans) assigns a smaller id and the WAL becomes unreplayable.
15550 #[test]
15551 fn dense_rewrite_error_rolls_back_speculative_interns() {
15552 let dir = tmp_dir("dense-rewrite-rollback");
15553 {
15554 let mut db = GraphDb::open(&dir).unwrap();
15555 db.insert_node("Person", "a", vec![]).unwrap();
15556
15557 // Bypass MutPreview validation to hit the rewrite's own error path
15558 // (same shape as an id-exhaustion failure mid-rewrite). The
15559 // InsertEdge arm interns the edge type before it resolves keys.
15560 let err = db.rewrite_wal_dense(vec![WalRecord::InsertEdge {
15561 edge_type: "ORPHAN_TYPE".into(),
15562 src_key: "missing".into(),
15563 dst_key: "a".into(),
15564 }]);
15565 assert!(err.is_err(), "rewrite of a missing src key must fail");
15566 assert_eq!(
15567 db.syms.get("ORPHAN_TYPE"),
15568 None,
15569 "failed rewrite must roll back speculative interns"
15570 );
15571
15572 // A later successful mutation must produce a replayable WAL.
15573 db.set_prop("a", "later_field", Value::Int(2)).unwrap();
15574 }
15575 let db = GraphDb::open(&dir).expect("WAL must replay after failed rewrite");
15576 assert_eq!(db.get_prop("a", "later_field"), Some(Value::Int(2)));
15577 let _ = std::fs::remove_dir_all(&dir);
15578 }
15579}