Skip to main content

lex_vcs/
op_log.rs

1//! Persistence + DAG queries for the operation log.
2//!
3//! Layout: `<root>/ops/<op_id>.json` — one canonical-JSON file
4//! per [`OperationRecord`]. Atomic writes via tempfile + rename.
5//! Idempotent: writing an existing op_id is a no-op (content
6//! addressing guarantees the bytes match).
7//!
8//! # Packfiles (#261 slice 1)
9//!
10//! Loose-file storage is fine to ~10k ops; past that the
11//! filesystem starts to thrash. [`OpLog::repack`] consolidates
12//! loose files into deterministic, content-addressed packfiles:
13//!
14//! - `<dir>/pack-<hash>.pack`: each record framed as `[8-byte BE
15//!   length][canonical JSON]`, ops sorted by op_id within the pack.
16//! - `<dir>/pack-<hash>.idx`: JSON map of `op_id` → byte offset
17//!   into the `.pack` (offset of the length header).
18//!
19//! Pack name is the SHA-256 of the sorted op_ids, newline-joined,
20//! so the same input set always produces the same pack hash —
21//! a re-run of `lex op repack` is a no-op.
22//!
23//! [`OpLog::get`] tries loose first, falls back to scanning all
24//! `.idx` files in the directory. The write path
25//! ([`OpLog::put`]) only ever writes loose; ops migrate into
26//! packs via the explicit [`OpLog::repack`] call.
27
28use crate::canonical::hash_bytes;
29use crate::operation::{OpId, OperationRecord};
30use std::collections::{BTreeMap, BTreeSet, VecDeque};
31use std::fs;
32use std::io::{self, Read, Seek, SeekFrom, Write};
33use std::path::{Path, PathBuf};
34
35pub struct OpLog {
36    dir: PathBuf,
37}
38
39impl OpLog {
40    pub fn open(root: &Path) -> io::Result<Self> {
41        let dir = root.join("ops");
42        fs::create_dir_all(&dir)?;
43        Ok(Self { dir })
44    }
45
46    fn path(&self, op_id: &OpId) -> PathBuf {
47        self.dir.join(format!("{op_id}.json"))
48    }
49
50    /// Persist a record. Idempotent on existing op_ids (the bytes
51    /// must match by content addressing).
52    ///
53    /// Crash safety: the tempfile's data is fsync'd before rename,
54    /// so a successful return implies a durable file at the final
55    /// path. The containing directory is not fsync'd; on a crash
56    /// between rename and the directory's metadata flush, the file
57    /// can be lost. For a content-addressed log this is acceptable
58    /// — a lost record can be re-derived from the same source — but
59    /// callers that *also* persist references to the op_id (e.g.
60    /// branch heads) should fsync those refs after `put` returns.
61    pub fn put(&self, rec: &OperationRecord) -> io::Result<()> {
62        let path = self.path(&rec.op_id);
63        if path.exists() {
64            return Ok(());
65        }
66        let bytes = serde_json::to_vec(rec)
67            .map_err(|e| io::Error::new(io::ErrorKind::InvalidData, e))?;
68        let tmp = path.with_extension("json.tmp");
69        let mut f = fs::File::create(&tmp)?;
70        f.write_all(&bytes)?;
71        f.sync_all()?;
72        fs::rename(&tmp, &path)?;
73        Ok(())
74    }
75
76    pub fn get(&self, op_id: &OpId) -> io::Result<Option<OperationRecord>> {
77        let path = self.path(op_id);
78        if path.exists() {
79            let bytes = fs::read(&path)?;
80            let rec: OperationRecord = serde_json::from_slice(&bytes)
81                .map_err(|e| io::Error::new(io::ErrorKind::InvalidData, e))?;
82            return Ok(Some(rec));
83        }
84        // Loose miss — scan packfiles. Each `.idx` is a tiny JSON
85        // map; small constant cost per pack. For larger stores we
86        // could maintain an in-memory cache keyed off pack mtimes,
87        // but slice 1 keeps it simple — measure before optimizing.
88        for pack_idx in self.list_pack_indices()? {
89            let idx = PackIndex::load(&pack_idx)?;
90            if let Some(&offset) = idx.ops.get(op_id) {
91                let pack_path = pack_idx.with_extension("pack");
92                return read_packed_op(&pack_path, offset).map(Some);
93            }
94        }
95        Ok(None)
96    }
97
98    /// Walk the directory for `pack-*.idx` files. Order is whatever
99    /// the filesystem gives us — `get` doesn't depend on it (op_ids
100    /// are unique by content addressing, so the right pack wins).
101    fn list_pack_indices(&self) -> io::Result<Vec<PathBuf>> {
102        let mut out = Vec::new();
103        for entry in fs::read_dir(&self.dir)? {
104            let entry = entry?;
105            let name = match entry.file_name().into_string() {
106                Ok(s) => s,
107                Err(_) => continue,
108            };
109            if name.starts_with("pack-") && name.ends_with(".idx") {
110                out.push(entry.path());
111            }
112        }
113        Ok(out)
114    }
115
116    /// Consolidate loose op records into a deterministic, content-
117    /// addressed packfile (#261 slice 1). Returns the number of
118    /// ops moved into the new pack.
119    ///
120    /// `threshold` is the minimum number of loose ops required to
121    /// trigger a repack — under that, returns `0` and leaves the
122    /// log alone. The idea: small stores stay loose; only repack
123    /// when the file count starts to matter.
124    ///
125    /// Determinism: the pack name is the SHA-256 of the sorted
126    /// op_ids (newline-joined), so two independent runs against the
127    /// same set of loose ops produce a byte-identical pack.
128    /// Re-running on an empty loose directory is a no-op.
129    ///
130    /// Crash safety: the `.pack.tmp` and `.idx.tmp` files are
131    /// fsync'd before rename; loose files are deleted only after
132    /// both renames succeed. A crash mid-repack leaves both loose
133    /// and partial-pack files; a subsequent `get` finds the loose
134    /// version, and a subsequent `repack` cleans up.
135    pub fn repack(&self, threshold: usize) -> io::Result<usize> {
136        let loose: Vec<(OpId, PathBuf)> = self.list_loose_files()?;
137        if loose.len() < threshold {
138            return Ok(0);
139        }
140        // Sort ops deterministically by op_id (lex order). The pack
141        // hash is the SHA-256 of those op_ids joined by newlines —
142        // same input → same name.
143        let mut ops: Vec<(OpId, Vec<u8>)> = Vec::with_capacity(loose.len());
144        for (op_id, path) in &loose {
145            let bytes = fs::read(path)?;
146            ops.push((op_id.clone(), bytes));
147        }
148        ops.sort_by(|a, b| a.0.cmp(&b.0));
149        let mut name_input = Vec::new();
150        for (id, _) in &ops {
151            name_input.extend_from_slice(id.as_bytes());
152            name_input.push(b'\n');
153        }
154        let pack_hash = hash_bytes(&name_input);
155        let pack_path = self.dir.join(format!("pack-{pack_hash}.pack"));
156        let idx_path = self.dir.join(format!("pack-{pack_hash}.idx"));
157        if pack_path.exists() && idx_path.exists() {
158            // Same input set — pack already exists. Just clean up
159            // the loose duplicates.
160            let count = ops.len();
161            for (_, path) in &loose {
162                let _ = fs::remove_file(path);
163            }
164            return Ok(count);
165        }
166
167        // Write `<pack>.pack.tmp` framed as [8-byte BE length][JSON]
168        // for each record; record offsets for the index.
169        let pack_tmp = pack_path.with_extension("pack.tmp");
170        let idx_tmp = idx_path.with_extension("idx.tmp");
171        let mut offsets: BTreeMap<OpId, u64> = BTreeMap::new();
172        {
173            let mut f = fs::File::create(&pack_tmp)?;
174            let mut cursor: u64 = 0;
175            for (op_id, bytes) in &ops {
176                offsets.insert(op_id.clone(), cursor);
177                let len = bytes.len() as u64;
178                f.write_all(&len.to_be_bytes())?;
179                f.write_all(bytes)?;
180                cursor += 8 + len;
181            }
182            f.sync_all()?;
183        }
184        // Write the index. JSON for inspectability and
185        // forward-compat (we can add fields without breaking
186        // readers).
187        let idx = PackIndex { version: 1, ops: offsets };
188        idx.save(&idx_tmp)?;
189
190        fs::rename(&pack_tmp, &pack_path)?;
191        fs::rename(&idx_tmp, &idx_path)?;
192
193        // Now safe to delete the loose files — pack is durable.
194        let count = ops.len();
195        for (_, path) in &loose {
196            let _ = fs::remove_file(path);
197        }
198        Ok(count)
199    }
200
201    /// Remove every op_id in `victims` from the log, across both
202    /// loose files and packfiles (#261 slice 2). Used by
203    /// `lex op gc` after a retention plan identifies which ops to
204    /// drop. Idempotent — calling twice with the same set is a
205    /// no-op on the second pass.
206    ///
207    /// Pack handling: any pack containing one or more victims is
208    /// rewritten to a new content-addressed pack with only the
209    /// surviving ops; the old pack and its index file are deleted.
210    /// A pack whose every op is a victim is deleted outright.
211    ///
212    /// Returns the count of ops actually removed (loose files
213    /// deleted + packed ops dropped). Pre-existing absences don't
214    /// contribute.
215    pub fn evict(&self, victims: &BTreeSet<OpId>) -> io::Result<usize> {
216        if victims.is_empty() {
217            return Ok(0);
218        }
219        let mut removed = 0;
220        // Loose files: just delete the matching `<op_id>.json`.
221        for (op_id, path) in self.list_loose_files()? {
222            if victims.contains(&op_id) {
223                match fs::remove_file(&path) {
224                    Ok(()) => removed += 1,
225                    Err(e) if e.kind() == io::ErrorKind::NotFound => {}
226                    Err(e) => return Err(e),
227                }
228            }
229        }
230        // Packs: rewrite each affected pack with only surviving ops.
231        for pack_idx in self.list_pack_indices()? {
232            let idx = PackIndex::load(&pack_idx)?;
233            let pack_path = pack_idx.with_extension("pack");
234            let touched = idx.ops.keys().any(|op_id| victims.contains(op_id));
235            if !touched {
236                continue;
237            }
238            // Read every surviving op out, then drop the old pack.
239            let mut survivors: Vec<(OpId, Vec<u8>)> = Vec::new();
240            for (op_id, &offset) in &idx.ops {
241                if victims.contains(op_id) {
242                    removed += 1;
243                    continue;
244                }
245                let rec = read_packed_op(&pack_path, offset)?;
246                let bytes = serde_json::to_vec(&rec)
247                    .map_err(|e| io::Error::new(io::ErrorKind::InvalidData, e))?;
248                survivors.push((op_id.clone(), bytes));
249            }
250            // Delete old pack + idx first; we re-emit a fresh
251            // (different-hash) pack from the survivors below.
252            let _ = fs::remove_file(&pack_path);
253            let _ = fs::remove_file(&pack_idx);
254            if survivors.is_empty() {
255                continue;
256            }
257            self.write_pack_from_survivors(survivors)?;
258        }
259        Ok(removed)
260    }
261
262    /// Helper: write a new content-addressed pack from
263    /// already-serialized op bytes. Same shape as
264    /// [`Self::repack`]'s output path; factored out so
265    /// [`Self::evict`] can reuse it.
266    fn write_pack_from_survivors(
267        &self,
268        mut ops: Vec<(OpId, Vec<u8>)>,
269    ) -> io::Result<()> {
270        ops.sort_by(|a, b| a.0.cmp(&b.0));
271        let mut name_input = Vec::new();
272        for (id, _) in &ops {
273            name_input.extend_from_slice(id.as_bytes());
274            name_input.push(b'\n');
275        }
276        let pack_hash = hash_bytes(&name_input);
277        let pack_path = self.dir.join(format!("pack-{pack_hash}.pack"));
278        let idx_path = self.dir.join(format!("pack-{pack_hash}.idx"));
279        if pack_path.exists() && idx_path.exists() {
280            return Ok(());
281        }
282        let pack_tmp = pack_path.with_extension("pack.tmp");
283        let idx_tmp = idx_path.with_extension("idx.tmp");
284        let mut offsets: BTreeMap<OpId, u64> = BTreeMap::new();
285        {
286            let mut f = fs::File::create(&pack_tmp)?;
287            let mut cursor: u64 = 0;
288            for (op_id, bytes) in &ops {
289                offsets.insert(op_id.clone(), cursor);
290                let len = bytes.len() as u64;
291                f.write_all(&len.to_be_bytes())?;
292                f.write_all(bytes)?;
293                cursor += 8 + len;
294            }
295            f.sync_all()?;
296        }
297        let idx = PackIndex { version: 1, ops: offsets };
298        idx.save(&idx_tmp)?;
299        fs::rename(&pack_tmp, &pack_path)?;
300        fs::rename(&idx_tmp, &idx_path)?;
301        Ok(())
302    }
303
304    /// Enumerate every loose `<op_id>.json` in the ops directory.
305    /// Used by [`Self::repack`] and [`Self::list_all`].
306    fn list_loose_files(&self) -> io::Result<Vec<(OpId, PathBuf)>> {
307        let mut out = Vec::new();
308        for entry in fs::read_dir(&self.dir)? {
309            let entry = entry?;
310            let name = match entry.file_name().into_string() {
311                Ok(s) => s,
312                Err(_) => continue,
313            };
314            if let Some(id) = name.strip_suffix(".json") {
315                if !id.starts_with("pack-") {
316                    out.push((id.to_string(), entry.path()));
317                }
318            }
319        }
320        Ok(out)
321    }
322
323    /// Remove a record from the log. Used by [`crate::migrate`] to
324    /// delete the old `<op_id>.json` files after a format migration
325    /// has written their replacements. Idempotent on missing files.
326    ///
327    /// **Not** part of the day-to-day op-log API — the log is
328    /// append-only by design (#129). The only legitimate caller is
329    /// the migration tool, which is supervising a destructive,
330    /// `--confirm`-gated batch.
331    pub fn delete(&self, op_id: &OpId) -> io::Result<()> {
332        let path = self.path(op_id);
333        match fs::remove_file(&path) {
334            Ok(()) => Ok(()),
335            Err(e) if e.kind() == io::ErrorKind::NotFound => Ok(()),
336            Err(e) => Err(e),
337        }
338    }
339
340    /// Walk parents transitively. Newest-first, BFS, dedup'd by op_id.
341    /// Stops at parentless ops or after `limit` records.
342    pub fn walk_back(
343        &self,
344        head: &OpId,
345        limit: Option<usize>,
346    ) -> io::Result<Vec<OperationRecord>> {
347        let mut out = Vec::new();
348        let mut seen = BTreeSet::new();
349        let mut frontier: VecDeque<OpId> = VecDeque::from([head.clone()]);
350        while let Some(id) = frontier.pop_back() {
351            if !seen.insert(id.clone()) {
352                continue;
353            }
354            if let Some(rec) = self.get(&id)? {
355                // Push parents before recording so traversal order is
356                // a stable BFS-by-discovery: children-first, then their
357                // parents, parents of those, etc.
358                for p in &rec.op.parents {
359                    if !seen.contains(p) {
360                        frontier.push_front(p.clone());
361                    }
362                }
363                out.push(rec);
364                if let Some(n) = limit {
365                    if out.len() >= n {
366                        break;
367                    }
368                }
369            }
370        }
371        Ok(out)
372    }
373
374    /// Same set as walk_back but oldest-first. Used by branch_head
375    /// for left-to-right transition replay.
376    pub fn walk_forward(
377        &self,
378        head: &OpId,
379        limit: Option<usize>,
380    ) -> io::Result<Vec<OperationRecord>> {
381        let mut all = self.walk_back(head, None)?;
382        all.reverse();
383        if let Some(n) = limit {
384            all.truncate(n);
385        }
386        Ok(all)
387    }
388
389    /// Like [`Self::walk_forward`], but bounded: walk from `head` back
390    /// toward genesis and stop as soon as `since` is reached, without
391    /// visiting `since`'s own parents or including `since` itself in the
392    /// result. Returns oldest-first, suitable for incrementally
393    /// extending a transition map already computed as of `since`.
394    ///
395    /// Returns `Ok(None)` if `since` is never reached (not an ancestor
396    /// of `head` — e.g. after a branch reset or a merge that reordered
397    /// history): callers should fall back to a full `walk_forward` in
398    /// that case, since there is nothing valid to incrementally extend.
399    ///
400    /// This is the piece `walk_forward` itself doesn't provide: its own
401    /// `limit` truncates the *result* after a full walk_back to genesis
402    /// has already completed (see its body above), so it can't turn an
403    /// O(N)-in-total-history walk into an O(ops since a checkpoint) one.
404    /// `head == since` returns `Ok(Some(vec![]))` without touching the
405    /// op log at all.
406    pub fn walk_forward_since(
407        &self,
408        head: &OpId,
409        since: &OpId,
410    ) -> io::Result<Option<Vec<OperationRecord>>> {
411        if head == since {
412            return Ok(Some(Vec::new()));
413        }
414        let mut out = Vec::new();
415        let mut seen = BTreeSet::new();
416        let mut frontier: VecDeque<OpId> = VecDeque::from([head.clone()]);
417        let mut found = false;
418        while let Some(id) = frontier.pop_back() {
419            if !seen.insert(id.clone()) {
420                continue;
421            }
422            if id == *since {
423                found = true;
424                continue; // boundary: don't include it, don't descend into its parents
425            }
426            if let Some(rec) = self.get(&id)? {
427                for p in &rec.op.parents {
428                    if !seen.contains(p) {
429                        frontier.push_front(p.clone());
430                    }
431                }
432                out.push(rec);
433            }
434        }
435        if !found {
436            return Ok(None);
437        }
438        out.reverse();
439        Ok(Some(out))
440    }
441
442    /// Common ancestor of two op_ids in the DAG.
443    ///
444    /// On tree-shaped histories and chain merges this is the
445    /// **lowest** common ancestor — the closest shared op. On
446    /// criss-cross merges (two ops each with two parents from
447    /// independent histories) there can be multiple
448    /// incomparable common ancestors; this picks one
449    /// deterministically (the first hit when traversing `b`'s
450    /// ancestors newest-first), but not via a recursive merge.
451    /// `None` if no shared ancestor exists.
452    ///
453    /// Tier-1 merge in #129 covers linear and tree-shaped
454    /// histories; criss-cross resolution is deferred to a
455    /// future tier (Git's `recursive` strategy is the reference).
456    pub fn lca(&self, a: &OpId, b: &OpId) -> io::Result<Option<OpId>> {
457        let a_anc: BTreeSet<OpId> = self
458            .walk_back(a, None)?
459            .into_iter()
460            .map(|r| r.op_id)
461            .collect();
462        // Walk b's ancestors newest-first; first hit is the deepest
463        // common ancestor on tree-shaped histories. In criss-cross
464        // DAGs this picks deterministically but not via recursive
465        // resolution — see the doc comment above.
466        for rec in self.walk_back(b, None)? {
467            if a_anc.contains(&rec.op_id) {
468                return Ok(Some(rec.op_id));
469            }
470        }
471        Ok(None)
472    }
473
474    /// Every record in the log. Order is whatever the directory
475    /// listing produces — undefined and not stable. Used by the
476    /// [`crate::predicate`] evaluator when no narrower candidate
477    /// set is available.
478    pub fn list_all(&self) -> io::Result<Vec<OperationRecord>> {
479        let mut out = Vec::new();
480        let mut seen: BTreeSet<OpId> = BTreeSet::new();
481        // Loose first so dedup wins for them on collision (loose
482        // and pack should never both exist for the same op_id post-
483        // repack, but during an interrupted repack both can be
484        // present transiently).
485        for (id, _) in self.list_loose_files()? {
486            if let Some(rec) = self.get(&id)? {
487                if seen.insert(rec.op_id.clone()) {
488                    out.push(rec);
489                }
490            }
491        }
492        for pack_idx in self.list_pack_indices()? {
493            let idx = PackIndex::load(&pack_idx)?;
494            let pack_path = pack_idx.with_extension("pack");
495            for (op_id, &offset) in &idx.ops {
496                if seen.insert(op_id.clone()) {
497                    out.push(read_packed_op(&pack_path, offset)?);
498                }
499            }
500        }
501        Ok(out)
502    }
503
504    /// Ops in `head`'s history that are not in `base`'s history.
505    /// `base = None` means "include all of head's history" (used for
506    /// independent-histories case where the LCA is None).
507    pub fn ops_since(
508        &self,
509        head: &OpId,
510        base: Option<&OpId>,
511    ) -> io::Result<Vec<OperationRecord>> {
512        let exclude: BTreeSet<OpId> = match base {
513            Some(b) => self
514                .walk_back(b, None)?
515                .into_iter()
516                .map(|r| r.op_id)
517                .collect(),
518            None => BTreeSet::new(),
519        };
520        Ok(self
521            .walk_back(head, None)?
522            .into_iter()
523            .filter(|r| !exclude.contains(&r.op_id))
524            .collect())
525    }
526}
527
528/// Sidecar index for a packfile. Maps `op_id` to the byte offset
529/// of the record's length header inside the `.pack`. JSON for
530/// inspectability and forward-compat.
531#[derive(serde::Serialize, serde::Deserialize)]
532struct PackIndex {
533    version: u32,
534    ops: BTreeMap<OpId, u64>,
535}
536
537impl PackIndex {
538    fn load(path: &Path) -> io::Result<Self> {
539        let bytes = fs::read(path)?;
540        serde_json::from_slice(&bytes)
541            .map_err(|e| io::Error::new(io::ErrorKind::InvalidData, e))
542    }
543
544    fn save(&self, path: &Path) -> io::Result<()> {
545        let bytes = serde_json::to_vec(self)
546            .map_err(|e| io::Error::new(io::ErrorKind::InvalidData, e))?;
547        let mut f = fs::File::create(path)?;
548        f.write_all(&bytes)?;
549        f.sync_all()?;
550        Ok(())
551    }
552}
553
554/// Read one record from a packfile at `offset`. The record is
555/// framed as `[8-byte BE length][canonical JSON]`.
556fn read_packed_op(pack_path: &Path, offset: u64) -> io::Result<OperationRecord> {
557    let mut f = fs::File::open(pack_path)?;
558    f.seek(SeekFrom::Start(offset))?;
559    let mut len_buf = [0u8; 8];
560    f.read_exact(&mut len_buf)?;
561    let len = u64::from_be_bytes(len_buf) as usize;
562    let mut buf = vec![0u8; len];
563    f.read_exact(&mut buf)?;
564    serde_json::from_slice(&buf)
565        .map_err(|e| io::Error::new(io::ErrorKind::InvalidData, e))
566}
567
568#[cfg(test)]
569mod tests {
570    use super::*;
571    use crate::operation::{Operation, OperationKind, StageTransition};
572    use std::collections::{BTreeMap, BTreeSet};
573
574    fn add_op() -> OperationRecord {
575        let op = Operation::new(
576            OperationKind::AddFunction {
577                sig_id: "fac::Int->Int".into(),
578                stage_id: "abc123".into(),
579                effects: BTreeSet::new(),
580                budget_cost: None,
581                in_file: None,
582            },
583            [],
584        );
585        OperationRecord::new(
586            op,
587            StageTransition::Create {
588                sig_id: "fac::Int->Int".into(),
589                stage_id: "abc123".into(),
590            },
591        )
592    }
593
594    fn modify_op(parent: &OpId, sig: &str, from: &str, to: &str) -> OperationRecord {
595        let op = Operation::new(
596            OperationKind::ModifyBody {
597                sig_id: sig.into(),
598                from_stage_id: from.into(),
599                to_stage_id: to.into(),
600                from_budget: None,
601                to_budget: None,
602                to_sig_id: None,
603            },
604            [parent.clone()],
605        );
606        OperationRecord::new(
607            op,
608            StageTransition::Replace {
609                sig_id: sig.into(),
610                from: from.into(),
611                to: to.into(),
612            },
613        )
614    }
615
616    #[test]
617    fn put_then_get_round_trips() {
618        let tmp = tempfile::tempdir().unwrap();
619        let log = OpLog::open(tmp.path()).unwrap();
620        let rec = add_op();
621        log.put(&rec).unwrap();
622        let back = log.get(&rec.op_id).unwrap().unwrap();
623        assert_eq!(back, rec);
624    }
625
626    #[test]
627    fn put_is_idempotent() {
628        let tmp = tempfile::tempdir().unwrap();
629        let log = OpLog::open(tmp.path()).unwrap();
630        let rec = add_op();
631        log.put(&rec).unwrap();
632        log.put(&rec).unwrap(); // second write is a no-op
633        assert!(log.get(&rec.op_id).unwrap().is_some());
634    }
635
636    #[test]
637    fn get_missing_returns_none() {
638        let tmp = tempfile::tempdir().unwrap();
639        let log = OpLog::open(tmp.path()).unwrap();
640        assert!(log.get(&"deadbeef".to_string()).unwrap().is_none());
641    }
642
643    #[test]
644    fn walk_back_returns_newest_first() {
645        let tmp = tempfile::tempdir().unwrap();
646        let log = OpLog::open(tmp.path()).unwrap();
647        let a = add_op();
648        log.put(&a).unwrap();
649        let b = modify_op(&a.op_id, "fac::Int->Int", "abc123", "def456");
650        log.put(&b).unwrap();
651        let c = modify_op(&b.op_id, "fac::Int->Int", "def456", "789aaa");
652        log.put(&c).unwrap();
653
654        let walked = log.walk_back(&c.op_id, None).unwrap();
655        let ids: Vec<_> = walked.iter().map(|r| r.op_id.as_str()).collect();
656        assert_eq!(
657            ids,
658            vec![c.op_id.as_str(), b.op_id.as_str(), a.op_id.as_str()]
659        );
660    }
661
662    #[test]
663    fn walk_forward_returns_oldest_first() {
664        let tmp = tempfile::tempdir().unwrap();
665        let log = OpLog::open(tmp.path()).unwrap();
666        let a = add_op();
667        log.put(&a).unwrap();
668        let b = modify_op(&a.op_id, "fac::Int->Int", "abc123", "def456");
669        log.put(&b).unwrap();
670
671        let walked = log.walk_forward(&b.op_id, None).unwrap();
672        let ids: Vec<_> = walked.iter().map(|r| r.op_id.as_str()).collect();
673        assert_eq!(ids, vec![a.op_id.as_str(), b.op_id.as_str()]);
674    }
675
676    #[test]
677    fn walk_forward_since_returns_only_ops_after_the_boundary() {
678        let tmp = tempfile::tempdir().unwrap();
679        let log = OpLog::open(tmp.path()).unwrap();
680        let a = add_op();
681        log.put(&a).unwrap();
682        let b = modify_op(&a.op_id, "fac::Int->Int", "abc123", "def456");
683        log.put(&b).unwrap();
684        let c = modify_op(&b.op_id, "fac::Int->Int", "def456", "789aaa");
685        log.put(&c).unwrap();
686
687        // Everything strictly after `a`: b, c, oldest-first.
688        let since_a = log.walk_forward_since(&c.op_id, &a.op_id).unwrap().unwrap();
689        let ids: Vec<_> = since_a.iter().map(|r| r.op_id.as_str()).collect();
690        assert_eq!(ids, vec![b.op_id.as_str(), c.op_id.as_str()]);
691
692        // Everything strictly after `b`: just c.
693        let since_b = log.walk_forward_since(&c.op_id, &b.op_id).unwrap().unwrap();
694        let ids: Vec<_> = since_b.iter().map(|r| r.op_id.as_str()).collect();
695        assert_eq!(ids, vec![c.op_id.as_str()]);
696    }
697
698    #[test]
699    fn walk_forward_since_head_equals_since_returns_empty_without_touching_the_log() {
700        let tmp = tempfile::tempdir().unwrap();
701        let log = OpLog::open(tmp.path()).unwrap();
702        let a = add_op();
703        log.put(&a).unwrap();
704
705        // Deliberately pass an op_id that was never `put` -- if this
706        // took the "walk and look for it" path it would return `None`
707        // (not found). The `head == since` fast path must short-circuit
708        // before ever touching the log.
709        let ghost = "never-written-anywhere".to_string();
710        let result = log.walk_forward_since(&ghost, &ghost).unwrap();
711        assert_eq!(result, Some(Vec::new()));
712    }
713
714    #[test]
715    fn walk_forward_since_returns_none_when_boundary_is_not_an_ancestor() {
716        let tmp = tempfile::tempdir().unwrap();
717        let log = OpLog::open(tmp.path()).unwrap();
718        let a = add_op();
719        log.put(&a).unwrap();
720        // A second root with different content (distinct sig_id), so it
721        // gets a different content-addressed op_id and shares no
722        // history with `a` -- `add_op()` alone is parameterless and
723        // would collide with itself.
724        let op = Operation::new(
725            OperationKind::AddFunction {
726                sig_id: "unrelated::Str->Str".into(),
727                stage_id: "zzz999".into(),
728                effects: BTreeSet::new(),
729                budget_cost: None,
730                in_file: None,
731            },
732            [],
733        );
734        let unrelated = OperationRecord::new(
735            op,
736            StageTransition::Create {
737                sig_id: "unrelated::Str->Str".into(),
738                stage_id: "zzz999".into(),
739            },
740        );
741        log.put(&unrelated).unwrap();
742        assert_ne!(a.op_id, unrelated.op_id, "test setup must produce two distinct ops");
743
744        let result = log.walk_forward_since(&a.op_id, &unrelated.op_id).unwrap();
745        assert_eq!(
746            result, None,
747            "unrelated op_id is not an ancestor of `a` -- callers must fall back to a full walk"
748        );
749    }
750
751    #[test]
752    fn lca_finds_common_ancestor() {
753        let tmp = tempfile::tempdir().unwrap();
754        let log = OpLog::open(tmp.path()).unwrap();
755        let root = add_op();
756        log.put(&root).unwrap();
757        let left = modify_op(&root.op_id, "fac::Int->Int", "abc123", "left1");
758        log.put(&left).unwrap();
759        let right = modify_op(&root.op_id, "fac::Int->Int", "abc123", "right1");
760        log.put(&right).unwrap();
761
762        let lca = log.lca(&left.op_id, &right.op_id).unwrap();
763        assert_eq!(lca, Some(root.op_id));
764    }
765
766    #[test]
767    fn lca_none_for_independent_histories() {
768        let tmp = tempfile::tempdir().unwrap();
769        let log = OpLog::open(tmp.path()).unwrap();
770        let a = add_op();
771        log.put(&a).unwrap();
772        // A second parentless op (different sig, so different op_id).
773        let b = OperationRecord::new(
774            Operation::new(
775                OperationKind::AddFunction {
776                    sig_id: "double::Int->Int".into(),
777                    stage_id: "ddd111".into(),
778                    effects: BTreeSet::new(),
779                    budget_cost: None,
780                    in_file: None,
781                },
782                [],
783            ),
784            StageTransition::Create {
785                sig_id: "double::Int->Int".into(),
786                stage_id: "ddd111".into(),
787            },
788        );
789        log.put(&b).unwrap();
790
791        assert_eq!(log.lca(&a.op_id, &b.op_id).unwrap(), None);
792    }
793
794    #[test]
795    fn ops_since_excludes_base_history() {
796        let tmp = tempfile::tempdir().unwrap();
797        let log = OpLog::open(tmp.path()).unwrap();
798        let a = add_op();
799        log.put(&a).unwrap();
800        let b = modify_op(&a.op_id, "fac::Int->Int", "abc123", "def456");
801        log.put(&b).unwrap();
802        let c = modify_op(&b.op_id, "fac::Int->Int", "def456", "789aaa");
803        log.put(&c).unwrap();
804
805        let since: Vec<_> = log
806            .ops_since(&c.op_id, Some(&a.op_id))
807            .unwrap()
808            .into_iter()
809            .map(|r| r.op_id)
810            .collect();
811        assert_eq!(since.len(), 2);
812        assert!(since.contains(&b.op_id));
813        assert!(since.contains(&c.op_id));
814        assert!(!since.contains(&a.op_id));
815    }
816
817    #[test]
818    fn repack_consolidates_loose_files_into_a_pack() {
819        let tmp = tempfile::tempdir().unwrap();
820        let log = OpLog::open(tmp.path()).unwrap();
821        let a = add_op();
822        log.put(&a).unwrap();
823        let b = modify_op(&a.op_id, "fac::Int->Int", "abc123", "def456");
824        log.put(&b).unwrap();
825
826        let n = log.repack(0).unwrap();  // threshold 0 = always
827        assert_eq!(n, 2);
828        let ops_dir = tmp.path().join("ops");
829        let loose: Vec<_> = fs::read_dir(&ops_dir).unwrap()
830            .filter_map(|e| e.ok())
831            .filter(|e| e.path().extension().is_some_and(|x| x == "json"))
832            .filter(|e| !e.file_name().to_string_lossy().starts_with("pack-"))
833            .collect();
834        assert!(loose.is_empty(), "loose .json files should be deleted");
835        let packs: Vec<_> = fs::read_dir(&ops_dir).unwrap()
836            .filter_map(|e| e.ok())
837            .filter(|e| e.path().extension().is_some_and(|x| x == "pack"))
838            .collect();
839        assert_eq!(packs.len(), 1);
840
841        // After repack, get() must still return both ops via the pack.
842        assert_eq!(log.get(&a.op_id).unwrap().unwrap(), a);
843        assert_eq!(log.get(&b.op_id).unwrap().unwrap(), b);
844    }
845
846    #[test]
847    fn repack_below_threshold_is_a_noop() {
848        let tmp = tempfile::tempdir().unwrap();
849        let log = OpLog::open(tmp.path()).unwrap();
850        log.put(&add_op()).unwrap();
851        let n = log.repack(10).unwrap();
852        assert_eq!(n, 0);
853    }
854
855    #[test]
856    fn repack_is_deterministic_on_same_input() {
857        // Two stores with the same loose ops repack to the same
858        // pack hash — content addressing all the way down.
859        let make_log = || {
860            let tmp = tempfile::tempdir().unwrap();
861            let log = OpLog::open(tmp.path()).unwrap();
862            let a = add_op();
863            log.put(&a).unwrap();
864            let b = modify_op(&a.op_id, "fac::Int->Int", "abc123", "def456");
865            log.put(&b).unwrap();
866            log.repack(0).unwrap();
867            (tmp, log)
868        };
869        let (tmp1, _log1) = make_log();
870        let (tmp2, _log2) = make_log();
871        let pack_name = |dir: &std::path::Path| -> String {
872            fs::read_dir(dir.join("ops")).unwrap()
873                .filter_map(|e| e.ok())
874                .find(|e| e.path().extension().is_some_and(|x| x == "pack"))
875                .unwrap()
876                .file_name().into_string().unwrap()
877        };
878        assert_eq!(pack_name(tmp1.path()), pack_name(tmp2.path()));
879    }
880
881    #[test]
882    fn walk_back_works_across_loose_and_packed_ops() {
883        // Pack the older history, leave newer ops loose. walk_back
884        // must traverse seamlessly.
885        let tmp = tempfile::tempdir().unwrap();
886        let log = OpLog::open(tmp.path()).unwrap();
887        let a = add_op();
888        log.put(&a).unwrap();
889        let b = modify_op(&a.op_id, "fac::Int->Int", "abc123", "b1");
890        log.put(&b).unwrap();
891        log.repack(0).unwrap();
892        // Now add a newer op as a loose file.
893        let c = modify_op(&b.op_id, "fac::Int->Int", "b1", "c1");
894        log.put(&c).unwrap();
895
896        let walked = log.walk_back(&c.op_id, None).unwrap();
897        let ids: Vec<_> = walked.iter().map(|r| r.op_id.as_str()).collect();
898        assert_eq!(ids, vec![c.op_id.as_str(), b.op_id.as_str(), a.op_id.as_str()]);
899    }
900
901    #[test]
902    fn list_all_dedups_across_loose_and_pack() {
903        let tmp = tempfile::tempdir().unwrap();
904        let log = OpLog::open(tmp.path()).unwrap();
905        let a = add_op();
906        log.put(&a).unwrap();
907        log.repack(0).unwrap();
908        // Re-put the same op as a loose file (simulate an
909        // interrupted repack). list_all should still report
910        // exactly one record per op_id.
911        log.put(&a).unwrap();
912
913        let all = log.list_all().unwrap();
914        assert_eq!(all.len(), 1);
915        assert_eq!(all[0].op_id, a.op_id);
916    }
917
918    #[test]
919    fn walk_back_orders_ancestors_after_descendants() {
920        // Build a small DAG with a merge:
921        //
922        //     a
923        //    / \
924        //   b   c
925        //    \ /
926        //     m  (merge with parents [b, c])
927        //
928        // The merge engine relies on the property that any ancestor of
929        // X appears strictly after X in the walk_back output. Pin it.
930        let tmp = tempfile::tempdir().unwrap();
931        let log = OpLog::open(tmp.path()).unwrap();
932        let a = add_op();
933        log.put(&a).unwrap();
934        let b = modify_op(&a.op_id, "fac::Int->Int", "abc123", "b1");
935        log.put(&b).unwrap();
936        let c = OperationRecord::new(
937            Operation::new(
938                OperationKind::ModifyBody {
939                    sig_id: "double::Int->Int".into(),
940                    from_stage_id: "ddd000".into(),
941                    to_stage_id: "c1".into(),
942                    from_budget: None,
943                    to_budget: None,
944                    to_sig_id: None,
945                },
946                [a.op_id.clone()],
947            ),
948            StageTransition::Replace {
949                sig_id: "double::Int->Int".into(),
950                from: "ddd000".into(),
951                to: "c1".into(),
952            },
953        );
954        log.put(&c).unwrap();
955        let m = OperationRecord::new(
956            Operation::new(
957                OperationKind::Merge { resolved: 0 },
958                [b.op_id.clone(), c.op_id.clone()],
959            ),
960            StageTransition::Merge { entries: BTreeMap::new() },
961        );
962        log.put(&m).unwrap();
963
964        let walked = log.walk_back(&m.op_id, None).unwrap();
965        let pos = |id: &str| walked.iter().position(|r| r.op_id == id).unwrap();
966        let (m_pos, b_pos, c_pos, a_pos) =
967            (pos(&m.op_id), pos(&b.op_id), pos(&c.op_id), pos(&a.op_id));
968        // Each ancestor must appear strictly after its descendants.
969        assert!(m_pos < b_pos, "merge before its parent b");
970        assert!(m_pos < c_pos, "merge before its parent c");
971        assert!(b_pos < a_pos, "b before its parent a");
972        assert!(c_pos < a_pos, "c before its parent a");
973    }
974}