Skip to main content

vtcode_memory/
event_log.rs

1//! Append-only per-session `ThreadEvent` log plus index and manifest.
2
3use std::collections::{BTreeMap, HashMap, VecDeque};
4use std::fs::File;
5use std::io::{BufRead, Read, Seek, SeekFrom, Write};
6use std::path::{Path, PathBuf};
7use std::sync::atomic::{AtomicBool, Ordering};
8use std::sync::{Arc, Mutex, OnceLock, Weak};
9
10use chrono::Utc;
11use serde::{Deserialize, Serialize};
12use vtcode_commons::VtCodePaths;
13use vtcode_exec_events::{EVENT_SCHEMA_VERSION, ThreadEvent, ThreadItemDetails, VersionedThreadEvent};
14
15use crate::error::SessionStoreError;
16
17/// Sidecar lock file flock-held while a session's event-log handles are open.
18pub(crate) const SESSION_LOCK_FILE: &str = "session.lock";
19use crate::manifest::{ManifestStore, PendingCapRewrite};
20use crate::session_dir;
21
22/// Default maximum number of events retained per session before the oldest
23/// completed turns are evicted.
24pub const DEFAULT_MAX_EVENTS: usize = 10_000;
25
26/// Maximum serialized event bytes retained before an append forces a write.
27/// Turn boundaries and reads still flush immediately.
28const MAX_WRITE_BUFFER_BYTES: usize = 64 * 1024;
29const MAX_EVICTION_GROUNDED_FACTS: usize = 32;
30const MAX_EVICTION_GROUNDED_FACT_BYTES: usize = 512;
31
32/// Callback used to persist a summary of events before they are evicted.
33///
34/// The callback runs after the event bytes have been flushed and decoded, but
35/// before the canonical log is rewritten. A failure leaves the original log
36/// and in-memory index untouched, so retention never silently discards
37/// history.
38pub type EvictionSummaryHook = Arc<dyn Fn(&[ThreadEvent]) -> Result<(), SessionStoreError> + Send + Sync>;
39
40/// Minimal envelope used while rebuilding the turn index.
41///
42/// The index only needs the event discriminator. Deserializing a complete
43/// [`VersionedThreadEvent`] here would allocate every nested tool argument,
44/// output, and thread item even though none of that payload is retained.
45#[derive(Debug, Deserialize)]
46struct VersionedEventKind<'a> {
47    #[serde(rename = "schema_version", borrow)]
48    _schema_version: &'a str,
49    #[serde(borrow)]
50    event: EventKind<'a>,
51}
52
53#[derive(Debug, Deserialize)]
54struct EventKind<'a> {
55    #[serde(rename = "type", borrow)]
56    kind: &'a str,
57}
58
59/// Zero-clone serialization envelope for `ThreadEvent`.
60///
61/// Produces JSON byte-identical to `VersionedThreadEvent` but borrows the
62/// event by reference instead of cloning it. `append` is called for every
63/// runtime event, and `ThreadEvent` can carry large tool outputs / thread
64/// items — cloning just to feed `serde_json::to_string` was pure waste.
65#[derive(Serialize)]
66struct BorrowedVersionedEvent<'a> {
67    schema_version: &'a str,
68    event: &'a ThreadEvent,
69}
70
71/// Turn-lifecycle discriminator extracted from either a `ThreadEvent` (at
72/// append time) or a raw `&str` kind (during scan).  This is the single
73/// representation that both code paths feed into
74/// [`LogState::apply_lifecycle_event`], eliminating a duplicated state machine.
75#[derive(Debug, Clone, Copy, PartialEq, Eq)]
76enum LifecycleKind {
77    ThreadStarted,
78    ThreadCompleted,
79    TurnStarted,
80    TurnCompleted,
81    TurnFailed,
82    Other,
83}
84
85impl LifecycleKind {
86    /// Discriminate from a runtime `ThreadEvent` at append time.
87    #[inline]
88    fn from_event(event: &ThreadEvent) -> Self {
89        match event {
90            ThreadEvent::ThreadStarted(_) => Self::ThreadStarted,
91            ThreadEvent::ThreadCompleted(_) => Self::ThreadCompleted,
92            ThreadEvent::TurnStarted(_) => Self::TurnStarted,
93            ThreadEvent::TurnCompleted(_) => Self::TurnCompleted,
94            ThreadEvent::TurnFailed(_) => Self::TurnFailed,
95            _ => Self::Other,
96        }
97    }
98
99    /// Discriminate from a raw event-type string at scan time.
100    #[inline]
101    fn from_kind(kind: &str) -> Self {
102        match kind {
103            "thread.started" => Self::ThreadStarted,
104            "thread.completed" => Self::ThreadCompleted,
105            "turn.started" => Self::TurnStarted,
106            "turn.completed" => Self::TurnCompleted,
107            "turn.failed" => Self::TurnFailed,
108            _ => Self::Other,
109        }
110    }
111}
112
113/// In-memory state protected by a mutex (cheap; appends are infrequent relative
114/// to model inference).
115struct LogState {
116    manifest: SessionManifest,
117    index: TurnIndex,
118    /// Whether we are currently inside a turn (between TurnStarted and
119    /// TurnCompleted/TurnFailed). Used to update the last index entry's
120    /// offsets as intermediate events arrive.
121    in_turn: bool,
122    /// Running byte offset of the next append. Avoids a `stat` syscall per
123    /// event (the previous implementation re-statted the file twice on every
124    /// `append`); initialized from the file length on `open`.
125    next_offset: u64,
126    /// Buffered pending writes to batch syscalls. Events are appended here
127    /// and flushed to disk at turn boundaries or before read operations.
128    write_buf: Vec<u8>,
129}
130
131#[derive(Debug, Clone, Copy)]
132struct CapEvictionPlan {
133    truncate_offset: u64,
134    evicted_event_count: u64,
135    evicted_turn_count: usize,
136}
137
138impl LogState {
139    fn new(session_id: &str) -> Self {
140        Self {
141            manifest: SessionManifest::new(session_id),
142            index: TurnIndex::default(),
143            in_turn: false,
144            next_offset: 0,
145            write_buf: Vec::with_capacity(65536),
146        }
147    }
148
149    /// Serialize `event` directly into the reusable write buffer with rollback
150    /// on failure.
151    ///
152    /// This encapsulates the invariant that `write_buf` never contains a
153    /// partial JSON document: if `serde_json::to_writer` fails mid-write the
154    /// buffer is truncated back to its pre-serialization boundary.  Returns
155    /// the `(start, end)` byte offsets of the serialized event so the caller
156    /// can feed them to [`Self::apply_lifecycle_event`].
157    fn serialize_event(&mut self, event: &ThreadEvent) -> Result<(u64, u64), SessionStoreError> {
158        let start = self.next_offset;
159        let buf_len_before = self.write_buf.len();
160        if let Err(err) = serde_json::to_writer(
161            &mut self.write_buf,
162            &BorrowedVersionedEvent { schema_version: EVENT_SCHEMA_VERSION, event },
163        ) {
164            self.write_buf.truncate(buf_len_before);
165            return Err(err.into());
166        }
167        self.write_buf.push(b'\n');
168        let written = self.write_buf.len() - buf_len_before;
169        let end = start + written as u64;
170        self.next_offset = end;
171        Ok((start, end))
172    }
173
174    /// Update the in-memory turn index and manifest counters for a single
175    /// event.
176    ///
177    /// This is the single implementation of the turn-lifecycle state machine;
178    /// both the append path (via [`LifecycleKind::from_event`]) and the scan
179    /// path (via [`LifecycleKind::from_kind`]) route through here, eliminating
180    /// a previously duplicated match block.
181    ///
182    /// Returns `true` when the event closes a turn boundary
183    /// (`TurnCompleted` / `TurnFailed`) so the caller can persist metadata
184    /// at the appropriate time (append persists immediately; scan persists
185    /// once after the full scan).
186    fn apply_lifecycle_event(&mut self, kind: LifecycleKind, start: u64, end: u64) -> bool {
187        let is_boundary = match kind {
188            LifecycleKind::ThreadStarted => {
189                self.manifest.status = "active".to_string();
190                false
191            }
192            LifecycleKind::ThreadCompleted => {
193                self.manifest.status = "completed".to_string();
194                true
195            }
196            LifecycleKind::TurnStarted => {
197                self.manifest.status = "active".to_string();
198                self.in_turn = true;
199                let n = self.manifest.turn_count + 1;
200                self.index.entries.push_back(TurnIndexEntry {
201                    turn_number: n,
202                    start_offset: start,
203                    end_offset: end,
204                    event_count: 1,
205                    ts: now_rfc3339(),
206                });
207                false
208            }
209            LifecycleKind::TurnCompleted | LifecycleKind::TurnFailed => {
210                if self.in_turn {
211                    if let Some(entry) = self.index.entries.back_mut() {
212                        entry.end_offset = end;
213                        entry.event_count += 1;
214                        // `turn_number` remains monotonic when older completed
215                        // turns have been evicted. The manifest is the source
216                        // for the next ordinal, so never replace it with the
217                        // retained index length.
218                        self.manifest.turn_count = self.manifest.turn_count.max(entry.turn_number);
219                    }
220                    self.in_turn = false;
221                }
222                true
223            }
224            LifecycleKind::Other => {
225                if self.in_turn
226                    && let Some(entry) = self.index.entries.back_mut()
227                {
228                    entry.end_offset = end;
229                    entry.event_count += 1;
230                }
231                false
232            }
233        };
234        // Persist the open-turn marker alongside the manifest. The marker is
235        // deliberately optional for backwards compatibility: an older
236        // manifest without it forces a scan on reopen so the state can be
237        // reconstructed from the canonical event log.
238        self.manifest.in_turn = Some(self.in_turn);
239        is_boundary
240    }
241
242    /// Plan a cap-enforcement eviction: pop the oldest completed turns from
243    /// the index until `event_count` is within `max_events`.
244    ///
245    /// Returns the byte offset at which the file should be rewritten and the
246    /// counts needed to apply the eviction after successful I/O. Returns
247    /// `None` when no eviction is needed.
248    fn plan_cap_eviction(&self, max_events: usize) -> Option<CapEvictionPlan> {
249        if max_events == 0 || self.manifest.event_count <= max_events as u64 {
250            return None;
251        }
252        let mut evicted_event_count = 0u64;
253        let mut truncate_offset = 0u64;
254        let mut evicted_turn_count = 0;
255        for (entry_index, oldest) in self.index.entries.iter().enumerate() {
256            // Never evict the active turn. Its entry has a provisional end
257            // offset and will be closed by a later completion/failure event;
258            // removing it here would make that event unindexed and lose the
259            // in-flight turn from reconstruction. If the completed history
260            // alone cannot bring the log under the cap, retain the active
261            // turn until it reaches a terminal boundary.
262            if self.in_turn && entry_index + 1 == self.index.entries.len() {
263                break;
264            }
265            if self.manifest.event_count.saturating_sub(evicted_event_count) <= max_events as u64 {
266                break;
267            }
268            truncate_offset = oldest.end_offset;
269            evicted_event_count += oldest.event_count;
270            evicted_turn_count += 1;
271        }
272        if truncate_offset == 0 || evicted_turn_count == 0 {
273            None
274        } else {
275            Some(CapEvictionPlan {
276                truncate_offset,
277                evicted_event_count,
278                evicted_turn_count,
279            })
280        }
281    }
282
283    fn apply_cap_eviction(&mut self, plan: CapEvictionPlan, next_offset: u64) {
284        for _ in 0..plan.evicted_turn_count {
285            let _ = self.index.entries.pop_front();
286        }
287        for entry in &mut self.index.entries {
288            entry.start_offset = entry.start_offset.saturating_sub(plan.truncate_offset);
289            entry.end_offset = entry.end_offset.saturating_sub(plan.truncate_offset);
290        }
291        self.next_offset = next_offset;
292        self.manifest.event_count = self.manifest.event_count.saturating_sub(plan.evicted_event_count);
293        self.manifest.retained_turn_base = Some(self.retained_turn_base_after_eviction(0));
294    }
295
296    fn retained_turn_base_after_eviction(&self, evicted_turn_count: usize) -> u64 {
297        self.index
298            .entries
299            .get(evicted_turn_count)
300            .map(|entry| entry.turn_number)
301            .or_else(|| self.manifest.turn_count.checked_add(1))
302            .unwrap_or(1)
303            .max(1)
304    }
305}
306
307/// State and file handle shared by every owner of one session in this process.
308///
309/// A keyed operation lock alone is insufficient: two independently opened
310/// handles could still carry stale turn counters and overwrite each other's
311/// metadata after taking the lock. Sharing the mutable state and append file
312/// makes the lock a true session boundary while preserving the value-type
313/// `SessionEventLog` API.
314struct SessionShared {
315    file: Mutex<Option<File>>,
316    state: Mutex<LogState>,
317    eviction_lock: Mutex<()>,
318    initialized: AtomicBool,
319    /// Exclusive flock on the session's `session.lock`, held for as long as
320    /// any handle to this session exists. Retention reads it as a liveness
321    /// signal so an open-but-idle session is never marked or evicted.
322    liveness_lock: Option<File>,
323}
324
325/// Return the process-wide shared state for one session's canonical event file.
326///
327/// The weak registry avoids retaining closed sessions forever while still
328/// making repeated `open` calls converge on one file handle and turn state.
329fn shared_session(events_path: &Path, session_id: &str) -> Result<Arc<SessionShared>, SessionStoreError> {
330    static SESSION_SHARED: OnceLock<Mutex<HashMap<PathBuf, Weak<SessionShared>>>> = OnceLock::new();
331
332    let key = events_path
333        .parent()
334        .and_then(|parent| vtcode_commons::paths::canonicalize(parent).ok())
335        .and_then(|parent| events_path.file_name().map(|name| parent.join(name)))
336        .unwrap_or_else(|| events_path.to_path_buf());
337    let registry = SESSION_SHARED.get_or_init(|| Mutex::new(HashMap::new()));
338    let mut shared_by_path = match registry.lock() {
339        Ok(locks) => locks,
340        Err(poisoned) => poisoned.into_inner(),
341    };
342    // Do not retain dead weak entries for every session ever opened by a
343    // long-running process. The registry is only an in-process coordination
344    // aid, so removing entries with no live owners is safe.
345    shared_by_path.retain(|_, shared| shared.strong_count() > 0);
346    if let Some(shared) = shared_by_path.get(&key).and_then(Weak::upgrade) {
347        return Ok(shared);
348    }
349
350    let file = VtCodePaths::open_private_append_file(events_path)
351        .map_err(|error| SessionStoreError::io(events_path.to_path_buf(), std::io::Error::other(error)))?;
352    let shared = Arc::new(SessionShared {
353        file: Mutex::new(Some(file)),
354        state: Mutex::new(LogState::new(session_id)),
355        eviction_lock: Mutex::new(()),
356        initialized: AtomicBool::new(false),
357        liveness_lock: acquire_liveness_lock(events_path),
358    });
359    shared_by_path.insert(key, Arc::downgrade(&shared));
360    Ok(shared)
361}
362
363/// Best-effort exclusive flock on the session's `session.lock`.
364///
365/// The lock is the liveness signal retention reads (`session_dir_is_live`): a
366/// held lock means a live process still has the session open. The crate has
367/// no logging surface and this is an advisory signal, not persistence, so a
368/// failure to create/open/lock degrades to unlocked (previous retention
369/// behavior) instead of failing the session open or swallowing a persistence
370/// error. The fd is released automatically when the shared handle drops.
371fn acquire_liveness_lock(events_path: &Path) -> Option<File> {
372    let dir = events_path.parent()?;
373    let lock_path = dir.join(SESSION_LOCK_FILE);
374    VtCodePaths::write_private_file_atomic_if_absent(&lock_path, b"").ok()?;
375    let file = std::fs::OpenOptions::new().read(true).write(true).open(&lock_path).ok()?;
376    match file.try_lock() {
377        Ok(()) => Some(file),
378        Err(_) => None,
379    }
380}
381
382/// Canonical append-only event log for a single session.
383///
384/// All session history is reconstructable from this log. Live conversation
385/// state is never read back into context from here; the log is only consumed
386/// for revert, compaction, analytics, and long-term-learning queries.
387pub struct SessionEventLog {
388    events_path: PathBuf,
389    manifest_store: ManifestStore,
390    shared: Arc<SessionShared>,
391    max_events: usize,
392    eviction_summary_hook: EvictionSummaryHook,
393}
394
395impl SessionEventLog {
396    /// Open the log for `session_id`, creating the session directory tree and
397    /// rebuilding the index from `events.jsonl` if it already exists.
398    pub(crate) fn open(workspace: &Path, session_id: &str, max_events: usize) -> Result<Self, SessionStoreError> {
399        let dir = session_dir(workspace, session_id);
400        let hook = default_eviction_summary_hook(dir.join(crate::DERIVED_DIR), session_id.to_string());
401        Self::open_with_eviction_summary(workspace, session_id, max_events, hook)
402    }
403
404    /// Open a log with an explicit eviction-summary callback.
405    ///
406    /// This is useful for hosts that keep derived memory in another store and
407    /// for deterministic failure-path tests. The callback must persist its
408    /// summary before returning `Ok(())`.
409    pub fn open_with_eviction_summary(
410        workspace: &Path,
411        session_id: &str,
412        max_events: usize,
413        eviction_summary_hook: EvictionSummaryHook,
414    ) -> Result<Self, SessionStoreError> {
415        let dir = session_dir(workspace, session_id);
416        crate::ensure_private_directory(&crate::sessions_root(workspace))?;
417        crate::ensure_private_directory(&dir)?;
418        crate::ensure_private_directory(&dir.join(crate::DERIVED_DIR))?;
419        crate::ensure_private_directory(&dir.join("index"))?;
420        let events_path = dir.join("events.jsonl");
421        let manifest_store = ManifestStore::new(dir.clone());
422        let pending_rewrite = manifest_store.load_pending_cap_rewrite()?;
423        let shared = shared_session(&events_path, session_id)?;
424        let log = Self {
425            events_path: events_path.clone(),
426            manifest_store,
427            shared,
428            max_events,
429            eviction_summary_hook,
430        };
431        let _eviction_guard = log.shared.eviction_lock.lock().map_err(poison)?;
432        // Try the fast path: read the persisted manifest + index and skip
433        // the O(n) scan when they are present and consistent.
434        if !log.shared.initialized.load(Ordering::Acquire) {
435            let manifest_opt = log.manifest_store.load_manifest()?;
436            let index_opt = log.manifest_store.load_turn_index()?;
437            let file_len = log.event_file_metadata_len()?;
438            let pending_rewrite_matches_file = pending_rewrite.as_ref().is_some_and(|pending| {
439                pending.new_file_len == file_len && pending.new_file_len < pending.previous_file_len
440            });
441            match (&manifest_opt, &index_opt) {
442                (Some(manifest), Some(index))
443                    if !pending_rewrite_matches_file
444                        && manifest.in_turn.is_some()
445                        && manifest.persisted_file_len == Some(file_len)
446                        && index.is_valid_for_file(file_len)
447                        && index.is_consistent_with_manifest(manifest) =>
448                {
449                    let mut st = log.shared.state.lock().map_err(poison)?;
450                    st.in_turn = manifest.in_turn.unwrap_or(false);
451                    st.manifest = manifest.clone();
452                    st.index = index.clone();
453                    st.next_offset = file_len;
454                }
455                _ => {
456                    let scan_turn_base = infer_scan_turn_base(
457                        manifest_opt.as_ref(),
458                        index_opt.as_ref(),
459                        pending_rewrite.as_ref().filter(|_| pending_rewrite_matches_file),
460                        file_len,
461                    );
462                    {
463                        let mut st = log.shared.state.lock().map_err(poison)?;
464                        if let Some(previous) = manifest_opt.as_ref() {
465                            st.manifest = previous.clone();
466                        }
467                        // The canonical event file is authoritative after any
468                        // stale/corrupt metadata. Preserve the ordinal of the
469                        // first retained turn while rebuilding all counters.
470                        st.manifest.turn_count = scan_turn_base.saturating_sub(1);
471                        st.manifest.event_count = 0;
472                        st.manifest.status = "active".to_string();
473                        st.manifest.in_turn = Some(false);
474                        st.manifest.retained_turn_base = Some(scan_turn_base);
475                        st.index = TurnIndex::default();
476                        st.in_turn = false;
477                    }
478                    log.scan()?;
479                    let mut st = log.shared.state.lock().map_err(poison)?;
480                    st.next_offset = file_len;
481                    log.persist_meta_locked(&mut st)?;
482                }
483            }
484            if pending_rewrite.is_some() {
485                // A marker whose file length did not match either side of the
486                // rewrite is stale, while a matching marker has now been
487                // incorporated into the repaired metadata. In both cases it
488                // is safe to remove it after the open path has persisted the
489                // authoritative state.
490                log.manifest_store.clear_pending_cap_rewrite()?;
491            }
492            log.shared.initialized.store(true, Ordering::Release);
493        }
494        drop(_eviction_guard);
495        Ok(log)
496    }
497
498    /// Append an event to the log and update the in-memory index/manifest.
499    pub fn append(&self, event: &ThreadEvent) -> Result<(), SessionStoreError> {
500        let _eviction_guard = self.shared.eviction_lock.lock().map_err(poison)?;
501        let mut st = self.shared.state.lock().map_err(poison)?;
502
503        // Serialize into the write buffer with rollback on failure — the
504        // invariant that `write_buf` never contains partial JSON is
505        // encapsulated in `serialize_event`.
506        let (start, end) = st.serialize_event(event)?;
507
508        st.manifest.event_count += 1;
509        st.manifest.updated_at = now_rfc3339();
510
511        // Route through the single turn-lifecycle state machine.  When the
512        // event closes a turn, persist metadata immediately so a reopen
513        // after a mid-turn crash sees a consistent index.
514        let is_turn_boundary = st.apply_lifecycle_event(LifecycleKind::from_event(event), start, end);
515        if is_turn_boundary {
516            self.persist_meta_locked(&mut st)?;
517        }
518
519        if st.write_buf.len() >= MAX_WRITE_BUFFER_BYTES {
520            // Persist metadata with the bounded byte flush so a reopen after
521            // a mid-turn crash does not trust an index that predates these
522            // already-written events.
523            self.persist_meta_locked(&mut st)?;
524        }
525        drop(st);
526        self.enforce_event_cap()
527    }
528
529    /// Enforce the per-session event cap by evicting the oldest completed
530    /// turns when the log exceeds [`Self::max_events`]. Returns `Ok(())` even
531    /// when no truncation is needed or the cap is disabled (`max_events == 0`).
532    fn enforce_event_cap(&self) -> Result<(), SessionStoreError> {
533        let mut st = self.shared.state.lock().map_err(poison)?;
534
535        // `plan_cap_eviction` encapsulates the index arithmetic and returns
536        // `None` when the cap is disabled or not yet exceeded.
537        let Some(plan) = st.plan_cap_eviction(self.max_events) else {
538            return Ok(());
539        };
540
541        // Keep ordinary appends in memory until a turn boundary or an
542        // explicit read. Cap enforcement is the one append-time path that
543        // needs the complete on-disk file before rewriting it.
544        self.flush_write_buf_locked(&mut st)?;
545
546        let (evicted, remaining, previous_file_len) = {
547            let mut file_slot = self.shared.file.lock().map_err(poison)?;
548            let file = file_slot.as_mut().ok_or_else(|| self.event_file_unavailable())?;
549            let file_len = file
550                .metadata()
551                .map_err(|error| SessionStoreError::io(&self.events_path, error))?
552                .len();
553            if plan.truncate_offset > file_len {
554                return Err(SessionStoreError::io(
555                    &self.events_path,
556                    std::io::Error::new(std::io::ErrorKind::InvalidData, "cap offset exceeds event log length"),
557                ));
558            }
559            file.seek(SeekFrom::Start(0))
560                .map_err(|e| SessionStoreError::io(&self.events_path, e))?;
561            let mut evicted = vec![
562                0u8;
563                usize::try_from(plan.truncate_offset).map_err(|error| {
564                    SessionStoreError::io(
565                        &self.events_path,
566                        std::io::Error::new(std::io::ErrorKind::InvalidData, error),
567                    )
568                })?
569            ];
570            file.read_exact(&mut evicted)
571                .map_err(|e| SessionStoreError::io(&self.events_path, e))?;
572            file.seek(SeekFrom::Start(plan.truncate_offset))
573                .map_err(|e| SessionStoreError::io(&self.events_path, e))?;
574            let mut remaining = Vec::new();
575            file.read_to_end(&mut remaining)
576                .map_err(|e| SessionStoreError::io(&self.events_path, e))?;
577            (evicted, remaining, file_len)
578        };
579        let retained_turn_base = st.retained_turn_base_after_eviction(plan.evicted_turn_count);
580        drop(st);
581
582        // The turn index counts only records inside indexed turns. The bytes
583        // removed by a cap rewrite may also contain valid session-level
584        // records (for example `thread.started`) before the first turn, so
585        // reconcile the manifest against the actual persisted prefix rather
586        // than the turn-only estimate from `plan_cap_eviction`.
587        let mut plan = plan;
588        plan.evicted_event_count = count_persisted_event_records(&evicted);
589        let evicted_events = decode_events(&evicted);
590        (self.eviction_summary_hook)(&evicted_events)?;
591
592        let new_file_len = u64::try_from(remaining.len()).map_err(|error| {
593            SessionStoreError::io(&self.events_path, std::io::Error::new(std::io::ErrorKind::InvalidData, error))
594        })?;
595        self.manifest_store.write_pending_cap_rewrite(&PendingCapRewrite {
596            previous_file_len,
597            new_file_len,
598            retained_turn_base,
599        })?;
600        let next_offset = self.replace_event_file_contents(&remaining)?;
601        let mut st = self.shared.state.lock().map_err(poison)?;
602        st.apply_cap_eviction(plan, next_offset);
603        // The rewrite changed byte offsets and retained counts; persist the
604        // derived metadata before exposing the append as successful.
605        self.persist_meta_locked(&mut st)?;
606        self.manifest_store.clear_pending_cap_rewrite()?;
607        Ok(())
608    }
609
610    /// Reconstruct every event belonging to `turn`.
611    pub(crate) fn reconstruct_turn(&self, turn: u64) -> Result<Vec<ThreadEvent>, SessionStoreError> {
612        // Keep the index snapshot and byte-range read together with cap
613        // rewriting. Otherwise an eviction can replace the file between these
614        // steps and leave the snapshot offsets pointing into unrelated events.
615        let _eviction_guard = self.shared.eviction_lock.lock().map_err(poison)?;
616        let entry = {
617            let st = self.shared.state.lock().map_err(poison)?;
618            st.index
619                .entries
620                .iter()
621                .find(|e| e.turn_number == turn)
622                .cloned()
623                .ok_or(SessionStoreError::TurnNotFound { session: st.manifest.session_id.clone(), turn })?
624        };
625        {
626            let mut st = self.shared.state.lock().map_err(poison)?;
627            self.flush_write_buf_locked(&mut st)?;
628        }
629        let buf = {
630            let mut file_slot = self.shared.file.lock().map_err(poison)?;
631            let file = file_slot.as_mut().ok_or_else(|| self.event_file_unavailable())?;
632            file.seek(SeekFrom::Start(entry.start_offset))
633                .map_err(|e| SessionStoreError::io(&self.events_path, e))?;
634            let len = usize::try_from(entry.end_offset.checked_sub(entry.start_offset).ok_or_else(|| {
635                SessionStoreError::io(
636                    &self.events_path,
637                    std::io::Error::new(std::io::ErrorKind::InvalidData, "turn index offsets are out of order"),
638                )
639            })?)
640            .map_err(|error| {
641                SessionStoreError::io(&self.events_path, std::io::Error::new(std::io::ErrorKind::InvalidData, error))
642            })?;
643            let mut buf = vec![0u8; len];
644            file.read_exact(&mut buf)
645                .map_err(|e| SessionStoreError::io(&self.events_path, e))?;
646            buf
647        };
648        let text = String::from_utf8_lossy(&buf);
649        let mut events = Vec::new();
650        for line in text.lines() {
651            let line = line.trim();
652            if line.is_empty() {
653                continue;
654            }
655            // The index scan only validates the event envelope (plus the
656            // lifecycle shape) so it can rebuild cheaply. A line accepted by
657            // the scan can therefore still fail full decoding here; skip it
658            // instead of failing the whole reconstruction (revert, compaction,
659            // and analytics must not break on a single malformed record).
660            let v: VersionedThreadEvent = match serde_json::from_str(line) {
661                Ok(v) => v,
662                Err(_) => continue,
663            };
664            events.push(v.into_event());
665        }
666        Ok(events)
667    }
668
669    /// Number of turns recorded.
670    #[must_use]
671    pub(crate) fn turn_count(&self) -> u64 {
672        self.shared.state.lock().map_err(poison).map_or(0, |s| s.manifest.turn_count)
673    }
674
675    /// Number of events recorded.
676    #[must_use]
677    pub fn event_count(&self) -> u64 {
678        self.shared.state.lock().map_err(poison).map_or(0, |s| s.manifest.event_count)
679    }
680
681    /// Flush pending event bytes and metadata to the session store.
682    pub fn flush(&self) -> Result<(), SessionStoreError> {
683        let _eviction_guard = self.shared.eviction_lock.lock().map_err(poison)?;
684        let mut st = self.shared.state.lock().map_err(poison)?;
685        self.persist_meta_locked(&mut st)
686    }
687
688    /// Snapshot of the session manifest.
689    #[must_use]
690    pub fn manifest(&self) -> SessionManifest {
691        self.shared
692            .state
693            .lock()
694            .map_err(poison)
695            .map(|s| s.manifest.clone())
696            .unwrap_or_else(|_| SessionManifest::new(""))
697    }
698
699    /// Snapshot of the turn index.
700    #[must_use]
701    pub fn turn_index(&self) -> TurnIndex {
702        self.shared
703            .state
704            .lock()
705            .map_err(poison)
706            .map(|s| s.index.clone())
707            .unwrap_or_default()
708    }
709
710    /// Flush metadata for callers that explicitly close a log handle.
711    ///
712    /// Terminal status is intentionally controlled only by a persisted
713    /// `thread.completed` event. This method does not synthesize lifecycle
714    /// state for callers that merely release a store handle.
715    pub(crate) fn complete(&self) -> Result<(), SessionStoreError> {
716        let _eviction_guard = self.shared.eviction_lock.lock().map_err(poison)?;
717        let mut st = self.shared.state.lock().map_err(poison)?;
718        st.manifest.updated_at = now_rfc3339();
719        self.persist_meta_locked(&mut st)
720    }
721
722    /// Rebuild index + manifest by scanning `events.jsonl` (authoritative).
723    ///
724    /// Reads the file line-by-line via `BufReader` to avoid loading the entire
725    /// log into memory. Long-lived sessions can otherwise produce multi-megabyte
726    /// logs that spike memory on every reopen.
727    fn scan(&self) -> Result<(), SessionStoreError> {
728        let mut st = self.shared.state.lock().map_err(poison)?;
729        let file = self
730            .shared
731            .file
732            .lock()
733            .map_err(poison)?
734            .as_ref()
735            .ok_or_else(|| self.event_file_unavailable())?
736            .try_clone()
737            .map_err(|e| SessionStoreError::io(&self.events_path, e))?;
738        let mut reader = std::io::BufReader::new(file);
739        reader
740            .seek(SeekFrom::Start(0))
741            .map_err(|e| SessionStoreError::io(&self.events_path, e))?;
742        let mut buf = Vec::new();
743        let mut pos = 0u64;
744        let mut first_ts: Option<String> = None;
745        loop {
746            buf.clear();
747            let n = reader
748                .read_until(b'\n', &mut buf)
749                .map_err(|e| SessionStoreError::io(&self.events_path, e))?;
750            if n == 0 {
751                break;
752            }
753            let line_end = pos + n as u64;
754            let trimmed = std::str::from_utf8(&buf).unwrap_or("").trim();
755            if !trimmed.is_empty()
756                && let Ok(v) = serde_json::from_str::<VersionedEventKind<'_>>(trimmed)
757            {
758                let kind = v.event.kind;
759                if requires_full_lifecycle_validation(kind) && !valid_lifecycle_payload(kind, trimmed) {
760                    pos = line_end;
761                    continue;
762                }
763                st.manifest.event_count += 1;
764                // `thread.started` is not part of the turn lifecycle — it
765                // only seeds `created_at` on the first occurrence.
766                if kind == "thread.started" && first_ts.is_none() {
767                    first_ts = Some(now_rfc3339());
768                }
769                // Route turn-lifecycle events through the same state machine
770                // as `append`, eliminating a previously duplicated match block.
771                st.apply_lifecycle_event(LifecycleKind::from_kind(kind), pos, line_end);
772            }
773            pos = line_end;
774        }
775        // Keep the open-turn state reconstructed from the canonical event log.
776        // This lets a reopened session continue a turn that was flushed before
777        // its completion event was written.
778        st.manifest.in_turn = Some(st.in_turn);
779        if let Some(ts) = first_ts
780            && st.manifest.created_at.is_empty()
781        {
782            st.manifest.created_at = ts;
783        }
784        Ok(())
785    }
786
787    fn persist_meta_locked(&self, st: &mut LogState) -> Result<(), SessionStoreError> {
788        self.flush_write_buf_locked(st)?;
789        // The manifest is only eligible for the fast reopen path when it
790        // describes the complete on-disk event file. Drop intentionally
791        // flushes bytes without metadata, so a length mismatch safely forces
792        // the authoritative scan on the next open.
793        st.manifest.persisted_file_len = Some(st.next_offset);
794        // Publish the derived index first. If a process stops between these
795        // two atomic renames, the older manifest still carries a stale file
796        // length and forces a scan instead of allowing the new manifest to
797        // pair with an older, apparently valid index.
798        self.manifest_store.write_turn_index(&st.index)?;
799        self.manifest_store.write_manifest(&st.manifest)?;
800        Ok(())
801    }
802
803    /// Flush the in-memory write buffer to the underlying file.
804    fn flush_write_buf_locked(&self, st: &mut LogState) -> Result<(), SessionStoreError> {
805        if st.write_buf.is_empty() {
806            return Ok(());
807        }
808        let mut file_slot = self.shared.file.lock().map_err(poison)?;
809        let file = file_slot.as_mut().ok_or_else(|| self.event_file_unavailable())?;
810        let previous_len = file.metadata().map_err(|e| SessionStoreError::io(&self.events_path, e))?.len();
811        if let Err(error) = file.write_all(&st.write_buf) {
812            if file.set_len(previous_len).is_err() {
813                st.write_buf.clear();
814            }
815            return Err(SessionStoreError::io(&self.events_path, error));
816        }
817        if let Err(error) = file.sync_data() {
818            st.write_buf.clear();
819            return Err(SessionStoreError::io(&self.events_path, error));
820        }
821        st.write_buf.clear();
822        Ok(())
823    }
824
825    fn event_file_metadata_len(&self) -> Result<u64, SessionStoreError> {
826        let file_slot = self.shared.file.lock().map_err(poison)?;
827        file_slot
828            .as_ref()
829            .ok_or_else(|| self.event_file_unavailable())?
830            .metadata()
831            .map(|metadata| metadata.len())
832            .map_err(|error| SessionStoreError::io(&self.events_path, error))
833    }
834
835    fn open_event_file(&self) -> Result<File, SessionStoreError> {
836        VtCodePaths::open_private_append_file(&self.events_path)
837            .map_err(|error| SessionStoreError::io(&self.events_path, std::io::Error::other(error)))
838    }
839
840    fn replace_event_file_contents(&self, contents: &[u8]) -> Result<u64, SessionStoreError> {
841        let old_file = {
842            let mut file_slot = self.shared.file.lock().map_err(poison)?;
843            file_slot.take().ok_or_else(|| self.event_file_unavailable())?
844        };
845        drop(old_file);
846
847        if let Err(error) = VtCodePaths::write_private_file_atomic(&self.events_path, contents)
848            .map_err(|error| SessionStoreError::io(&self.events_path, std::io::Error::other(error)))
849        {
850            let restored = self.open_event_file();
851            if let Ok(file) = restored {
852                let mut file_slot = self.shared.file.lock().map_err(poison)?;
853                *file_slot = Some(file);
854                return Err(error);
855            }
856            return Err(SessionStoreError::io(
857                &self.events_path,
858                std::io::Error::other(format!("{error}; failed to restore event log handle")),
859            ));
860        }
861
862        let replacement = self.open_event_file()?;
863        let next_offset = replacement
864            .metadata()
865            .map_err(|error| SessionStoreError::io(&self.events_path, error))?
866            .len();
867        let mut file_slot = self.shared.file.lock().map_err(poison)?;
868        *file_slot = Some(replacement);
869        Ok(next_offset)
870    }
871
872    fn event_file_unavailable(&self) -> SessionStoreError {
873        SessionStoreError::io(&self.events_path, std::io::Error::other("event log file is unavailable"))
874    }
875}
876
877/// Recover the ordinal of the first retained turn when metadata is stale.
878///
879/// A cap rewrite atomically replaces the event file before it publishes the
880/// shortened index and manifest. If the process crashes in that interval,
881/// the old index still records the pre-rewrite offsets. The difference between
882/// its persisted length and the current file length is exactly the removed
883/// prefix, so the first old index entry at that boundary supplies the retained
884/// turn base. Other stale metadata falls back to the durable base field.
885fn infer_scan_turn_base(
886    manifest: Option<&SessionManifest>,
887    index: Option<&TurnIndex>,
888    pending: Option<&PendingCapRewrite>,
889    file_len: u64,
890) -> u64 {
891    let marker_base = pending
892        .filter(|pending| pending.new_file_len == file_len && pending.new_file_len < pending.previous_file_len)
893        .map(|pending| pending.retained_turn_base);
894    let rewritten_base = manifest.and_then(|manifest| {
895        let previous_len = manifest.persisted_file_len?;
896        if previous_len <= file_len {
897            return None;
898        }
899        let removed_prefix = previous_len - file_len;
900        let index = index.filter(|index| index.is_valid_for_file(previous_len))?;
901        let first_retained = index
902            .entries
903            .iter()
904            .find(|entry| entry.start_offset >= removed_prefix && entry.end_offset <= previous_len)
905            .map(|entry| entry.turn_number);
906        first_retained.or_else(|| {
907            // If no indexed turn starts in the shortened file, the rewrite
908            // evicted every previously indexed turn. Preserve the next
909            // ordinal from the stale manifest so a subsequent append cannot
910            // reuse an already-observed turn number.
911            (removed_prefix >= previous_len).then(|| manifest.turn_count.saturating_add(1))
912        })
913    });
914
915    let legacy_index_base = index
916        .filter(|index| index.is_valid_for_file(file_len))
917        .and_then(|index| index.entries.front().map(|entry| entry.turn_number));
918
919    marker_base
920        .or(rewritten_base)
921        .or_else(|| manifest.and_then(|manifest| manifest.retained_turn_base))
922        .or(legacy_index_base)
923        .unwrap_or(1)
924        .max(1)
925}
926
927fn requires_full_lifecycle_validation(kind: &str) -> bool {
928    matches!(kind, "thread.started" | "thread.completed" | "turn.started" | "turn.completed" | "turn.failed")
929}
930
931fn valid_lifecycle_payload(kind: &str, line: &str) -> bool {
932    if serde_json::from_str::<VersionedThreadEvent>(line).is_err() {
933        return false;
934    }
935    if kind != "turn.completed" {
936        return true;
937    }
938    let Ok(value) = serde_json::from_str::<serde_json::Value>(line) else {
939        return false;
940    };
941    value
942        .get("event")
943        .and_then(|event| event.get("usage"))
944        .is_some_and(serde_json::Value::is_object)
945}
946
947fn decode_events(bytes: &[u8]) -> Vec<ThreadEvent> {
948    bytes
949        .split(|byte| *byte == b'\n')
950        .filter_map(|line| {
951            let line = std::str::from_utf8(line).ok()?.trim();
952            if line.is_empty() {
953                return None;
954            }
955            serde_json::from_str::<VersionedThreadEvent>(line)
956                .ok()
957                .map(VersionedThreadEvent::into_event)
958        })
959        .collect()
960}
961
962/// Count records that the authoritative scan would include in the manifest.
963/// This deliberately parses the lightweight event envelope instead of using
964/// the turn index: a cap rewrite can remove session-level records that never
965/// belong to an indexed turn.
966fn count_persisted_event_records(bytes: &[u8]) -> u64 {
967    bytes
968        .split(|byte| *byte == b'\n')
969        .filter_map(|line| std::str::from_utf8(line).ok())
970        .map(str::trim)
971        .filter(|line| !line.is_empty())
972        .filter(|line| {
973            let Ok(value) = serde_json::from_str::<VersionedEventKind<'_>>(line) else {
974                return false;
975            };
976            let kind = value.event.kind;
977            !requires_full_lifecycle_validation(kind) || valid_lifecycle_payload(kind, line)
978        })
979        .count() as u64
980}
981
982#[derive(Debug, Serialize)]
983struct EvictionSummary {
984    session_id: String,
985    evicted_event_count: usize,
986    event_types: BTreeMap<String, u64>,
987    grounded_facts: Vec<String>,
988    created_at: String,
989}
990
991/// Extract a bounded, deterministic set of facts from canonical event
992/// payloads. This is intentionally structural rather than model-generated:
993/// eviction must remain synchronous, reproducible, and safe when the model is
994/// unavailable. Only completed item snapshots and terminal thread errors are
995/// considered, so streaming deltas and raw tool output cannot flood the
996/// derived summary.
997fn extract_grounded_facts(events: &[ThreadEvent]) -> Vec<String> {
998    let mut facts = Vec::new();
999    for event in events {
1000        let candidates: Vec<String> = match event {
1001            ThreadEvent::ItemCompleted(completed) => item_facts(&completed.item.details),
1002            ThreadEvent::TurnFailed(failed) => vec![failed.message.clone()],
1003            ThreadEvent::TurnBlocked(blocked) => vec![blocked.message.clone()],
1004            ThreadEvent::Error(error) => vec![error.message.clone()],
1005            _ => Vec::new(),
1006        };
1007        for candidate in candidates {
1008            let fact = normalize_eviction_fact(&candidate);
1009            if fact.is_empty() || facts.iter().any(|existing| existing == &fact) {
1010                continue;
1011            }
1012            facts.push(fact);
1013            if facts.len() == MAX_EVICTION_GROUNDED_FACTS {
1014                return facts;
1015            }
1016        }
1017    }
1018    facts
1019}
1020
1021fn item_facts(details: &ThreadItemDetails) -> Vec<String> {
1022    match details {
1023        ThreadItemDetails::AgentMessage(item) => vec![item.text.clone()],
1024        ThreadItemDetails::Plan(item) => vec![item.text.clone()],
1025        ThreadItemDetails::FileChange(item) => item
1026            .changes
1027            .iter()
1028            .map(|change| {
1029                let kind = match change.kind {
1030                    vtcode_exec_events::PatchChangeKind::Add => "add",
1031                    vtcode_exec_events::PatchChangeKind::Delete => "delete",
1032                    vtcode_exec_events::PatchChangeKind::Update => "update",
1033                };
1034                format!("file {kind}: {}", change.path)
1035            })
1036            .collect(),
1037        ThreadItemDetails::Harness(item) => item.message.clone().into_iter().collect(),
1038        _ => Vec::new(),
1039    }
1040}
1041
1042fn normalize_eviction_fact(value: &str) -> String {
1043    let normalized = value.split_whitespace().collect::<Vec<_>>().join(" ");
1044    if normalized.len() <= MAX_EVICTION_GROUNDED_FACT_BYTES {
1045        return normalized;
1046    }
1047    let mut end = MAX_EVICTION_GROUNDED_FACT_BYTES - '…'.len_utf8();
1048    while !normalized.is_char_boundary(end) {
1049        end -= 1;
1050    }
1051    format!("{}…", &normalized[..end])
1052}
1053
1054fn default_eviction_summary_hook(derived_dir: PathBuf, session_id: String) -> EvictionSummaryHook {
1055    Arc::new(move |events| {
1056        let mut event_types = BTreeMap::new();
1057        for event in events {
1058            let kind = serde_json::to_value(event)
1059                .ok()
1060                .and_then(|value| value.get("type").and_then(|value| value.as_str()).map(str::to_owned))
1061                .unwrap_or_else(|| "unknown".to_owned());
1062            *event_types.entry(kind).or_insert(0) += 1;
1063        }
1064        let summary = EvictionSummary {
1065            session_id: session_id.clone(),
1066            evicted_event_count: events.len(),
1067            event_types,
1068            grounded_facts: extract_grounded_facts(events),
1069            created_at: now_rfc3339(),
1070        };
1071        let path = derived_dir.join(format!("eviction-summary-{}.json", uuid::Uuid::new_v4().simple()));
1072        let bytes = serde_json::to_vec(&summary)?;
1073        VtCodePaths::write_private_file_atomic(&path, &bytes)
1074            .map_err(|error| SessionStoreError::io(path, std::io::Error::other(error)))
1075    })
1076}
1077
1078impl Drop for SessionEventLog {
1079    fn drop(&mut self) {
1080        if let Ok(_eviction_guard) = self.shared.eviction_lock.lock()
1081            && let Ok(mut st) = self.shared.state.lock()
1082        {
1083            // The fallible `flush` method is the authoritative shutdown path;
1084            // Drop only provides a best-effort byte flush for callers that do
1085            // not explicitly close the log. Rewriting metadata here could
1086            // overwrite a manifest update made by another owner after the
1087            // last append.
1088            let _ = self.flush_write_buf_locked(&mut st);
1089        }
1090    }
1091}
1092
1093/// Locate the next newline at or after `from`, returning a past-the-end index.
1094fn poison<T>(_e: std::sync::PoisonError<T>) -> SessionStoreError {
1095    SessionStoreError::Io {
1096        path: PathBuf::new(),
1097        source: std::io::Error::other("session store lock poisoned"),
1098    }
1099}
1100
1101fn now_rfc3339() -> String {
1102    Utc::now().to_rfc3339()
1103}
1104
1105/// Session-level metadata persisted to `manifest.json`.
1106#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
1107pub struct SessionManifest {
1108    /// Stable session identifier (directory name).
1109    pub session_id: String,
1110    /// Layout schema version (`SESSION_STORE_SCHEMA_VERSION`).
1111    schema_version: u32,
1112    /// RFC3339 creation timestamp.
1113    pub created_at: String,
1114    /// RFC3339 last-update timestamp.
1115    pub updated_at: String,
1116    /// Number of completed turns.
1117    pub turn_count: u64,
1118    /// Total number of events recorded.
1119    pub event_count: u64,
1120    /// Lifecycle status (`active` | `completed`).
1121    pub status: String,
1122    /// Whether the canonical log currently ends inside an open turn.
1123    ///
1124    /// This is optional on read so manifests written before open-turn
1125    /// persistence was introduced trigger a safe event-log scan instead of
1126    /// silently losing lifecycle state.
1127    #[serde(default)]
1128    in_turn: Option<bool>,
1129    /// Byte length covered by the persisted manifest and turn index.
1130    ///
1131    /// This is optional for compatibility with manifests written before the
1132    /// fast-path freshness guard existed; those manifests are rebuilt from the
1133    /// canonical event log on reopen.
1134    #[serde(default)]
1135    persisted_file_len: Option<u64>,
1136    /// Ordinal of the first turn retained in the canonical log.
1137    ///
1138    /// Cap eviction removes completed turns but must keep later turn numbers
1139    /// monotonic. The field lets an authoritative scan restore those ordinals
1140    /// even when the derived index is stale or missing.
1141    #[serde(default)]
1142    retained_turn_base: Option<u64>,
1143}
1144
1145impl SessionManifest {
1146    /// Create a fresh manifest for a session.
1147    #[must_use]
1148    pub(crate) fn new(session_id: &str) -> Self {
1149        let ts = now_rfc3339();
1150        Self {
1151            session_id: session_id.to_string(),
1152            schema_version: crate::SESSION_STORE_SCHEMA_VERSION,
1153            created_at: ts.clone(),
1154            updated_at: ts,
1155            turn_count: 0,
1156            event_count: 0,
1157            status: "active".to_string(),
1158            in_turn: Some(false),
1159            persisted_file_len: Some(0),
1160            retained_turn_base: Some(1),
1161        }
1162    }
1163}
1164
1165/// Byte-offset index of a single turn within `events.jsonl`.
1166#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
1167pub struct TurnIndexEntry {
1168    /// Turn ordinal (1-based).
1169    turn_number: u64,
1170    /// Byte offset of the turn's first event.
1171    start_offset: u64,
1172    /// Byte offset just past the turn's last event.
1173    end_offset: u64,
1174    /// Number of events in the turn.
1175    event_count: u64,
1176    /// RFC3339 timestamp of turn start.
1177    ts: String,
1178}
1179
1180/// Ordered index of all turns in a session.
1181#[derive(Debug, Clone, Default, Serialize, Deserialize, PartialEq, Eq)]
1182pub struct TurnIndex {
1183    /// Turn entries in ordinal order.
1184    entries: VecDeque<TurnIndexEntry>,
1185}
1186
1187impl TurnIndex {
1188    /// Number of indexed turns.
1189    #[must_use]
1190    pub fn len(&self) -> usize {
1191        self.entries.len()
1192    }
1193
1194    /// Whether the index is empty.
1195    #[must_use]
1196    pub fn is_empty(&self) -> bool {
1197        self.entries.is_empty()
1198    }
1199
1200    fn is_valid_for_file(&self, file_len: u64) -> bool {
1201        let mut previous_end = 0u64;
1202        self.entries.iter().all(|entry| {
1203            let valid = entry.event_count > 0
1204                && entry.start_offset >= previous_end
1205                && entry.start_offset <= entry.end_offset
1206                && entry.end_offset <= file_len;
1207            if valid {
1208                previous_end = entry.end_offset;
1209            }
1210            valid
1211        })
1212    }
1213
1214    fn is_consistent_with_manifest(&self, manifest: &SessionManifest) -> bool {
1215        let expected_last_turn = if manifest.in_turn == Some(true) {
1216            manifest.turn_count.saturating_add(1)
1217        } else {
1218            manifest.turn_count
1219        };
1220        let entries_are_contiguous = self
1221            .entries
1222            .iter()
1223            .map(|entry| entry.turn_number)
1224            .try_fold(None::<u64>, |previous, turn_number| {
1225                if previous.is_some_and(|previous| turn_number != previous.saturating_add(1)) {
1226                    return Err(());
1227                }
1228                Ok(Some(turn_number))
1229            })
1230            .is_ok();
1231        if !entries_are_contiguous {
1232            return false;
1233        }
1234
1235        let expected_first_turn = manifest.retained_turn_base.unwrap_or(1).max(1);
1236        match (self.entries.front(), self.entries.back()) {
1237            (Some(first), Some(last)) => {
1238                first.turn_number == expected_first_turn && last.turn_number == expected_last_turn
1239            }
1240            (None, None) => {
1241                expected_last_turn == 0
1242                    || manifest
1243                        .retained_turn_base
1244                        .is_some_and(|retained_turn_base| retained_turn_base > manifest.turn_count)
1245            }
1246            _ => false,
1247        }
1248    }
1249}
1250
1251#[cfg(test)]
1252mod borrowed_envelope_tests {
1253    use super::{BorrowedVersionedEvent, EVENT_SCHEMA_VERSION};
1254    use vtcode_exec_events::{
1255        ThreadEvent, ThreadStartedEvent, TurnCompletedEvent, TurnStartedEvent, Usage, VersionedThreadEvent,
1256    };
1257
1258    /// The borrowed envelope must produce JSON byte-identical to
1259    /// `VersionedThreadEvent::new(event.clone())`. This guards against drift if
1260    /// either the envelope or the canonical wrapper is modified.
1261    #[test]
1262    fn borrowed_envelope_matches_versioned_envelope() {
1263        for event in [
1264            ThreadEvent::ThreadStarted(ThreadStartedEvent { thread_id: "thread".to_string() }),
1265            ThreadEvent::TurnStarted(TurnStartedEvent::default()),
1266            ThreadEvent::TurnCompleted(TurnCompletedEvent {
1267                usage: Usage::default(),
1268                in_progress_exec_sessions: Vec::new(),
1269            }),
1270        ] {
1271            let canonical =
1272                serde_json::to_string(&VersionedThreadEvent::new(event.clone())).expect("canonical serialize");
1273            let borrowed = serde_json::to_string(&BorrowedVersionedEvent {
1274                schema_version: EVENT_SCHEMA_VERSION,
1275                event: &event,
1276            })
1277            .expect("borrowed serialize");
1278            assert_eq!(canonical, borrowed, "JSON differs for {event:?}");
1279        }
1280    }
1281}
1282
1283#[cfg(test)]
1284mod lifecycle_state_machine_tests {
1285    use super::{LifecycleKind, LogState};
1286    use vtcode_exec_events::{
1287        ThreadCompletedEvent, ThreadCompletionSubtype, ThreadEvent, ThreadStartedEvent, TurnCompletedEvent,
1288        TurnFailedEvent, TurnStartedEvent, Usage,
1289    };
1290
1291    fn fresh_state() -> LogState {
1292        LogState::new("test-session")
1293    }
1294
1295    #[test]
1296    fn lifecycle_kind_from_event_covers_all_variants() {
1297        assert_eq!(
1298            LifecycleKind::from_event(&ThreadEvent::TurnStarted(TurnStartedEvent::default())),
1299            LifecycleKind::TurnStarted
1300        );
1301        assert_eq!(
1302            LifecycleKind::from_event(&ThreadEvent::TurnCompleted(TurnCompletedEvent {
1303                usage: Usage::default(),
1304                in_progress_exec_sessions: Vec::new()
1305            })),
1306            LifecycleKind::TurnCompleted
1307        );
1308        assert_eq!(
1309            LifecycleKind::from_event(&ThreadEvent::TurnFailed(TurnFailedEvent {
1310                message: "err".to_string(),
1311                usage: None,
1312            })),
1313            LifecycleKind::TurnFailed
1314        );
1315        assert_eq!(
1316            LifecycleKind::from_event(&ThreadEvent::ThreadStarted(ThreadStartedEvent { thread_id: "x".to_string() })),
1317            LifecycleKind::ThreadStarted
1318        );
1319        assert_eq!(
1320            LifecycleKind::from_event(&ThreadEvent::ThreadCompleted(Box::new(ThreadCompletedEvent {
1321                thread_id: "x".to_string(),
1322                session_id: "x".to_string(),
1323                subtype: ThreadCompletionSubtype::Success,
1324                outcome_code: "completed".to_string(),
1325                result: None,
1326                stop_reason: None,
1327                usage: Usage::default(),
1328                total_cost_usd: None,
1329                num_turns: 1,
1330            }))),
1331            LifecycleKind::ThreadCompleted
1332        );
1333        assert_eq!(LifecycleKind::from_kind("thread.started"), LifecycleKind::ThreadStarted);
1334        assert_eq!(LifecycleKind::from_kind("thread.completed"), LifecycleKind::ThreadCompleted);
1335    }
1336
1337    #[test]
1338    fn lifecycle_kind_from_str_matches_event_discriminator() {
1339        assert_eq!(LifecycleKind::from_kind("turn.started"), LifecycleKind::TurnStarted);
1340        assert_eq!(LifecycleKind::from_kind("turn.completed"), LifecycleKind::TurnCompleted);
1341        assert_eq!(LifecycleKind::from_kind("turn.failed"), LifecycleKind::TurnFailed);
1342        assert_eq!(LifecycleKind::from_kind("tool.called"), LifecycleKind::Other);
1343        assert_eq!(LifecycleKind::from_kind("thread.started"), LifecycleKind::ThreadStarted);
1344        assert_eq!(LifecycleKind::from_kind("thread.completed"), LifecycleKind::ThreadCompleted);
1345    }
1346
1347    #[test]
1348    fn turn_started_pushes_index_entry_and_sets_in_turn() {
1349        let mut st = fresh_state();
1350        st.manifest.status = "completed".to_string();
1351        let is_boundary = st.apply_lifecycle_event(LifecycleKind::TurnStarted, 0, 100);
1352        assert!(!is_boundary, "TurnStarted is not a turn boundary");
1353        assert!(st.in_turn);
1354        assert_eq!(st.manifest.status, "active");
1355        assert_eq!(st.index.entries.len(), 1);
1356        let entry = &st.index.entries[0];
1357        assert_eq!(entry.turn_number, 1);
1358        assert_eq!(entry.start_offset, 0);
1359        assert_eq!(entry.end_offset, 100);
1360        assert_eq!(entry.event_count, 1);
1361    }
1362
1363    #[test]
1364    fn intermediate_events_extend_current_turn() {
1365        let mut st = fresh_state();
1366        st.apply_lifecycle_event(LifecycleKind::TurnStarted, 0, 100);
1367        // Simulate two intermediate events.
1368        let is_b1 = st.apply_lifecycle_event(LifecycleKind::Other, 100, 200);
1369        let is_b2 = st.apply_lifecycle_event(LifecycleKind::Other, 200, 300);
1370        assert!(!is_b1 && !is_b2);
1371        assert!(st.in_turn);
1372        assert_eq!(st.index.entries.len(), 1);
1373        let entry = &st.index.entries[0];
1374        assert_eq!(entry.end_offset, 300);
1375        assert_eq!(entry.event_count, 3);
1376    }
1377
1378    #[test]
1379    fn turn_completed_closes_turn_and_returns_boundary() {
1380        let mut st = fresh_state();
1381        st.apply_lifecycle_event(LifecycleKind::TurnStarted, 0, 100);
1382        st.apply_lifecycle_event(LifecycleKind::Other, 100, 200);
1383        let is_boundary = st.apply_lifecycle_event(LifecycleKind::TurnCompleted, 200, 300);
1384        assert!(is_boundary);
1385        assert!(!st.in_turn);
1386        assert_eq!(st.manifest.turn_count, 1);
1387        assert_eq!(st.manifest.status, "active");
1388        st.apply_lifecycle_event(LifecycleKind::ThreadCompleted, 300, 400);
1389        assert_eq!(st.manifest.status, "completed");
1390        let entry = &st.index.entries[0];
1391        assert_eq!(entry.end_offset, 300);
1392        assert_eq!(entry.event_count, 3);
1393    }
1394
1395    #[test]
1396    fn turn_failed_closes_turn_without_terminal_thread_status() {
1397        let mut st = fresh_state();
1398        st.apply_lifecycle_event(LifecycleKind::TurnStarted, 0, 100);
1399        let is_boundary = st.apply_lifecycle_event(LifecycleKind::TurnFailed, 100, 200);
1400        assert!(is_boundary);
1401        assert!(!st.in_turn);
1402        assert_eq!(st.manifest.turn_count, 1);
1403        assert_eq!(st.manifest.status, "active");
1404        st.apply_lifecycle_event(LifecycleKind::ThreadCompleted, 200, 300);
1405        assert_eq!(st.manifest.status, "completed");
1406    }
1407
1408    #[test]
1409    fn turn_completed_without_turn_started_is_idempotent() {
1410        let mut st = fresh_state();
1411        // Receiving TurnCompleted without a preceding TurnStarted should not
1412        // panic or corrupt the index; terminal status remains active until the
1413        // thread lifecycle itself completes.
1414        let is_boundary = st.apply_lifecycle_event(LifecycleKind::TurnCompleted, 0, 100);
1415        assert!(is_boundary);
1416        assert!(!st.in_turn);
1417        assert_eq!(st.manifest.turn_count, 0, "no turn was started");
1418        assert_eq!(st.manifest.status, "active");
1419        st.apply_lifecycle_event(LifecycleKind::ThreadCompleted, 100, 200);
1420        assert_eq!(st.manifest.status, "completed");
1421        assert!(st.index.entries.is_empty());
1422    }
1423
1424    #[test]
1425    fn multiple_turns_get_incrementing_ordinals() {
1426        let mut st = fresh_state();
1427        for n in 1..=3 {
1428            st.apply_lifecycle_event(LifecycleKind::TurnStarted, n * 100, n * 100 + 50);
1429            st.apply_lifecycle_event(LifecycleKind::TurnCompleted, n * 100 + 50, n * 100 + 100);
1430        }
1431        assert_eq!(st.index.entries.len(), 3);
1432        for (i, entry) in st.index.entries.iter().enumerate() {
1433            assert_eq!(entry.turn_number, (i + 1) as u64);
1434        }
1435        assert_eq!(st.manifest.turn_count, 3);
1436    }
1437}
1438
1439#[cfg(test)]
1440mod cap_eviction_tests {
1441    use super::{LogState, TurnIndexEntry};
1442
1443    /// Build a `LogState` with `turns` fake turns, each having `events_per_turn`
1444    /// events, starting at byte offset 0.
1445    fn state_with_turns(turns: usize, events_per_turn: u64) -> LogState {
1446        let mut st = LogState::new("cap-test");
1447        st.manifest.event_count = (turns as u64) * events_per_turn;
1448        let mut offset = 0u64;
1449        for n in 1..=turns {
1450            st.index.entries.push_back(TurnIndexEntry {
1451                turn_number: n as u64,
1452                start_offset: offset,
1453                end_offset: offset + events_per_turn * 10,
1454                event_count: events_per_turn,
1455                ts: "2026-01-01T00:00:00Z".to_string(),
1456            });
1457            offset += events_per_turn * 10;
1458        }
1459        st
1460    }
1461
1462    #[test]
1463    fn no_eviction_when_under_cap() {
1464        let st = state_with_turns(3, 2); // 6 events
1465        assert!(st.plan_cap_eviction(10).is_none());
1466        assert_eq!(st.index.entries.len(), 3, "no turns should be evicted");
1467    }
1468
1469    #[test]
1470    fn no_eviction_when_cap_disabled() {
1471        let st = state_with_turns(5, 2); // 10 events
1472        assert!(st.plan_cap_eviction(0).is_none());
1473        assert_eq!(st.index.entries.len(), 5);
1474    }
1475
1476    #[test]
1477    fn evicts_oldest_turns_to_meet_cap() {
1478        // 5 turns × 2 events = 10 events; cap = 6 → need to evict 2 turns (4 events).
1479        let st = state_with_turns(5, 2);
1480        let plan = st.plan_cap_eviction(6).expect("eviction planned");
1481        assert_eq!(plan.evicted_event_count, 4, "should evict 4 events (2 turns)");
1482        assert_eq!(st.index.entries.len(), 5, "planning must not mutate state");
1483        // Truncate offset is the end of the last evicted turn.
1484        assert_eq!(plan.truncate_offset, 40); // 2 turns × 20 bytes each
1485
1486        // Applying the plan leaves turns 3, 4, 5.
1487        let mut st = st;
1488        st.apply_cap_eviction(plan, 60);
1489        assert_eq!(st.index.entries.len(), 3, "should keep 3 turns");
1490        assert_eq!(st.index.entries[0].turn_number, 3);
1491        assert_eq!(st.index.entries[2].turn_number, 5);
1492    }
1493
1494    #[test]
1495    fn evicts_all_turns_when_cap_smaller_than_one_turn() {
1496        // 3 turns × 5 events = 15 events; cap = 3 → evict turns until ≤ 3 remain.
1497        // Each turn has 5 events, so evicting 2 turns leaves 5 (>3), evicting
1498        // 3 turns leaves 0.
1499        let st = state_with_turns(3, 5);
1500        let plan = st.plan_cap_eviction(3).expect("eviction planned");
1501        assert_eq!(plan.evicted_event_count, 15, "all events evicted");
1502        let mut st = st;
1503        st.apply_cap_eviction(plan, 0);
1504        assert_eq!(st.index.entries.len(), 0);
1505    }
1506}