froe 0.12.0

Reader and offline maintenance toolkit for Apache Jackrabbit Oak segment-tar (TarMK) repositories: parse archives and records, extract node data, compact, back up, and recover.
Documentation
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
763
764
765
766
767
768
769
770
771
772
773
774
775
776
777
778
779
780
781
782
783
784
785
786
787
788
789
790
791
792
793
794
795
796
797
798
799
800
801
802
803
804
805
806
807
808
809
810
811
812
813
814
815
816
817
818
819
820
821
822
823
824
825
826
827
828
829
830
831
832
833
834
835
836
837
838
839
840
841
842
843
844
845
846
847
848
849
850
851
852
853
854
855
856
857
858
859
860
861
862
863
864
865
866
867
868
869
870
871
872
873
874
875
876
877
878
879
880
881
882
883
884
885
886
887
888
889
890
891
892
893
894
895
896
897
898
899
900
901
902
903
904
905
906
907
908
909
910
911
912
913
914
915
916
917
918
919
920
921
922
923
924
925
926
927
928
929
930
931
932
933
934
935
936
937
938
939
940
941
942
943
944
945
946
947
948
949
950
951
952
953
954
955
956
957
958
959
960
961
962
963
964
965
966
967
//! What a planned run reports: the actions it would take, why an
//! archive or journal line is stale, and what the applied run did.

use super::options::MaintenanceTask;
use super::planning::{
    CheckpointPlan, DirectoryFingerprint, JournalPlan, PlannedFileRemoval, StaleArchive,
};
use crate::segment::identifier::SegmentIdentifier;
use crate::segment::record::RecordIdentifier;
use crate::writer::compaction::CompactionKind;
use crate::writer::segment_builder::GarbageCollectionGeneration;
use crate::writer::store_writer::StandaloneSegmentCompactionPlan;
use std::collections::HashSet;
use std::path::{Path, PathBuf};

/// Why an archive file can be removed without losing an active segment.
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
#[non_exhaustive]
pub enum StaleArchiveReason {
    /// A different letter of the same archive number has the newest valid
    /// index and is the active reader winner.
    Superseded,
    /// The file is empty and therefore contains no recoverable segment.
    EmptyIncomplete,
}

impl std::fmt::Display for StaleArchiveReason {
    fn fmt(&self, formatter: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
        formatter.write_str(match self {
            Self::Superseded => "superseded by the active archive generation",
            Self::EmptyIncomplete => "empty incomplete archive",
        })
    }
}

/// Why one physical journal line is removable.
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
#[non_exhaustive]
pub enum JournalRemovalReason {
    /// The tolerant journal reader skips a line that contains no ASCII space.
    ParserSkippedNoSpace,
    /// The first space-delimited field is not a valid record identifier.
    InvalidRecordIdentifier,
    /// The record identifier names a segment that is not present.
    MissingSegment,
    /// The non-current historical node revision does not fully traverse.
    UnreadableRevision,
    /// The revision resolves, but an explicit retention bound keeps only
    /// newer revisions. Removing the line is what releases its closure from
    /// the history keep-veto; without it the line stays a tracing root and
    /// the segments behind it stay protected.
    BeyondRetention,
}

impl std::fmt::Display for JournalRemovalReason {
    fn fmt(&self, formatter: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
        formatter.write_str(match self {
            Self::ParserSkippedNoSpace => "parser-skipped (no ASCII space)",
            Self::InvalidRecordIdentifier => "invalid record identifier",
            Self::MissingSegment => "missing segment",
            Self::UnreadableRevision => "unreadable historical revision",
            Self::BeyondRetention => "beyond the journal retention bound",
        })
    }
}

/// One physical journal line selected for removal.
///
/// The preview is an exact, bounded prefix of the line excluding its line
/// terminator. It is bytes rather than text so invalid UTF-8 remains auditable;
/// terminal applications must escape it before display.
#[derive(Clone, Debug, PartialEq, Eq)]
#[non_exhaustive]
pub struct JournalLineRemoval {
    pub(super) line_number: usize,
    pub(super) record_identifier: Option<RecordIdentifier>,
    pub(super) reason: JournalRemovalReason,
    pub(super) preview: Vec<u8>,
    pub(super) preview_truncated: bool,
}

impl JournalLineRemoval {
    /// One-based physical line number in the journal snapshot.
    #[must_use]
    pub fn line_number(&self) -> usize {
        self.line_number
    }

    /// Parsed record identifier, when the line contained one.
    #[must_use]
    pub fn record_identifier(&self) -> Option<RecordIdentifier> {
        self.record_identifier
    }

    /// Structured reason for removing this line.
    #[must_use]
    pub fn reason(&self) -> JournalRemovalReason {
        self.reason
    }

    /// Exact bounded prefix of the line, excluding its terminator.
    #[must_use]
    pub fn preview_bytes(&self) -> &[u8] {
        &self.preview
    }

    /// Whether bytes after [`Self::preview_bytes`] were omitted.
    #[must_use]
    pub fn preview_truncated(&self) -> bool {
        self.preview_truncated
    }
}

pub(super) const ALREADY_ABSENT_DELETION_DETAIL: &str =
    "file was already absent when deletion was attempted";

#[derive(Clone, Copy, Debug, PartialEq, Eq)]
pub(super) enum FileDeletionFailureKind {
    Retained,
    AlreadyAbsent,
}

/// A planned deletion that this cleanup could not perform or confirm itself.
///
/// The target usually remains for a later retry. It can instead have already
/// been absent when the guarded unlink was reached; use
/// [`Self::target_was_already_absent`] to distinguish that auditable race.
#[derive(Clone, Debug, PartialEq, Eq)]
#[non_exhaustive]
pub struct FileDeletionFailure {
    pub(super) file_name: String,
    pub(super) error: String,
    pub(super) kind: FileDeletionFailureKind,
}

impl FileDeletionFailure {
    pub(super) fn retained(file_name: String, error: impl Into<String>) -> Self {
        Self {
            file_name,
            error: error.into(),
            kind: FileDeletionFailureKind::Retained,
        }
    }

    pub(super) fn already_absent(file_name: String, error: impl Into<String>) -> Self {
        Self {
            file_name,
            error: error.into(),
            kind: FileDeletionFailureKind::AlreadyAbsent,
        }
    }

    /// Exact managed file name involved in the partial deletion result.
    ///
    /// The path need not remain when [`Self::target_was_already_absent`]
    /// returns `true`.
    #[must_use]
    pub fn file_name(&self) -> &str {
        &self.file_name
    }

    /// Operating-system or consistency detail from the incomplete or
    /// externally satisfied deletion.
    #[must_use]
    pub fn error(&self) -> &str {
        &self.error
    }

    /// Whether another actor had already removed the exact planned pathname
    /// when cleanup reached its guarded deletion.
    #[must_use]
    pub fn target_was_already_absent(&self) -> bool {
        self.kind == FileDeletionFailureKind::AlreadyAbsent
    }
}

/// One concrete, deterministically ordered cleanup action.
#[derive(Clone, Debug, PartialEq, Eq)]
#[non_exhaustive]
pub enum CompactionAction {
    /// Rebuild the index of an active archive that has none.
    ///
    /// Reported by the read-only preview, which cannot do more than name the
    /// work: the repair itself happens under the repository lock, and every
    /// index-dependent decision — the segment sweep, checkpoint removal —
    /// can only be planned once it has. The one locked plan the operator
    /// confirms is therefore always larger than a dry-run preview that
    /// named this.
    RepairArchiveIndex {
        /// Archive file name the rebuilt archive is installed under: the
        /// lowest non-empty generation letter of its number.
        file_name: String,
        /// Other generation letters of the same number, whose contents are
        /// merged into the rebuild and which are then retired to `.bak`
        /// names. Named because confirmation is scoped to the files a plan
        /// printed, and these leave the archive namespace.
        retired_file_names: Vec<String>,
        /// Why the existing index was rejected.
        reason: String,
        /// Whole-file bytes across every letter that will be read, which is
        /// what ends up retained under `.bak` names.
        bytes: u64,
    },
    /// Rewrite the journal while retaining readable record lines verbatim.
    PruneJournal {
        /// Total physical lines removed.
        lines: usize,
        /// Lines the tolerant reader already skips.
        parser_ignored: usize,
        /// Syntactic record lines whose head segment is absent.
        missing_segments: usize,
        /// Non-current historical node roots that do not fully traverse.
        unreadable_revisions: usize,
        /// Resolvable revisions older than an explicit retention bound.
        beyond_retention: usize,
    },
    /// Atomically raise `store.version` from 1 to 2 before writing v2 data.
    UpgradeManifest,
    /// Remove checkpoints in one head update.
    RemoveCheckpoints {
        /// Exact checkpoint names, sorted and deduplicated.
        names: Vec<String>,
        /// Names selected because their valid timestamp has expired.
        expired: usize,
        /// Additional names selected by the opt-in `/:async` rule.
        unreferenced: usize,
    },
    /// Unlink a fully reclaimable active archive.
    RemoveReclaimableArchive {
        /// Current archive file name.
        file_name: String,
        /// Segments made unavailable by the unlink.
        segments: usize,
        /// Current whole-file bytes.
        bytes: u64,
    },
    /// Rewrite an active archive to its next letter with only survivors.
    RewriteArchive {
        /// Source archive file name.
        file_name: String,
        /// Exclusively created replacement name.
        replacement_name: String,
        /// Segments omitted from the replacement.
        segments: usize,
        /// TAR-entry bytes eligible for reclamation.
        eligible_bytes: u64,
    },
    /// Remove an inactive archive generation or empty incomplete archive.
    RemoveStaleArchive {
        /// Exact archive file name.
        file_name: String,
        /// Proof supporting removal.
        reason: StaleArchiveReason,
        /// Current whole-file bytes.
        bytes: u64,
    },
    /// Remove a provably redundant interrupted-operation staging file.
    RemoveTemporary {
        /// Exact file name.
        file_name: String,
        /// Current whole-file bytes.
        bytes: u64,
    },
    /// Retire every journal revision but the one the copy publishes.
    ///
    /// Named separately from `PruneJournal`, which describes removing lines
    /// that cannot resolve. This removes lines that resolve perfectly well,
    /// by policy, because the segments behind them are what the run reclaims.
    RetireJournalHistory {
        /// Physical journal lines present before the run, all of which are
        /// replaced by the single line naming the compacted head.
        revisions: usize,
    },
    /// Omit orphaned version histories from the copy — the run's one
    /// content mutation, listed so confirmation covers it explicitly.
    PurgeOrphanedVersionHistories {
        /// Histories the copy omits.
        histories: u64,
        /// Nodes those histories hold.
        nodes: u64,
        /// Checkpoints the run retains, whose snapshots keep the purged
        /// histories' storage alive until they expire.
        retained_checkpoints: u64,
    },
    /// Retire the output an interrupted earlier compaction left behind.
    ///
    /// Segments stamped ahead of the head are, by construction, the copy of a
    /// run that died before it committed. No ordinary rule removes them, so a
    /// killed run would otherwise leave residue that every later run steps
    /// around while it holds bulk segments alive.
    RetireInterruptedCompactionResidue {
        /// Data segments found ahead of the head.
        segments: usize,
    },
    /// Deep-copy the head, and every checkpoint this run retains, into a
    /// fresh garbage-collection generation.
    CopyHeadIntoFreshGeneration {
        /// Distinct node records the current head reaches.
        ///
        /// What the copy rewrites is this minus whatever only a retired
        /// checkpoint reaches, so it is an exact statement about the store
        /// rather than a prediction about the copy.
        head_nodes: u64,
        /// The generation the copy writes into.
        target_generation: GarbageCollectionGeneration,
        /// Whether this is a full or a tail compaction.
        kind: CompactionKind,
    },
    /// Remove an explicitly authorized old recovery backup.
    RemoveRecoveryBackup {
        /// Exact file name.
        file_name: String,
        /// Current whole-file bytes.
        bytes: u64,
    },
}

/// Segments this run identified as reclaimable and then declined to remove.
///
/// Every count here is garbage the mark phase proved removable; the archive
/// sweep kept it anyway, because rewriting the archive that holds it would
/// not repay the rewrite. Reporting it is what separates "this store holds no
/// garbage" from "this store holds garbage that is not worth moving".
#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)]
pub(super) struct RetainedReclaimable {
    /// Segments kept by Oak's 25% savings gate.
    pub(super) below_savings_gate: usize,
    /// Segments kept because the archive exhausted the `a`–`z` namespace.
    pub(super) at_last_generation: usize,
    /// Segments kept because another generation pathname is occupied.
    pub(super) blocked_by_occupied_generation: usize,
    /// TAR entry bytes those segments occupy, summed across every reason.
    pub(super) bytes: u64,
}

impl RetainedReclaimable {
    /// Segments identified as reclaimable and left in place, all reasons.
    fn segments(self) -> usize {
        self.below_savings_gate
            .saturating_add(self.at_last_generation)
            .saturating_add(self.blocked_by_occupied_generation)
    }
}

/// What the journal-history keep-veto protects, and what it costs.
///
/// froe retains every readable journal revision as a tracing root, which Oak
/// does not do: Oak judges data segments by their index generation triple
/// alone. The veto is strictly conservative, so it can never delete anything
/// Oak would keep — but on a long-lived store it is normally the single
/// largest reason a cleanup reclaims nothing, and nothing in the run used to
/// say so.
#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)]
pub(super) struct HistoryProtection {
    /// Data segments reachable only from a historical journal revision, and
    /// not from the current head.
    pub(super) history_only_segments: usize,
    /// Segments this same sweep would physically free with the veto lifted
    /// and nothing else changed. Measured by replanning rather than reasoned
    /// about: the veto holds bulk segments only through the data segments
    /// that reference them, and releasing more of an archive can carry it
    /// over the 25% rewrite gate. Counting protected data segments alone
    /// understates this by orders of magnitude on a store whose history
    /// holds inline binaries.
    pub(super) would_be_reclaimable_segments: usize,
    /// Bytes those segments occupy, whole archive files included where the
    /// unvetoed sweep would unlink one outright.
    pub(super) would_be_reclaimable_bytes: u64,
}

/// The external binaries the verified head references, counted while the
/// planning walk was reading every property anyway.
///
/// Compaction can never touch these bytes — they live in the blob store —
/// and a plan that says nothing about them invites the operator to expect
/// blob-store savings from a segment-store tool. Distinctness is tracked
/// per blob identifier (hash-keyed for identifier formats that carry no
/// content hash of their own), and bytes are summed from the length suffix
/// Oak's file blob stores embed in their identifiers; an identifier without
/// one is counted but cannot contribute bytes.
#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)]
pub struct ExternalBinaryFootprint {
    /// Distinct external blob identifiers the head references.
    pub distinct_references: u64,
    /// Bytes summed over the identifiers that carry a length suffix.
    pub measured_bytes: u64,
    /// Distinct identifiers without a parsable length suffix.
    pub unmeasured_references: u64,
}

/// What the planner established about orphaned version histories:
/// `nt:versionHistory` subtrees whose `jcr:versionableUuid` matches no live
/// `jcr:uuid` outside version storage. Reachable content — so no structural
/// sweep may touch them — yet garbage by the only definition that matters
/// once the versionable is gone, and the figures here are what they pin.
/// Detection runs on every plan; removal is the separate, explicitly
/// selected purge.
#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)]
pub struct OrphanedVersionHistoryReport {
    /// Histories whose versionables no longer exist.
    pub orphaned_histories: u64,
    /// Nodes those histories hold, the history nodes included.
    pub orphaned_nodes: u64,
    /// Inline binary bytes stored in their properties.
    pub inline_binary_bytes: u64,
    /// External binary references stored in their properties — bytes that
    /// live in the blob store and return only through blob-store garbage
    /// collection once a purge unreferences them.
    pub external_references: u64,
    /// Bulk segments a purge would stop referencing — an upper bound on
    /// what it releases, not a promise. Attribution is per certified
    /// record, so a record shared between a purged history and one the
    /// store keeps (Oak's writer dedups identical frozen subtrees into
    /// shared records) keeps its blocks alive without this figure knowing,
    /// and a retained checkpoint's snapshot can pin blocks until it
    /// expires. The sweep that follows the copy frees exactly what is
    /// actually unreferenced, whatever this predicted.
    pub released_bulk_segments: u64,
    /// Indexed bytes of those bulk segments — the same upper-bound
    /// semantics as [`Self::released_bulk_segments`].
    pub released_bulk_bytes: u64,
    /// The orphans' share of the copy's node records, scaled from the
    /// head's average bytes per node. An estimate for the plan, realized
    /// and reported exactly by the copy that runs.
    pub node_record_bytes_estimate: u64,
    /// Histories whose `jcr:versionableUuid` did not parse and therefore
    /// could not be classified.
    pub malformed_identifiers: u64,
    /// Checkpoints the run retains. Their snapshots can pin some of the
    /// released bulk until they expire — pinning the walks cannot see when
    /// the snapshot shares the head's version-storage records — so a
    /// nonzero count is the caveat on [`Self::released_bulk_bytes`].
    pub retained_checkpoints: u64,
}

/// The purge the plan carries when one is selected: the subtree roots the
/// copy omits, and the counts the summary restates.
#[derive(Clone, Debug, PartialEq, Eq)]
pub(super) struct VersionHistoryPurge {
    pub(super) omitted_records: Vec<RecordIdentifier>,
    /// The ancestors on the path from the content root down to each omitted
    /// record. Their rewritten form depends on the scope — the head's copy
    /// loses the omitted subtrees, a checkpoint snapshot's keeps them — so
    /// the copy memoizes them per scope rather than globally.
    pub(super) context_dependent_records: Vec<RecordIdentifier>,
    pub(super) histories: u64,
    pub(super) nodes: u64,
    /// Checkpoints the run retains, whose snapshots keep the purged
    /// histories' storage alive until they expire.
    pub(super) retained_checkpoints: u64,
}

/// A strictly read-only cleanup analysis.
#[derive(Clone, Debug, PartialEq, Eq)]
pub struct CompactionPlan {
    pub(super) directory: PathBuf,
    pub(super) tasks: Vec<MaintenanceTask>,
    pub(super) current_head: RecordIdentifier,
    pub(super) actions: Vec<CompactionAction>,
    pub(super) warnings: Vec<String>,
    pub(super) estimated_reclaimable_bytes: u64,
    pub(super) estimated_archive_rewrite_source_bytes: u64,
    pub(super) retained_reclaimable: RetainedReclaimable,
    pub(super) history_protection: HistoryProtection,
    pub(super) fingerprint: DirectoryFingerprint,
    pub(super) journal: JournalPlan,
    pub(super) checkpoints: CheckpointPlan,
    pub(super) checkpoint_archive_number: Option<u32>,
    pub(super) stale_archives: Vec<StaleArchive>,
    pub(super) temporaries: Vec<PlannedFileRemoval>,
    pub(super) recovery_backups: Vec<PlannedFileRemoval>,
    pub(super) segment_plan: Option<StandaloneSegmentCompactionPlan>,
    /// The sweep that retires an interrupted earlier compaction's output,
    /// applied before this run's own copy.
    pub(super) residue_sweep: Option<StandaloneSegmentCompactionPlan>,
    pub(super) reference_generation: GarbageCollectionGeneration,
    pub(super) protected_history_segments: HashSet<SegmentIdentifier>,
    pub(super) manifest_upgrade: bool,
    /// What the planned copy is expected to write into the fresh
    /// generation, when a copy is planned and the head closure was traced.
    pub(super) predicted_copy_output_bytes: Option<u64>,
    /// The external binaries the verified head references.
    pub(super) external_binary_footprint: ExternalBinaryFootprint,
    /// The compaction this run will actually perform: the selected kind,
    /// unless the convergence gate proved the copy pointless and dropped it.
    pub(super) effective_compaction_kind: Option<CompactionKind>,
    /// Whether the planner proved the head already fully compacted.
    pub(super) already_fully_compacted: bool,
    /// What the planner established about orphaned version histories.
    pub(super) orphaned_version_histories: OrphanedVersionHistoryReport,
    /// The purge this run performs, when one is selected and non-empty.
    pub(super) version_history_purge: Option<VersionHistoryPurge>,
}

impl CompactionPlan {
    /// Canonical absolute repository directory this plan describes.
    #[must_use]
    pub fn directory(&self) -> &Path {
        &self.directory
    }

    /// Selected cleanup categories in deterministic order.
    #[must_use]
    #[cfg(test)]
    pub(crate) fn tasks(&self) -> &[MaintenanceTask] {
        &self.tasks
    }

    /// Exact current head verified while planning.
    #[must_use]
    pub fn current_head(&self) -> RecordIdentifier {
        self.current_head
    }

    /// Concrete mutations in deterministic display order.
    #[must_use]
    pub fn actions(&self) -> &[CompactionAction] {
        &self.actions
    }

    /// Exact physical journal lines selected for removal. This is empty unless
    /// the journal task was selected; internal journal analysis still
    /// runs for the safety of other tasks.
    #[must_use]
    pub fn journal_line_removals(&self) -> &[JournalLineRemoval] {
        if self.tasks.contains(&MaintenanceTask::Journal) {
            &self.journal.removals
        } else {
            &[]
        }
    }

    /// Non-fatal deferrals and malformed metadata retained for safety.
    ///
    /// Also carries one advisory line per index definition Oak has flagged
    /// for reindex — `pending reindex: <path> (<type>; …)` — because an
    /// operator compacting before an AEM restart is holding the store open
    /// at the moment that fact is cheap to read and expensive to miss. It
    /// is advisory in the strict sense: no action is added to the plan, no
    /// byte the run writes changes, and nothing about the run is decided by
    /// it.
    #[must_use]
    pub fn warnings(&self) -> &[String] {
        &self.warnings
    }

    /// Conservative sum of whole files and TAR entry bytes selected for
    /// removal. Archive overhead and deletion failures can make the actual
    /// result differ.
    #[must_use]
    pub fn estimated_reclaimable_bytes(&self) -> u64 {
        self.estimated_reclaimable_bytes
    }

    /// The bytes the planned copy is expected to write into the fresh
    /// generation, when a copy is planned and the head closure was traced.
    ///
    /// This is the head closure's indexed data bytes — exact for a head a
    /// previous compaction wrote (the deterministic copy is a fixed point),
    /// an upper bound for a fragmented head (the dense copy writes less),
    /// and slightly low only for the first compaction of a store another
    /// writer produced, whose nodes gain stable-identifier blocks in the
    /// copy. Reported so an estimate can state the run's net effect rather
    /// than only its reclaimed side: a swap that removes one generation and
    /// writes an equal one has a net effect of nothing, and an estimate
    /// that hides the written side over-promises by exactly this figure.
    #[must_use]
    pub fn predicted_copy_output_bytes(&self) -> Option<u64> {
        self.predicted_copy_output_bytes
    }

    /// The compaction this run will actually perform. `None` either because
    /// none was selected or because the convergence gate proved the head is
    /// already fully compacted and dropped the selected copy — in which
    /// case [`Self::already_fully_compacted`] says so.
    #[must_use]
    pub fn effective_compaction_kind(&self) -> Option<CompactionKind> {
        self.effective_compaction_kind
    }

    /// Whether the planner proved every data segment the head reaches
    /// already carries the head's own compacted generation triple, with no
    /// checkpoint to omit, no version history selected for purge, and the
    /// journal already retired — the state a completed full compaction
    /// leaves, in which a repeat copy would only replace a generation with
    /// an identical one.
    #[must_use]
    pub fn already_fully_compacted(&self) -> bool {
        self.already_fully_compacted
    }

    /// What the planner established about orphaned version histories.
    #[must_use]
    pub fn orphaned_version_histories(&self) -> OrphanedVersionHistoryReport {
        self.orphaned_version_histories
    }

    /// The external binaries the verified head references — blob-store
    /// content compaction can never reclaim, reported so an operator's
    /// blob-store expectations land on blob-store garbage collection
    /// rather than on this run.
    #[must_use]
    pub fn external_binary_footprint(&self) -> ExternalBinaryFootprint {
        self.external_binary_footprint
    }

    /// Sum of the current source-file sizes for archives that will be
    /// rewritten. Source mappings stay open through the sweep, so the
    /// filesystem may need cumulative additional space of this order. This is
    /// an operational proxy for archive rewriting, not a bound on other cleanup
    /// files or filesystem allocation overhead.
    #[must_use]
    pub fn estimated_archive_rewrite_source_bytes(&self) -> u64 {
        self.estimated_archive_rewrite_source_bytes
    }

    /// Segments proved reclaimable that this run will nevertheless leave in
    /// place, because rewriting the archives holding them is not worthwhile
    /// or not possible. Nonzero alongside a zero reclaimable estimate means
    /// the store holds garbage this cleanup declined, not that it holds none.
    #[must_use]
    pub fn retained_reclaimable_segments(&self) -> usize {
        self.retained_reclaimable.segments()
    }

    /// TAR entry bytes occupied by [`Self::retained_reclaimable_segments`].
    #[must_use]
    pub fn retained_reclaimable_bytes(&self) -> u64 {
        self.retained_reclaimable.bytes
    }

    /// Data segments kept alive only because a historical journal revision
    /// still reaches them. Zero unless the segment task ran.
    #[must_use]
    pub fn history_protected_segments(&self) -> usize {
        self.history_protection.history_only_segments
    }

    /// Those of [`Self::history_protected_segments`] that Oak's generation
    /// predicate would have reclaimed, and the bytes they occupy. This is
    /// what retiring the journal history — a full compaction — would make
    /// eligible; standalone cleanup never will.
    #[must_use]
    pub fn history_protected_reclaimable(&self) -> (usize, u64) {
        (
            self.history_protection.would_be_reclaimable_segments,
            self.history_protection.would_be_reclaimable_bytes,
        )
    }

    /// Whether application would request any mutation.
    #[must_use]
    pub fn is_empty(&self) -> bool {
        self.actions.is_empty()
    }
}

/// What a run's deep copy produced.
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
#[non_exhaustive]
pub struct CompactedGeneration {
    /// Distinct node records the copy rewrote.
    pub nodes: u64,
    /// The garbage-collection generation the copy wrote into.
    pub generation: GarbageCollectionGeneration,
}

/// Result of a prepared maintenance application and its final fresh
/// verification.
#[derive(Clone, Debug, PartialEq, Eq)]
#[non_exhaustive]
pub struct CompactionOutcome {
    /// Head before cleanup.
    pub head_before: RecordIdentifier,
    /// Freshly reopened and verified head after cleanup.
    pub head_after: RecordIdentifier,
    /// Number of checkpoints removed in one logical commit.
    pub removed_checkpoints: u64,
    /// Journal physical lines removed.
    pub removed_journal_lines: usize,
    /// Active archives rewritten.
    pub rewritten_archives: usize,
    /// Active fully reclaimable archives unlinked.
    pub removed_reclaimable_archives: usize,
    /// Superseded/empty archive files unlinked.
    pub removed_stale_archives: usize,
    /// Proven staging files removed.
    pub removed_temporaries: usize,
    /// Opt-in recovery backups removed.
    pub removed_recovery_backups: usize,
    /// Archive indexes rebuilt before planning, under the repository lock.
    pub repaired_archives: usize,
    /// Recognized deletion targets this cleanup did not unlink itself.
    ///
    /// Most entries remain for retry; entries reported as already absent need
    /// no further deletion attempt.
    pub files_not_deleted: Vec<String>,
    /// Bytes in recognized archive files before application.
    pub archive_bytes_before: u64,
    /// Bytes in recognized archive files after application.
    pub archive_bytes_after: u64,
    /// Bytes still held by retained recovery backups after application.
    ///
    /// These sit outside [`Self::archive_bytes_after`], which counts only
    /// active archive names. A run that rebuilds an index retires the
    /// original under a `.bak` name, so the directory grows by this much
    /// while the archive figures report no change at all.
    pub retained_recovery_backup_bytes: u64,
    /// Distinct node records copied into the fresh generation, and the
    /// generation they were copied into. `None` when the run did not compact.
    pub compacted: Option<CompactedGeneration>,
    pub(super) removed_segments: usize,
    pub(super) journal_backup_path: Option<PathBuf>,
    pub(super) deletion_failures: Vec<FileDeletionFailure>,
}

impl CompactionOutcome {
    /// Orphan segments removed from the active archive set.
    #[must_use]
    pub fn removed_segments(&self) -> usize {
        self.removed_segments
    }

    /// Durable byte-exact journal backup created by this cleanup, if any.
    #[must_use]
    pub fn journal_backup_path(&self) -> Option<&Path> {
        self.journal_backup_path.as_deref()
    }

    /// Planned deletions this cleanup did not perform itself, making the
    /// result partial even when another actor already removed a target.
    #[must_use]
    pub fn deletion_failures(&self) -> &[FileDeletionFailure] {
        &self.deletion_failures
    }

    /// Whether this cleanup itself completed every planned deletion without
    /// an auditable partial result.
    #[must_use]
    pub fn is_complete(&self) -> bool {
        self.deletion_failures.is_empty()
    }
}

#[cfg(test)]
mod tests {
    use super::*;
    use crate::store::Repository;

    use crate::writer::maintenance::options::*;

    use crate::writer::maintenance::prepared::*;

    use crate::writer::maintenance::test_support::*;
    use std::io::Write as _;
    use std::num::NonZeroUsize;

    #[test]
    fn dangling_journal_line_is_pruned_with_backup_and_archives_untouched() {
        let directory = TestDirectory::repository("dangling-journal");
        let missing = SegmentIdentifier::new(7, 0xA000_0000_0000_0007);
        let journal_path = directory.path.join("journal.log");
        let retained_journal = std::fs::read(&journal_path).expect("read retained journal");
        let mut journal = std::fs::OpenOptions::new()
            .append(true)
            .open(&journal_path)
            .expect("open journal");
        writeln!(journal, "{missing}:0 root 123").expect("append dangling line");
        drop(journal);
        let archive_before =
            std::fs::read(directory.path.join("data00000a.tar")).expect("read archive");
        std::fs::write(
            directory.path.join("manifest"),
            b"custom.property=untouched\nstore.version=1\n",
        )
        .expect("version-one manifest");
        let manifest_before = std::fs::read(directory.path.join("manifest")).expect("manifest");
        let options = CompactionOptions::default().with_tasks([MaintenanceTask::Journal]);

        let plan = plan_compaction(&directory.path, &options).expect("plan");
        assert_eq!(plan.tasks(), &[MaintenanceTask::Journal]);
        assert_eq!(plan.journal_line_removals().len(), 1);
        let removal = &plan.journal_line_removals()[0];
        assert_eq!(
            removal.record_identifier().map(|record| record.segment),
            Some(missing)
        );
        assert_eq!(removal.reason(), JournalRemovalReason::MissingSegment);
        assert!(
            removal
                .preview_bytes()
                .starts_with(missing.to_string().as_bytes())
        );
        assert!(!removal.preview_truncated());
        assert!(plan.actions().iter().any(|action| matches!(
            action,
            CompactionAction::PruneJournal {
                missing_segments: 1,
                ..
            }
        )));
        let outcome = compact(&directory.path, options).expect("apply");

        assert_eq!(outcome.removed_journal_lines, 1);
        let expected_backup =
            canonical_fixture_directory(&directory.path).join("journal.log.bak.000");
        assert_eq!(
            outcome.journal_backup_path(),
            Some(expected_backup.as_path())
        );
        assert!(outcome.is_complete());
        assert!(directory.path.join("journal.log.bak.000").is_file());
        assert!(
            !std::fs::read_to_string(&journal_path)
                .expect("journal")
                .contains(&missing.to_string())
        );
        assert_eq!(
            std::fs::read(&journal_path).expect("rewritten journal"),
            retained_journal,
            "the retained physical journal line must be byte-exact"
        );
        assert_eq!(
            std::fs::read(directory.path.join("data00000a.tar")).expect("archive"),
            archive_before
        );
        assert_eq!(
            std::fs::read(directory.path.join("manifest")).expect("manifest"),
            manifest_before
        );
        Repository::open(&directory.path).expect("healthy repository");
    }
    #[test]
    fn deletion_absence_state_does_not_depend_on_diagnostic_text() {
        let retained = super::FileDeletionFailure::retained(
            "data00000a.tar".to_owned(),
            ALREADY_ABSENT_DELETION_DETAIL,
        );
        let absent = super::FileDeletionFailure::already_absent(
            "data00001a.tar".to_owned(),
            "a deliberately different ENOENT diagnostic",
        );

        assert!(!retained.target_was_already_absent());
        assert!(absent.target_was_already_absent());
    }
    #[test]
    fn a_journal_retention_bound_retires_the_history_the_veto_protects() {
        let (directory, old_head, new_head) = history_veto_fixture("history-veto-retention");
        let protected = plan_compaction(
            &directory.path,
            &CompactionOptions::default().with_tasks([MaintenanceTask::Segments]),
        )
        .expect("unbounded plan");
        assert!(protected.history_protected_reclaimable().0 != 0);
        assert!(
            !protected.actions().iter().any(|action| matches!(
                action,
                CompactionAction::RemoveReclaimableArchive { file_name, .. }
                    if file_name == "data00000a.tar"
            )),
            "without a bound the veto must keep the bootstrap archive"
        );

        let bounded = CompactionOptions::default()
            .with_tasks([MaintenanceTask::Segments, MaintenanceTask::Journal])
            .with_journal_revision_retention(NonZeroUsize::new(1).expect("one revision"));
        let plan = plan_compaction(&directory.path, &bounded).expect("bounded plan");

        // The older line is pruned for the retention reason, not for damage.
        assert!(
            plan.journal_line_removals().iter().any(|removal| {
                removal.reason() == JournalRemovalReason::BeyondRetention
                    && removal.record_identifier() == Some(old_head)
            }),
            "the superseded revision must be removed as beyond retention"
        );
        // Releasing that root is what makes the archive eligible.
        assert!(
            plan.actions().iter().any(|action| matches!(
                action,
                CompactionAction::RemoveReclaimableArchive { file_name, .. }
                    if file_name == "data00000a.tar"
            )),
            "the bound must release the bootstrap archive to Oak's predicate"
        );
        assert!(plan.estimated_reclaimable_bytes() != 0);

        let outcome = compact(&directory.path, bounded).expect("bounded cleanup");
        assert_eq!(outcome.head_after, new_head);
        assert!(!directory.path.join("data00000a.tar").exists());
        let repository = Repository::open(&directory.path).expect("healthy final repository");
        assert_eq!(repository.head_record_identifier(), new_head);
        // The journal keeps exactly the bound's worth of revisions, and the
        // retired history is genuinely gone rather than merely unrooted.
        let journal =
            std::fs::read_to_string(directory.path.join("journal.log")).expect("read journal");
        assert_eq!(
            journal.lines().count(),
            1,
            "a bound of one leaves one journal line"
        );
        assert!(
            crate::tooling::verify_node_tree(&repository, old_head).is_err(),
            "the retired revision must no longer resolve"
        );
    }
    #[test]
    fn a_bound_counts_only_revisions_that_actually_resolve() {
        // A line whose segment exists but whose tree does not verify used to
        // fill a slot in the bound and then be removed as unreadable anyway,
        // so `N = 2` kept one revision and irreversibly retired a readable
        // one to make room for it. Every earlier retention test used N = 1,
        // which cannot expose this: the head is always the newest resolvable
        // line and always verifies.
        let (directory, old_head, new_head) = history_veto_fixture("retention-counts-readable");

        // A journal line naming a record that resolves to a segment but not
        // to a readable node tree: the head's own segment, at a record
        // number that is not a node record.
        let unreadable = RecordIdentifier::new(new_head.segment, new_head.record_number + 1);
        // Second newest, not newest: the newest line is the head, and a
        // head that is not a node record is refused long before any bound.
        let journal_path = directory.path.join("journal.log");
        let journal = std::fs::read_to_string(&journal_path).expect("read journal");
        let mut lines: Vec<&str> = journal.lines().collect();
        let head_line = lines.pop().expect("a head line");
        let unreadable_line = format!("{unreadable} root 0");
        lines.push(&unreadable_line);
        lines.push(head_line);
        std::fs::write(&journal_path, format!("{}\n", lines.join("\n")))
            .expect("insert unreadable line");

        let options = CompactionOptions::default()
            .with_tasks([MaintenanceTask::Segments, MaintenanceTask::Journal])
            .with_journal_revision_retention(NonZeroUsize::new(2).expect("two revisions"));
        let plan = plan_compaction(&directory.path, &options).expect("bounded plan");

        // The unreadable line goes, as it always did. What must not happen is
        // the older *readable* revision going with it to satisfy a bound the
        // unreadable line was counted against.
        assert!(
            !plan.journal_line_removals().iter().any(|removal| {
                removal.reason() == JournalRemovalReason::BeyondRetention
                    && removal.record_identifier() == Some(old_head)
            }),
            "a readable revision was retired to make room for an unreadable one: {:?}",
            plan.journal_line_removals()
        );
    }
    #[test]
    fn a_bound_larger_than_the_journal_removes_nothing() {
        let (directory, _old_head, _new_head) = history_veto_fixture("history-veto-wide-bound");
        let options = CompactionOptions::default()
            .with_tasks([MaintenanceTask::Segments, MaintenanceTask::Journal])
            .with_journal_revision_retention(NonZeroUsize::new(64).expect("wide bound"));
        let plan = plan_compaction(&directory.path, &options).expect("wide plan");
        assert!(
            !plan
                .journal_line_removals()
                .iter()
                .any(|removal| { removal.reason() == JournalRemovalReason::BeyondRetention }),
            "a bound wider than the journal must retire nothing"
        );
        assert!(plan.history_protected_reclaimable().0 != 0);
    }
}