core_api/restore.rs
1//! Seeding a store directory from a backup.
2//!
3//! The mirror of [`GraphDb::backup_to`](crate::GraphDb::backup_to), and it
4//! lives beside it for the same reason: a restore is a file-level copy of a
5//! store directory followed by an open that proves the copy is a store.
6//!
7//! This was the CLI's `--restore-from` implementation until v0.6.10, when the
8//! Python binding needed it too. `cli::restore_if_empty` is now a thin wrapper
9//! over [`restore_if_empty`] here, with the CLI's error type; the semantics are
10//! unchanged.
11
12use crate::{GraphDb, GraphError, Result};
13use std::path::{Path, PathBuf};
14
15/// What [`restore_if_empty`] did.
16#[derive(Debug, PartialEq, Eq)]
17pub enum RestoreOutcome {
18 /// `db_dir` already holds a store; nothing was copied.
19 AlreadyPresent,
20 /// Seeded from `from`.
21 Restored {
22 from: PathBuf,
23 files: Vec<String>,
24 bytes: u64,
25 },
26 /// `from` held no backup: nothing under it looks like a store.
27 Empty,
28}
29
30/// Every file [`GraphDb::backup_to`](crate::GraphDb::backup_to) copies that is
31/// not a WAL archive.
32///
33/// Kept in the same order, so a restore writes them the way a backup wrote
34/// them. Archives are found by name at copy time, since their count varies.
35const RESTORE_FILES: [&str; 6] = [
36 "snapshot.bin",
37 "snapshot.bin.bak",
38 "wal.bin",
39 "wal.floor",
40 "wal.genesis",
41 "roles.json",
42];
43
44/// Is there a store in `dir` already?
45///
46/// A store is "present" when `dir` holds `snapshot.bin` or a non-empty
47/// `wal.bin`. An empty `wal.bin` is what a crashed first boot leaves behind,
48/// and seeding over it is the whole point of `--restore-from`.
49pub fn holds_a_store(dir: &Path) -> bool {
50 if dir.join("snapshot.bin").is_file() {
51 return true;
52 }
53 std::fs::metadata(dir.join("wal.bin"))
54 .map(|m| m.is_file() && m.len() > 0)
55 .unwrap_or(false)
56}
57
58/// How recently a backup directory was written: the newest mtime among the
59/// files that make it a store.
60///
61/// `snapshot.bin` alone is not enough to rank by, because a store that has
62/// never snapshotted backs up as `wal.bin` and nothing else — which is exactly
63/// what `mushroomdb demo` then `mushroomdb backup` produces.
64fn backup_mtime(dir: &Path) -> Option<std::time::SystemTime> {
65 ["snapshot.bin", "wal.bin"]
66 .iter()
67 .filter_map(|n| {
68 std::fs::metadata(dir.join(n))
69 .and_then(|m| m.modified())
70 .ok()
71 })
72 .max()
73}
74
75/// Pick the backup under `from`, if there is one.
76///
77/// `from` is either a backup directory itself (it holds a store) or a
78/// directory of them, in which case the immediate subdirectory named `latest`
79/// wins outright if it holds one — so a symlink or a rolling copy can name
80/// itself — and otherwise the newest by mtime wins.
81///
82/// "Holds a store" is [`holds_a_store`], the same predicate that decides
83/// whether `db_dir` needs seeding. A backup of a never-snapshotted store is
84/// `wal.bin` and nothing else, and it carries every commit; ranking on
85/// `snapshot.bin` alone would skip it and start empty.
86fn choose_backup(from: &Path) -> Option<PathBuf> {
87 if holds_a_store(from) {
88 return Some(from.to_path_buf());
89 }
90 let entries = std::fs::read_dir(from).ok()?;
91 let mut best: Option<(std::time::SystemTime, PathBuf)> = None;
92 for entry in entries.flatten() {
93 let dir = entry.path();
94 if !holds_a_store(&dir) {
95 continue;
96 }
97 if dir.file_name().map(|n| n == "latest").unwrap_or(false) {
98 return Some(dir);
99 }
100 let Some(mtime) = backup_mtime(&dir) else {
101 continue;
102 };
103 // Ties break on the path, so a vault of same-second backups still
104 // picks the same one on every boot.
105 let better = match &best {
106 None => true,
107 Some((best_mtime, best_dir)) => (mtime, &dir) > (*best_mtime, best_dir),
108 };
109 if better {
110 best = Some((mtime, dir));
111 }
112 }
113 best.map(|(_, dir)| dir)
114}
115
116/// Seed `db_dir` from the newest backup under `from`, if `db_dir` has no store.
117///
118/// A store is "present" when `db_dir` holds `snapshot.bin` or a non-empty
119/// `wal.bin`. `from` is either a backup directory itself (it holds a store by
120/// that same test) or a directory of them, in which case the immediate
121/// subdirectory named `latest` wins if it holds one, else the newest by mtime.
122///
123/// The restore is all-or-nothing. The backup is copied into a staging
124/// directory **inside** `db_dir` and opened there — the same CRC and replay
125/// checks any open runs — and only a copy that opened is moved into place. Any
126/// *error* removes the staging directory (or, once files have started moving,
127/// undoes the moves already made) and leaves `db_dir` exactly as it was, so
128/// the error names the paths, the operator can fix the backup, and the next
129/// boot restores rather than reporting [`RestoreOutcome::AlreadyPresent`] over
130/// a half-written store. That unwind is process-local: a crash between the
131/// two renames that install the staged files (not a returned error, but the
132/// process dying) can leave `db_dir` holding one file but not the other. The
133/// next boot sees that partial store as already present and reports
134/// [`RestoreOutcome::AlreadyPresent`] rather than restoring over it — clear
135/// the directory and restore again.
136///
137/// Staging lives inside `db_dir` on purpose: `db_dir` is typically the mount
138/// point, so a sibling directory could land on another filesystem and turn the
139/// final moves into cross-device copies.
140///
141/// # Errors
142///
143/// Whatever creating `db_dir`, copying a file, opening the staged copy, or
144/// moving it into place returned, as [`GraphError::Io`] carrying the message
145/// verbatim — the paths are in the message, and a caller that must tell the
146/// causes apart has them in the text rather than in the variant.
147pub fn restore_if_empty(db_dir: &Path, from: &Path) -> Result<RestoreOutcome> {
148 if holds_a_store(db_dir) {
149 return Ok(RestoreOutcome::AlreadyPresent);
150 }
151 let Some(backup) = choose_backup(from) else {
152 return Ok(RestoreOutcome::Empty);
153 };
154
155 std::fs::create_dir_all(db_dir)
156 .map_err(|e| failed(format!("restore into {}: {e}", db_dir.display())))?;
157
158 // Named for this process, so two `serve` processes racing onto one fresh
159 // volume stage into separate directories rather than over each other.
160 let staging = db_dir.join(format!(".restore-{}", std::process::id()));
161 let _ = std::fs::remove_dir_all(&staging); // a previous run that was killed
162 std::fs::create_dir_all(&staging)
163 .map_err(|e| failed(format!("restore into {}: {e}", staging.display())))?;
164
165 let outcome = stage_and_install(db_dir, &staging, &backup);
166 // Whether it worked or not: the staging directory never outlives the call.
167 // On success it holds only what opening the copy created (`LOCK`), since
168 // the restored files were moved out of it.
169 let _ = std::fs::remove_dir_all(&staging);
170 outcome
171}
172
173/// A restore failure, carrying `msg` and nothing else.
174///
175/// `std::io::Error::other` displays as its message alone, so the text a caller
176/// sees is the text built here — which is what lets the CLI keep printing the
177/// messages it always has.
178fn failed(msg: String) -> GraphError {
179 GraphError::Io(std::io::Error::other(msg))
180}
181
182/// Copy `backup` into `staging`, prove it opens, then move it into `db_dir`.
183///
184/// Split out of [`restore_if_empty`] so every early return runs through one
185/// cleanup of `staging` at the call site.
186fn stage_and_install(db_dir: &Path, staging: &Path, backup: &Path) -> Result<RestoreOutcome> {
187 let mut names: Vec<String> = RESTORE_FILES.iter().map(|n| n.to_string()).collect();
188 let mut archives: Vec<String> = std::fs::read_dir(backup)
189 .map_err(|e| failed(format!("restore from {}: {e}", backup.display())))?
190 .flatten()
191 .filter_map(|e| e.file_name().into_string().ok())
192 .filter(|n| n.starts_with("wal.") && n.ends_with(".archive"))
193 .collect();
194 archives.sort();
195 names.extend(archives);
196
197 let mut files = Vec::new();
198 let mut bytes = 0u64;
199 for name in names {
200 let src = backup.join(&name);
201 if !src.is_file() {
202 continue;
203 }
204 let n = std::fs::copy(&src, staging.join(&name))
205 .map_err(|e| failed(format!("restore {} from {}: {e}", name, backup.display())))?;
206 bytes += n;
207 files.push(name);
208 }
209
210 // Prove the copy opens before `serve` gets it. A copy that does not open is
211 // a hard failure, not a silent empty start — and because it opened in
212 // staging, nothing of it reaches `db_dir`.
213 GraphDb::<core_storage::fs::RealFs>::open(staging).map_err(|e| {
214 failed(format!(
215 "restore into {} from {} failed: the copy does not open: {e}",
216 db_dir.display(),
217 backup.display()
218 ))
219 })?;
220
221 // The copy is good. Move it in. A rename within one directory tree is the
222 // closest thing to atomic the filesystem offers; if one still fails, undo
223 // the moves already made so `db_dir` is left as it was found.
224 let mut moved: Vec<&String> = Vec::new();
225 for name in &files {
226 if let Err(e) = std::fs::rename(staging.join(name), db_dir.join(name)) {
227 for done in &moved {
228 let _ = std::fs::remove_file(db_dir.join(done));
229 }
230 return Err(failed(format!(
231 "restore into {} from {}: installing {name}: {e}",
232 db_dir.display(),
233 backup.display()
234 )));
235 }
236 moved.push(name);
237 }
238
239 Ok(RestoreOutcome::Restored {
240 from: backup.to_path_buf(),
241 files,
242 bytes,
243 })
244}