Skip to main content

treeship_core/keys/
mod.rs

1use std::{
2    collections::HashMap,
3    fs,
4    io::{self, Read, Write},
5    path::{Path, PathBuf},
6    sync::{Arc, RwLock},
7};
8
9use aes_gcm::{
10    aead::{Aead, KeyInit, OsRng as AeadOsRng, Payload},
11    AeadCore, Aes256Gcm, Key as AesKey, Nonce,
12};
13use rand::{rngs::OsRng, RngCore};
14use serde::{Deserialize, Serialize};
15use sha2::{Digest as Sha2Digest, Sha256};
16use zeroize::Zeroizing;
17
18use crate::attestation::{Ed25519Signer, Signer};
19
20// --- Public types ---
21
22pub type KeyId = String;
23
24/// Public information about a stored key. Never contains private material.
25#[derive(Debug, Clone, Serialize, Deserialize)]
26pub struct KeyInfo {
27    pub id: KeyId,
28    pub algorithm: String, // "ed25519"
29    pub is_default: bool,
30    pub created_at: String, // RFC 3339
31    /// First 8 bytes of sha256(public_key), hex-encoded.
32    pub fingerprint: String,
33    pub public_key: Vec<u8>, // raw 32-byte Ed25519 public key
34    /// RFC 3339 timestamp after which signatures by this key should be
35    /// considered stale. `None` means the key has not been rotated and is
36    /// indefinitely valid. Set automatically by `Store::rotate` to
37    /// `now + grace_period` on the predecessor key.
38    #[serde(default, skip_serializing_if = "Option::is_none")]
39    pub valid_until: Option<String>,
40    /// If this key was rotated to a successor, the successor's key id.
41    /// Lets verifiers walk a rotation chain forward when validating an old
42    /// receipt against the current keystore. `None` means this is the head
43    /// of its chain.
44    #[serde(default, skip_serializing_if = "Option::is_none")]
45    pub successor_key_id: Option<KeyId>,
46}
47
48/// Outcome of a `Store::rotate` call.
49#[derive(Debug, Clone)]
50pub struct RotationResult {
51    /// The key that was rotated. Its `valid_until` is now set.
52    pub predecessor: KeyInfo,
53    /// The freshly minted successor key.
54    pub successor: KeyInfo,
55    /// RFC 3339 timestamp until which the predecessor remains valid for
56    /// signature verification under the grace period. Equal to
57    /// `predecessor.valid_until.unwrap()`.
58    pub grace_period_until: String,
59}
60
61/// Errors from keystore operations.
62#[derive(Debug)]
63pub enum KeyError {
64    Io(io::Error),
65    Json(serde_json::Error),
66    Crypto(String),
67    NotFound(KeyId),
68    EmptyKeyId,
69    NoDefaultKey,
70    /// Private key file has insecure permissions (group- or world-readable).
71    /// Carries the path and the observed octal mode so the caller can show
72    /// an actionable error. Set `TREESHIP_ALLOW_INSECURE_KEY_PERMS=1` to
73    /// bypass during testing or controlled environments.
74    InsecureKeyPerms {
75        path: PathBuf,
76        mode: u32,
77    },
78}
79
80impl std::fmt::Display for KeyError {
81    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
82        match self {
83            Self::Io(e) => write!(f, "keys io: {}", e),
84            Self::Json(e) => write!(f, "keys json: {}", e),
85            Self::Crypto(e) => write!(f, "keys crypto: {}", e),
86            Self::NotFound(k) => write!(f, "key not found: {}", k),
87            Self::EmptyKeyId => write!(f, "key id must not be empty"),
88            Self::NoDefaultKey => write!(f, "no default key — run treeship init"),
89            Self::InsecureKeyPerms { path, mode } => write!(
90                f,
91                "private key {} has insecure permissions (mode {:o}); \
92                 run `treeship doctor --fix` or chmod 600 the file. \
93                 Set TREESHIP_ALLOW_INSECURE_KEY_PERMS=1 to bypass.",
94                path.display(),
95                mode & 0o777,
96            ),
97        }
98    }
99}
100
101impl std::error::Error for KeyError {}
102impl From<io::Error> for KeyError {
103    fn from(e: io::Error) -> Self {
104        Self::Io(e)
105    }
106}
107impl From<serde_json::Error> for KeyError {
108    fn from(e: serde_json::Error) -> Self {
109        Self::Json(e)
110    }
111}
112
113// --- On-disk formats ---
114
115/// The encrypted representation of one keypair on disk.
116#[derive(Serialize, Deserialize, Clone)]
117struct EncryptedEntry {
118    id: KeyId,
119    algorithm: String,
120    created_at: String,
121    public_key: Vec<u8>,
122    /// AES-256-GCM ciphertext of the 32-byte Ed25519 secret scalar.
123    enc_priv_key: Vec<u8>,
124    /// 12-byte GCM nonce used when encrypting.
125    nonce: Vec<u8>,
126    /// RFC 3339 timestamp after which signatures by this key should be
127    /// considered stale. `None` means the key is indefinitely valid.
128    /// Defaulted on deserialization so pre-0.9.5 entry files still load.
129    #[serde(default, skip_serializing_if = "Option::is_none")]
130    valid_until: Option<String>,
131    /// Successor key id if this key was rotated. Defaulted on
132    /// deserialization for pre-0.9.5 entry files.
133    #[serde(default, skip_serializing_if = "Option::is_none")]
134    successor_key_id: Option<KeyId>,
135}
136
137/// The manifest file: which keys exist and which is the default.
138#[derive(Serialize, Deserialize, Default)]
139struct Manifest {
140    default_key_id: Option<KeyId>,
141    key_ids: Vec<KeyId>,
142}
143
144// --- Store ---
145
146/// Local encrypted keystore.
147///
148/// Private keys are encrypted with AES-256-GCM (RustCrypto `aes-gcm`
149/// 0.10) before writing to disk. The encryption key is derived from a
150/// machine-specific secret so key files are useless if copied to
151/// another machine.
152///
153/// Pre-v0.10.3 keystores used a homemade SHA-256-CTR + HMAC-SHA-256
154/// construction (TS-2026-001) and are transparently migrated to the
155/// new AEAD format on first decrypt; see `encrypt_for_disk_v2` /
156/// `decrypt_from_disk` for the format dispatcher.
157///
158/// A future version will delegate to OS credential stores (Secure
159/// Enclave / TPM 2.0).
160pub struct Store {
161    dir: PathBuf,
162    machine_key: [u8; 32],
163    /// Decrypt-only fallback machine keys, tried in order when the primary
164    /// fails. These cover every wrapping an existing keystore may carry:
165    /// the v1 hostname+username key under the current hostname, the same
166    /// under the raw (non-canonicalized) path when the path contains a
167    /// symlink, and — on macOS — the v1 key under `scutil LocalHostName`
168    /// variants, because macOS renames `kern.hostname` out from under a
169    /// running machine (network collisions, DHCP) while `LocalHostName`
170    /// keeps the name the keystore was written under. Never used to
171    /// encrypt: any entry that decrypts via a fallback is transparently
172    /// rewrapped under the primary. See `open` and `signer`.
173    fallback_machine_keys: Vec<[u8; 32]>,
174    /// In-memory cache — avoids disk reads on hot paths.
175    cache: Arc<RwLock<HashMap<KeyId, EncryptedEntry>>>,
176}
177
178impl Store {
179    /// Opens or creates a keystore at `dir`.
180    pub fn open(dir: impl AsRef<Path>) -> Result<Self, KeyError> {
181        let dir = dir.as_ref().to_path_buf();
182        fs::create_dir_all(&dir)?;
183
184        // Canonicalize the keystore path before deriving the machine key. The
185        // derivation hashes the store path into the key, so the SAME logical
186        // directory must produce the SAME path string every time -- otherwise
187        // `init` and a later command can hash different strings for one
188        // directory (e.g. macOS `/var` -> `/private/var`, or a symlinked
189        // `$HOME`) and decryption fails with a misleading "wrong machine" MAC
190        // error. canonicalize resolves symlinks to a stable absolute path;
191        // create_dir_all above guarantees it exists.
192        //
193        // The raw-path key is retained as a DECRYPT-ONLY fallback so any
194        // keystore written before this change (encrypted under the raw path)
195        // still opens -- this hardening must never lock an existing user out.
196        // Encryption always uses the canonical key, so entries migrate to it
197        // as they are rewritten.
198        let canonical = fs::canonicalize(&dir).unwrap_or_else(|_| dir.clone());
199
200        // Primary (encrypt) key: hardware-stable when the machine offers a
201        // stable identifier (/etc/machine-id, IOPlatformSerialNumber), so a
202        // hostname rename can never invalidate the keystore again. Machines
203        // with neither identifier keep the v1 derivation, whose seed-file
204        // fallback is CO-LOCATED with the keystore -- switching those to the
205        // stable derivation would move their seed to the global
206        // ~/.treeship/.internal/ and silently break project-local keystore
207        // isolation (the v0.9.6 property).
208        // AUD-19: the PRIMARY (encrypt) key derives from a SECRET OsRng seed,
209        // not from guessable machine identifiers. Before this the wrapping key
210        // was `SHA256(machine-id/serial/hostname ‖ store_path)` — all guessable
211        // — so an exfiltrated `keys/*.json` was forgeable by anyone who knew
212        // the victim's hostname or machine-id. The old guessable derivations
213        // are now DECRYPT-ONLY fallbacks below; the first successful fallback
214        // decrypt rewraps the entry under this seed key (see `signer`), so an
215        // existing keystore migrates transparently on first open.
216        //
217        // Durability note: the machine_seed file is now load-bearing — losing
218        // it makes an already-migrated keystore unrecoverable (the price of
219        // real at-rest confidentiality; the old guessable key was recoverable
220        // precisely because it was guessable).
221        let machine_key = derive_seed_primary_key(&canonical)?;
222
223        // Decrypt-only fallbacks, most-likely first. Existing keystores are
224        // wrapped under one of these; the first successful decrypt rewraps
225        // the entry under the primary (see `signer`), so the fallbacks are
226        // a migration path, not a permanent second key.
227        let mut fallback_machine_keys: Vec<[u8; 32]> = Vec::new();
228        // The former PRIMARY: hardware-stable key (machine-id / serial).
229        if let Some(k) = stable_hardware_key(&canonical) {
230            fallback_machine_keys.push(k);
231        }
232        // v1 under the current hostname (every pre-migration keystore).
233        if let Ok(k) = derive_machine_key(&canonical) {
234            fallback_machine_keys.push(k);
235        }
236        // v1 under the raw path, for keystores written before path
237        // canonicalization through a symlink.
238        if canonical != dir {
239            if let Ok(k) = derive_machine_key(&dir) {
240                fallback_machine_keys.push(k);
241            }
242        }
243        // macOS: v1 under the mDNS LocalHostName variants. When macOS
244        // renames kern.hostname (the usual keystore-bricking event),
245        // LocalHostName typically still holds the name the store was
246        // written under, so these recover a drifted keystore with no
247        // user action.
248        if let Ok(user) = std::env::var("USER") {
249            for h in local_hostname_variants() {
250                fallback_machine_keys.push(derive_machine_key_v1_from_parts(&h, &user, &canonical));
251            }
252        }
253        // Order-preserving dedupe (Vec::dedup only folds adjacent repeats),
254        // and drop any candidate equal to the primary -- retrying the same
255        // key can only re-fail.
256        let mut seen: Vec<[u8; 32]> = vec![machine_key];
257        fallback_machine_keys.retain(|k| {
258            if seen.contains(k) {
259                false
260            } else {
261                seen.push(*k);
262                true
263            }
264        });
265
266        Ok(Self {
267            dir,
268            machine_key,
269            fallback_machine_keys,
270            cache: Arc::new(RwLock::new(HashMap::new())),
271        })
272    }
273
274    /// Generates a new Ed25519 keypair, encrypts and stores it.
275    /// If `set_default` is true (or there is no current default), makes
276    /// this key the default signing key.
277    pub fn generate(&self, set_default: bool) -> Result<KeyInfo, KeyError> {
278        let key_id = new_key_id();
279
280        let signer =
281            Ed25519Signer::generate(&key_id).map_err(|e| KeyError::Crypto(e.to_string()))?;
282
283        // `secret` is a Zeroizing<[u8; 32]> -- the caller-side copy of the
284        // signer's secret scalar is wiped on scope exit. `signer` is dropped
285        // at end of fn, which wipes its own copy via the Drop impl in
286        // attestation::signer.
287        let secret = signer.secret_bytes();
288        let pub_key = signer.public_key_bytes();
289
290        let enc = encrypt_for_disk_v2(
291            &self.machine_key,
292            key_id.as_str(),
293            &pub_key,
294            secret.as_slice(),
295        )
296        .map_err(KeyError::Crypto)?;
297
298        let entry = EncryptedEntry {
299            id: key_id.clone(),
300            algorithm: "ed25519".into(),
301            created_at: crate::statements::unix_to_rfc3339(unix_now()),
302            public_key: pub_key.clone(),
303            enc_priv_key: enc,
304            // v2 ciphertexts carry their nonce inline (bytes [2..14]).
305            // The separate `nonce` field is retained for v1 legacy
306            // compatibility; for fresh v2 entries we serialize an empty
307            // vec so the JSON stays well-formed.
308            nonce: Vec::new(),
309            valid_until: None,
310            successor_key_id: None,
311        };
312
313        self.write_entry(&entry)?;
314
315        // Update manifest.
316        let mut manifest = self.read_manifest()?;
317        manifest.key_ids.push(key_id.clone());
318        if set_default || manifest.default_key_id.is_none() {
319            manifest.default_key_id = Some(key_id.clone());
320        }
321        self.write_manifest(&manifest)?;
322
323        // Populate cache.
324        self.cache.write().unwrap().insert(key_id.clone(), entry);
325
326        Ok(KeyInfo {
327            id: key_id.clone(),
328            algorithm: "ed25519".into(),
329            is_default: manifest.default_key_id.as_deref() == Some(key_id.as_str()),
330            created_at: crate::statements::unix_to_rfc3339(unix_now()),
331            fingerprint: fingerprint(&pub_key),
332            public_key: pub_key,
333            valid_until: None,
334            successor_key_id: None,
335        })
336    }
337
338    /// Rotate the current default key (or a specific key) to a freshly
339    /// generated successor.
340    ///
341    /// Mints a new Ed25519 keypair, links the predecessor to it via
342    /// `successor_key_id`, and stamps the predecessor with a `valid_until`
343    /// of `now + grace_period`. The grace window lets verifiers continue to
344    /// accept signatures from the predecessor while clients catch up to
345    /// the new public key.
346    ///
347    /// If `set_default` is true (the typical case -- you rotate because you
348    /// want to start signing with the new key immediately), the successor
349    /// becomes the default. Pass `false` to stage a rotation for review
350    /// without flipping the active signer.
351    ///
352    /// `predecessor_id` may be `None` to rotate the current default. Pass
353    /// an explicit id to rotate a non-default key (e.g. a per-environment
354    /// secondary).
355    ///
356    /// Note on threat model: this is a graceful rotation primitive, not a
357    /// revocation primitive. If the predecessor key is suspected compromised
358    /// the grace_period should be `Duration::ZERO` (or use a future
359    /// `revoke()` call once that lands) so the predecessor's `valid_until`
360    /// is in the past and any verifier honoring the metadata refuses
361    /// further signatures from it.
362    pub fn rotate(
363        &self,
364        predecessor_id: Option<&str>,
365        grace_period: std::time::Duration,
366        set_default: bool,
367    ) -> Result<RotationResult, KeyError> {
368        // Resolve predecessor: explicit id, else the current default.
369        let pred_id = match predecessor_id {
370            Some(id) => id.to_string(),
371            None => self.default_key_id()?,
372        };
373
374        // Refuse to rotate a key that has already been rotated -- the
375        // chain head is the only valid rotation source. This makes the
376        // operation idempotent in the face of accidental re-runs.
377        let pred_entry_existing = self.load_entry(&pred_id)?;
378        if let Some(existing) = &pred_entry_existing.successor_key_id {
379            return Err(KeyError::Crypto(format!(
380                "key {pred_id} has already been rotated to {existing}; \
381                 rotate the chain head instead"
382            )));
383        }
384
385        // Mint the successor. We deliberately do NOT call `self.generate()`
386        // because that path also updates the manifest's default. We need a
387        // single transactional update that sets both predecessor metadata
388        // AND (optionally) the new default in one manifest write.
389        let succ_id = new_key_id();
390        let signer =
391            Ed25519Signer::generate(&succ_id).map_err(|e| KeyError::Crypto(e.to_string()))?;
392        // `succ_secret` is a Zeroizing<[u8; 32]>; the caller-side copy is
393        // wiped on scope exit, and `signer` is dropped at end of fn (which
394        // wipes its own copy via the attestation::signer Drop impl).
395        let succ_secret = signer.secret_bytes();
396        let succ_pub_key = signer.public_key_bytes();
397        let succ_enc = encrypt_for_disk_v2(
398            &self.machine_key,
399            succ_id.as_str(),
400            &succ_pub_key,
401            succ_secret.as_slice(),
402        )
403        .map_err(KeyError::Crypto)?;
404
405        let succ_created = crate::statements::unix_to_rfc3339(unix_now());
406        let succ_entry = EncryptedEntry {
407            id: succ_id.clone(),
408            algorithm: "ed25519".into(),
409            created_at: succ_created.clone(),
410            public_key: succ_pub_key.clone(),
411            enc_priv_key: succ_enc,
412            // v2 ciphertexts carry their nonce inline; the legacy
413            // `nonce` field is left empty for fresh writes.
414            nonce: Vec::new(),
415            valid_until: None,
416            successor_key_id: None,
417        };
418
419        // Stamp the predecessor with the grace deadline and link forward.
420        let valid_until = crate::statements::unix_to_rfc3339(unix_now() + grace_period.as_secs());
421        let mut pred_entry = pred_entry_existing;
422        pred_entry.valid_until = Some(valid_until.clone());
423        pred_entry.successor_key_id = Some(succ_id.clone());
424
425        // Write order matters for partial-failure recovery. Persist the
426        // successor entry FIRST, then stamp the predecessor pointing at
427        // it. If we wrote the predecessor first and then the successor
428        // write failed, the predecessor's successor_key_id would dangle
429        // at a key that doesn't exist on disk -- and the
430        // already-been-rotated guard would refuse to retry. With this
431        // order:
432        //   - successor write fails: nothing observable changed; retry clean.
433        //   - predecessor write fails: orphan successor key file on disk
434        //     (not yet referenced by manifest or by any other key); retry
435        //     generates a new successor and the orphan is harmless.
436        //   - manifest write fails: predecessor + successor both on disk,
437        //     manifest stale; retry's already-rotated guard catches the
438        //     half-finished state and surfaces a clear error.
439        self.write_entry(&succ_entry)?;
440        self.write_entry(&pred_entry)?;
441
442        // Refresh the cache to mirror the on-disk state we just wrote --
443        // BEFORE the manifest update. If the manifest write fails, the
444        // cache must still match disk so a same-process retry sees the
445        // half-rotated state and the already-rotated guard fires
446        // correctly. Doing this AFTER write_manifest would leave a
447        // window where disk reflects the rotation but the in-memory
448        // cache still serves the unstamped predecessor, and a retry
449        // from the same Store instance would generate a duplicate
450        // successor -- defeating the whole point of the guard.
451        {
452            let mut cache = self.cache.write().unwrap();
453            cache.insert(pred_entry.id.clone(), pred_entry.clone());
454            cache.insert(succ_id.clone(), succ_entry.clone());
455        }
456
457        // Update the manifest: register the new key, optionally promote it.
458        let mut manifest = self.read_manifest()?;
459        manifest.key_ids.push(succ_id.clone());
460        if set_default {
461            manifest.default_key_id = Some(succ_id.clone());
462        }
463        self.write_manifest(&manifest)?;
464
465        let default_id = manifest.default_key_id.clone();
466        let predecessor = KeyInfo {
467            id: pred_entry.id.clone(),
468            algorithm: pred_entry.algorithm.clone(),
469            is_default: default_id.as_deref() == Some(pred_entry.id.as_str()),
470            created_at: pred_entry.created_at.clone(),
471            fingerprint: fingerprint(&pred_entry.public_key),
472            public_key: pred_entry.public_key.clone(),
473            valid_until: pred_entry.valid_until.clone(),
474            successor_key_id: pred_entry.successor_key_id.clone(),
475        };
476        let successor = KeyInfo {
477            id: succ_id.clone(),
478            algorithm: "ed25519".into(),
479            is_default: default_id.as_deref() == Some(succ_id.as_str()),
480            created_at: succ_created,
481            fingerprint: fingerprint(&succ_pub_key),
482            public_key: succ_pub_key,
483            valid_until: None,
484            successor_key_id: None,
485        };
486
487        Ok(RotationResult {
488            predecessor,
489            successor,
490            grace_period_until: valid_until,
491        })
492    }
493
494    /// Walk the rotation chain forward from `id`, returning the ordered
495    /// list of key ids: `[id, successor_of_id, ...]`. The first element is
496    /// always `id` itself. Stops at a key with no `successor_key_id`.
497    pub fn successor_chain(&self, id: &str) -> Result<Vec<KeyId>, KeyError> {
498        let mut chain = Vec::new();
499        let mut cursor = id.to_string();
500        // Cap iterations at the manifest size to defend against a corrupt
501        // chain that loops back on itself. A well-formed chain is bounded
502        // by the number of keys in the keystore.
503        let max_steps = self.read_manifest()?.key_ids.len() + 1;
504        for _ in 0..max_steps {
505            chain.push(cursor.clone());
506            let entry = self.load_entry(&cursor)?;
507            match entry.successor_key_id {
508                Some(next) => cursor = next,
509                None => return Ok(chain),
510            }
511        }
512        Err(KeyError::Crypto(format!(
513            "rotation chain starting at {id} exceeds keystore size; suspected loop"
514        )))
515    }
516
517    /// Returns the `KeyInfo` for every key whose `valid_until` is either
518    /// unset or strictly after `at_unix_secs`. The result includes both
519    /// rotated-but-still-in-grace predecessors and never-rotated keys.
520    /// Useful for building a verifier's accept-set as of a given time.
521    pub fn valid_keys_at(&self, at_unix_secs: u64) -> Result<Vec<KeyInfo>, KeyError> {
522        let cutoff_rfc = crate::statements::unix_to_rfc3339(at_unix_secs);
523        Ok(self
524            .list()?
525            .into_iter()
526            .filter(|k| match &k.valid_until {
527                None => true,
528                Some(until) => until.as_str() > cutoff_rfc.as_str(),
529            })
530            .collect())
531    }
532
533    /// Returns a boxed `Signer` for the current default key.
534    pub fn default_signer(&self) -> Result<Box<dyn Signer>, KeyError> {
535        let manifest = self.read_manifest()?;
536        let id = manifest.default_key_id.ok_or(KeyError::NoDefaultKey)?;
537        self.signer(&id)
538    }
539
540    /// Returns a boxed `Signer` for a specific key ID.
541    ///
542    /// Refuses to load if the on-disk key file has insecure permissions
543    /// (any group or world bits). This is the choke point for *all*
544    /// signing — public-key reads and successor lookups go through
545    /// `read_entry` / `public_key` and are not affected.
546    ///
547    /// Bypass with `TREESHIP_ALLOW_INSECURE_KEY_PERMS=1` for controlled
548    /// environments (CI sandboxes, recovery flows). The bypass should
549    /// not be set in normal operation.
550    ///
551    /// TOCTOU note: the perm-check and the ciphertext read run against
552    /// the SAME file descriptor (open once, fstat, then read from that
553    /// fd). The previous shape — `check_key_file_perms(path)` followed
554    /// by `load_entry(id)` (which called `fs::read(path)`) — opened the
555    /// file twice. An attacker with write access to `~/.treeship/keys/`
556    /// could swap the file between the two opens: first present an
557    /// owner-only file to pass the perm gate, then replace it with a
558    /// different (loose-perm) file containing an attacker-controlled
559    /// scalar before the second `open`. The single-fd shape closes that
560    /// window because the inode is pinned by the open file descriptor;
561    /// path-level swaps after the open don't affect what we read. This
562    /// matches the pattern in `session/event_log.rs::open_lock_file`.
563    pub fn signer(&self, id: &str) -> Result<Box<dyn Signer>, KeyError> {
564        let entry = self.read_entry_with_perm_check(id)?;
565
566        // Dispatcher: v2 ciphertexts start with magic 0x54, version 0x02
567        // and use real AES-256-GCM. Older entries fall through to the
568        // legacy SHA-256-CTR+HMAC path (`decrypt_legacy_v1`) and are
569        // transparently re-encrypted in the new format below.
570        let was_legacy = is_legacy_v1(&entry.enc_priv_key);
571        let mut used_fallback = false;
572        let secret = match decrypt_from_disk(
573            &self.machine_key,
574            &entry.id,
575            &entry.public_key,
576            &entry.enc_priv_key,
577            &entry.nonce,
578        ) {
579            Ok(secret) => secret,
580            Err(primary_err) => {
581                // The entry may be wrapped under an older machine-key
582                // derivation: the v1 hostname key (any pre-stable keystore),
583                // the raw-path key (pre-canonicalization through a symlink),
584                // or a v1 key under a hostname macOS has since renamed away
585                // (the LocalHostName candidates). Try each in order; the
586                // first hit marks the entry for rewrapping under the
587                // primary. All misses surface the PRIMARY error, enriched,
588                // so the diagnosis is unchanged for normal failures.
589                let mut recovered = None;
590                for candidate in &self.fallback_machine_keys {
591                    if let Ok(secret) = decrypt_from_disk(
592                        candidate,
593                        &entry.id,
594                        &entry.public_key,
595                        &entry.enc_priv_key,
596                        &entry.nonce,
597                    ) {
598                        recovered = Some(secret);
599                        used_fallback = true;
600                        break;
601                    }
602                }
603                match recovered {
604                    Some(secret) => secret,
605                    None => return Err(self.enrich_crypto_error(primary_err)),
606                }
607            }
608        };
609
610        // L3: wrap the on-stack copy of the decrypted secret in a
611        // `Zeroizing` so the byte buffer is wiped on drop. `secret`
612        // itself is already a `Zeroizing<Vec<u8>>` returned by
613        // `decrypt_from_disk`, but `try_into::<[u8; 32]>` produces an
614        // independent stack-allocated array that the Vec's Drop will
615        // not cover. Without this wrapper, returning from `signer()`
616        // would leave the secret scalar in stale stack memory until
617        // a future stack frame happens to overwrite it.
618        let secret_arr: Zeroizing<[u8; 32]> = Zeroizing::new(
619            secret
620                .as_slice()
621                .try_into()
622                .map_err(|_| KeyError::Crypto("decrypted key is wrong length".into()))?,
623        );
624
625        // Transparent migration: if this entry was still in the legacy
626        // v1 format (the broken SHA-256-CTR construction from
627        // TS-2026-001), re-encrypt it with v2 AES-256-GCM and rewrite
628        // the file. We do this best-effort -- a migration failure here
629        // must NOT block signing for the current call, since the
630        // in-memory secret is already valid. The next decrypt on a
631        // fresh process will retry.
632        if was_legacy || used_fallback {
633            if let Err(e) = self.migrate_entry_to_primary(&entry, &secret_arr) {
634                // Surface the failure as a tracing-style stderr note
635                // rather than an error -- the user's signing flow is
636                // unaffected, and we'd rather them know about it than
637                // wedge the call.
638                eprintln!(
639                    "treeship: keystore entry {} could not be rewrapped \
640                     under the current machine key ({}); will retry next \
641                     load",
642                    entry.id, e
643                );
644            }
645        }
646
647        let signer = Ed25519Signer::from_bytes(&entry.id, &secret_arr)
648            .map_err(|e| KeyError::Crypto(e.to_string()))?;
649
650        Ok(Box::new(signer))
651    }
652
653    /// Re-encrypt a legacy v1 entry with the new v2 AEAD and persist
654    /// it. Updates the in-memory cache so subsequent loads in the same
655    /// process see the migrated entry. Idempotent; safe to invoke
656    /// concurrently because the migration is serialized by a per-entry
657    /// advisory lock on `<entry>.migrate.lock` (TS-2026-001 H3).
658    ///
659    /// We lock a *sentinel* file rather than the entry file itself,
660    /// because the entry file is renamed-into-place during the atomic
661    /// write inside `write_entry`. Holding a flock on the entry's inode
662    /// while a sibling process renames a new inode into its path is
663    /// nonsensical (the lock would survive on the now-orphaned inode);
664    /// the sentinel sidecar has a stable identity for the whole
665    /// migration window.
666    ///
667    /// Same blocking-flock pattern as `packages/core/src/session/event_log.rs`
668    /// (Lane F): exclusive lock, then a same-thread re-read to settle
669    /// "did a peer already migrate while I was waiting?" cleanly.
670    fn migrate_entry_to_primary(
671        &self,
672        old_entry: &EncryptedEntry,
673        secret: &[u8; 32],
674    ) -> Result<(), KeyError> {
675        let entry_path = self.entry_path(&old_entry.id);
676        let lock_path = entry_path.with_extension("migrate.lock");
677
678        // Open (or create) the sentinel lock file with restrictive perms
679        // and take an exclusive flock. We intentionally use the blocking
680        // `lock_exclusive` -- not `try_lock_exclusive` -- because the
681        // migration window is short (a single AEAD encrypt + atomic
682        // rename) and the worst case under contention is one writer
683        // serialized behind another. Pulling the
684        // try-with-bounded-retry pattern in here would buy us nothing:
685        // the second writer's re-read after the lock releases would
686        // observe the now-v2 entry and short-circuit.
687        let lock_file = open_migration_lock_file(&lock_path).map_err(KeyError::Io)?;
688
689        #[cfg(not(target_family = "wasm"))]
690        {
691            use fs2::FileExt;
692            lock_file.lock_exclusive().map_err(KeyError::Io)?;
693        }
694
695        // Under the lock: did a peer already complete the migration
696        // while we were waiting? If so, our work is done -- we must
697        // NOT rewrite, because we'd overwrite a peer's freshly-rotated
698        // v2 ciphertext with our own (semantically equivalent, but
699        // unnecessary I/O and an unnecessary cache update).
700        if let Ok(current) = self.read_entry(&old_entry.id) {
701            // "Already migrated" now means: v2 format AND decryptable under
702            // the PRIMARY machine key. The format check alone is not enough
703            // since this path also rewraps v2 entries that a fallback
704            // machine key decrypted (hostname drift, raw-path legacy); the
705            // primary-decrypt probe is what proves a peer finished the job.
706            let already_primary = !is_legacy_v1(&current.enc_priv_key)
707                && decrypt_from_disk(
708                    &self.machine_key,
709                    &current.id,
710                    &current.public_key,
711                    &current.enc_priv_key,
712                    &current.nonce,
713                )
714                .is_ok();
715            if already_primary {
716                // Peer already migrated. Refresh the cache so subsequent
717                // loads in this process see the rewrapped entry rather
718                // than the stale copy our caller passed in.
719                if let Ok(mut cache) = self.cache.write() {
720                    cache.insert(current.id.clone(), current);
721                }
722                // Lock drops at function exit; sentinel file remains on
723                // disk as a harmless inode (no migration data, idempotent
724                // for future invocations).
725                return Ok(());
726            }
727        }
728
729        let new_ciphertext = encrypt_for_disk_v2(
730            &self.machine_key,
731            &old_entry.id,
732            &old_entry.public_key,
733            secret,
734        )
735        .map_err(KeyError::Crypto)?;
736
737        let migrated = EncryptedEntry {
738            id: old_entry.id.clone(),
739            algorithm: old_entry.algorithm.clone(),
740            created_at: old_entry.created_at.clone(),
741            public_key: old_entry.public_key.clone(),
742            enc_priv_key: new_ciphertext,
743            // v2 carries the nonce inline; clear the legacy field.
744            nonce: Vec::new(),
745            valid_until: old_entry.valid_until.clone(),
746            successor_key_id: old_entry.successor_key_id.clone(),
747        };
748
749        self.write_entry(&migrated)?;
750        if let Ok(mut cache) = self.cache.write() {
751            cache.insert(migrated.id.clone(), migrated);
752        }
753
754        // Best-effort cleanup of the sentinel lock file. We hold the
755        // lock until function exit (drop), so by the time we reach
756        // here it is safe to unlink the inode -- future migrations
757        // for this entry will succeed via the early-return path
758        // because the entry is now v2. Leaving the sentinel behind is
759        // also harmless; on Unix removing a flocked file is allowed
760        // and the lock is released on fd drop regardless.
761        let _ = std::fs::remove_file(&lock_path);
762
763        // Keep the lock_file binding alive to function exit so the
764        // flock is held across write_entry + remove_file. Explicit
765        // drop makes the intent obvious to readers.
766        drop(lock_file);
767        Ok(())
768    }
769
770    /// Wrap a bare crypto error (typically "MAC verification failed ..." from
771    /// the AES-GCM decrypt path) with a diagnostic and an actionable recovery
772    /// path.
773    ///
774    /// The common failure mode in the wild is a pre-0.9.x keystore whose
775    /// machine-key derivation was seed-file-based. Later versions derive
776    /// the machine key from hostname+username (macOS) or /etc/machine-id
777    /// (Linux), so old ciphertexts can't be MAC-verified with the new key.
778    /// Detecting that case is best-effort: the presence of a legacy seed
779    /// file (`.machineseed` or `machine_seed` inside the keys dir) is a
780    /// strong hint. If we see one, call it out explicitly.
781    fn enrich_crypto_error(&self, raw: String) -> KeyError {
782        // Only enrich on MAC failures -- other errors (I/O, wrong length) are
783        // surfaced as-is because their remediation differs.
784        if !raw.contains("MAC verification failed") {
785            return KeyError::Crypto(raw);
786        }
787
788        let legacy_seed_dot = self.dir.join(".machineseed");
789        let legacy_seed = self.dir.join("machine_seed");
790        let has_legacy_seed = legacy_seed_dot.exists() || legacy_seed.exists();
791
792        let diagnosis = if has_legacy_seed {
793            "your keystore was created by an older Treeship version whose \
794             machine-key derivation has since changed. The ciphertext is \
795             intact but cannot be decrypted under the current derivation."
796        } else {
797            "the keystore cannot be decrypted under any known machine-key \
798             derivation (hardware id, current hostname, mDNS LocalHostName, \
799             raw path). Usual causes: the key file was copied from a \
800             different machine, the username changed, or the file was \
801             corrupted."
802        };
803
804        // Name the keystore that actually failed, not the default one.
805        //
806        // This used to read $HOME/.treeship unconditionally, which is the
807        // store root only in the default layout. `self.dir` is the keys
808        // directory (KeyStore::open is always called with cfg.keys_dir), so
809        // under `--config`, or any custom keys_dir, the old message told the
810        // user to move a keystore that was working and leave the broken one
811        // in place: advice that damages a good store and fixes nothing.
812        //
813        // `mv` on a store directory is exactly the operation that has
814        // scrambled state here before -- see resolve_dirs in the CLI's
815        // config.rs, where one machine reached six .bak directories and
816        // three ship identities. Aiming it at the wrong directory is not a
817        // cosmetic flaw in the message; it is the message handing over a
818        // footgun pointed somewhere else.
819        let keys_dir = self.dir.display().to_string();
820
821        // A non-default store needs --config on the way back in, or
822        // `treeship init` re-creates the *default* store and the user is
823        // still broken, now with an extra keystore.
824        //
825        // --force is required rather than optional: moving the keys aside
826        // leaves config.json behind, and plain `init` then refuses with
827        // "already initialized" while `attest` says "no default key -- run
828        // treeship init". Verified: without --force the two commands send
829        // the user in a circle.
830        // Compare canonicalized paths, not strings: on macOS $HOME and the
831        // resolved store path can differ by the /tmp -> /private/tmp symlink,
832        // and a string compare then reports a default store as custom. The
833        // custom branch is still correct when that happens (it just names a
834        // --config that was already implied), so this only sharpens the
835        // message; it cannot make it wrong.
836        let canon = |p: &Path| std::fs::canonicalize(p).unwrap_or_else(|_| p.to_path_buf());
837        let default_keys = std::env::var("HOME")
838            .ok()
839            .map(|h| canon(&PathBuf::from(h).join(".treeship").join("keys")));
840        let init_cmd = if default_keys.as_deref() == Some(canon(&self.dir).as_path()) {
841            "treeship init --force".to_string()
842        } else {
843            let root = self
844                .dir
845                .parent()
846                .map(|p| p.display().to_string())
847                .unwrap_or_else(|| keys_dir.clone());
848            format!("treeship --config {root}/config.json init --force")
849        };
850
851        // The outer KeyError::Crypto Display impl already prepends
852        // "keys crypto: "; don't double it. Start with the raw MAC error
853        // so the user still sees the underlying cryptographic reason,
854        // then follow with the human-readable diagnosis and recovery.
855        let msg = format!(
856            "{raw}\n\n  \
857             Diagnosis: {diagnosis}\n\n  \
858             Recovery (reversible -- the old keystore is moved aside, not \
859             deleted, and moving it back restores the previous state):\n\n    \
860             mv {keys_dir} {keys_dir}.bak.$(date +%s)\n    \
861             {init_cmd}\n\n  \
862             After this you sign under a new key. Receipts already signed by \
863             the old key verify as `unknown key` until that keystore is \
864             restored, so keep the .bak directory. Sealed .treeship packages \
865             are unaffected -- they embed the public key they were signed \
866             with.\n"
867        );
868
869        KeyError::Crypto(msg)
870    }
871
872    /// Returns the default key ID.
873    pub fn default_key_id(&self) -> Result<KeyId, KeyError> {
874        self.read_manifest()?
875            .default_key_id
876            .ok_or(KeyError::NoDefaultKey)
877    }
878
879    /// Lists all keys.
880    pub fn list(&self) -> Result<Vec<KeyInfo>, KeyError> {
881        let manifest = self.read_manifest()?;
882        let default = manifest.default_key_id.as_deref().unwrap_or("");
883
884        manifest
885            .key_ids
886            .iter()
887            .map(|id| {
888                let entry = self.load_entry(id)?;
889                Ok(KeyInfo {
890                    id: entry.id.clone(),
891                    algorithm: entry.algorithm.clone(),
892                    is_default: entry.id == default,
893                    created_at: entry.created_at.clone(),
894                    fingerprint: fingerprint(&entry.public_key),
895                    public_key: entry.public_key.clone(),
896                    valid_until: entry.valid_until.clone(),
897                    successor_key_id: entry.successor_key_id.clone(),
898                })
899            })
900            .collect()
901    }
902
903    /// Sets the default signing key.
904    pub fn set_default(&self, id: &str) -> Result<(), KeyError> {
905        // Verify the key exists before updating the manifest.
906        self.load_entry(id)?;
907        let mut manifest = self.read_manifest()?;
908        manifest.default_key_id = Some(id.to_string());
909        self.write_manifest(&manifest)
910    }
911
912    /// Returns the public key bytes for a key ID.
913    pub fn public_key(&self, id: &str) -> Result<Vec<u8>, KeyError> {
914        Ok(self.load_entry(id)?.public_key)
915    }
916
917    /// Encrypt an arbitrary secret for at-rest storage OUTSIDE the keystore
918    /// (for example, the hub DPoP signing key that lives in `config.json`).
919    ///
920    /// The secret is sealed under this machine's key with the same
921    /// AES-256-GCM v2 framing the keystore uses for private keys, so a
922    /// stolen `config.json` is useless on another machine — the same
923    /// guarantee AGENTS.md §7 already makes for the ship key. `context` is
924    /// bound as AEAD associated data: a blob sealed for one context (e.g.
925    /// `"hub-dpop:v1:<hub_id>"`) will not decrypt under another, so a local
926    /// attacker cannot swap a ciphertext between two hub connections in the
927    /// same file. Store the returned bytes base64-encoded; recover the
928    /// plaintext with [`KeyStore::decrypt_secret`] using the same `context`.
929    pub fn encrypt_secret(&self, context: &str, plaintext: &[u8]) -> Result<Vec<u8>, KeyError> {
930        // public_key is empty: for a non-keystore secret there is no
931        // associated pubkey to bind, but `context` (carried as the AAD
932        // entry_id) plus the framing prefix still bind machine + purpose.
933        encrypt_for_disk_v2(&self.machine_key, context, &[], plaintext).map_err(KeyError::Crypto)
934    }
935
936    /// Decrypt a blob produced by [`KeyStore::encrypt_secret`] with the same
937    /// `context`. Tries the primary machine key first, then the same
938    /// migration fallbacks used for keystore entries, so a machine whose
939    /// hostname/username drifted still recovers the secret. A wrong
940    /// `context`, a tampered blob, or a different machine each fail closed
941    /// with a MAC error rather than returning wrong bytes.
942    pub fn decrypt_secret(&self, context: &str, blob: &[u8]) -> Result<Vec<u8>, KeyError> {
943        match decrypt_v2(&self.machine_key, context, &[], blob) {
944            Ok(pt) => Ok(pt),
945            Err(primary_err) => {
946                for candidate in &self.fallback_machine_keys {
947                    if let Ok(pt) = decrypt_v2(candidate, context, &[], blob) {
948                        return Ok(pt);
949                    }
950                }
951                Err(KeyError::Crypto(primary_err))
952            }
953        }
954    }
955
956    // --- private ---
957
958    fn load_entry(&self, id: &str) -> Result<EncryptedEntry, KeyError> {
959        // Check cache first.
960        if let Ok(cache) = self.cache.read() {
961            if let Some(entry) = cache.get(id) {
962                return Ok(entry.clone());
963            }
964        }
965        self.read_entry(id)
966    }
967
968    fn entry_path(&self, id: &str) -> PathBuf {
969        self.dir.join(format!("{}.json", id))
970    }
971
972    fn write_entry(&self, entry: &EncryptedEntry) -> Result<(), KeyError> {
973        let path = self.entry_path(&entry.id);
974        let json = serde_json::to_vec_pretty(entry)?;
975        write_file_600(&path, &json)?;
976        Ok(())
977    }
978
979    fn read_entry(&self, id: &str) -> Result<EncryptedEntry, KeyError> {
980        let path = self.entry_path(id);
981        if !path.exists() {
982            return Err(KeyError::NotFound(id.to_string()));
983        }
984        let bytes = fs::read(&path)?;
985        let entry: EncryptedEntry = serde_json::from_slice(&bytes)?;
986        Ok(entry)
987    }
988
989    /// Single-open, race-free counterpart to `read_entry` for the
990    /// signing path. Opens the key file ONCE, fstat's the file
991    /// descriptor to check perms, then reads the JSON from the SAME
992    /// descriptor. The path is never re-resolved after the open, so an
993    /// attacker who swaps `<id>.json` on disk between the perm check
994    /// and the ciphertext read cannot influence the bytes we decrypt.
995    ///
996    /// Cache: this path intentionally skips the in-memory entry cache.
997    /// The cache is read-mostly and seeded by `load_entry`, which is
998    /// fine for public-key lookups but defeats the perm gate (a cached
999    /// entry would let `signer()` return without ever consulting the
1000    /// on-disk perms). The signing path is rare enough that the extra
1001    /// disk read is not a hot spot.
1002    fn read_entry_with_perm_check(&self, id: &str) -> Result<EncryptedEntry, KeyError> {
1003        let path = self.entry_path(id);
1004
1005        // Open once. NotFound surfaces as `KeyError::NotFound` to
1006        // match the legacy `read_entry` shape; any other I/O error
1007        // (permission denied at the *open* layer, EIO, etc.)
1008        // propagates via the `From<io::Error>` impl.
1009        let mut file = match fs::File::open(&path) {
1010            Ok(f) => f,
1011            Err(e) if e.kind() == io::ErrorKind::NotFound => {
1012                return Err(KeyError::NotFound(id.to_string()));
1013            }
1014            Err(e) => return Err(KeyError::Io(e)),
1015        };
1016
1017        // Perm check on the open fd. On Unix `File::metadata` is
1018        // documented to call `fstat` on the underlying fd, which pins
1019        // the inode -- a subsequent path swap on disk cannot change
1020        // what we see. The bypass env var continues to short-circuit.
1021        check_open_key_file_perms(&path, &file)?;
1022
1023        // Read the full ciphertext envelope from the same fd.
1024        let mut bytes = Vec::new();
1025        file.read_to_end(&mut bytes)?;
1026
1027        let entry: EncryptedEntry = serde_json::from_slice(&bytes)?;
1028        Ok(entry)
1029    }
1030
1031    fn manifest_path(&self) -> PathBuf {
1032        self.dir.join("manifest.json")
1033    }
1034
1035    fn read_manifest(&self) -> Result<Manifest, KeyError> {
1036        let path = self.manifest_path();
1037        if !path.exists() {
1038            return Ok(Manifest::default());
1039        }
1040        let bytes = fs::read(&path)?;
1041        Ok(serde_json::from_slice(&bytes)?)
1042    }
1043
1044    fn write_manifest(&self, m: &Manifest) -> Result<(), KeyError> {
1045        let json = serde_json::to_vec_pretty(m)?;
1046        write_file_600(&self.manifest_path(), &json)?;
1047        Ok(())
1048    }
1049}
1050
1051// --- Crypto helpers ---
1052//
1053// AEAD choice: AES-256-GCM via the RustCrypto `aes-gcm` 0.10 crate.
1054// Reasons:
1055//   - Matches the original (documented but never implemented) intent of
1056//     the keystore, so audit reports and SECURITY.md don't need to be
1057//     re-anchored on a different primitive.
1058//   - Well-audited, widely deployed, no platform gotchas.
1059//   - `chacha20poly1305` would have been a defensible alternative
1060//     (slightly better software performance), but the migration cost of
1061//     changing the documented primitive while we already have to ship a
1062//     migration for the broken construction is not worth it.
1063//
1064// On-disk v2 format (`encrypt_for_disk_v2`):
1065//   [ magic = 0x54 ('T') ]   1 byte
1066//   [ version = 0x02     ]   1 byte
1067//   [ nonce              ]  12 bytes (random per encryption)
1068//   [ ciphertext || tag  ]  N + 16 bytes (tag appended by aead crate)
1069//
1070// The first byte (0x54) is a structural sentinel so we can dispatch on
1071// the format without relying on length heuristics. v1 ciphertexts start
1072// with the first byte of their random nonce, so the chance of an
1073// accidental v1 entry that looks like v2 is ~1/2^16 (matching both magic
1074// AND version byte) and we still re-validate by AEAD-decrypting; if the
1075// AEAD fails on something that looks like v2, we fall back to v1.
1076
1077const KEYSTORE_MAGIC: u8 = 0x54; // 'T'
1078const KEYSTORE_VERSION_V2: u8 = 0x02;
1079
1080/// Build the v2 keystore AEAD AAD.
1081///
1082/// The AAD binds two things into the GCM tag beyond ciphertext+nonce:
1083///
1084/// 1. **Framing prefix** (`[KEYSTORE_MAGIC, KEYSTORE_VERSION_V2]`) so
1085///    flipping the magic or version byte on disk surfaces as a MAC
1086///    failure rather than dispatcher confusion (the M2 audit finding).
1087/// 2. **Entry identity** (`entry_id` and `public_key`) so an attacker
1088///    with write access to `~/.treeship/keys/` cannot copy entry A's
1089///    `enc_priv_key` ciphertext into entry B's JSON envelope. Without
1090///    this binding, the swap would decrypt cleanly (same machine key,
1091///    same framing-only AAD) and the signer for advertised key id A
1092///    would silently sign with key B's secret scalar — un-binding
1093///    `KeyInfo.public_key` from the actual scalar in use. This closes
1094///    the "intra-keystore swap" class flagged in the post-merge audit
1095///    of TS-2026-001.
1096///
1097/// Every variable-length field is length-prefixed with a big-endian
1098/// u32 before its bytes. Concatenating variable-length fields without
1099/// length prefixes is a forgery class (an attacker who controls field
1100/// boundaries can shift bytes between fields and present a different
1101/// `(entry_id, public_key)` pair whose AAD-bytes serialize identically).
1102/// `entry_id` is a fixed-prefix `key_<hex>` string in practice, but we
1103/// length-prefix it anyway to defend against future id schemes.
1104///
1105/// The AAD must be byte-identical on encrypt and decrypt. Future
1106/// versions (V3+) get their own builder; the dispatcher picks which
1107/// to use based on the framing prefix.
1108fn build_aad_v2(entry_id: &str, public_key: &[u8]) -> Vec<u8> {
1109    let mut aad = Vec::with_capacity(2 + 4 + entry_id.len() + 4 + public_key.len());
1110    aad.push(KEYSTORE_MAGIC);
1111    aad.push(KEYSTORE_VERSION_V2);
1112    aad.extend_from_slice(&(entry_id.len() as u32).to_be_bytes());
1113    aad.extend_from_slice(entry_id.as_bytes());
1114    aad.extend_from_slice(&(public_key.len() as u32).to_be_bytes());
1115    aad.extend_from_slice(public_key);
1116    aad
1117}
1118
1119/// AES-256-GCM (the real one) encrypt for at-rest keystore storage.
1120/// Returns the framed v2 blob ready to drop into `EncryptedEntry::enc_priv_key`.
1121///
1122/// Output: `[magic, version, nonce(12), ciphertext || tag(16)]`.
1123///
1124/// The AEAD's Associated Authenticated Data binds:
1125/// - the framing prefix (M2 — flipping magic/version surfaces as MAC failure)
1126/// - the entry id and public key (post-merge audit fix-up — closes the
1127///   intra-keystore swap class where a local attacker copies entry A's
1128///   `enc_priv_key` into entry B's JSON envelope).
1129///
1130/// See `build_aad_v2` for the exact layout. `entry_id` and `public_key`
1131/// must match what gets serialized into the `EncryptedEntry` JSON;
1132/// `decrypt_for_disk_v2` reads them back from the deserialized entry
1133/// to recompute the AAD.
1134fn encrypt_for_disk_v2(
1135    key: &[u8; 32],
1136    entry_id: &str,
1137    public_key: &[u8],
1138    plaintext: &[u8],
1139) -> Result<Vec<u8>, String> {
1140    // Wrap the in-memory AEAD key in Zeroizing so the local stack copy
1141    // is wiped on drop. The aes-gcm cipher object owns its own internal
1142    // expanded key schedule; that's outside our control, but the raw
1143    // 32-byte buffer at this scope is ours to clear.
1144    let key_buf: Zeroizing<[u8; 32]> = Zeroizing::new(*key);
1145    let aead_key: &AesKey<Aes256Gcm> = AesKey::<Aes256Gcm>::from_slice(key_buf.as_slice());
1146    let cipher = Aes256Gcm::new(aead_key);
1147
1148    // 96-bit random nonce from the OS CSPRNG.
1149    let nonce = Aes256Gcm::generate_nonce(&mut AeadOsRng);
1150
1151    let aad = build_aad_v2(entry_id, public_key);
1152    let ciphertext = cipher
1153        .encrypt(
1154            &nonce,
1155            Payload {
1156                msg: plaintext,
1157                aad: aad.as_slice(),
1158            },
1159        )
1160        .map_err(|e| format!("aead encrypt failed: {e}"))?;
1161
1162    let mut out = Vec::with_capacity(2 + 12 + ciphertext.len());
1163    out.push(KEYSTORE_MAGIC);
1164    out.push(KEYSTORE_VERSION_V2);
1165    out.extend_from_slice(nonce.as_slice());
1166    out.extend_from_slice(&ciphertext);
1167    Ok(out)
1168}
1169
1170/// AES-256-GCM decrypt of a v2 framed blob. Uses the same AAD binding
1171/// as `encrypt_for_disk_v2`:
1172///   - framing prefix (so a tampered magic/version surfaces as MAC failure)
1173///   - entry id + public key (so swapping `enc_priv_key` between entries
1174///     in the same keystore surfaces as MAC failure).
1175///
1176/// `entry_id` and `public_key` come from the `EncryptedEntry` JSON
1177/// envelope that holds `blob`. The caller is responsible for passing the
1178/// *envelope's* id and pubkey, not values from some other source — that
1179/// is precisely what binds the ciphertext to its envelope.
1180fn decrypt_v2(
1181    key: &[u8; 32],
1182    entry_id: &str,
1183    public_key: &[u8],
1184    blob: &[u8],
1185) -> Result<Vec<u8>, String> {
1186    // Minimum: magic(1) + version(1) + nonce(12) + tag(16) = 30 bytes.
1187    if blob.len() < 30 {
1188        return Err("v2 ciphertext too short".into());
1189    }
1190    if blob[0] != KEYSTORE_MAGIC || blob[1] != KEYSTORE_VERSION_V2 {
1191        return Err("v2 ciphertext has wrong magic/version".into());
1192    }
1193    let nonce_bytes = &blob[2..14];
1194    let ct = &blob[14..];
1195
1196    let key_buf: Zeroizing<[u8; 32]> = Zeroizing::new(*key);
1197    let aead_key: &AesKey<Aes256Gcm> = AesKey::<Aes256Gcm>::from_slice(key_buf.as_slice());
1198    let cipher = Aes256Gcm::new(aead_key);
1199    let nonce = Nonce::from_slice(nonce_bytes);
1200
1201    let aad = build_aad_v2(entry_id, public_key);
1202    cipher
1203        .decrypt(
1204            nonce,
1205            Payload {
1206                msg: ct,
1207                aad: aad.as_slice(),
1208            },
1209        )
1210        .map_err(|_| "MAC verification failed — key file may be corrupt or wrong machine".into())
1211}
1212
1213/// Returns true iff `blob` is shaped like a v1 (legacy) ciphertext.
1214/// Used by the dispatcher to decide whether a successful decrypt should
1215/// trigger a transparent re-encrypt to v2.
1216fn is_legacy_v1(blob: &[u8]) -> bool {
1217    // A v2 blob always starts with [magic, version]. Anything else
1218    // (including the empty enc_priv_key case during partial writes) is
1219    // treated as legacy and routed through the v1 path, which will fail
1220    // cleanly on garbage.
1221    !(blob.len() >= 2 && blob[0] == KEYSTORE_MAGIC && blob[1] == KEYSTORE_VERSION_V2)
1222}
1223
1224/// Top-level decrypt dispatcher used by the keystore. Tries v2 if the
1225/// blob carries the magic+version prefix, otherwise falls through to the
1226/// legacy v1 path. If a blob looks like v2 but AEAD verification fails,
1227/// we also try v1 — this defends against the (negligible) probability
1228/// that a legacy ciphertext's random first two bytes happen to collide
1229/// with our magic+version.
1230///
1231/// M1 (TS-2026-001 audit): when the blob is v2-shaped and BOTH the v2
1232/// AEAD and the v1 fallback fail, surface the v2 error rather than the
1233/// v1 error. v1's failure on a v2-shaped blob is mechanical (wrong
1234/// MAC computed under the wrong construction) and tells the user
1235/// nothing useful; v2's failure is the actually-relevant signal
1236/// (MAC verification under the documented AEAD). The previous code
1237/// would mask the meaningful error with a confused legacy error
1238/// message that pointed at the wrong remediation.
1239fn decrypt_from_disk(
1240    key: &[u8; 32],
1241    entry_id: &str,
1242    public_key: &[u8],
1243    enc_data: &[u8],
1244    legacy_nonce_field: &[u8],
1245) -> Result<Zeroizing<Vec<u8>>, String> {
1246    if !is_legacy_v1(enc_data) {
1247        match decrypt_v2(key, entry_id, public_key, enc_data) {
1248            Ok(pt) => return Ok(Zeroizing::new(pt)),
1249            Err(v2_err) => {
1250                // Collision fallback. v1 entries had random first bytes;
1251                // there's a vanishing chance one looks like v2 framing.
1252                // Try v1 first; if it succeeds we have a legitimate
1253                // legacy entry whose framing happens to look v2-shaped.
1254                // If v1 also fails, surface the v2 error (the
1255                // semantically meaningful one) rather than v1's
1256                // mechanical-junk failure.
1257                return match decrypt_legacy_v1(key, enc_data, legacy_nonce_field) {
1258                    Ok(pt) => Ok(Zeroizing::new(pt)),
1259                    Err(_) => Err(v2_err),
1260                };
1261            }
1262        }
1263    }
1264    decrypt_legacy_v1(key, enc_data, legacy_nonce_field).map(Zeroizing::new)
1265}
1266
1267/// DEPRECATED: legacy at-rest decryption for keystores written before
1268/// v0.10.3. This is the SHA-256-CTR + HMAC-SHA-256 construction that
1269/// was mis-labelled as AES-256-GCM (TS-2026-001). The CTR keystream is
1270/// also degenerate (the same `enc_key` byte is reused once per
1271/// plaintext byte, since `block[i % 32]` indexes the same SHA-256 output
1272/// modulo 32), so the construction is NOT a real stream cipher even
1273/// ignoring the AEAD mislabelling.
1274///
1275/// Kept ONLY to migrate existing on-disk keystores forward to the v2
1276/// AEAD format. Never call this for new writes. The encrypt counterpart
1277/// has been removed from the v2 codepath — the only place v1
1278/// ciphertexts come from is files written by older Treeship versions.
1279pub fn aes_gcm_decrypt(
1280    key: &[u8; 32],
1281    enc_data: &[u8],
1282    _nonce_unused: &[u8],
1283) -> Result<Vec<u8>, String> {
1284    // Preserved as a public symbol because the `treeship-vi` sibling
1285    // crate calls it directly. vi only ever produces v1 ciphertexts
1286    // (its `aes_gcm_encrypt` shim calls `legacy_v1_encrypt`) and has
1287    // no concept of the `EncryptedEntry` envelope that carries the
1288    // entry id + public key the v2 AAD now requires. Route this shim
1289    // directly through the legacy v1 path so vi's call site keeps
1290    // working byte-for-byte; vi's eventual migration release will
1291    // adopt its own AEAD path with its own envelope binding.
1292    decrypt_legacy_v1(key, enc_data, _nonce_unused)
1293}
1294
1295/// DEPRECATED: legacy at-rest encryption. Same caveats as
1296/// `aes_gcm_decrypt`. Kept ONLY as a public symbol for compatibility
1297/// with the `treeship-vi` sibling crate; the core keystore no longer
1298/// produces v1 ciphertexts.
1299///
1300/// New code MUST use `encrypt_for_disk_v2`. This function still
1301/// produces v1-format output so the vi crate's on-disk format remains
1302/// byte-stable until it migrates on its own cadence.
1303pub fn aes_gcm_encrypt(key: &[u8; 32], plaintext: &[u8]) -> Result<(Vec<u8>, Vec<u8>), String> {
1304    legacy_v1_encrypt(key, plaintext)
1305}
1306
1307/// Legacy v1 encrypt. SHA-256-CTR + HMAC-SHA-256. DO NOT USE for new
1308/// writes — present only so vi-keystore callers keep working until
1309/// they migrate. See `aes_gcm_encrypt` doc-comment for the security
1310/// caveats.
1311fn legacy_v1_encrypt(key: &[u8; 32], plaintext: &[u8]) -> Result<(Vec<u8>, Vec<u8>), String> {
1312    use sha2::Sha256;
1313
1314    let mut nonce = [0u8; 12];
1315    // v0.10.4 P1 audit: nonce reuse breaks AEAD. Read directly from the OS
1316    // CSPRNG via OsRng rather than the userland thread_rng, which can mis-seed
1317    // across forks / on some WASM targets. Legacy v1 write path is kept for
1318    // treeship-vi byte-stability but still needs sound nonces.
1319    OsRng.fill_bytes(&mut nonce);
1320
1321    let mut enc_key_input = key.to_vec();
1322    enc_key_input.extend_from_slice(&nonce);
1323    enc_key_input.extend_from_slice(b"enc");
1324    let enc_key = Sha256::digest(&enc_key_input);
1325
1326    let mut mac_key_input = key.to_vec();
1327    mac_key_input.extend_from_slice(&nonce);
1328    mac_key_input.extend_from_slice(b"mac");
1329    let mac_key = Sha256::digest(&mac_key_input);
1330
1331    let ciphertext: Vec<u8> = plaintext
1332        .iter()
1333        .enumerate()
1334        .map(|(i, &b)| {
1335            let mut block_input = enc_key.to_vec();
1336            block_input.extend_from_slice(&(i as u64).to_le_bytes());
1337            let block = Sha256::digest(&block_input);
1338            b ^ block[i % 32]
1339        })
1340        .collect();
1341
1342    let mut mac_input = mac_key.to_vec();
1343    mac_input.extend_from_slice(&nonce);
1344    mac_input.extend_from_slice(&ciphertext);
1345    let mac = Sha256::digest(&mac_input);
1346
1347    let mut out = Vec::with_capacity(12 + 32 + ciphertext.len());
1348    out.extend_from_slice(&nonce);
1349    out.extend_from_slice(&mac);
1350    out.extend_from_slice(&ciphertext);
1351
1352    Ok((out, nonce.to_vec()))
1353}
1354
1355/// Legacy v1 decrypt. SHA-256-CTR + HMAC-SHA-256. See the module-level
1356/// notes on TS-2026-001 for why this is broken; kept only to migrate
1357/// existing keystores forward.
1358fn decrypt_legacy_v1(
1359    key: &[u8; 32],
1360    enc_data: &[u8],
1361    _nonce_unused: &[u8],
1362) -> Result<Vec<u8>, String> {
1363    if enc_data.len() < 44 {
1364        return Err("ciphertext too short".into());
1365    }
1366    use sha2::Sha256;
1367
1368    let nonce = &enc_data[..12];
1369    let stored_mac = &enc_data[12..44];
1370    let ciphertext = &enc_data[44..];
1371
1372    let nonce_arr: [u8; 12] = nonce.try_into().unwrap();
1373
1374    let mut enc_key_input = key.to_vec();
1375    enc_key_input.extend_from_slice(&nonce_arr);
1376    enc_key_input.extend_from_slice(b"enc");
1377    let enc_key = Sha256::digest(&enc_key_input);
1378
1379    let mut mac_key_input = key.to_vec();
1380    mac_key_input.extend_from_slice(&nonce_arr);
1381    mac_key_input.extend_from_slice(b"mac");
1382    let mac_key = Sha256::digest(&mac_key_input);
1383
1384    let mut mac_input = mac_key.to_vec();
1385    mac_input.extend_from_slice(&nonce_arr);
1386    mac_input.extend_from_slice(ciphertext);
1387    let computed_mac = Sha256::digest(&mac_input);
1388
1389    let mac_ok = stored_mac
1390        .iter()
1391        .zip(computed_mac.iter())
1392        .fold(0u8, |acc, (a, b)| acc | (a ^ b))
1393        == 0;
1394
1395    if !mac_ok {
1396        return Err("MAC verification failed — key file may be corrupt or wrong machine".into());
1397    }
1398
1399    let plaintext: Vec<u8> = ciphertext
1400        .iter()
1401        .enumerate()
1402        .map(|(i, &b)| {
1403            let mut block_input = enc_key.to_vec();
1404            block_input.extend_from_slice(&(i as u64).to_le_bytes());
1405            let block = Sha256::digest(&block_input);
1406            b ^ block[i % 32]
1407        })
1408        .collect();
1409
1410    Ok(plaintext)
1411}
1412
1413// --- Machine key derivation ---
1414
1415pub fn derive_machine_key(store_dir: &Path) -> Result<[u8; 32], KeyError> {
1416    // 1. Linux: /etc/machine-id (stable across reboots)
1417    if let Ok(id) = fs::read_to_string("/etc/machine-id") {
1418        let trimmed = id.trim();
1419        if !trimmed.is_empty() {
1420            let mut h = Sha256::new();
1421            h.update(trimmed.as_bytes());
1422            h.update(store_dir.to_string_lossy().as_bytes());
1423            return Ok(h.finalize().into());
1424        }
1425    }
1426
1427    // 2. macOS: hostname + username derivation (v1, backward compatible).
1428    //
1429    // TODO(v0.7.0): Migrate to IOPlatformSerialNumber-based derivation.
1430    // The serial number is more stable (survives hostname and username
1431    // changes), but switching now would silently invalidate all existing
1432    // keys on macOS. A proper migration needs to:
1433    //   1. Try the new derivation first.
1434    //   2. On decryption failure, fall back to hostname+username.
1435    //   3. If legacy succeeds, re-encrypt with the new key and save.
1436    // Until that migration tooling is in place, keep hostname+username
1437    // as the primary derivation so existing users are not locked out.
1438    #[cfg(target_os = "macos")]
1439    {
1440        let hostname = std::process::Command::new("hostname")
1441            .output()
1442            .map(|o| String::from_utf8_lossy(&o.stdout).trim().to_string())
1443            .unwrap_or_default();
1444        let username = std::env::var("USER").unwrap_or_default();
1445        if !hostname.is_empty() && !username.is_empty() {
1446            return Ok(derive_machine_key_v1_from_parts(
1447                &hostname, &username, store_dir,
1448            ));
1449        }
1450    }
1451
1452    // 3. Fallback: random seed file. Co-located with the keystore so a
1453    //    project-local keystore (/proj/.treeship/keys/) keeps its seed at
1454    //    /proj/.treeship/machine_seed -- never reaching for ~/.treeship.
1455    //    A global keystore (~/.treeship/keys/) co-locates to
1456    //    ~/.treeship/machine_seed, which is byte-identical to the
1457    //    pre-v0.9.6 location, so existing global keystores keep working.
1458    //
1459    //    Backward-compat read order:
1460    //      1. <store_dir>/../machine_seed  (the new co-located path)
1461    //      2. ~/.treeship/machine_seed     (the old hardcoded path)
1462    //    Write order on first creation:
1463    //      1. <store_dir>/../machine_seed  if the parent exists/is writable
1464    //      2. ~/.treeship/machine_seed     as a last resort
1465    //
1466    //    This makes project-local config truly self-contained: an
1467    //    isolated /proj keystore can decrypt its own keys even when
1468    //    the user's ~/.treeship is corrupt or on a different machine,
1469    //    closing the trust-fabric isolation gap that blocked
1470    //    project-local smoke tests.
1471    let seed = read_or_create_machine_seed(store_dir)?;
1472
1473    let mut h = Sha256::new();
1474    h.update(b"treeship-machine-key-fallback:");
1475    h.update(seed.trim().as_bytes());
1476    h.update(b":");
1477    h.update(store_dir.to_string_lossy().as_bytes());
1478    Ok(h.finalize().into())
1479}
1480
1481/// Read the secret machine seed, creating it (OsRng, mode 0600) on first use.
1482/// Returns the hex-encoded seed string.
1483///
1484/// The seed is co-located with the keystore so a project-local keystore
1485/// (`/proj/.treeship/keys/`) keeps its seed at `/proj/.treeship/machine_seed`
1486/// and a global keystore at `~/.treeship/machine_seed` (byte-identical to the
1487/// pre-v0.9.6 location, so existing global keystores keep working).
1488///
1489/// AUD-19: as of the seed-primary wrapping change this is the PRIMARY entropy
1490/// for every at-rest signing-key wrap on every platform — not just a container
1491/// fallback — so it is security-critical. AUD-25: an existing seed file must be
1492/// a regular file the current user owns, with no group/world access; anything
1493/// else (a planted seed, loose perms, a symlink) is refused fail-closed rather
1494/// than trusted as key material.
1495fn read_or_create_machine_seed(store_dir: &Path) -> Result<String, KeyError> {
1496    let local_seed_path = store_dir.parent().map(|p| p.join("machine_seed"));
1497    let home = std::env::var("HOME")
1498        .map(std::path::PathBuf::from)
1499        .map_err(|_| KeyError::Crypto("HOME not set".to_string()))?;
1500    let global_seed_path = home.join(".treeship").join("machine_seed");
1501
1502    // Read order: co-located seed first, then the legacy global path.
1503    if let Some(local) = local_seed_path.as_ref().filter(|p| p.exists()) {
1504        check_seed_file_secure(local)?;
1505        return fs::read_to_string(local).map_err(KeyError::Io);
1506    }
1507    if global_seed_path.exists() {
1508        check_seed_file_secure(&global_seed_path)?;
1509        return fs::read_to_string(&global_seed_path).map_err(KeyError::Io);
1510    }
1511
1512    // First use: mint a 32-byte seed straight from the OS CSPRNG.
1513    let mut bytes = [0u8; 32];
1514    OsRng.fill_bytes(&mut bytes);
1515    let seed_hex = hex_encode(&bytes);
1516
1517    // Prefer creating the seed co-located with the keystore; fall back to the
1518    // global path only when the keystore has no usable parent (store_dir is
1519    // "/" or similar pathological input).
1520    let target = match local_seed_path.as_ref() {
1521        Some(p) => {
1522            let _ = fs::create_dir_all(p.parent().unwrap_or(Path::new(".")));
1523            p.clone()
1524        }
1525        None => {
1526            let _ = fs::create_dir_all(global_seed_path.parent().unwrap_or(Path::new(".")));
1527            global_seed_path.clone()
1528        }
1529    };
1530    fs::write(&target, &seed_hex).map_err(KeyError::Io)?;
1531    #[cfg(unix)]
1532    {
1533        use std::os::unix::fs::PermissionsExt;
1534        let _ = fs::set_permissions(&target, fs::Permissions::from_mode(0o600));
1535    }
1536    Ok(seed_hex)
1537}
1538
1539/// AUD-25: refuse to trust a machine_seed that is not a regular file owned by
1540/// the current user with no group/world access. On non-unix this only checks
1541/// that it is a regular file (no ownership model to consult). The
1542/// `TREESHIP_ALLOW_INSECURE_KEY_PERMS=1` escape hatch mirrors the keystore's.
1543fn check_seed_file_secure(path: &Path) -> Result<(), KeyError> {
1544    let meta = fs::symlink_metadata(path).map_err(KeyError::Io)?;
1545    if !meta.file_type().is_file() {
1546        // A symlink or special file could redirect the read to attacker bytes.
1547        return Err(KeyError::Crypto(format!(
1548            "machine_seed at {} is not a regular file (refusing to use it as key material)",
1549            path.display()
1550        )));
1551    }
1552    #[cfg(unix)]
1553    {
1554        use std::os::unix::fs::MetadataExt;
1555        use std::os::unix::fs::PermissionsExt;
1556        let bypass = std::env::var_os("TREESHIP_ALLOW_INSECURE_KEY_PERMS")
1557            .map(|v| v == "1")
1558            .unwrap_or(false);
1559        if !bypass {
1560            let mode = meta.permissions().mode() & 0o777;
1561            if mode & 0o077 != 0 {
1562                return Err(KeyError::InsecureKeyPerms {
1563                    path: path.to_path_buf(),
1564                    mode,
1565                });
1566            }
1567            if meta.uid() != nix_geteuid() {
1568                return Err(KeyError::Crypto(format!(
1569                    "machine_seed at {} is not owned by the current user (refusing to use a planted seed)",
1570                    path.display()
1571                )));
1572            }
1573        }
1574    }
1575    Ok(())
1576}
1577
1578/// Current effective uid, via libc. Kept tiny and unix-gated so the seed check
1579/// does not pull a new dependency.
1580#[cfg(unix)]
1581fn nix_geteuid() -> u32 {
1582    // SAFETY: geteuid is always successful and has no preconditions.
1583    unsafe { libc_geteuid() }
1584}
1585
1586#[cfg(unix)]
1587extern "C" {
1588    #[link_name = "geteuid"]
1589    fn libc_geteuid() -> u32;
1590}
1591
1592/// AUD-19: the PRIMARY at-rest wrapping key. Derives from the SECRET OsRng
1593/// machine seed (high-entropy, mode 0600), with the machine identifier mixed
1594/// in only as a binding salt — never as the sole entropy. This is what makes
1595/// an exfiltrated `keys/*.json` un-forgeable: before this the wrapping key was
1596/// `SHA256(guessable_machine_id ‖ store_path)`, so anyone who knew the victim's
1597/// hostname / machine-id could recompute it. Now the secret seed is required.
1598/// Two hosts with identical machine-id / hostname / user but different seeds
1599/// cannot decrypt each other's keystore.
1600fn derive_seed_primary_key(store_dir: &Path) -> Result<[u8; 32], KeyError> {
1601    let seed = read_or_create_machine_seed(store_dir)?;
1602    let mut h = Sha256::new();
1603    h.update(b"treeship-seed-primary-v1:");
1604    h.update(seed.trim().as_bytes()); // the secret (all the entropy)
1605    h.update(b":");
1606    h.update(machine_id_salt().as_bytes()); // binding salt only (non-secret)
1607    h.update(b":");
1608    h.update(store_dir.to_string_lossy().as_bytes());
1609    Ok(h.finalize().into())
1610}
1611
1612/// A best-effort stable machine identifier used ONLY as a binding salt in
1613/// [`derive_seed_primary_key`] (so a keystore is bound to the machine that
1614/// wrote it, in addition to the secret seed). Non-secret and possibly empty;
1615/// the security comes from the seed, not this.
1616fn machine_id_salt() -> String {
1617    if let Ok(id) = fs::read_to_string("/etc/machine-id") {
1618        let t = id.trim();
1619        if !t.is_empty() {
1620            return t.to_string();
1621        }
1622    }
1623    // Hostname is a weak, drift-prone salt but fine as a last resort — it only
1624    // binds, it does not gate (all the security is in the secret seed).
1625    std::process::Command::new("hostname")
1626        .output()
1627        .map(|o| String::from_utf8_lossy(&o.stdout).trim().to_string())
1628        .unwrap_or_default()
1629}
1630
1631/// The v1 hostname+username machine-key derivation, as a pure function of
1632/// its inputs. This is the exact construction the macOS branch of
1633/// [`derive_machine_key`] has always used; extracting it lets `Store::open`
1634/// derive decrypt-only fallback candidates for hostnames the machine no
1635/// longer reports (macOS renames `kern.hostname` on network collisions,
1636/// which used to brick the keystore) without shelling `hostname` twice.
1637pub fn derive_machine_key_v1_from_parts(
1638    hostname: &str,
1639    username: &str,
1640    store_dir: &Path,
1641) -> [u8; 32] {
1642    let mut h = Sha256::new();
1643    h.update(b"treeship-machine-key:");
1644    h.update(hostname.as_bytes());
1645    h.update(b":");
1646    h.update(username.as_bytes());
1647    h.update(b":");
1648    h.update(store_dir.to_string_lossy().as_bytes());
1649    h.finalize().into()
1650}
1651
1652/// Hostname candidates a drifted macOS keystore may be wrapped under.
1653///
1654/// `scutil --get LocalHostName` holds the user-visible mDNS name, which
1655/// usually retains the value `hostname` reported when the keystore was
1656/// written even after macOS renames `kern.hostname` (DHCP, name-collision
1657/// auto-renames). `hostname` historically reported it with and without the
1658/// `.local` suffix depending on network state, so both variants are
1659/// candidates. Non-macOS platforms have no such drift (machine-id is
1660/// stable) and return no candidates.
1661#[cfg(target_os = "macos")]
1662fn local_hostname_variants() -> Vec<String> {
1663    let lh = std::process::Command::new("scutil")
1664        .args(["--get", "LocalHostName"])
1665        .output()
1666        .map(|o| String::from_utf8_lossy(&o.stdout).trim().to_string())
1667        .unwrap_or_default();
1668    if lh.is_empty() {
1669        return Vec::new();
1670    }
1671    vec![format!("{lh}.local"), lh]
1672}
1673
1674#[cfg(not(target_os = "macos"))]
1675fn local_hostname_variants() -> Vec<String> {
1676    Vec::new()
1677}
1678
1679/// The hardware-identifier half of [`derive_machine_key_stable`]: machine-id
1680/// (Linux) or IOPlatformSerialNumber (macOS), `None` when the machine offers
1681/// neither. Split out so `Store::open` can pick a hardware-stable PRIMARY
1682/// key without inheriting the stable derivation's seed-file fallback, whose
1683/// seed lives under the global `~/.treeship/.internal/` and would break
1684/// project-local keystore isolation (the v1 seed is co-located with the
1685/// keystore on purpose).
1686fn stable_hardware_key(store_dir: &Path) -> Option<[u8; 32]> {
1687    if let Ok(id) = fs::read_to_string("/etc/machine-id") {
1688        let trimmed = id.trim();
1689        if !trimmed.is_empty() {
1690            let mut h = Sha256::new();
1691            h.update(b"treeship-machine-key-v2:");
1692            h.update(trimmed.as_bytes());
1693            h.update(b":");
1694            h.update(store_dir.to_string_lossy().as_bytes());
1695            return Some(h.finalize().into());
1696        }
1697    }
1698
1699    #[cfg(target_os = "macos")]
1700    {
1701        if let Ok(output) = std::process::Command::new("ioreg")
1702            .args(["-rd1", "-c", "IOPlatformExpertDevice"])
1703            .output()
1704        {
1705            let stdout = String::from_utf8_lossy(&output.stdout);
1706            for line in stdout.lines() {
1707                if line.contains("IOPlatformSerialNumber") {
1708                    if let Some(serial) = line.split('"').nth(3) {
1709                        if !serial.is_empty() {
1710                            let mut h = Sha256::new();
1711                            h.update(b"treeship-machine-key-v2:");
1712                            h.update(serial.as_bytes());
1713                            h.update(b":");
1714                            h.update(store_dir.to_string_lossy().as_bytes());
1715                            return Some(h.finalize().into());
1716                        }
1717                    }
1718                }
1719            }
1720        }
1721    }
1722
1723    None
1724}
1725
1726/// Stable machine key derivation for NEW keys (VI P-256, etc).
1727/// Uses hardware identifiers that survive hostname/user changes.
1728/// For legacy ship Ed25519 keys, use `derive_machine_key()` instead.
1729pub fn derive_machine_key_stable(store_dir: &Path) -> Result<[u8; 32], KeyError> {
1730    // 1./2. Hardware identifiers: /etc/machine-id (Linux) or
1731    //    IOPlatformSerialNumber (macOS) -- stable across hostname changes,
1732    //    user renames, non-interactive shells. Shared with `Store::open`'s
1733    //    primary-key selection via `stable_hardware_key`.
1734    if let Some(k) = stable_hardware_key(store_dir) {
1735        return Ok(k);
1736    }
1737
1738    // 3. Fallback: persistent random seed in ~/.treeship/.internal/
1739    //    Separate from key material. Mode 0600.
1740    let home = std::env::var("HOME")
1741        .map(std::path::PathBuf::from)
1742        .map_err(|_| KeyError::Crypto("HOME not set".to_string()))?;
1743    let seed_dir = home.join(".treeship").join(".internal");
1744    let _ = fs::create_dir_all(&seed_dir);
1745    #[cfg(unix)]
1746    {
1747        use std::os::unix::fs::PermissionsExt;
1748        let _ = fs::set_permissions(&seed_dir, fs::Permissions::from_mode(0o700));
1749    }
1750
1751    let seed_path = seed_dir.join("machine_seed_v2");
1752    let seed = if seed_path.exists() {
1753        fs::read_to_string(&seed_path).map_err(KeyError::Io)?
1754    } else {
1755        let mut bytes = [0u8; 32];
1756        // v0.10.4 P1 audit: machine_seed_v2 backs the v2 machine-key
1757        // fallback. Same OsRng rationale as the v1 seed above.
1758        OsRng.fill_bytes(&mut bytes);
1759        let seed_hex = hex_encode(&bytes);
1760        fs::write(&seed_path, &seed_hex).map_err(KeyError::Io)?;
1761        #[cfg(unix)]
1762        {
1763            use std::os::unix::fs::PermissionsExt;
1764            let _ = fs::set_permissions(&seed_path, fs::Permissions::from_mode(0o600));
1765        }
1766        seed_hex
1767    };
1768
1769    let mut h = Sha256::new();
1770    h.update(b"treeship-machine-key-v2-fallback:");
1771    h.update(seed.trim().as_bytes());
1772    h.update(b":");
1773    h.update(store_dir.to_string_lossy().as_bytes());
1774    Ok(h.finalize().into())
1775}
1776
1777// --- Utility ---
1778
1779fn new_key_id() -> KeyId {
1780    let mut b = [0u8; 8];
1781    // v0.10.4 P1 audit: key_id is mixed into AAD by encrypt_for_disk_v2, so
1782    // collisions or low-entropy ids would weaken the AAD binding. Use OsRng
1783    // directly so the id is OS-CSPRNG-quality even under fork or odd targets.
1784    OsRng.fill_bytes(&mut b);
1785    format!("key_{}", hex_encode(&b))
1786}
1787
1788fn fingerprint(pub_key: &[u8]) -> String {
1789    let h = Sha256::digest(pub_key);
1790    hex_encode(&h[..8])
1791}
1792
1793fn hex_encode(b: &[u8]) -> String {
1794    b.iter().fold(String::new(), |mut s, byte| {
1795        s.push_str(&format!("{:02x}", byte));
1796        s
1797    })
1798}
1799
1800/// Verify a private-key file has restrictive permissions before loading
1801/// it for signing. Returns `Ok(())` on non-Unix platforms, when the
1802/// `TREESHIP_ALLOW_INSECURE_KEY_PERMS=1` escape hatch is set, or when
1803/// the file is not group/world accessible. Otherwise returns
1804/// `KeyError::InsecureKeyPerms` with the offending path and mode.
1805///
1806/// **TOCTOU caveat:** this path-based check has an unavoidable race
1807/// window between the `stat` and any subsequent `open` of the same
1808/// path. New signing-path callers MUST use
1809/// `check_open_key_file_perms` (fstat on an already-open fd) instead;
1810/// this function is retained only for non-signing callers that
1811/// already accept the race (e.g. `treeship doctor` scanning the
1812/// keystore directory).
1813#[allow(dead_code)]
1814fn check_key_file_perms(path: &Path) -> Result<(), KeyError> {
1815    #[cfg(unix)]
1816    {
1817        use std::os::unix::fs::PermissionsExt;
1818        if std::env::var_os("TREESHIP_ALLOW_INSECURE_KEY_PERMS")
1819            .map(|v| v == "1")
1820            .unwrap_or(false)
1821        {
1822            return Ok(());
1823        }
1824        // Missing files are reported by the caller as NotFound -- don't
1825        // mask that with a perm error.
1826        let meta = match fs::metadata(path) {
1827            Ok(m) => m,
1828            Err(_) => return Ok(()),
1829        };
1830        let mode = meta.permissions().mode();
1831        if mode & 0o077 != 0 {
1832            return Err(KeyError::InsecureKeyPerms {
1833                path: path.to_path_buf(),
1834                mode,
1835            });
1836        }
1837    }
1838    let _ = path;
1839    Ok(())
1840}
1841
1842/// Race-free perm gate: runs `fstat` on an already-open `File` and
1843/// rejects if the mode has any group or world bits. Use this from the
1844/// signing path: open the key file once, hand the resulting `File` to
1845/// this function, then read from the SAME `File` -- the inode is
1846/// pinned by the open fd, so a path-level swap between perm-check and
1847/// read cannot influence what we end up decrypting.
1848///
1849/// `path` is carried only for error reporting; it is never re-opened.
1850/// The `TREESHIP_ALLOW_INSECURE_KEY_PERMS=1` bypass is honored
1851/// identically to `check_key_file_perms` so existing CI workflows keep
1852/// working.
1853#[allow(unused_variables)]
1854fn check_open_key_file_perms(path: &Path, file: &fs::File) -> Result<(), KeyError> {
1855    #[cfg(unix)]
1856    {
1857        use std::os::unix::fs::PermissionsExt;
1858        if std::env::var_os("TREESHIP_ALLOW_INSECURE_KEY_PERMS")
1859            .map(|v| v == "1")
1860            .unwrap_or(false)
1861        {
1862            return Ok(());
1863        }
1864        // `File::metadata` on Unix calls `fstat(fd)` -- it does NOT
1865        // re-resolve the path, so the result describes the same inode
1866        // we will read from. This is the structural property that
1867        // makes the gate race-free.
1868        let meta = file.metadata()?;
1869        let mode = meta.permissions().mode();
1870        if mode & 0o077 != 0 {
1871            return Err(KeyError::InsecureKeyPerms {
1872                path: path.to_path_buf(),
1873                mode,
1874            });
1875        }
1876    }
1877    Ok(())
1878}
1879
1880impl Store {
1881    /// Repair file permissions on the keystore directory and every file
1882    /// inside it: dir to 0700, key entry files and manifest to 0600.
1883    /// Used by `treeship doctor --fix`. No-op on non-Unix.
1884    ///
1885    /// Returns the list of (path, old_mode, new_mode) tuples for paths
1886    /// that were actually changed, so the caller can report what it did.
1887    pub fn fix_perms(&self) -> Result<Vec<(PathBuf, u32, u32)>, KeyError> {
1888        let mut changed: Vec<(PathBuf, u32, u32)> = Vec::new();
1889        #[cfg(unix)]
1890        {
1891            use std::os::unix::fs::PermissionsExt;
1892
1893            let dir_meta = fs::metadata(&self.dir)?;
1894            let dir_mode = dir_meta.permissions().mode() & 0o777;
1895            if dir_mode != 0o700 {
1896                fs::set_permissions(&self.dir, fs::Permissions::from_mode(0o700))?;
1897                changed.push((self.dir.clone(), dir_mode, 0o700));
1898            }
1899
1900            for entry in fs::read_dir(&self.dir)? {
1901                let entry = entry?;
1902                let path = entry.path();
1903                if !entry.file_type()?.is_file() {
1904                    continue;
1905                }
1906                let mode = entry.metadata()?.permissions().mode() & 0o777;
1907                if mode != 0o600 {
1908                    fs::set_permissions(&path, fs::Permissions::from_mode(0o600))?;
1909                    changed.push((path, mode, 0o600));
1910                }
1911            }
1912        }
1913        Ok(changed)
1914    }
1915}
1916
1917/// Open (or create) the per-entry migration sentinel lock file with
1918/// owner-only permissions (0o600 on Unix). The handle returned can be
1919/// passed to `fs2::FileExt::lock_exclusive` to serialize concurrent
1920/// v1->v2 migrations of the same entry across processes/threads
1921/// (TS-2026-001 H3).
1922///
1923/// On Unix the mode is set at creation via `OpenOptionsExt::mode` so the
1924/// sentinel never has a moment of looser perms. On non-Unix platforms the
1925/// file inherits parent ACLs (the keystore dir is owner-scoped already).
1926#[cfg(unix)]
1927fn open_migration_lock_file(path: &Path) -> Result<fs::File, io::Error> {
1928    use std::os::unix::fs::OpenOptionsExt;
1929    fs::OpenOptions::new()
1930        .create(true)
1931        .read(true)
1932        .write(true)
1933        .truncate(false)
1934        .mode(0o600)
1935        .open(path)
1936}
1937
1938#[cfg(not(unix))]
1939fn open_migration_lock_file(path: &Path) -> Result<fs::File, io::Error> {
1940    fs::OpenOptions::new()
1941        .create(true)
1942        .read(true)
1943        .write(true)
1944        .truncate(false)
1945        .open(path)
1946}
1947
1948/// Atomically write `data` to `path` with owner-only (0o600) permissions on
1949/// Unix.
1950///
1951/// TS-2026-001 H1 + H2: the prior implementation was truncate-then-write,
1952/// which destroys the original file if the process crashes mid-write. For
1953/// the keystore that's catastrophic -- a crash during transparent v1->v2
1954/// migration would leave a zero-byte (or partial) key entry on disk and
1955/// the private key would be unrecoverable. This implementation writes to
1956/// a sibling tmp file in the same directory, fsyncs the bytes through to
1957/// the platter, then performs a POSIX-atomic same-filesystem `rename(2)`.
1958/// A crash before the rename leaves the original file intact; the tmp
1959/// file is harmless garbage that the next successful write will overwrite.
1960///
1961/// The 0o600 mode is set at file *creation* via `OpenOptionsExt::mode`
1962/// so there is no window in which the file exists with looser perms.
1963/// The prior `set_permissions` post-write call is dropped because it was
1964/// redundant and gave the appearance (but not the substance) of safety.
1965fn write_file_600(path: &Path, data: &[u8]) -> Result<(), KeyError> {
1966    // Place the tmp file in the same directory as the final path so the
1967    // rename stays on the same filesystem (cross-FS renames are not atomic
1968    // and degrade to copy+unlink, defeating the whole point).
1969    let tmp_path = path.with_extension("tmp");
1970
1971    // Best-effort cleanup of any stale tmp from a prior crash before we
1972    // start writing. Ignored on error -- if it doesn't exist that's fine,
1973    // and if it can't be removed the OpenOptions call below will surface
1974    // the underlying error.
1975    let _ = fs::remove_file(&tmp_path);
1976
1977    let write_result: Result<(), KeyError> = (|| {
1978        #[cfg(unix)]
1979        let open = {
1980            use std::os::unix::fs::OpenOptionsExt;
1981            fs::OpenOptions::new()
1982                .write(true)
1983                .create(true)
1984                .truncate(true)
1985                .mode(0o600)
1986                .open(&tmp_path)
1987        };
1988        #[cfg(not(unix))]
1989        let open = fs::OpenOptions::new()
1990            .write(true)
1991            .create(true)
1992            .truncate(true)
1993            .open(&tmp_path);
1994
1995        let mut f = open?;
1996        f.write_all(data)?;
1997        // sync_all flushes both data AND metadata, so on a crash after
1998        // the rename, fsck/journal recovery sees the new bytes -- not a
1999        // ghost inode with stale content.
2000        f.sync_all()?;
2001        Ok(())
2002    })();
2003
2004    if let Err(e) = write_result {
2005        // Best-effort cleanup so the next write isn't surprised by a
2006        // half-written tmp. Errors here are not surfaced: the original
2007        // write error is what the caller needs to see.
2008        let _ = fs::remove_file(&tmp_path);
2009        return Err(e);
2010    }
2011
2012    // Atomic same-filesystem rename. On Unix this is a single
2013    // rename(2) syscall guaranteed by POSIX to be atomic with respect
2014    // to other observers. On Windows std::fs::rename is implemented
2015    // via MoveFileEx with MOVEFILE_REPLACE_EXISTING (atomic on NTFS,
2016    // best-effort elsewhere). After this returns Ok, the new bytes are
2017    // visible at `path` and the tmp file no longer exists.
2018    if let Err(e) = fs::rename(&tmp_path, path) {
2019        let _ = fs::remove_file(&tmp_path);
2020        return Err(KeyError::Io(e));
2021    }
2022
2023    // fsync the parent directory so the rename's directory-entry update
2024    // is itself persisted. The previous code only fsynced the tmp
2025    // file's contents (via sync_all on the file handle) -- on ext4/xfs
2026    // with default mount options, the rename can return to userspace
2027    // before the dirent metadata has been written to the journal. A
2028    // power loss in that window leaves the directory entry pointing at
2029    // the OLD inode (or, worse, missing entirely if both old and new
2030    // were unlinked from the parent), even though both the data bytes
2031    // and the rename syscall ostensibly completed. The H1 doc-comment
2032    // above promised stronger durability than the code delivered;
2033    // fsyncing the parent dir closes that gap.
2034    //
2035    // Best-effort on Unix: a directory open + sync_all is the standard
2036    // pattern (see e.g. SQLite's atomic-commit, leveldb, lmdb). On
2037    // platforms where opening a directory for sync isn't supported, we
2038    // silently skip -- the rename is still atomic-with-respect-to-
2039    // observers, we just don't guarantee crash-durability of the
2040    // dirent update.
2041    #[cfg(unix)]
2042    {
2043        if let Some(parent) = path.parent() {
2044            // Errors here are non-fatal: the rename succeeded and the
2045            // common case (no power loss before the next fs flush) is
2046            // correct. We surface a failure to open/sync the dir only
2047            // if the rename itself succeeded, since otherwise the
2048            // caller would mistake a durability hint for a write
2049            // failure. swallow silently rather than return.
2050            if let Ok(dir) = fs::File::open(parent) {
2051                let _ = dir.sync_all();
2052            }
2053        }
2054    }
2055
2056    Ok(())
2057}
2058
2059fn unix_now() -> u64 {
2060    use std::time::{SystemTime, UNIX_EPOCH};
2061    SystemTime::now()
2062        .duration_since(UNIX_EPOCH)
2063        .unwrap_or_default()
2064        .as_secs()
2065}
2066
2067#[cfg(test)]
2068mod tests {
2069    use super::*;
2070
2071    fn temp_dir_path() -> PathBuf {
2072        let mut p = std::env::temp_dir();
2073        p.push(format!("treeship-test-{}", {
2074            let mut b = [0u8; 4];
2075            // v0.10.4 P1 audit: thread_rng acceptable here. This is a
2076            // test-only temp-dir suffix to avoid collisions between parallel
2077            // test runs. Not a cryptographic input; entropy quality irrelevant.
2078            rand::thread_rng().fill_bytes(&mut b);
2079            hex_encode(&b)
2080        }));
2081        // Nest the store under a per-test parent (mirrors production's
2082        // `~/.treeship/keys`). The machine_seed lives in `store_dir.parent()`;
2083        // without this nesting the parent would be the SHARED system temp dir
2084        // and every test would share (and race on) one seed. `dir` still points
2085        // at the store directory, so test logic that inspects entry files is
2086        // unaffected.
2087        p.push("keys");
2088        p
2089    }
2090
2091    fn make_store() -> (Store, PathBuf) {
2092        let dir = temp_dir_path();
2093        let store = Store::open(&dir).unwrap();
2094        (store, dir)
2095    }
2096
2097    fn cleanup(dir: PathBuf) {
2098        // Remove the store dir AND its per-test parent (which holds machine_seed).
2099        let _ = fs::remove_dir_all(&dir);
2100        if let Some(parent) = dir.parent() {
2101            let _ = fs::remove_dir_all(parent);
2102        }
2103    }
2104
2105    // AUD-19: the wrapping key must derive from the SECRET machine_seed, not
2106    // from guessable machine identifiers. Two "hosts" with identical
2107    // machine-id / hostname / user but a DIFFERENT seed must not be able to
2108    // decrypt each other's keystore. We simulate the second host by swapping
2109    // the seed file under an otherwise-identical machine environment.
2110    #[test]
2111    fn different_seed_cannot_decrypt_even_with_identical_machine_id() {
2112        // Pre-create the co-located seed instead of relying on first-open to
2113        // mint one. `read_or_create_machine_seed` reads
2114        // `<store>/../machine_seed` first, then falls back to
2115        // `~/.treeship/machine_seed` before creating anything -- and that
2116        // global file exists on any machine that has run `treeship init`. So
2117        // the create branch never ran for a real developer, no seed appeared
2118        // beside the temp store, and this test failed for everyone who had
2119        // actually used the tool while passing on clean CI. Seeding the local
2120        // path up front makes the test hermetic and keeps it focused on the
2121        // property it exists to prove (AUD-19), not on where seeds get minted.
2122        let dir = temp_dir_path();
2123        let parent = dir.parent().expect("temp store has a parent").to_path_buf();
2124        fs::create_dir_all(&parent).unwrap();
2125        let seed_path = parent.join("machine_seed");
2126        fs::write(&seed_path, hex_encode(&[0x11u8; 32])).unwrap();
2127        #[cfg(unix)]
2128        {
2129            use std::os::unix::fs::PermissionsExt;
2130            // check_seed_file_secure refuses anything group/world readable.
2131            fs::set_permissions(&seed_path, fs::Permissions::from_mode(0o600)).unwrap();
2132        }
2133
2134        let store = Store::open(&dir).unwrap();
2135        store.generate(true).unwrap();
2136        // The default key is now on disk, wrapped under the seed-primary key.
2137        store.default_signer().expect("own seed must decrypt");
2138
2139        // Swap the seed for a different one. machine-id / hostname / user are
2140        // unchanged (same test host) — only the secret seed differs.
2141        fs::write(&seed_path, hex_encode(&[0xABu8; 32])).unwrap();
2142        #[cfg(unix)]
2143        {
2144            use std::os::unix::fs::PermissionsExt;
2145            fs::set_permissions(&seed_path, fs::Permissions::from_mode(0o600)).unwrap();
2146        }
2147
2148        // A fresh Store sees the new seed. The guessable machine-id/hostname
2149        // fallbacks are identical to the original host's, but they never
2150        // wrapped this entry (it is seed-wrapped), so decryption MUST fail.
2151        let store2 = Store::open(&dir).unwrap();
2152        assert!(
2153            store2.default_signer().is_err(),
2154            "a different machine_seed (same machine-id/hostname) must NOT decrypt the keystore"
2155        );
2156        cleanup(dir);
2157    }
2158
2159    // AUD-25: a machine_seed that is a symlink (which could redirect the read
2160    // to attacker-controlled bytes) is refused rather than trusted as key
2161    // material.
2162    #[cfg(unix)]
2163    #[test]
2164    fn planted_symlink_seed_is_refused() {
2165        let (_store, dir) = make_store(); // creates a real seed
2166        let seed_path = dir.parent().unwrap().join("machine_seed");
2167        let _ = fs::remove_file(&seed_path);
2168        // Replace the seed with a symlink to some attacker file.
2169        let attacker = dir.parent().unwrap().join("attacker_bytes");
2170        fs::write(&attacker, hex_encode(&[0x11u8; 32])).unwrap();
2171        std::os::unix::fs::symlink(&attacker, &seed_path).unwrap();
2172        // Opening must refuse the symlinked seed (fail closed), not follow it.
2173        assert!(
2174            Store::open(&dir).and_then(|s| s.default_signer()).is_err(),
2175            "a symlinked machine_seed must be refused"
2176        );
2177        cleanup(dir);
2178    }
2179
2180    #[test]
2181    fn generate_key() {
2182        let (store, dir) = make_store();
2183        let info = store.generate(true).unwrap();
2184        assert!(info.id.starts_with("key_"));
2185        assert_eq!(info.algorithm, "ed25519");
2186        assert!(!info.fingerprint.is_empty());
2187        assert_eq!(info.public_key.len(), 32);
2188        cleanup(dir);
2189    }
2190
2191    #[test]
2192    fn default_signer_works() {
2193        let (store, dir) = make_store();
2194        store.generate(true).unwrap();
2195        let signer = store.default_signer().unwrap();
2196        assert!(!signer.key_id().is_empty());
2197        let pae = crate::attestation::pae("text/plain", b"test");
2198        let sig = signer.sign(&pae).unwrap();
2199        assert_eq!(sig.len(), 64);
2200        cleanup(dir);
2201    }
2202
2203    /// Regression: a keystore whose path contains a symlink must decrypt
2204    /// consistently no matter which path string is used to open it. Before
2205    /// path canonicalization in `open`, deriving the machine key from the raw
2206    /// path string produced different keys for the same directory (e.g. open
2207    /// via a symlink vs. the real path), surfacing as a misleading
2208    /// "MAC verification failed -- wrong machine" error on a perfectly good
2209    /// keystore on the same machine.
2210    #[cfg(unix)]
2211    #[test]
2212    fn machine_key_stable_across_symlinked_path() {
2213        let real = temp_dir_path();
2214        fs::create_dir_all(&real).unwrap();
2215        let link = temp_dir_path();
2216        fs::create_dir_all(link.parent().unwrap()).unwrap();
2217        std::os::unix::fs::symlink(&real, &link).unwrap();
2218
2219        // Mint a default key via the SYMLINK path.
2220        {
2221            let store = Store::open(&link).unwrap();
2222            store.generate(true).unwrap();
2223        }
2224
2225        // Re-open via the REAL (canonical) path and decrypt. Pre-fix this
2226        // failed because the raw path strings ("link" vs "real") hashed to
2227        // different machine keys.
2228        let via_real = Store::open(&real).unwrap();
2229        via_real
2230            .default_signer()
2231            .expect("decrypt via the canonical path must succeed");
2232
2233        // And via the symlink again (fresh Store, re-derives the key).
2234        let via_link = Store::open(&link).unwrap();
2235        via_link
2236            .default_signer()
2237            .expect("decrypt via the symlink path must succeed");
2238
2239        fs::remove_file(&link).ok();
2240        cleanup(real);
2241    }
2242
2243    /// A keystore encrypted under the RAW path key (as the pre-canonicalization
2244    /// code wrote it) must still open after the change -- the legacy fallback
2245    /// must never lock an existing user out.
2246    #[cfg(unix)]
2247    #[test]
2248    fn legacy_raw_path_key_still_decrypts() {
2249        let real = temp_dir_path();
2250        fs::create_dir_all(&real).unwrap();
2251        let link = temp_dir_path();
2252        fs::create_dir_all(link.parent().unwrap()).unwrap();
2253        std::os::unix::fs::symlink(&real, &link).unwrap();
2254
2255        // Simulate a pre-fix keystore: encrypt a key under the machine key
2256        // derived from the RAW (symlink) path, bypassing canonicalization.
2257        let key_id = new_key_id();
2258        let signer = Ed25519Signer::generate(&key_id).unwrap();
2259        let raw_key = derive_machine_key(&link).unwrap();
2260        let canon_key = derive_machine_key(&fs::canonicalize(&link).unwrap()).unwrap();
2261        assert_ne!(raw_key, canon_key, "symlink must change the raw path key");
2262        let enc = encrypt_for_disk_v2(
2263            &raw_key,
2264            key_id.as_str(),
2265            &signer.public_key_bytes(),
2266            signer.secret_bytes().as_slice(),
2267        )
2268        .unwrap();
2269        let entry = EncryptedEntry {
2270            id: key_id.clone(),
2271            algorithm: "ed25519".into(),
2272            created_at: crate::statements::unix_to_rfc3339(unix_now()),
2273            public_key: signer.public_key_bytes(),
2274            enc_priv_key: enc,
2275            nonce: Vec::new(),
2276            valid_until: None,
2277            successor_key_id: None,
2278        };
2279
2280        // The store opened via the symlink has the canonical key as primary and
2281        // the raw-path key as the legacy fallback. The entry above is encrypted
2282        // under the raw key, so decryption must fall back rather than fail.
2283        let store = Store::open(&link).unwrap();
2284        store.write_entry(&entry).unwrap();
2285        let got = store
2286            .signer(key_id.as_str())
2287            .expect("legacy raw-path key must decrypt via the fallback");
2288        assert_eq!(got.public_key_bytes(), signer.public_key_bytes());
2289
2290        fs::remove_file(&link).ok();
2291        cleanup(real);
2292    }
2293
2294    /// A keystore wrapped under the v1 hostname+username machine key (every
2295    /// keystore written before the stable-primary change) must decrypt via
2296    /// the fallback chain AND be transparently rewrapped under the primary,
2297    /// so a later hostname rename can no longer brick it. This is the
2298    /// migration path for the recurring real-world failure where macOS
2299    /// renames `kern.hostname` (network collision, DHCP) and the keystore
2300    /// dies with "MAC verification failed -- wrong machine".
2301    #[test]
2302    fn v1_wrapped_keystore_decrypts_and_rewraps_under_primary() {
2303        let dir = temp_dir_path();
2304        fs::create_dir_all(&dir).unwrap();
2305        let canonical = fs::canonicalize(&dir).unwrap();
2306
2307        // AUD-19: the primary is now the SECRET seed-derived key. A legacy
2308        // entry wrapped under the old guessable v1 key must migrate to it.
2309        // Migration always applies now (there is always a seed primary),
2310        // regardless of whether a hardware-stable id exists.
2311        let primary = derive_seed_primary_key(&canonical).unwrap();
2312        let v1_key = derive_machine_key(&canonical).unwrap();
2313        assert_ne!(
2314            primary, v1_key,
2315            "seed-primary and v1 derivations must differ"
2316        );
2317
2318        // Simulate the pre-fix keystore: entry wrapped under the v1 key.
2319        let key_id = new_key_id();
2320        let signer = Ed25519Signer::generate(&key_id).unwrap();
2321        let enc = encrypt_for_disk_v2(
2322            &v1_key,
2323            key_id.as_str(),
2324            &signer.public_key_bytes(),
2325            signer.secret_bytes().as_slice(),
2326        )
2327        .unwrap();
2328        let entry = EncryptedEntry {
2329            id: key_id.clone(),
2330            algorithm: "ed25519".into(),
2331            created_at: crate::statements::unix_to_rfc3339(unix_now()),
2332            public_key: signer.public_key_bytes(),
2333            enc_priv_key: enc,
2334            nonce: Vec::new(),
2335            valid_until: None,
2336            successor_key_id: None,
2337        };
2338
2339        let store = Store::open(&dir).unwrap();
2340        store.write_entry(&entry).unwrap();
2341        let got = store
2342            .signer(key_id.as_str())
2343            .expect("v1-wrapped entry must decrypt via the fallback chain");
2344        assert_eq!(got.public_key_bytes(), signer.public_key_bytes());
2345
2346        // The successful fallback decrypt must have rewrapped the on-disk
2347        // entry under the PRIMARY key: after migration the entry decrypts
2348        // with the primary directly and no longer with the old v1 key.
2349        let migrated = store.read_entry(key_id.as_str()).unwrap();
2350        assert!(
2351            decrypt_from_disk(
2352                &primary,
2353                &migrated.id,
2354                &migrated.public_key,
2355                &migrated.enc_priv_key,
2356                &migrated.nonce,
2357            )
2358            .is_ok(),
2359            "entry must be rewrapped under the primary machine key"
2360        );
2361        assert!(
2362            decrypt_from_disk(
2363                &v1_key,
2364                &migrated.id,
2365                &migrated.public_key,
2366                &migrated.enc_priv_key,
2367                &migrated.nonce,
2368            )
2369            .is_err(),
2370            "rewrapped entry must no longer decrypt under the old v1 key"
2371        );
2372
2373        cleanup(dir);
2374    }
2375
2376    /// A keystore wrapped under a v1 key derived from the mDNS
2377    /// LocalHostName (the hostname the machine reported before macOS
2378    /// renamed `kern.hostname` away from it) must decrypt via the
2379    /// LocalHostName fallback candidates. This is the exact drift shape
2380    /// that repeatedly bricked real keystores.
2381    #[cfg(target_os = "macos")]
2382    #[test]
2383    fn local_hostname_variant_recovers_drifted_keystore() {
2384        let variants = local_hostname_variants();
2385        let Some(old_hostname) = variants.first() else {
2386            // No LocalHostName on this machine; nothing to test.
2387            return;
2388        };
2389        let Ok(user) = std::env::var("USER") else {
2390            return;
2391        };
2392
2393        let dir = temp_dir_path();
2394        fs::create_dir_all(&dir).unwrap();
2395        let canonical = fs::canonicalize(&dir).unwrap();
2396
2397        let drifted_key = derive_machine_key_v1_from_parts(old_hostname, &user, &canonical);
2398
2399        let key_id = new_key_id();
2400        let signer = Ed25519Signer::generate(&key_id).unwrap();
2401        let enc = encrypt_for_disk_v2(
2402            &drifted_key,
2403            key_id.as_str(),
2404            &signer.public_key_bytes(),
2405            signer.secret_bytes().as_slice(),
2406        )
2407        .unwrap();
2408        let entry = EncryptedEntry {
2409            id: key_id.clone(),
2410            algorithm: "ed25519".into(),
2411            created_at: crate::statements::unix_to_rfc3339(unix_now()),
2412            public_key: signer.public_key_bytes(),
2413            enc_priv_key: enc,
2414            nonce: Vec::new(),
2415            valid_until: None,
2416            successor_key_id: None,
2417        };
2418
2419        let store = Store::open(&dir).unwrap();
2420        store.write_entry(&entry).unwrap();
2421        let got = store.signer(key_id.as_str()).expect(
2422            "keystore wrapped under the LocalHostName-derived v1 key must \
2423             decrypt via the drift-recovery candidates",
2424        );
2425        assert_eq!(got.public_key_bytes(), signer.public_key_bytes());
2426
2427        cleanup(dir);
2428    }
2429
2430    #[test]
2431    fn encrypt_decrypt_roundtrip() {
2432        // Routes the legacy public API through the dispatcher; v1
2433        // ciphertexts must still decrypt correctly.
2434        let key = [42u8; 32];
2435        let plaintext = b"super secret private key material here!";
2436        let (enc, nonce) = aes_gcm_encrypt(&key, plaintext).unwrap();
2437        let dec = aes_gcm_decrypt(&key, &enc, &nonce).unwrap();
2438        assert_eq!(dec, plaintext);
2439    }
2440
2441    #[test]
2442    fn decrypt_wrong_key_fails() {
2443        let key = [42u8; 32];
2444        let wrong = [99u8; 32];
2445        let (enc, nonce) = aes_gcm_encrypt(&key, b"secret").unwrap();
2446        assert!(aes_gcm_decrypt(&wrong, &enc, &nonce).is_err());
2447    }
2448
2449    // --- v2 AEAD tests (TS-2026-001 fix) -----------------------------------
2450
2451    // Fixed entry id + pubkey for the unit-level v2 tests below. The AAD
2452    // builder binds these into the GCM tag, so encrypt and decrypt must
2453    // see identical values. Using constants keeps each test focused on
2454    // its own bit-flip / tamper assertion without dragging Store setup
2455    // into the picture.
2456    const TEST_ENTRY_ID: &str = "key_unit_test_entry_0001";
2457    const TEST_PUBLIC_KEY: &[u8; 32] = &[0xAA; 32];
2458
2459    #[test]
2460    fn v2_encrypt_decrypt_roundtrip() {
2461        let key = [7u8; 32];
2462        let plaintext = b"super secret private key material here!";
2463        let blob = encrypt_for_disk_v2(&key, TEST_ENTRY_ID, TEST_PUBLIC_KEY, plaintext).unwrap();
2464        // Structural check on the framing.
2465        assert_eq!(blob[0], KEYSTORE_MAGIC, "magic byte");
2466        assert_eq!(blob[1], KEYSTORE_VERSION_V2, "version byte");
2467        assert_eq!(
2468            blob.len(),
2469            2 + 12 + plaintext.len() + 16,
2470            "magic+version+nonce+ct+tag length"
2471        );
2472
2473        let dec = decrypt_from_disk(&key, TEST_ENTRY_ID, TEST_PUBLIC_KEY, &blob, &[]).unwrap();
2474        assert_eq!(&*dec, plaintext);
2475    }
2476
2477    // ── KeyStore::encrypt_secret / decrypt_secret (AUD-02) ─────────────
2478    // The hub DPoP key used to be written to config.json as plaintext hex.
2479    // These lock the machine-bound, context-bound at-rest sealing that
2480    // replaces it.
2481
2482    #[test]
2483    fn encrypt_secret_roundtrips_and_hides_plaintext() {
2484        let (store, dir) = make_store();
2485        // A 32-byte Ed25519 secret, the actual thing we are protecting.
2486        let secret = [0x42u8; 32];
2487        let ctx = "hub-dpop:v1:hub_abc";
2488        let blob = store.encrypt_secret(ctx, &secret).unwrap();
2489
2490        // Fail-before-fix invariant: the sealed blob must NOT contain the
2491        // raw secret bytes. (The old code stored them verbatim.)
2492        assert!(
2493            !blob.windows(secret.len()).any(|w| w == secret),
2494            "sealed blob must not contain the raw secret"
2495        );
2496
2497        let recovered = store.decrypt_secret(ctx, &blob).unwrap();
2498        assert_eq!(
2499            recovered.as_slice(),
2500            &secret,
2501            "roundtrip must recover the secret"
2502        );
2503        cleanup(dir);
2504    }
2505
2506    #[test]
2507    fn decrypt_secret_wrong_context_fails_closed() {
2508        let (store, dir) = make_store();
2509        let secret = [0x11u8; 32];
2510        let blob = store.encrypt_secret("hub-dpop:v1:hub_A", &secret).unwrap();
2511        // A blob sealed for hub A must not open under hub B's context —
2512        // this is what prevents an intra-file ciphertext swap.
2513        let r = store.decrypt_secret("hub-dpop:v1:hub_B", &blob);
2514        assert!(
2515            r.is_err(),
2516            "wrong context must fail closed, not return wrong bytes"
2517        );
2518        cleanup(dir);
2519    }
2520
2521    #[test]
2522    fn v2_decrypt_wrong_key_fails() {
2523        let key = [7u8; 32];
2524        let wrong = [99u8; 32];
2525        let blob = encrypt_for_disk_v2(&key, TEST_ENTRY_ID, TEST_PUBLIC_KEY, b"secret").unwrap();
2526        // Wrong key with v2 framing: AEAD must reject. Dispatcher will
2527        // try v1 fallback (which also fails on garbage), so the final
2528        // error surfaces as a MAC failure rather than wrong plaintext.
2529        let result = decrypt_from_disk(&wrong, TEST_ENTRY_ID, TEST_PUBLIC_KEY, &blob, &[]);
2530        assert!(result.is_err(), "wrong key must fail");
2531    }
2532
2533    #[test]
2534    fn v2_tamper_ciphertext_fails() {
2535        let key = [7u8; 32];
2536        let mut blob = encrypt_for_disk_v2(
2537            &key,
2538            TEST_ENTRY_ID,
2539            TEST_PUBLIC_KEY,
2540            b"super secret private key",
2541        )
2542        .unwrap();
2543        // Flip one bit inside the ciphertext body (after the 14-byte
2544        // framing). GCM authenticates ciphertext + nonce; any flip must
2545        // fail.
2546        let last = blob.len() - 5;
2547        blob[last] ^= 0x01;
2548        let result = decrypt_from_disk(&key, TEST_ENTRY_ID, TEST_PUBLIC_KEY, &blob, &[]);
2549        assert!(result.is_err(), "tampered ciphertext must fail to decrypt");
2550    }
2551
2552    #[test]
2553    fn v2_tamper_nonce_fails() {
2554        let key = [7u8; 32];
2555        let mut blob = encrypt_for_disk_v2(
2556            &key,
2557            TEST_ENTRY_ID,
2558            TEST_PUBLIC_KEY,
2559            b"super secret private key",
2560        )
2561        .unwrap();
2562        // Flip a bit in the nonce (bytes [2..14]).
2563        blob[5] ^= 0x01;
2564        let result = decrypt_from_disk(&key, TEST_ENTRY_ID, TEST_PUBLIC_KEY, &blob, &[]);
2565        assert!(result.is_err(), "tampered nonce must fail to decrypt");
2566    }
2567
2568    #[test]
2569    fn v2_tamper_tag_fails() {
2570        let key = [7u8; 32];
2571        let mut blob = encrypt_for_disk_v2(
2572            &key,
2573            TEST_ENTRY_ID,
2574            TEST_PUBLIC_KEY,
2575            b"super secret private key",
2576        )
2577        .unwrap();
2578        // Flip a bit in the trailing GCM tag (last 16 bytes).
2579        let len = blob.len();
2580        blob[len - 1] ^= 0x80;
2581        let result = decrypt_from_disk(&key, TEST_ENTRY_ID, TEST_PUBLIC_KEY, &blob, &[]);
2582        assert!(result.is_err(), "tampered GCM tag must fail to decrypt");
2583    }
2584
2585    #[test]
2586    fn v2_nonces_are_unique_across_writes() {
2587        // Sanity check: two encryptions of identical plaintext under the
2588        // same key must produce different blobs (random per-write nonce).
2589        // Without this property, AES-GCM is catastrophically broken.
2590        let key = [7u8; 32];
2591        let blob_a =
2592            encrypt_for_disk_v2(&key, TEST_ENTRY_ID, TEST_PUBLIC_KEY, b"identical").unwrap();
2593        let blob_b =
2594            encrypt_for_disk_v2(&key, TEST_ENTRY_ID, TEST_PUBLIC_KEY, b"identical").unwrap();
2595        assert_ne!(
2596            blob_a, blob_b,
2597            "two v2 encryptions of the same plaintext must differ"
2598        );
2599        assert_ne!(&blob_a[2..14], &blob_b[2..14], "nonces must differ");
2600
2601        // L1 (TS-2026-001 audit): draw 10k nonces in a row and assert
2602        // every one is distinct. A duplicate at this volume would be a
2603        // strong (10k^2 / 2^96 ~ 2^-65 floor) signal that the OS CSPRNG
2604        // backing aead::OsRng is misbehaving on this build. Cheap, fast,
2605        // and catches a regression class (PRNG mis-seeding,
2606        // accidentally-deterministic nonce, RNG getting forked across
2607        // threads without re-seed) that the 2-sample check above can't.
2608        const N: usize = 10_000;
2609        let mut nonces: std::collections::HashSet<Vec<u8>> =
2610            std::collections::HashSet::with_capacity(N);
2611        for _ in 0..N {
2612            let blob = encrypt_for_disk_v2(&key, TEST_ENTRY_ID, TEST_PUBLIC_KEY, b"x").unwrap();
2613            // bytes [2..14] are the 12-byte GCM nonce.
2614            nonces.insert(blob[2..14].to_vec());
2615        }
2616        assert_eq!(
2617            nonces.len(),
2618            N,
2619            "all {} v2 nonces must be unique; collision => RNG defect",
2620            N
2621        );
2622    }
2623
2624    #[test]
2625    fn v2_tamper_version_byte_fails() {
2626        // M2: flipping the version byte must cause decryption to fail.
2627        // The framing sanity check catches obvious flips immediately;
2628        // the AAD-binding test below covers the case where the framing
2629        // sanity check would otherwise pass.
2630        let key = [7u8; 32];
2631        let mut blob = encrypt_for_disk_v2(
2632            &key,
2633            TEST_ENTRY_ID,
2634            TEST_PUBLIC_KEY,
2635            b"super secret private key",
2636        )
2637        .unwrap();
2638        assert_eq!(blob[1], KEYSTORE_VERSION_V2);
2639        blob[1] = 0xff;
2640        assert!(
2641            decrypt_v2(&key, TEST_ENTRY_ID, TEST_PUBLIC_KEY, &blob).is_err(),
2642            "altered version byte must be rejected"
2643        );
2644    }
2645
2646    #[test]
2647    fn v2_aad_binding_detects_framing_substitution() {
2648        // M2 direct check: encrypt a payload with v2 AAD, then construct
2649        // a blob whose framing claims to be v2 but whose ciphertext was
2650        // computed under a different AAD (empty). decrypt_v2 must
2651        // reject with MAC failure rather than returning the plaintext.
2652        let key = [7u8; 32];
2653        let plaintext = b"M2 AAD bound material";
2654
2655        // Compute a v2-framed blob without supplying AAD -- mimics what
2656        // the *pre-M2* code would have produced. This is the exact
2657        // attack surface AAD closes: an old blob whose framing is v2
2658        // but whose tag was computed empty.
2659        use aes_gcm::aead::Aead;
2660        let key_buf: Zeroizing<[u8; 32]> = Zeroizing::new(key);
2661        let aead_key: &AesKey<Aes256Gcm> = AesKey::<Aes256Gcm>::from_slice(key_buf.as_slice());
2662        let cipher = Aes256Gcm::new(aead_key);
2663        let nonce = Aes256Gcm::generate_nonce(&mut AeadOsRng);
2664        let ct_no_aad = cipher.encrypt(&nonce, plaintext.as_slice()).unwrap();
2665
2666        let mut forged = Vec::with_capacity(2 + 12 + ct_no_aad.len());
2667        forged.push(KEYSTORE_MAGIC);
2668        forged.push(KEYSTORE_VERSION_V2);
2669        forged.extend_from_slice(nonce.as_slice());
2670        forged.extend_from_slice(&ct_no_aad);
2671
2672        // Framing sanity passes. AAD does not. decrypt_v2 must reject.
2673        assert_eq!(forged[0], KEYSTORE_MAGIC);
2674        assert_eq!(forged[1], KEYSTORE_VERSION_V2);
2675        let result = decrypt_v2(&key, TEST_ENTRY_ID, TEST_PUBLIC_KEY, &forged);
2676        assert!(
2677            result.is_err(),
2678            "ciphertext computed without AAD must fail to decrypt now that AAD is bound"
2679        );
2680    }
2681
2682    #[test]
2683    fn dispatcher_surfaces_v2_error_on_corrupted_v2_blob() {
2684        // M1: a v2-shaped blob whose AEAD verification fails (and
2685        // whose v1 fallback also fails, since the bytes are garbage
2686        // under both constructions) must surface the v2 MAC error, not
2687        // the v1 "ciphertext too short" / random-junk error. The user
2688        // sees a meaningful message that points at the right
2689        // remediation.
2690        let key = [7u8; 32];
2691        let mut blob = encrypt_for_disk_v2(&key, TEST_ENTRY_ID, TEST_PUBLIC_KEY, b"hello").unwrap();
2692        // Flip a byte in the GCM tag (last 16 bytes) so the v2 AEAD
2693        // rejects but the framing still classifies as v2.
2694        let last = blob.len() - 1;
2695        blob[last] ^= 0x01;
2696
2697        let err = decrypt_from_disk(&key, TEST_ENTRY_ID, TEST_PUBLIC_KEY, &blob, &[]).unwrap_err();
2698        // The dispatcher should bubble the v2 error string up. v2's
2699        // error message contains "MAC verification failed"; v1's
2700        // shape on garbage data is either "ciphertext too short" or
2701        // a different MAC error. Match on the v2-specific tail.
2702        assert!(
2703            err.contains("MAC verification failed"),
2704            "dispatcher must surface the v2 MAC error on corrupted v2 blob, got: {err}"
2705        );
2706    }
2707
2708    #[test]
2709    fn legacy_v1_ciphertext_still_decrypts_via_dispatcher() {
2710        // Simulates an on-disk keystore written by Treeship <= v0.10.2:
2711        // the dispatcher must successfully route legacy ciphertexts
2712        // through the v1 path so existing users are not locked out.
2713        let key = [13u8; 32];
2714        let plaintext = b"pre-v0.10.3 keystore entry";
2715        let (legacy_blob, legacy_nonce) = legacy_v1_encrypt(&key, plaintext).unwrap();
2716
2717        // Sanity: legacy blob does NOT start with v2 framing.
2718        assert!(
2719            is_legacy_v1(&legacy_blob),
2720            "legacy_v1_encrypt output must classify as legacy"
2721        );
2722
2723        // Dispatcher must accept it. AAD inputs are irrelevant for the
2724        // v1 path (it doesn't use them), but the signature requires them
2725        // — pass the same placeholder constants used elsewhere.
2726        let dec = decrypt_from_disk(
2727            &key,
2728            TEST_ENTRY_ID,
2729            TEST_PUBLIC_KEY,
2730            &legacy_blob,
2731            &legacy_nonce,
2732        )
2733        .unwrap();
2734        assert_eq!(&*dec, plaintext);
2735    }
2736
2737    #[test]
2738    fn store_signer_migrates_legacy_entry_to_v2() {
2739        // End-to-end: write a key entry with the legacy v1 ciphertext
2740        // (as if upgrading from v0.10.2), call `signer()`, then verify
2741        // the on-disk entry has been rewritten in v2 format.
2742        let (store, dir) = make_store();
2743
2744        // Generate normally (this writes v2). Then re-encrypt the
2745        // secret in v1 format and overwrite the entry on disk to
2746        // simulate the upgrade scenario.
2747        let info = store.generate(true).unwrap();
2748        let entry_path = store.entry_path(&info.id);
2749
2750        // Pull the v2 entry off disk, decrypt to recover the secret,
2751        // then re-encode in legacy v1 format and write it back.
2752        let v2_entry: EncryptedEntry =
2753            serde_json::from_slice(&fs::read(&entry_path).unwrap()).unwrap();
2754        let secret = decrypt_from_disk(
2755            &store.machine_key,
2756            &v2_entry.id,
2757            &v2_entry.public_key,
2758            &v2_entry.enc_priv_key,
2759            &v2_entry.nonce,
2760        )
2761        .unwrap();
2762        let (legacy_blob, legacy_nonce) = legacy_v1_encrypt(&store.machine_key, &secret).unwrap();
2763        let legacy_entry = EncryptedEntry {
2764            id: v2_entry.id.clone(),
2765            algorithm: v2_entry.algorithm.clone(),
2766            created_at: v2_entry.created_at.clone(),
2767            public_key: v2_entry.public_key.clone(),
2768            enc_priv_key: legacy_blob,
2769            nonce: legacy_nonce,
2770            valid_until: v2_entry.valid_until.clone(),
2771            successor_key_id: v2_entry.successor_key_id.clone(),
2772        };
2773        fs::write(
2774            &entry_path,
2775            serde_json::to_vec_pretty(&legacy_entry).unwrap(),
2776        )
2777        .unwrap();
2778
2779        // Reload with a fresh Store so the cache doesn't paper over the
2780        // on-disk change.
2781        let store2 = Store::open(&dir).unwrap();
2782        // Loading the signer must succeed (legacy path works) AND
2783        // trigger the transparent migration to v2.
2784        let _signer = store2.signer(&info.id).unwrap();
2785
2786        let after: EncryptedEntry =
2787            serde_json::from_slice(&fs::read(&entry_path).unwrap()).unwrap();
2788        assert!(
2789            !is_legacy_v1(&after.enc_priv_key),
2790            "post-migration entry must be in v2 format"
2791        );
2792        assert_eq!(after.enc_priv_key[0], KEYSTORE_MAGIC);
2793        assert_eq!(after.enc_priv_key[1], KEYSTORE_VERSION_V2);
2794        assert!(
2795            after.nonce.is_empty(),
2796            "v2 entries serialize an empty legacy nonce field"
2797        );
2798
2799        // L2 (TS-2026-001 audit): the framing check above proves the
2800        // migrator *wrote* a v2-shaped blob, but a downstream
2801        // assert_eq! on framing alone doesn't prove the v2 ciphertext
2802        // is actually a working AEAD encryption of the right secret.
2803        // Load the signer one more time through a fresh Store; this
2804        // routes through the dispatcher's v2-first branch and would
2805        // fail loudly if the migration had produced garbage.
2806        let store3 = Store::open(&dir).unwrap();
2807        let _signer = store3
2808            .signer(&info.id)
2809            .expect("post-migration v2 decrypt works");
2810
2811        cleanup(dir);
2812    }
2813
2814    #[test]
2815    fn persist_and_reload() {
2816        let (store, dir) = make_store();
2817        let info = store.generate(true).unwrap();
2818
2819        // Open a new Store instance pointing to the same directory.
2820        let store2 = Store::open(&dir).unwrap();
2821        let signer = store2.signer(&info.id).unwrap();
2822        assert_eq!(signer.key_id(), info.id);
2823
2824        // The reloaded signer must produce signatures verifiable with
2825        // the same public key.
2826        let verifier = {
2827            use crate::attestation::Verifier;
2828            use ed25519_dalek::VerifyingKey;
2829            let pk_bytes: [u8; 32] = info.public_key.try_into().unwrap();
2830            let vk = VerifyingKey::from_bytes(&pk_bytes).unwrap();
2831            let mut v = Verifier::new(std::collections::HashMap::new());
2832            v.add_key(info.id.clone(), vk);
2833            v
2834        };
2835
2836        use crate::attestation::sign;
2837        use crate::statements::ActionStatement;
2838        let stmt = ActionStatement::new("agent://test", "tool.call");
2839        let pt = crate::statements::payload_type("action");
2840        let signed = sign(&pt, &stmt, signer.as_ref()).unwrap();
2841        verifier.verify(&signed.envelope).unwrap();
2842
2843        cleanup(dir);
2844    }
2845
2846    #[test]
2847    fn list_keys() {
2848        let (store, dir) = make_store();
2849        store.generate(true).unwrap();
2850        store.generate(false).unwrap();
2851
2852        let keys = store.list().unwrap();
2853        assert_eq!(keys.len(), 2);
2854        assert_eq!(keys.iter().filter(|k| k.is_default).count(), 1);
2855        cleanup(dir);
2856    }
2857
2858    #[test]
2859    fn no_default_key_errors() {
2860        let (store, dir) = make_store();
2861        assert!(store.default_signer().is_err());
2862        cleanup(dir);
2863    }
2864
2865    #[test]
2866    fn rotate_mints_successor_and_links_predecessor() {
2867        let (store, dir) = make_store();
2868        let pred = store.generate(true).unwrap();
2869        assert!(pred.valid_until.is_none(), "fresh key has no expiry");
2870        assert!(
2871            pred.successor_key_id.is_none(),
2872            "fresh key has no successor"
2873        );
2874
2875        let result = store
2876            .rotate(None, std::time::Duration::from_secs(3600), true)
2877            .unwrap();
2878
2879        // Predecessor metadata is updated.
2880        assert_eq!(result.predecessor.id, pred.id);
2881        assert!(
2882            result.predecessor.valid_until.is_some(),
2883            "predecessor must get valid_until after rotation"
2884        );
2885        assert_eq!(
2886            result.predecessor.successor_key_id.as_deref(),
2887            Some(result.successor.id.as_str()),
2888            "predecessor must link forward to successor"
2889        );
2890        assert!(
2891            !result.predecessor.is_default,
2892            "after rotation with set_default=true, predecessor is no longer default"
2893        );
2894
2895        // Successor is fresh.
2896        assert_ne!(result.successor.id, pred.id);
2897        assert!(
2898            result.successor.valid_until.is_none(),
2899            "successor has no expiry yet"
2900        );
2901        assert!(
2902            result.successor.successor_key_id.is_none(),
2903            "successor is chain head"
2904        );
2905        assert!(result.successor.is_default, "successor is the new default");
2906
2907        // Same metadata visible via list().
2908        let listed = store.list().unwrap();
2909        assert_eq!(listed.len(), 2);
2910        let pred_listed = listed.iter().find(|k| k.id == pred.id).unwrap();
2911        assert!(pred_listed.valid_until.is_some());
2912        assert_eq!(
2913            pred_listed.successor_key_id.as_deref(),
2914            Some(result.successor.id.as_str())
2915        );
2916
2917        cleanup(dir);
2918    }
2919
2920    #[test]
2921    fn rotate_with_set_default_false_keeps_predecessor_active() {
2922        let (store, dir) = make_store();
2923        let pred = store.generate(true).unwrap();
2924
2925        let result = store
2926            .rotate(None, std::time::Duration::from_secs(3600), false)
2927            .unwrap();
2928
2929        // Predecessor is still default. Successor exists but is not default.
2930        assert!(result.predecessor.is_default);
2931        assert!(!result.successor.is_default);
2932        assert_eq!(store.default_key_id().unwrap(), pred.id);
2933
2934        cleanup(dir);
2935    }
2936
2937    #[test]
2938    fn rotate_predecessor_signing_still_works_during_grace_window() {
2939        let (store, dir) = make_store();
2940        let pred = store.generate(true).unwrap();
2941        let _ = store
2942            .rotate(None, std::time::Duration::from_secs(3600), true)
2943            .unwrap();
2944
2945        // Predecessor key must still be loadable and capable of signing
2946        // during its grace window. Verifiers can refuse on lifecycle, but
2947        // the keystore must not preemptively destroy material.
2948        let signer = store.signer(&pred.id).unwrap();
2949        let pae = crate::attestation::pae("text/plain", b"grace-window-payload");
2950        let sig = signer.sign(&pae).unwrap();
2951        assert_eq!(sig.len(), 64);
2952
2953        cleanup(dir);
2954    }
2955
2956    #[test]
2957    fn rotate_refuses_to_rotate_already_rotated_key() {
2958        let (store, dir) = make_store();
2959        store.generate(true).unwrap();
2960        let r1 = store
2961            .rotate(None, std::time::Duration::from_secs(60), true)
2962            .unwrap();
2963
2964        // Rotating the predecessor again must be refused -- it already
2965        // points at r1.successor. Caller should rotate the chain head.
2966        let err = store
2967            .rotate(
2968                Some(&r1.predecessor.id),
2969                std::time::Duration::from_secs(60),
2970                true,
2971            )
2972            .unwrap_err();
2973        match err {
2974            KeyError::Crypto(msg) => assert!(
2975                msg.contains("already been rotated"),
2976                "error must explain why: {msg}"
2977            ),
2978            other => panic!("expected Crypto error, got {other:?}"),
2979        }
2980        cleanup(dir);
2981    }
2982
2983    #[test]
2984    fn successor_chain_walks_forward() {
2985        let (store, dir) = make_store();
2986        let k0 = store.generate(true).unwrap();
2987        let r1 = store
2988            .rotate(None, std::time::Duration::from_secs(60), true)
2989            .unwrap();
2990        let r2 = store
2991            .rotate(None, std::time::Duration::from_secs(60), true)
2992            .unwrap();
2993
2994        let chain = store.successor_chain(&k0.id).unwrap();
2995        assert_eq!(
2996            chain,
2997            vec![
2998                k0.id.clone(),
2999                r1.successor.id.clone(),
3000                r2.successor.id.clone()
3001            ],
3002            "chain must be ordered head -> tail"
3003        );
3004
3005        // Mid-chain start: chain from r1.successor should drop k0.
3006        let mid = store.successor_chain(&r1.successor.id).unwrap();
3007        assert_eq!(mid, vec![r1.successor.id.clone(), r2.successor.id.clone()]);
3008
3009        // Tail: just itself.
3010        let tail = store.successor_chain(&r2.successor.id).unwrap();
3011        assert_eq!(tail, vec![r2.successor.id.clone()]);
3012
3013        cleanup(dir);
3014    }
3015
3016    #[test]
3017    fn valid_keys_at_filters_by_grace_window() {
3018        let (store, dir) = make_store();
3019        let _ = store.generate(true).unwrap();
3020        let result = store
3021            .rotate(None, std::time::Duration::from_secs(3600), true)
3022            .unwrap();
3023
3024        // At time-of-rotation, both keys must be valid -- predecessor is
3025        // mid-grace, successor is freshly minted.
3026        let now = unix_now();
3027        let valid_now = store.valid_keys_at(now).unwrap();
3028        assert_eq!(
3029            valid_now.len(),
3030            2,
3031            "both predecessor (in grace) and successor should be valid"
3032        );
3033
3034        // After the grace window expires, only the successor remains.
3035        let after_grace = unix_now() + 7200;
3036        let valid_after = store.valid_keys_at(after_grace).unwrap();
3037        assert_eq!(
3038            valid_after.len(),
3039            1,
3040            "after grace window only successor remains valid"
3041        );
3042        assert_eq!(valid_after[0].id, result.successor.id);
3043
3044        cleanup(dir);
3045    }
3046
3047    /// Regression: if the successor key file is missing on disk (because a
3048    /// prior rotate() crashed AFTER stamping the predecessor but BEFORE
3049    /// writing the successor), retrying must NOT be wedged. With the
3050    /// successor-first write order this scenario can't be reached by a
3051    /// single-process crash, but we still need to defend against an operator
3052    /// who manually deletes a successor file mid-life. The recovery path
3053    /// is: clear the predecessor's successor pointer (or restore the file
3054    /// from backup) and try again.
3055    /// Regression: even if the manifest write FAILED (say, disk full at
3056    /// the worst possible moment), the in-memory cache must reflect the
3057    /// stamped predecessor that already landed on disk -- otherwise a
3058    /// same-process retry would skip the already-rotated guard and mint
3059    /// a duplicate successor.
3060    ///
3061    /// We can't easily inject a manifest-write failure mid-test, but we
3062    /// can verify the precondition that makes the recovery work: after a
3063    /// successful rotate(), the cache holds the stamped predecessor (so
3064    /// any subsequent rotate would correctly refuse). Combined with the
3065    /// write order (cache update BEFORE manifest write in rotate()),
3066    /// this proves a manifest-write crash leaves the cache aligned with
3067    /// disk, not behind it.
3068    #[test]
3069    fn rotate_cache_reflects_stamped_predecessor_for_retry_safety() {
3070        let (store, dir) = make_store();
3071        let pred = store.generate(true).unwrap();
3072        let _ = store
3073            .rotate(None, std::time::Duration::from_secs(60), true)
3074            .unwrap();
3075
3076        // The cache must have the stamped predecessor; a same-process
3077        // retry of rotate(predecessor) MUST be refused. If the cache
3078        // were stale (still showing the unstamped predecessor), this
3079        // call would proceed and mint a duplicate successor.
3080        let err = store
3081            .rotate(Some(&pred.id), std::time::Duration::from_secs(60), true)
3082            .unwrap_err();
3083        match err {
3084            KeyError::Crypto(msg) => assert!(
3085                msg.contains("already been rotated"),
3086                "cache should reflect stamped predecessor; got: {msg}"
3087            ),
3088            other => panic!("expected Crypto error, got {other:?}"),
3089        }
3090
3091        cleanup(dir);
3092    }
3093
3094    #[test]
3095    fn rotated_predecessor_pointing_at_missing_successor_surfaces_clear_error() {
3096        let (store, dir) = make_store();
3097        store.generate(true).unwrap();
3098        let result = store
3099            .rotate(None, std::time::Duration::from_secs(60), true)
3100            .unwrap();
3101
3102        // Simulate operator-deleted successor file. The manifest still
3103        // references it, so a cold-cache reader trying to walk the chain
3104        // hits a clear NotFound for the missing key.
3105        let succ_path = store.entry_path(&result.successor.id);
3106        fs::remove_file(&succ_path).unwrap();
3107
3108        // Open a fresh Store instance so the cache doesn't paper over the
3109        // missing on-disk entry. successor_chain() walks via load_entry;
3110        // the missing file must produce KeyError::NotFound, not a panic
3111        // and not an infinite loop.
3112        let store2 = Store::open(&dir).unwrap();
3113        let err = store2.successor_chain(&result.predecessor.id).unwrap_err();
3114        match err {
3115            KeyError::NotFound(id) => assert_eq!(id, result.successor.id),
3116            other => panic!("expected NotFound error, got {other:?}"),
3117        }
3118
3119        cleanup(dir);
3120    }
3121
3122    /// Pre-0.9.5 entry files lack `valid_until` and `successor_key_id`.
3123    /// They must still deserialize cleanly and be visible via `list()` /
3124    /// `default_signer()` etc.
3125    #[test]
3126    fn legacy_entry_without_lifecycle_fields_loads() {
3127        let (store, dir) = make_store();
3128        let info = store.generate(true).unwrap();
3129
3130        // Re-serialize the on-disk entry without the new fields, simulating
3131        // a file created by a 0.9.4 or earlier CLI.
3132        let path = store.entry_path(&info.id);
3133        let raw = fs::read(&path).unwrap();
3134        let mut json: serde_json::Value = serde_json::from_slice(&raw).unwrap();
3135        let obj = json.as_object_mut().unwrap();
3136        obj.remove("valid_until");
3137        obj.remove("successor_key_id");
3138        fs::write(&path, serde_json::to_vec_pretty(&json).unwrap()).unwrap();
3139
3140        // A fresh Store (cold cache) must still load the entry and treat
3141        // the missing fields as None.
3142        let store2 = Store::open(&dir).unwrap();
3143        let listed = store2.list().unwrap();
3144        assert_eq!(listed.len(), 1);
3145        assert!(
3146            listed[0].valid_until.is_none(),
3147            "missing valid_until must default to None on legacy entry"
3148        );
3149        assert!(
3150            listed[0].successor_key_id.is_none(),
3151            "missing successor_key_id must default to None on legacy entry"
3152        );
3153        let signer = store2.default_signer().unwrap();
3154        assert_eq!(signer.key_id(), info.id);
3155
3156        cleanup(dir);
3157    }
3158
3159    // --- keystore permission hardening (PR 1) -------------------------------
3160
3161    // The perm tests below mutate the process-global env var
3162    // TREESHIP_ALLOW_INSECURE_KEY_PERMS. cargo test runs cases in
3163    // parallel by default, so without serialization one test can set
3164    // the bypass while another expects it unset and racefully fail.
3165    // This mutex serializes them; everything else in the file remains
3166    // parallel-safe.
3167    static ENV_LOCK: std::sync::Mutex<()> = std::sync::Mutex::new(());
3168
3169    #[test]
3170    #[cfg(unix)]
3171    fn write_entry_creates_file_with_0600() {
3172        use std::os::unix::fs::PermissionsExt;
3173        let (store, dir) = make_store();
3174        let info = store.generate(true).unwrap();
3175        let mode = fs::metadata(store.entry_path(&info.id))
3176            .unwrap()
3177            .permissions()
3178            .mode()
3179            & 0o777;
3180        assert_eq!(
3181            mode, 0o600,
3182            "freshly written key file must be 0600, got {:o}",
3183            mode
3184        );
3185        cleanup(dir);
3186    }
3187
3188    #[test]
3189    #[cfg(unix)]
3190    fn signer_refuses_world_readable_key() {
3191        use std::os::unix::fs::PermissionsExt;
3192        // Mutex prevents the bypass var from being toggled by a
3193        // sibling test mid-flight (cargo test parallel runner).
3194        let _g = ENV_LOCK.lock().unwrap_or_else(|e| e.into_inner());
3195        // Make sure the bypass var is not leaking from the host env.
3196        std::env::remove_var("TREESHIP_ALLOW_INSECURE_KEY_PERMS");
3197
3198        let (store, dir) = make_store();
3199        let info = store.generate(true).unwrap();
3200
3201        // Loosen perms on the key file -- simulates a checkout, scp, or
3202        // shared-volume mishap.
3203        let path = store.entry_path(&info.id);
3204        fs::set_permissions(&path, fs::Permissions::from_mode(0o644)).unwrap();
3205
3206        match store.signer(&info.id) {
3207            Err(KeyError::InsecureKeyPerms { path: p, mode }) => {
3208                assert_eq!(p, path);
3209                assert_eq!(mode & 0o777, 0o644);
3210            }
3211            other => panic!("expected InsecureKeyPerms, got {:?}", other.map(|_| "ok")),
3212        }
3213        cleanup(dir);
3214    }
3215
3216    #[test]
3217    #[cfg(unix)]
3218    fn signer_bypass_via_env_var() {
3219        use std::os::unix::fs::PermissionsExt;
3220        let _g = ENV_LOCK.lock().unwrap_or_else(|e| e.into_inner());
3221        let (store, dir) = make_store();
3222        let info = store.generate(true).unwrap();
3223        let path = store.entry_path(&info.id);
3224        fs::set_permissions(&path, fs::Permissions::from_mode(0o644)).unwrap();
3225
3226        std::env::set_var("TREESHIP_ALLOW_INSECURE_KEY_PERMS", "1");
3227        let result = store.signer(&info.id);
3228        std::env::remove_var("TREESHIP_ALLOW_INSECURE_KEY_PERMS");
3229
3230        assert!(
3231            result.is_ok(),
3232            "bypass env var must allow signing: {:?}",
3233            result.err()
3234        );
3235        cleanup(dir);
3236    }
3237
3238    // --- v0.10.4 P2: TOCTOU window in signer() perm-check ---------------
3239
3240    /// Structural / single-open proof: the on-disk key file is opened
3241    /// EXACTLY ONCE during `signer()`. The fix replaces the prior
3242    /// `check_key_file_perms(path) + load_entry(id) -> fs::read(path)`
3243    /// two-open shape with `read_entry_with_perm_check`, which opens
3244    /// once and fstat's the resulting fd. We can't reliably race the
3245    /// FS in a unit test, so instead we assert the structural
3246    /// invariant: after `signer()` succeeds, only the bytes that the
3247    /// open file descriptor saw at perm-check time can have been read.
3248    ///
3249    /// The simulation: stage an attacker-controlled "loose perms"
3250    /// envelope at the path, then call `signer()`. With the fixed
3251    /// single-open shape, perm-check on the open fd fails before any
3252    /// content is read -- we get `InsecureKeyPerms`, not a successful
3253    /// signer. The legacy two-open code would have observed the perm
3254    /// failure on the same loose file too, but the property we are
3255    /// pinning here is that the perm rejection comes from the SAME fd
3256    /// the read would have used (no chance for an intermediate swap).
3257    #[test]
3258    #[cfg(unix)]
3259    fn signer_rejects_post_check_swap() {
3260        use std::os::unix::fs::PermissionsExt;
3261        let _g = ENV_LOCK.lock().unwrap_or_else(|e| e.into_inner());
3262        std::env::remove_var("TREESHIP_ALLOW_INSECURE_KEY_PERMS");
3263
3264        let (store, dir) = make_store();
3265        let info = store.generate(true).unwrap();
3266        let path = store.entry_path(&info.id);
3267
3268        // Snapshot the legit (0o600) v2 ciphertext bytes so we can
3269        // confirm that even if an attacker were to swap THIS exact
3270        // content under a loose-perms file, the single-open gate
3271        // catches it on the fd.
3272        let original_bytes = fs::read(&path).unwrap();
3273        assert!(!original_bytes.is_empty(), "test sanity");
3274
3275        // Stage the swapped file: same envelope content (so the JSON
3276        // parses and AEAD would succeed if we got that far), but
3277        // loose perms. With the old two-open shape, an attacker could
3278        // present 0o600 to perm-check, then race in this 0o644
3279        // version before the read; with the new single-open shape,
3280        // we open once, fstat the fd, and reject before reading.
3281        fs::write(&path, &original_bytes).unwrap();
3282        fs::set_permissions(&path, fs::Permissions::from_mode(0o644)).unwrap();
3283
3284        match store.signer(&info.id) {
3285            Err(KeyError::InsecureKeyPerms { path: p, mode }) => {
3286                assert_eq!(p, path);
3287                assert_eq!(mode & 0o777, 0o644);
3288            }
3289            Err(other) => panic!(
3290                "expected InsecureKeyPerms from single-open fstat gate, got {:?}",
3291                other
3292            ),
3293            Ok(_) => panic!("expected InsecureKeyPerms from single-open fstat gate, got ok signer"),
3294        }
3295
3296        // The "structural" half of the test: invoke the helper
3297        // directly. It must reject on the open fd, never returning
3298        // an `EncryptedEntry`. This pins the no-second-open property
3299        // -- if a future refactor reintroduces a path-based read
3300        // after the perm gate, this assertion still holds (the gate
3301        // would still trip on the same loose fd) but the code review
3302        // diff is the real test for the structural invariant.
3303        let direct = store.read_entry_with_perm_check(&info.id);
3304        assert!(
3305            matches!(direct, Err(KeyError::InsecureKeyPerms { .. })),
3306            "read_entry_with_perm_check must reject before reading bytes; got {:?}",
3307            direct.map(|_| "ok")
3308        );
3309
3310        cleanup(dir);
3311    }
3312
3313    // --- TS-2026-001 H3 migration-lock concurrency test -----------------
3314
3315    /// H3: two threads calling `Store::signer` on the same legacy v1
3316    /// entry must both succeed, the on-disk entry must end up as a
3317    /// valid v2 entry (decryptable via the v2 path), and no `.tmp`
3318    /// fragment must be left in the keystore directory.
3319    ///
3320    /// Without the advisory lock around `migrate_entry_to_v2`, two
3321    /// concurrent migrators would race the read-modify-rename cycle:
3322    /// the loser's rename would clobber the winner's v2 entry with
3323    /// its own (also-valid) v2 entry, but in between the two
3324    /// renames a third reader could observe a v2 entry, decrypt
3325    /// successfully, then have its in-memory state invalidated by
3326    /// the second writer. The flock turns the race into a queue --
3327    /// both writers produce identical v2 plaintext, only one rename
3328    /// per entry is actually needed, and the second writer's
3329    /// post-lock recheck observes the v2 state and exits cleanly.
3330    #[test]
3331    fn concurrent_migration_serializes_correctly() {
3332        use std::sync::Arc;
3333        use std::thread;
3334
3335        // Set up a legacy v1 entry on disk -- same shape as the
3336        // store_signer_migrates_legacy_entry_to_v2 test, just shared
3337        // with two threads.
3338        let (store, dir) = make_store();
3339        let info = store.generate(true).unwrap();
3340        let entry_path = store.entry_path(&info.id);
3341
3342        let v2_entry: EncryptedEntry =
3343            serde_json::from_slice(&fs::read(&entry_path).unwrap()).unwrap();
3344        let secret = decrypt_from_disk(
3345            &store.machine_key,
3346            &v2_entry.id,
3347            &v2_entry.public_key,
3348            &v2_entry.enc_priv_key,
3349            &v2_entry.nonce,
3350        )
3351        .unwrap();
3352        let (legacy_blob, legacy_nonce) = legacy_v1_encrypt(&store.machine_key, &secret).unwrap();
3353        let legacy_entry = EncryptedEntry {
3354            id: v2_entry.id.clone(),
3355            algorithm: v2_entry.algorithm.clone(),
3356            created_at: v2_entry.created_at.clone(),
3357            public_key: v2_entry.public_key.clone(),
3358            enc_priv_key: legacy_blob,
3359            nonce: legacy_nonce,
3360            valid_until: v2_entry.valid_until.clone(),
3361            successor_key_id: v2_entry.successor_key_id.clone(),
3362        };
3363        fs::write(
3364            &entry_path,
3365            serde_json::to_vec_pretty(&legacy_entry).unwrap(),
3366        )
3367        .unwrap();
3368
3369        // Two independent Store instances racing on the same on-disk
3370        // legacy entry. Using independent Store instances forces the
3371        // lock-on-disk path to engage (a shared Store would serialize
3372        // through the internal RwLock cache and we'd be testing the
3373        // wrong thing).
3374        let dir_a = Arc::new(dir.clone());
3375        let dir_b = Arc::new(dir.clone());
3376        let id_a = info.id.clone();
3377        let id_b = info.id.clone();
3378
3379        let h1 = thread::spawn(move || -> Result<(), String> {
3380            let s = Store::open(&*dir_a).map_err(|e| e.to_string())?;
3381            let _signer = s.signer(&id_a).map_err(|e| e.to_string())?;
3382            Ok(())
3383        });
3384        let h2 = thread::spawn(move || -> Result<(), String> {
3385            let s = Store::open(&*dir_b).map_err(|e| e.to_string())?;
3386            let _signer = s.signer(&id_b).map_err(|e| e.to_string())?;
3387            Ok(())
3388        });
3389
3390        h1.join()
3391            .unwrap()
3392            .expect("thread 1 signer load must succeed");
3393        h2.join()
3394            .unwrap()
3395            .expect("thread 2 signer load must succeed");
3396
3397        // Post-condition: on-disk entry is v2 framed.
3398        let after: EncryptedEntry =
3399            serde_json::from_slice(&fs::read(&entry_path).unwrap()).unwrap();
3400        assert!(
3401            !is_legacy_v1(&after.enc_priv_key),
3402            "post-concurrent-migration entry must be in v2 format"
3403        );
3404        assert_eq!(after.enc_priv_key[0], KEYSTORE_MAGIC);
3405        assert_eq!(after.enc_priv_key[1], KEYSTORE_VERSION_V2);
3406
3407        // v2 decrypts cleanly. Use the post-migration entry's own id +
3408        // pubkey — the migration must have re-encrypted with those bound
3409        // into the AAD, or this assertion would surface a MAC failure.
3410        let dec = decrypt_v2(
3411            &store.machine_key,
3412            &after.id,
3413            &after.public_key,
3414            &after.enc_priv_key,
3415        )
3416        .expect("v2 entry must decrypt cleanly after concurrent migration");
3417        assert_eq!(
3418            dec.len(),
3419            32,
3420            "decrypted secret must be a 32-byte ed25519 scalar"
3421        );
3422
3423        // No stale .tmp file left behind.
3424        for entry in fs::read_dir(&dir).unwrap() {
3425            let p = entry.unwrap().path();
3426            assert!(
3427                p.extension().is_none_or(|e| e != "tmp"),
3428                "no .tmp fragment must remain after migration, found: {}",
3429                p.display()
3430            );
3431        }
3432
3433        cleanup(dir);
3434    }
3435
3436    // --- TS-2026-001 H1 + H2 atomic write tests ------------------------
3437
3438    /// H1: a partial failure between writing the tmp file and renaming
3439    /// it into place MUST leave the original on-disk file intact. We
3440    /// simulate the failure by pre-creating a tmp file (so the next
3441    /// write_file_600 would clobber it) and then independently verifying
3442    /// that an already-written key entry remains decryptable even after
3443    /// a fresh write_file_600 fails partway.
3444    ///
3445    /// We exercise the failure path by pointing the rename at an
3446    /// unwritable target. On Unix we make the *parent directory*
3447    /// read-only after the original key is in place, which causes the
3448    /// final fs::rename to fail with EACCES. The original key file is
3449    /// unaffected because rename(2) returns before touching the target.
3450    #[test]
3451    #[cfg(unix)]
3452    fn atomic_write_leaves_original_intact_on_partial_failure() {
3453        use std::os::unix::fs::PermissionsExt;
3454        let (store, dir) = make_store();
3455        let info = store.generate(true).unwrap();
3456        let entry_path = store.entry_path(&info.id);
3457
3458        // Capture the original bytes for byte-identity comparison.
3459        let original = fs::read(&entry_path).expect("entry file must exist");
3460        assert!(
3461            !original.is_empty(),
3462            "freshly generated entry must be non-empty"
3463        );
3464
3465        // Lock the directory: read+execute only, no write. fs::rename
3466        // into this directory will fail.
3467        let orig_dir_mode = fs::metadata(&dir).unwrap().permissions().mode() & 0o777;
3468        fs::set_permissions(&dir, fs::Permissions::from_mode(0o500)).unwrap();
3469
3470        // Attempt a fresh write to the SAME path -- must fail because
3471        // the directory is read-only, exercising the rename-failure
3472        // branch.
3473        let res = write_file_600(&entry_path, b"new junk that must not land");
3474        assert!(
3475            res.is_err(),
3476            "write_file_600 must fail when dir is read-only"
3477        );
3478
3479        // Restore perms so we can read back the entry.
3480        fs::set_permissions(&dir, fs::Permissions::from_mode(orig_dir_mode)).unwrap();
3481
3482        // The original key file must be byte-identical to what we
3483        // captured before the failed write.
3484        let after = fs::read(&entry_path).expect("entry file must still exist after failed write");
3485        assert_eq!(
3486            after, original,
3487            "failed atomic write must not corrupt the original file",
3488        );
3489
3490        // And the keystore must still produce a working signer from it.
3491        let store2 = Store::open(&dir).unwrap();
3492        let signer = store2
3493            .signer(&info.id)
3494            .expect("original key must still decrypt after a failed write");
3495        let pae = crate::attestation::pae("text/plain", b"survive");
3496        assert_eq!(signer.sign(&pae).unwrap().len(), 64);
3497
3498        // No stale tmp file left behind.
3499        let tmp = entry_path.with_extension("tmp");
3500        assert!(
3501            !tmp.exists(),
3502            "tmp file must be cleaned up after rename failure"
3503        );
3504
3505        cleanup(dir);
3506    }
3507
3508    /// H2: the entry file's mode is 0o600 at the moment of creation, set
3509    /// via OpenOptionsExt::mode rather than a post-write set_permissions
3510    /// (which had a tiny window of looser perms). Also confirms the tmp
3511    /// file is removed by the rename.
3512    #[test]
3513    #[cfg(unix)]
3514    fn mode_is_600_at_creation() {
3515        use std::os::unix::fs::PermissionsExt;
3516        let (store, dir) = make_store();
3517        let info = store.generate(true).unwrap();
3518        let entry_path = store.entry_path(&info.id);
3519
3520        let mode = fs::metadata(&entry_path).unwrap().permissions().mode() & 0o777;
3521        assert_eq!(
3522            mode, 0o600,
3523            "entry file must be 0600 at creation, got {:o}",
3524            mode
3525        );
3526
3527        let tmp = entry_path.with_extension("tmp");
3528        assert!(
3529            !tmp.exists(),
3530            "no .tmp file must be left behind after a successful atomic write"
3531        );
3532
3533        cleanup(dir);
3534    }
3535
3536    #[test]
3537    #[cfg(unix)]
3538    fn fix_perms_repairs_loose_modes() {
3539        use std::os::unix::fs::PermissionsExt;
3540        let (store, dir) = make_store();
3541        let info = store.generate(true).unwrap();
3542        let key_path = store.entry_path(&info.id);
3543
3544        fs::set_permissions(&dir, fs::Permissions::from_mode(0o755)).unwrap();
3545        fs::set_permissions(&key_path, fs::Permissions::from_mode(0o644)).unwrap();
3546
3547        let changes = store.fix_perms().unwrap();
3548        // dir + key file + manifest = 3 paths to fix (manifest may already be 0600
3549        // depending on Manifest write path; we only assert the loose ones moved).
3550        assert!(
3551            changes.iter().any(|(p, _, _)| p == &dir),
3552            "dir should be repaired"
3553        );
3554        assert!(
3555            changes.iter().any(|(p, _, _)| p == &key_path),
3556            "key file should be repaired"
3557        );
3558
3559        let dir_mode = fs::metadata(&dir).unwrap().permissions().mode() & 0o777;
3560        let key_mode = fs::metadata(&key_path).unwrap().permissions().mode() & 0o777;
3561        assert_eq!(dir_mode, 0o700);
3562        assert_eq!(key_mode, 0o600);
3563
3564        // After repair, signing must work again.
3565        store
3566            .signer(&info.id)
3567            .expect("signing must work after fix_perms");
3568
3569        cleanup(dir);
3570    }
3571
3572    // --- TS-2026-001 post-merge fix-up: entry-binding AAD ------------------
3573
3574    /// Post-merge audit fix: the v2 AAD now binds entry id + public key
3575    /// into the GCM tag. Without that binding, a local attacker with
3576    /// write access to ~/.treeship/keys/ could copy entry A's
3577    /// `enc_priv_key` ciphertext into entry B's JSON envelope; the
3578    /// decrypt would succeed (same machine key, same framing-only AAD)
3579    /// and the signer for advertised key id A would silently sign with
3580    /// key B's secret scalar.
3581    ///
3582    /// This test performs exactly that swap and asserts decryption now
3583    /// fails. Before the fix this test would silently pass with the
3584    /// wrong scalar -- a true regression guard.
3585    #[test]
3586    fn cross_entry_swap_fails_decryption() {
3587        let (store, dir) = make_store();
3588
3589        // Two independent keys in the same store, same machine key.
3590        let a = store.generate(true).unwrap();
3591        let b = store.generate(false).unwrap();
3592
3593        // Snapshot both on-disk envelopes.
3594        let path_a = store.entry_path(&a.id);
3595        let path_b = store.entry_path(&b.id);
3596        let entry_a: EncryptedEntry = serde_json::from_slice(&fs::read(&path_a).unwrap()).unwrap();
3597        let entry_b: EncryptedEntry = serde_json::from_slice(&fs::read(&path_b).unwrap()).unwrap();
3598
3599        // Sanity: both are v2 framed, and the ciphertexts differ.
3600        assert_eq!(entry_a.enc_priv_key[0], KEYSTORE_MAGIC);
3601        assert_eq!(entry_a.enc_priv_key[1], KEYSTORE_VERSION_V2);
3602        assert_eq!(entry_b.enc_priv_key[0], KEYSTORE_MAGIC);
3603        assert_eq!(entry_b.enc_priv_key[1], KEYSTORE_VERSION_V2);
3604        assert_ne!(
3605            entry_a.enc_priv_key, entry_b.enc_priv_key,
3606            "two freshly-generated entries must have distinct ciphertexts"
3607        );
3608
3609        // The attack: copy B's enc_priv_key into A's envelope. Leave
3610        // everything else (id, public_key, algorithm) as it was in A.
3611        // This is the file an attacker with write access to the keys
3612        // directory would produce.
3613        let mut tampered_a = entry_a.clone();
3614        tampered_a.enc_priv_key = entry_b.enc_priv_key.clone();
3615        // The v2 nonce travels inline with the ciphertext (bytes
3616        // [2..14] of enc_priv_key), so swapping the blob also swaps
3617        // the nonce; the separate JSON `nonce` field is empty for v2
3618        // entries either way.
3619        fs::write(&path_a, serde_json::to_vec_pretty(&tampered_a).unwrap()).unwrap();
3620
3621        // Fresh Store so the in-memory cache doesn't paper over the
3622        // on-disk tamper.
3623        let store2 = Store::open(&dir).unwrap();
3624        let err = match store2.signer(&a.id) {
3625            Ok(_) => panic!(
3626                "swapping B's ciphertext into A's envelope must fail decrypt; \
3627                 got Ok which means the signer would silently sign with key B"
3628            ),
3629            Err(e) => e,
3630        };
3631
3632        // The specific error must be a crypto/MAC failure, not (e.g.)
3633        // a NotFound or InsecureKeyPerms surface that could mask the
3634        // class of bug.
3635        match err {
3636            KeyError::Crypto(msg) => assert!(
3637                msg.contains("MAC verification failed"),
3638                "swap must surface MAC failure; got: {msg}"
3639            ),
3640            other => panic!("expected Crypto MAC error, got: {other:?}"),
3641        }
3642
3643        cleanup(dir);
3644    }
3645
3646    /// Companion to `cross_entry_swap_fails_decryption`: the id field
3647    /// is also bound into the AAD, so editing the JSON `id` while
3648    /// leaving the ciphertext alone must also fail. (An attacker who
3649    /// renames a stolen entry file onto a victim's id without
3650    /// re-encrypting would land here.)
3651    #[test]
3652    fn aad_tampered_entry_id_fails_decryption() {
3653        let (store, dir) = make_store();
3654        let info = store.generate(true).unwrap();
3655        let path = store.entry_path(&info.id);
3656
3657        let mut entry: EncryptedEntry = serde_json::from_slice(&fs::read(&path).unwrap()).unwrap();
3658        assert_eq!(
3659            entry.id, info.id,
3660            "sanity: id matches what generate returned"
3661        );
3662
3663        // Pretend the attacker forged an id. Note we write this back to
3664        // the SAME file path so Store::load_entry by the original id
3665        // finds it; if we changed the path too we'd just be testing
3666        // NotFound, which isn't the point.
3667        entry.id = "key_attacker_substituted_id".to_string();
3668        fs::write(&path, serde_json::to_vec_pretty(&entry).unwrap()).unwrap();
3669
3670        // Fresh Store so cache doesn't paper this over. Load via the
3671        // tampered id (matching what's in the JSON) so we exercise the
3672        // decrypt path rather than a path-vs-id mismatch.
3673        let store2 = Store::open(&dir).unwrap();
3674        // Drop the cache by opening fresh; load by the on-disk id.
3675        // The entry_path for "key_attacker_substituted_id" doesn't
3676        // exist, so we deliberately call the lower-level read by
3677        // path-of-original and assert decrypt fails via the dispatcher.
3678        // Easiest: bypass entry_path and invoke decrypt_from_disk with
3679        // the tampered id directly.
3680        let key_buf = store2.machine_key;
3681        let result = decrypt_from_disk(
3682            &key_buf,
3683            &entry.id,         // tampered id (bound into AAD)
3684            &entry.public_key, // original pubkey
3685            &entry.enc_priv_key,
3686            &entry.nonce,
3687        );
3688        assert!(
3689            result.is_err(),
3690            "AAD-bound entry id mismatch must fail decrypt; got Ok"
3691        );
3692
3693        cleanup(dir);
3694    }
3695}