runsync-transfer 0.1.0

High-throughput P2P file transfer engine: adaptive compression, end-to-end AEAD, parallel chunked pipeline over QUIC or any async transport.
Documentation
//! A cache of per-chunk hashes, so a file that has not changed is never read.
//!
//! Delta sync needs both ends to know the hash of every chunk. Computing those
//! means reading the whole file, which is fine once and absurd on the tenth
//! sync of a tree that never changed: re-sending an unchanged 3.4 GiB tree
//! moved 0.1 MiB but still spent 27 seconds hashing.
//!
//! So the hashes are remembered. An entry is trusted while the file's size and
//! modification time both match what they were when it was hashed, which is the
//! same bargain rsync and Syncthing make. It is a bargain, not a proof: a file
//! edited within the timestamp's resolution *and* left at the same length would
//! be missed. Two things bound the damage — the receiver verifies every
//! completed file against the sender's hash root before committing it, so a
//! stale entry surfaces as a failed transfer rather than a corrupt file, and
//! `Config::trust_mtime` turns the whole thing off for callers who would rather
//! pay the read.

use crate::error::Result;
use std::collections::HashMap;
use std::path::{Path, PathBuf};

const MAGIC: &[u8; 4] = b"RSIX";
const VERSION: u16 = 1;
/// Where the cache lives, relative to the root it describes.
pub const INDEX_FILE: &str = ".rst-index";

#[derive(Clone)]
struct Entry {
    size: u64,
    mtime: i64,
    chunk_size: u32,
    hashes: Vec<[u8; 32]>,
}

/// Chunk hashes for the files under one root.
#[derive(Default)]
pub struct ChunkIndex {
    path: PathBuf,
    entries: HashMap<String, Entry>,
    dirty: bool,
}

impl ChunkIndex {
    /// Load the index for `root`, or start empty if there is none to load.
    ///
    /// A corrupt or truncated index is treated as absent: the cost is re-
    /// hashing, and trusting damaged hashes would be the one outcome worth
    /// avoiding.
    pub fn load(root: &Path) -> Self {
        let path = root.join(INDEX_FILE);
        let mut index = Self {
            path,
            entries: HashMap::new(),
            dirty: false,
        };
        let Ok(raw) = std::fs::read(&index.path) else {
            return index;
        };
        if let Some(entries) = decode(&raw) {
            index.entries = entries;
        } else {
            tracing::warn!(path = %index.path.display(), "chunk index unreadable; rebuilding");
        }
        index
    }

    /// Hashes for a file, if the cache still describes it.
    pub fn get(&self, rel: &str, size: u64, mtime: i64, chunk_size: u32) -> Option<&[[u8; 32]]> {
        let e = self.entries.get(rel)?;
        (e.size == size && e.mtime == mtime && e.chunk_size == chunk_size)
            .then_some(e.hashes.as_slice())
    }

    pub fn insert(
        &mut self,
        rel: &str,
        size: u64,
        mtime: i64,
        chunk_size: u32,
        hashes: Vec<[u8; 32]>,
    ) {
        self.entries.insert(
            rel.to_string(),
            Entry {
                size,
                mtime,
                chunk_size,
                hashes,
            },
        );
        self.dirty = true;
    }

    /// Forget everything not named in `keep`, so a long-lived index does not
    /// accumulate entries for files that were deleted years ago.
    pub fn retain(&mut self, keep: &std::collections::HashSet<String>) {
        let before = self.entries.len();
        self.entries.retain(|k, _| keep.contains(k));
        if self.entries.len() != before {
            self.dirty = true;
        }
    }

    pub fn len(&self) -> usize {
        self.entries.len()
    }

    pub fn is_empty(&self) -> bool {
        self.entries.is_empty()
    }

    /// Write the index out, if anything changed.
    ///
    /// Through a temporary file and a rename, so an interrupted save leaves the
    /// previous index rather than a half-written one.
    pub fn save(&mut self) -> Result<()> {
        if !self.dirty {
            return Ok(());
        }
        if let Some(parent) = self.path.parent() {
            std::fs::create_dir_all(parent)?;
        }
        let body = encode(&self.entries);
        let tmp = self.path.with_extension("tmp");
        std::fs::write(&tmp, &body)?;
        std::fs::rename(&tmp, &self.path)?;
        self.dirty = false;
        Ok(())
    }
}

fn encode(entries: &HashMap<String, Entry>) -> Vec<u8> {
    let mut body = Vec::with_capacity(entries.len() * 96);
    body.extend_from_slice(&(entries.len() as u32).to_le_bytes());
    for (rel, e) in entries {
        let p = rel.as_bytes();
        body.extend_from_slice(&(p.len() as u16).to_le_bytes());
        body.extend_from_slice(p);
        body.extend_from_slice(&e.size.to_le_bytes());
        body.extend_from_slice(&e.mtime.to_le_bytes());
        body.extend_from_slice(&e.chunk_size.to_le_bytes());
        body.extend_from_slice(&(e.hashes.len() as u32).to_le_bytes());
        for h in &e.hashes {
            body.extend_from_slice(h);
        }
    }
    let mut out = Vec::with_capacity(body.len() + 38);
    out.extend_from_slice(MAGIC);
    out.extend_from_slice(&VERSION.to_le_bytes());
    out.extend_from_slice(blake3::hash(&body).as_bytes());
    out.extend_from_slice(&body);
    out
}

fn decode(raw: &[u8]) -> Option<HashMap<String, Entry>> {
    if raw.len() < 38 || &raw[0..4] != MAGIC {
        return None;
    }
    if u16::from_le_bytes(raw[4..6].try_into().ok()?) != VERSION {
        return None;
    }
    let body = &raw[38..];
    // A torn write is indistinguishable from a valid short index without this.
    if blake3::hash(body).as_bytes() != &raw[6..38] {
        return None;
    }

    let mut pos = 0usize;
    let mut take = |n: usize| -> Option<&[u8]> {
        let end = pos.checked_add(n)?;
        let s = body.get(pos..end)?;
        pos = end;
        Some(s)
    };
    let count = u32::from_le_bytes(take(4)?.try_into().ok()?) as usize;
    if count > 8_000_000 {
        return None;
    }
    let mut map = HashMap::with_capacity(count.min(4096));
    for _ in 0..count {
        let plen = u16::from_le_bytes(take(2)?.try_into().ok()?) as usize;
        let rel = String::from_utf8(take(plen)?.to_vec()).ok()?;
        let size = u64::from_le_bytes(take(8)?.try_into().ok()?);
        let mtime = i64::from_le_bytes(take(8)?.try_into().ok()?);
        let chunk_size = u32::from_le_bytes(take(4)?.try_into().ok()?);
        let n = u32::from_le_bytes(take(4)?.try_into().ok()?) as usize;
        if n > (1 << 26) {
            return None;
        }
        let mut hashes = Vec::with_capacity(n.min(4096));
        for _ in 0..n {
            let mut h = [0u8; 32];
            h.copy_from_slice(take(32)?);
            hashes.push(h);
        }
        map.insert(
            rel,
            Entry {
                size,
                mtime,
                chunk_size,
                hashes,
            },
        );
    }
    Some(map)
}

/// The modification time of `meta`, in **nanoseconds** since the epoch.
///
/// Nanoseconds rather than seconds, and the difference matters: at one-second
/// resolution a file rewritten within the same second at the same length would
/// be taken for unchanged, and on a fast machine that is not a rare event but
/// an ordinary one. At nanosecond resolution the window closes to whatever the
/// filesystem's timestamp granularity actually is.
pub fn mtime_of(meta: &std::fs::Metadata) -> i64 {
    meta.modified()
        .ok()
        .and_then(|t| t.duration_since(std::time::UNIX_EPOCH).ok())
        .map(|d| d.as_nanos().min(i64::MAX as u128) as i64)
        .unwrap_or(0)
}

#[cfg(test)]
mod tests {
    use super::*;

    fn hashes(n: usize) -> Vec<[u8; 32]> {
        (0..n).map(|i| [i as u8; 32]).collect()
    }

    #[test]
    fn survives_a_save_and_load() {
        let tmp = tempfile::tempdir().unwrap();
        let mut ix = ChunkIndex::load(tmp.path());
        assert!(ix.is_empty());
        ix.insert("a/b.bin", 1000, 42, 1024, hashes(4));
        ix.save().unwrap();

        let ix2 = ChunkIndex::load(tmp.path());
        assert_eq!(ix2.len(), 1);
        assert_eq!(ix2.get("a/b.bin", 1000, 42, 1024).unwrap().len(), 4);
    }

    /// The whole safety of the cache rests on this being fine-grained: at
    /// second resolution, two writes of the same length in the same second are
    /// indistinguishable, and the second one would be silently skipped.
    #[test]
    fn mtime_resolution_is_finer_than_a_second() {
        let tmp = tempfile::tempdir().unwrap();
        let p = tmp.path().join("f");
        std::fs::write(&p, b"first").unwrap();
        let a = mtime_of(&std::fs::metadata(&p).unwrap());
        std::thread::sleep(std::time::Duration::from_millis(5));
        std::fs::write(&p, b"secnd").unwrap();
        let b = mtime_of(&std::fs::metadata(&p).unwrap());
        assert_ne!(
            a, b,
            "two same-length writes 5ms apart were indistinguishable"
        );
    }

    #[test]
    fn a_changed_file_is_not_trusted() {
        let tmp = tempfile::tempdir().unwrap();
        let mut ix = ChunkIndex::load(tmp.path());
        ix.insert("f", 1000, 42, 1024, hashes(4));

        assert!(ix.get("f", 1000, 42, 1024).is_some());
        // Any of size, mtime or chunking differing invalidates the entry.
        assert!(ix.get("f", 1001, 42, 1024).is_none());
        assert!(ix.get("f", 1000, 43, 1024).is_none());
        assert!(ix.get("f", 1000, 42, 4096).is_none());
        assert!(ix.get("other", 1000, 42, 1024).is_none());
    }

    #[test]
    fn a_corrupt_index_is_discarded_rather_than_trusted() {
        let tmp = tempfile::tempdir().unwrap();
        let mut ix = ChunkIndex::load(tmp.path());
        ix.insert("f", 1000, 42, 1024, hashes(8));
        ix.save().unwrap();

        let p = tmp.path().join(INDEX_FILE);
        let mut raw = std::fs::read(&p).unwrap();
        let last = raw.len() - 1;
        raw[last] ^= 0xFF;
        std::fs::write(&p, &raw).unwrap();

        let ix2 = ChunkIndex::load(tmp.path());
        assert!(ix2.is_empty(), "damaged hashes must never be trusted");

        // Truncation too.
        std::fs::write(&p, &raw[..20]).unwrap();
        assert!(ChunkIndex::load(tmp.path()).is_empty());
        // And rubbish that is not an index at all.
        std::fs::write(&p, b"not an index").unwrap();
        assert!(ChunkIndex::load(tmp.path()).is_empty());
    }

    #[test]
    fn retain_drops_files_that_are_gone() {
        let tmp = tempfile::tempdir().unwrap();
        let mut ix = ChunkIndex::load(tmp.path());
        ix.insert("keep", 1, 1, 1024, hashes(1));
        ix.insert("gone", 1, 1, 1024, hashes(1));
        let keep: std::collections::HashSet<String> = ["keep".to_string()].into_iter().collect();
        ix.retain(&keep);
        assert_eq!(ix.len(), 1);
        assert!(ix.get("keep", 1, 1, 1024).is_some());
    }

    #[test]
    fn saving_is_a_no_op_when_nothing_changed() {
        let tmp = tempfile::tempdir().unwrap();
        let mut ix = ChunkIndex::load(tmp.path());
        ix.save().unwrap();
        assert!(!tmp.path().join(INDEX_FILE).exists(), "nothing to write");
    }
}