c2pa-warc 0.1.0

C2PA manifest embedding for WARC web archive files (ISO 28500)
Documentation
use std::collections::HashMap;

use crate::error::Error;

const VERSION: &str = "WARC/1.1";
const C2PA_CONTENT_TYPE: &str = "application/c2pa";

#[derive(Debug, Clone)]
pub struct WarcRecord {
    pub headers: HashMap<String, String>,
    pub body: Vec<u8>,
    pub raw_offset: usize,
    pub raw_length: usize,
}

impl WarcRecord {
    pub fn warc_type(&self) -> Option<&str> {
        self.headers.get("warc-type").map(|s| s.as_str())
    }

    pub fn content_type(&self) -> Option<&str> {
        self.headers.get("content-type").map(|s| s.as_str())
    }

    pub fn record_id(&self) -> Option<&str> {
        self.headers.get("warc-record-id").map(|s| s.as_str())
    }

    pub fn is_c2pa_manifest(&self) -> bool {
        self.warc_type() == Some("resource")
            && self.content_type() == Some(C2PA_CONTENT_TYPE)
    }
}

pub fn parse_records(data: &[u8]) -> Result<Vec<WarcRecord>, Error> {
    let mut records = Vec::new();
    let mut pos = 0;

    while pos < data.len() {
        while pos < data.len() && (data[pos] == b'\r' || data[pos] == b'\n') {
            pos += 1;
        }
        if pos >= data.len() {
            break;
        }

        let record_start = pos;

        if !data[pos..].starts_with(VERSION.as_bytes()) {
            return Err(Error::InvalidRecord(format!(
                "expected WARC/1.1 at offset {pos}"
            )));
        }

        let header_end = find_double_crlf(&data[pos..])
            .ok_or_else(|| Error::InvalidRecord("unterminated header".into()))?;
        let header_block = &data[pos..pos + header_end];
        pos += header_end + 4; // skip \r\n\r\n

        let headers = parse_headers(header_block)?;

        let content_length: usize = headers
            .get("content-length")
            .ok_or_else(|| Error::InvalidRecord("missing Content-Length".into()))?
            .parse()
            .map_err(|_| Error::InvalidRecord("invalid Content-Length".into()))?;

        if pos + content_length > data.len() {
            return Err(Error::InvalidRecord("body extends past end of data".into()));
        }

        let body = data[pos..pos + content_length].to_vec();
        pos += content_length;

        // Skip record terminator \r\n\r\n
        if data[pos..].starts_with(b"\r\n\r\n") {
            pos += 4;
        } else if data[pos..].starts_with(b"\n\n") {
            pos += 2;
        }

        let raw_length = pos - record_start;

        records.push(WarcRecord {
            headers,
            body,
            raw_offset: record_start,
            raw_length,
        });
    }

    Ok(records)
}

pub fn build_record(warc_type: &str, content_type: &str, record_id: &str, body: &[u8]) -> Vec<u8> {
    let date = "2026-01-01T00:00:00Z";
    let header = format!(
        "{VERSION}\r\nWARC-Type: {warc_type}\r\nWARC-Record-ID: <{record_id}>\r\nWARC-Target-URI: {record_id}\r\nWARC-Date: {date}\r\nContent-Type: {content_type}\r\nContent-Length: {}\r\n\r\n",
        body.len()
    );
    let mut out = header.into_bytes();
    out.extend_from_slice(body);
    out.extend_from_slice(b"\r\n\r\n");
    out
}

fn find_double_crlf(data: &[u8]) -> Option<usize> {
    data.windows(4)
        .position(|w| w == b"\r\n\r\n")
}

fn parse_headers(block: &[u8]) -> Result<HashMap<String, String>, Error> {
    let text = std::str::from_utf8(block)
        .map_err(|_| Error::InvalidRecord("non-UTF-8 header".into()))?;
    let mut headers = HashMap::new();

    for line in text.lines().skip(1) {
        if let Some((key, value)) = line.split_once(':') {
            headers.insert(
                key.trim().to_ascii_lowercase(),
                value.trim().to_string(),
            );
        }
    }

    Ok(headers)
}

#[cfg(test)]
mod tests {
    use super::*;

    #[test]
    fn parse_single_record() {
        let raw = b"WARC/1.1\r\nWARC-Type: resource\r\nWARC-Record-ID: <urn:uuid:abc>\r\nContent-Type: text/plain\r\nContent-Length: 5\r\n\r\nhello\r\n\r\n";
        let records = parse_records(raw).unwrap();
        assert_eq!(records.len(), 1);
        assert_eq!(records[0].body, b"hello");
        assert_eq!(records[0].warc_type(), Some("resource"));
    }

    #[test]
    fn parse_c2pa_record() {
        let manifest = b"\x00\x01\x02\x03";
        let record = build_record("resource", "application/c2pa", "urn:uuid:test-id", manifest);
        let records = parse_records(&record).unwrap();
        assert_eq!(records.len(), 1);
        assert!(records[0].is_c2pa_manifest());
        assert_eq!(records[0].body, manifest);
    }

    #[test]
    fn build_and_parse_roundtrip() {
        let body = b"test body content";
        let record = build_record("resource", "text/plain", "urn:uuid:123", body);
        let records = parse_records(&record).unwrap();
        assert_eq!(records[0].body, body);
    }
}