use anyhow::{Context, Result};
use sha2::{Digest, Sha256};
use std::path::Path;
use crate::config::{ArchiveConfig, ArchiveSource};
pub mod recent;
use crate::types::{Chunk, ChunkType};
pub use recent::{collect_recent, extract_body, extract_date_from_archive_path};
#[derive(Debug, Clone, Default)]
pub struct ArchiveMeta {
pub id: String,
pub timestamp: String,
pub fields: Vec<(String, String)>,
}
pub fn fetch_archive(config: &ArchiveConfig) -> Result<Vec<Chunk>> {
if !config.enabled {
return Ok(vec![]);
}
let mut all_chunks = Vec::new();
for source in &config.sources {
if source.path.is_empty() {
continue;
}
let root = Path::new(&source.path);
if root.exists() {
collect_records(root, root, source, &mut all_chunks)?;
}
}
Ok(all_chunks)
}
fn collect_records(
root: &Path,
dir: &Path,
source: &ArchiveSource,
chunks: &mut Vec<Chunk>,
) -> Result<()> {
let entries =
std::fs::read_dir(dir).with_context(|| format!("archive: read dir: {}", dir.display()))?;
for entry in entries {
let entry = entry?;
let path = entry.path();
if path.is_dir() {
collect_records(root, &path, source, chunks)?;
} else if path.extension().is_some_and(|e| e == "md") {
let content = std::fs::read_to_string(&path)
.with_context(|| format!("archive: read: {}", path.display()))?;
if let Some(chunk) = record_to_chunk(root, &path, &content, source) {
chunks.push(chunk);
}
}
}
Ok(())
}
fn record_to_chunk(
root: &Path,
path: &Path,
content: &str,
source: &ArchiveSource,
) -> Option<Chunk> {
let (meta, body) = parse_frontmatter(content, &source.schema)?;
let body = body.trim();
if body.is_empty() {
return None;
}
let rel_path = path.strip_prefix(root).unwrap_or(path);
let rel_str = rel_path.to_string_lossy();
let file_path = if looks_like_date_partitioned(&rel_str) {
format!("{}:{}", source.name, rel_str)
} else {
let date_prefix = timestamp_to_date_path(&meta.timestamp);
format!("{}:{}/{}", source.name, date_prefix, rel_str)
};
let lines: Vec<&str> = body.lines().collect();
let end_line = lines.len() as u32;
let mut hasher = Sha256::new();
hasher.update(format!("{}:{}:{}", source.name, meta.id, meta.timestamp).as_bytes());
let id = format!("{:x}", hasher.finalize());
let name_prefix = if source.name_field.is_empty() {
String::new()
} else {
meta.fields
.iter()
.find(|(k, _)| k == &source.name_field)
.map(|(_, v)| v.clone())
.unwrap_or_default()
};
let chunk_name = if name_prefix.is_empty() {
meta.id.clone()
} else {
format!("{}/{}", name_prefix, meta.id)
};
Some(Chunk {
id,
file_path,
chunk_type: ChunkType::Section,
name: Some(chunk_name),
start_line: 1,
end_line,
content: body.to_string(),
language: source.name.clone(),
tags: String::new(),
})
}
fn parse_frontmatter<'a>(content: &'a str, schema: &str) -> Option<(ArchiveMeta, &'a str)> {
let trimmed = content.trim_start();
if !trimmed.starts_with("---") {
return None;
}
let after_fence = trimmed[3..].find("\n---")?;
let fm_text = &trimmed[3..3 + after_fence];
if !fm_text.contains(schema) {
return None;
}
let body_start = 3 + after_fence + 4; let body = if body_start < trimmed.len() {
&trimmed[body_start..]
} else {
""
};
let meta = parse_meta_fields(fm_text);
Some((meta, body))
}
fn parse_meta_fields(fm: &str) -> ArchiveMeta {
let mut meta = ArchiveMeta::default();
let mut last_key = String::new();
for line in fm.lines() {
let trimmed = line.trim();
if trimmed.is_empty() || trimmed.starts_with("schema:") {
continue;
}
if let Some(item) = trimmed.strip_prefix("- ") {
if !last_key.is_empty() {
if let Some(entry) = meta.fields.iter_mut().find(|(k, _)| k == &last_key) {
if entry.1.is_empty() {
entry.1 = item.trim().to_string();
} else {
entry.1 = format!("{},{}", entry.1, item.trim());
}
}
}
continue;
}
if let Some(colon_pos) = trimmed.find(':') {
let key = trimmed[..colon_pos].trim().to_string();
let val = trimmed[colon_pos + 1..].trim().to_string();
if key == "id" {
meta.id = val.clone();
} else if key == "timestamp" {
meta.timestamp = val.clone();
}
let store_val = if val == "null" { String::new() } else { val };
last_key = key.clone();
meta.fields.push((key, store_val));
}
}
if meta.id.is_empty() && !meta.timestamp.is_empty() {
meta.id = meta.timestamp.clone();
}
meta
}
fn looks_like_date_partitioned(rel_path: &str) -> bool {
let parts: Vec<&str> = rel_path.splitn(4, '/').collect();
if parts.len() < 3 {
return false;
}
parts[0].len() == 4
&& parts[0].chars().all(|c| c.is_ascii_digit())
&& parts[1].len() == 2
&& parts[1].chars().all(|c| c.is_ascii_digit())
&& parts[2].len() == 2
&& parts[2].chars().all(|c| c.is_ascii_digit())
}
fn timestamp_to_date_path(timestamp: &str) -> String {
if timestamp.len() >= 10 {
let date_part = ×tamp[..10];
let parts: Vec<&str> = date_part.split('-').collect();
if parts.len() == 3 {
return format!("{}/{}/{}", parts[0], parts[1], parts[2]);
}
}
"unknown".to_string()
}
pub fn matches_schema(content: &str, schema: &str) -> bool {
let trimmed = content.trim_start();
if !trimmed.starts_with("---") {
return false;
}
if let Some(end) = trimmed[3..].find("\n---") {
let fm = &trimmed[3..3 + end];
fm.contains(schema)
} else {
false
}
}
#[cfg(test)]
mod tests {
use super::*;
const SAMPLE_HLA: &str = r#"---
schema: human-intent/v2
id: hi-01ARYZ6S41
timestamp: 2026-02-17T14:32:00Z
author: stiwi
source:
channel: telegram
context: dm
---
Deploy bobbin to node-5, not the old CT.
Make sure traefik points to the new host.
"#;
const SAMPLE_PENSIEVE: &str = r#"---
schema: agent-memory/v1
id: pm-01BXYZ7T52
timestamp: 2026-02-27T01:00:00Z
agent: aegis/crew/arnold
topics:
- mcp-tools
- performance
supersedes: null
confidence: high
---
Bobbin search latency improved from 200ms to 45ms after switching to hybrid mode.
The semantic_weight of 0.7 gives the best results for code search.
"#;
fn hla_source() -> ArchiveSource {
ArchiveSource {
name: "hla".to_string(),
path: "/archive".to_string(),
schema: "human-intent".to_string(),
name_field: "channel".to_string(),
}
}
fn pensieve_source() -> ArchiveSource {
ArchiveSource {
name: "pensieve".to_string(),
path: "/pensieve".to_string(),
schema: "agent-memory".to_string(),
name_field: "agent".to_string(),
}
}
#[test]
fn test_matches_schema() {
assert!(matches_schema(SAMPLE_HLA, "human-intent"));
assert!(!matches_schema(SAMPLE_HLA, "agent-memory"));
assert!(matches_schema(SAMPLE_PENSIEVE, "agent-memory"));
assert!(!matches_schema(SAMPLE_PENSIEVE, "human-intent"));
assert!(!matches_schema("# Not a record", "human-intent"));
}
#[test]
fn test_parse_hla_frontmatter() {
let (meta, body) = parse_frontmatter(SAMPLE_HLA, "human-intent").unwrap();
assert_eq!(meta.id, "hi-01ARYZ6S41");
assert_eq!(meta.timestamp, "2026-02-17T14:32:00Z");
assert!(body.contains("Deploy bobbin to node-5"));
let channel = meta.fields.iter().find(|(k, _)| k == "channel");
assert_eq!(channel.unwrap().1, "telegram");
}
#[test]
fn test_parse_pensieve_frontmatter() {
let (meta, body) = parse_frontmatter(SAMPLE_PENSIEVE, "agent-memory").unwrap();
assert_eq!(meta.id, "pm-01BXYZ7T52");
assert_eq!(meta.timestamp, "2026-02-27T01:00:00Z");
assert!(body.contains("Bobbin search latency"));
let agent = meta.fields.iter().find(|(k, _)| k == "agent");
assert_eq!(agent.unwrap().1, "aegis/crew/arnold");
let topics = meta.fields.iter().find(|(k, _)| k == "topics");
assert_eq!(topics.unwrap().1, "mcp-tools,performance");
}
#[test]
fn test_parse_frontmatter_wrong_schema() {
assert!(parse_frontmatter(SAMPLE_HLA, "agent-memory").is_none());
assert!(parse_frontmatter(SAMPLE_PENSIEVE, "human-intent").is_none());
}
#[test]
fn test_hla_record_to_chunk() {
let source = hla_source();
let root = Path::new("/archive");
let path = Path::new("/archive/2026/02/17/hi-01ARYZ6S41.md");
let chunk = record_to_chunk(root, path, SAMPLE_HLA, &source).unwrap();
assert_eq!(chunk.language, "hla");
assert_eq!(chunk.chunk_type, ChunkType::Section);
assert_eq!(chunk.file_path, "hla:2026/02/17/hi-01ARYZ6S41.md");
assert!(chunk.name.as_deref().unwrap().contains("hi-01ARYZ6S41"));
assert!(chunk.name.as_deref().unwrap().starts_with("telegram/"));
assert!(chunk.content.contains("Deploy bobbin to node-5"));
}
#[test]
fn test_pensieve_record_to_chunk() {
let source = pensieve_source();
let root = Path::new("/pensieve");
let path = Path::new("/pensieve/aegis-crew-arnold/pm-01BXYZ7T52.md");
let chunk = record_to_chunk(root, path, SAMPLE_PENSIEVE, &source).unwrap();
assert_eq!(chunk.language, "pensieve");
assert_eq!(chunk.chunk_type, ChunkType::Section);
assert!(chunk.file_path.starts_with("pensieve:2026/02/27/"));
assert!(chunk.file_path.contains("pm-01BXYZ7T52.md"));
assert_eq!(
chunk.name.as_deref().unwrap(),
"aegis/crew/arnold/pm-01BXYZ7T52"
);
assert!(chunk.content.contains("Bobbin search latency"));
}
#[test]
fn test_empty_body_returns_none() {
let source = hla_source();
let record = "---\nschema: human-intent/v2\nid: hi-test\n---\n";
let root = Path::new("/archive");
let path = Path::new("/archive/test.md");
assert!(record_to_chunk(root, path, record, &source).is_none());
}
#[test]
fn test_looks_like_date_partitioned() {
assert!(looks_like_date_partitioned("2026/02/17/hi-xxx.md"));
assert!(looks_like_date_partitioned("2026/12/01/file.md"));
assert!(!looks_like_date_partitioned("aegis-crew-arnold/pm-xxx.md"));
assert!(!looks_like_date_partitioned("records/pm-xxx.md"));
assert!(!looks_like_date_partitioned("file.md"));
}
#[test]
fn test_timestamp_to_date_path() {
assert_eq!(timestamp_to_date_path("2026-02-27T01:00:00Z"), "2026/02/27");
assert_eq!(timestamp_to_date_path("2026-12-01"), "2026/12/01");
assert_eq!(timestamp_to_date_path("bad"), "unknown");
assert_eq!(timestamp_to_date_path(""), "unknown");
}
#[test]
fn test_custom_source() {
let custom_record = r#"---
schema: field-notes/v1
id: fn-001
timestamp: 2026-03-01T12:00:00Z
project: alpha
---
Field observation: deployment went smoothly.
"#;
let source = ArchiveSource {
name: "fieldnotes".to_string(),
path: "/notes".to_string(),
schema: "field-notes".to_string(),
name_field: "project".to_string(),
};
let root = Path::new("/notes");
let path = Path::new("/notes/fn-001.md");
let chunk = record_to_chunk(root, path, custom_record, &source).unwrap();
assert_eq!(chunk.language, "fieldnotes");
assert_eq!(chunk.name.as_deref().unwrap(), "alpha/fn-001");
assert!(chunk.file_path.starts_with("fieldnotes:2026/03/01/"));
}
}