Skip to main content

kcode_kmap_size/
lib.rs

1//! Approximate the token footprint of the current text in a Kweb Kmap.
2
3use std::{ffi::OsStr, fs, path::Path, str::FromStr};
4
5use anyhow::{Context, ensure};
6use kcode_kweb_db::{Config, KwebDb, NodeId};
7
8#[derive(Debug, PartialEq, Eq)]
9pub struct KmapSize {
10    pub node_count: u64,
11    pub full_node_characters: u64,
12    pub full_node_words: u64,
13    pub full_node_tokens: u64,
14    pub long_description_characters: u64,
15    pub long_description_words: u64,
16    pub long_description_tokens: u64,
17}
18
19/// Measure the current text fields of every node in a Kweb root.
20///
21/// This opens the database, so callers must enforce any application-specific
22/// offline or mutual-exclusion policy before calling it.
23pub fn measure(path: &Path, config: Config) -> anyhow::Result<KmapSize> {
24    ensure!(
25        path.is_dir(),
26        "Kweb root {} does not exist or is not a directory",
27        path.display()
28    );
29    let database = KwebDb::open(path, config)
30        .map_err(anyhow::Error::new)
31        .with_context(|| format!("opening Kweb database {}", path.display()))?;
32    let mut size = KmapSize {
33        node_count: 0,
34        full_node_characters: 0,
35        full_node_words: 0,
36        full_node_tokens: 0,
37        long_description_characters: 0,
38        long_description_words: 0,
39        long_description_tokens: 0,
40    };
41    for id in node_ids(path)? {
42        let node = database.get_node(id).map_err(anyhow::Error::new)?;
43        let full = format!(
44            "{}{}{}",
45            node.data.short_name, node.data.short_description, node.data.long_description
46        );
47        size.node_count += 1;
48        size.full_node_characters += count_characters(&full);
49        size.full_node_words += count_words(&node.data.short_name)
50            + count_words(&node.data.short_description)
51            + count_words(&node.data.long_description);
52        size.long_description_characters += count_characters(&node.data.long_description);
53        size.long_description_words += count_words(&node.data.long_description);
54    }
55    size.full_node_tokens = size.full_node_characters.div_ceil(4);
56    size.long_description_tokens = size.long_description_characters.div_ceil(4);
57    Ok(size)
58}
59
60fn node_ids(root: &Path) -> anyhow::Result<Vec<NodeId>> {
61    let directory = root.join("nodes");
62    let mut ids = Vec::new();
63    for shard in
64        fs::read_dir(&directory).with_context(|| format!("reading {}", directory.display()))?
65    {
66        let shard = shard?;
67        if !shard.file_type()?.is_dir() {
68            continue;
69        }
70        let prefix = shard.file_name();
71        let prefix = prefix.to_str().context("Kweb node shard is not UTF-8")?;
72        for entry in fs::read_dir(shard.path())? {
73            let entry = entry?;
74            if !entry.file_type()?.is_file() || entry.path().extension() != Some(OsStr::new("kwn"))
75            {
76                continue;
77            }
78            let stem = entry
79                .path()
80                .file_stem()
81                .and_then(OsStr::to_str)
82                .context("Kweb node filename is not UTF-8")?
83                .to_owned();
84            ids.push(
85                NodeId::from_str(&format!("{prefix}{stem}"))
86                    .map_err(anyhow::Error::new)
87                    .context("decoding a Kweb node filename")?,
88            );
89        }
90    }
91    ids.sort_unstable();
92    Ok(ids)
93}
94
95fn count_characters(value: &str) -> u64 {
96    u64::try_from(value.chars().count()).unwrap_or(u64::MAX)
97}
98
99fn count_words(value: &str) -> u64 {
100    u64::try_from(value.split_whitespace().count()).unwrap_or(u64::MAX)
101}
102
103/// Render a human-readable report describing the estimate and its scope.
104pub fn render(size: &KmapSize) -> String {
105    format!(
106        "Kmap size estimate\n\nNodes: {}\nFull node text: ~{} tokens ({} words, {} characters)\nLong descriptions only: ~{} tokens ({} words, {} characters)\n\nEstimate: one token per 4 Unicode characters; node history, provenance, connections, and other tables are excluded.",
107        format_count(size.node_count),
108        format_count(size.full_node_tokens),
109        format_count(size.full_node_words),
110        format_count(size.full_node_characters),
111        format_count(size.long_description_tokens),
112        format_count(size.long_description_words),
113        format_count(size.long_description_characters),
114    )
115}
116
117fn format_count(value: u64) -> String {
118    let digits = value.to_string();
119    let first_group = digits.len() % 3;
120    let mut formatted = String::with_capacity(digits.len() + digits.len() / 3);
121    if first_group != 0 {
122        formatted.push_str(&digits[..first_group]);
123    }
124    for chunk in digits.as_bytes()[first_group..].chunks(3) {
125        if !formatted.is_empty() {
126            formatted.push(',');
127        }
128        formatted.push_str(std::str::from_utf8(chunk).expect("decimal digits are UTF-8"));
129    }
130    formatted
131}
132
133#[cfg(test)]
134mod tests {
135    use super::*;
136
137    #[test]
138    fn renders_readable_grouped_estimates_and_scope() {
139        let output = render(&KmapSize {
140            node_count: 12_345,
141            full_node_characters: 4_938_268,
142            full_node_words: 987_654,
143            full_node_tokens: 1_234_567,
144            long_description_characters: 3_950_616,
145            long_description_words: 800_000,
146            long_description_tokens: 987_654,
147        });
148        assert!(output.contains("Nodes: 12,345"));
149        assert!(output.contains("Full node text: ~1,234,567 tokens"));
150        assert!(output.contains("Long descriptions only: ~987,654 tokens"));
151        assert!(output.contains("history, provenance, connections, and other tables are excluded"));
152    }
153
154    #[test]
155    fn count_formatting_handles_small_and_grouped_values() {
156        assert_eq!(format_count(0), "0");
157        assert_eq!(format_count(12), "12");
158        assert_eq!(format_count(1_000), "1,000");
159        assert_eq!(format_count(1_234_567), "1,234,567");
160    }
161}