tokmat 0.3.3

Standalone high-performance Canadian address parsing engine core
Documentation
//! Typed token model loading and compiled lookup state.

use crate::word_definition::WordDefinition;
use pcre2::bytes::{Regex as Pcre2Regex, RegexBuilder as Pcre2RegexBuilder};
use std::collections::{HashMap, HashSet};
use std::path::Path;

pub type TokenDefinition = Vec<(String, String)>;
pub type TokenClassList = Vec<(String, HashSet<String>)>;

#[derive(Debug)]
pub struct TokenModel {
    token_definitions: TokenDefinition,
    token_class_list: TokenClassList,
    compiled_patterns: Vec<(String, Pcre2Regex)>,
    available_names: HashSet<String>,
    token_class_lookup: HashMap<String, String>,
    word_definition: WordDefinition,
    word_boundary: Pcre2Regex,
}

impl TokenModel {
    /// Build a compiled token model from loaded definitions and classes.
    ///
    /// # Panics
    ///
    /// Panics if a token definition regex is invalid after start/end anchors are applied.
    #[must_use]
    pub fn from_parts(
        token_definitions: TokenDefinition,
        token_class_list: TokenClassList,
    ) -> Self {
        let available_names: HashSet<String> = token_definitions
            .iter()
            .map(|(name, _)| name.clone())
            .collect();
        let compiled_patterns: Vec<(String, Pcre2Regex)> = token_definitions
            .iter()
            .map(|(name, pattern)| {
                let anchored = if pattern.starts_with('^') && pattern.ends_with('$') {
                    pattern.clone()
                } else {
                    format!(
                        "^{}$",
                        pattern.trim_start_matches('^').trim_end_matches('$')
                    )
                };
                (name.clone(), compile_token_pattern(&anchored))
            })
            .collect();

        let word_definition = WordDefinition::default();
        let word_boundary = word_definition.compiled_boundary();
        Self {
            token_class_lookup: build_token_class_lookup(Some(&token_class_list)),
            token_definitions,
            token_class_list,
            compiled_patterns,
            available_names,
            word_definition,
            word_boundary,
        }
    }

    /// Load the default wanParser model layout from a base directory.
    ///
    /// Expected layout:
    /// - `TOKENDEFINITION/TOKENDEFINITONS.param2`
    /// - `TOKENCLASS/*.param`
    ///
    /// # Errors
    ///
    /// Returns an I/O error when the model files cannot be read.
    pub fn load<P: AsRef<Path>>(base_path: P) -> std::io::Result<Self> {
        let base_path = base_path.as_ref();
        let token_definitions =
            load_token_definitions(base_path.join("TOKENDEFINITION/TOKENDEFINITONS.param2"))?;
        let token_class_list = load_token_class_list(base_path.join("TOKENCLASS"))?;
        let word_definition =
            load_word_definition(base_path.join("WORDDEFINITION.param"))?.unwrap_or_default();
        // NOTE: loading a model intentionally does NOT mutate the process-global
        // word definition — that global side effect made concurrently-loaded
        // models contaminate each other. The definition is retained on the model;
        // the tokenizer uses the model's own boundary (`word_boundary`), and
        // callers that drive the global-reading TEL/extractor path (e.g.
        // tokmat-polars) re-apply `word_definition()` per operation.
        let mut model = Self::from_parts(token_definitions, token_class_list);
        model.word_boundary = word_definition.compiled_boundary();
        model.word_definition = word_definition;
        Ok(model)
    }

    /// The word definition (character class that delimits tokens) for this model,
    /// loaded from `WORDDEFINITION.param` or the default `\w\-'`.
    #[must_use]
    pub const fn word_definition(&self) -> &WordDefinition {
        &self.word_definition
    }

    /// This model's compiled word-boundary regex, used by the per-model tokenize
    /// path so it does not depend on the process-global word definition.
    #[must_use]
    pub const fn word_boundary(&self) -> &Pcre2Regex {
        &self.word_boundary
    }

    #[must_use]
    pub const fn token_definitions(&self) -> &TokenDefinition {
        &self.token_definitions
    }

    #[must_use]
    pub const fn token_class_list(&self) -> &TokenClassList {
        &self.token_class_list
    }

    #[must_use]
    pub fn compiled_patterns(&self) -> &[(String, Pcre2Regex)] {
        &self.compiled_patterns
    }

    #[must_use]
    pub const fn available_names(&self) -> &HashSet<String> {
        &self.available_names
    }

    #[must_use]
    pub const fn token_class_lookup(&self) -> &HashMap<String, String> {
        &self.token_class_lookup
    }
}

/// Load `.param2` token definitions from disk.
///
/// # Errors
///
/// Returns an I/O error when the file cannot be read.
pub fn load_token_definitions<P: AsRef<Path>>(path: P) -> std::io::Result<TokenDefinition> {
    let content = std::fs::read_to_string(path)?;
    let mut definitions = Vec::new();

    for line in content.lines().map(str::trim) {
        if line.is_empty() || line.starts_with('#') {
            continue;
        }

        let name = extract_tag_value(line, "NAME");
        let value = extract_tag_value(line, "VALUE");

        if let (Some(name), Some(value)) = (name, value) {
            definitions.push((name, value));
        }
    }

    Ok(definitions)
}

/// Load token class lists from the `TOKENCLASS` directory.
///
/// # Errors
///
/// Returns an I/O error when the directory cannot be read.
pub fn load_token_class_list<P: AsRef<Path>>(dir_path: P) -> std::io::Result<TokenClassList> {
    let mut class_list = Vec::new();
    if !dir_path.as_ref().exists() {
        return Ok(class_list);
    }

    for entry in std::fs::read_dir(dir_path)? {
        let entry = entry?;
        let path = entry.path();
        if !(path.is_file() && path.extension().and_then(|value| value.to_str()) == Some("param")) {
            continue;
        }

        let class_name = path
            .file_stem()
            .and_then(|value| value.to_str())
            .ok_or_else(|| {
                std::io::Error::new(std::io::ErrorKind::InvalidData, "invalid class name")
            })?
            .to_string();
        let content = std::fs::read_to_string(&path)?;
        let values: HashSet<String> = content
            .lines()
            .map(str::trim)
            .filter(|line| !line.is_empty() && !line.starts_with('#'))
            .map(ToString::to_string)
            .collect();
        class_list.push((class_name, values));
    }

    Ok(class_list)
}

/// Load an optional word-definition override from a model.
///
/// The file contains the word character-class contents on its first non-empty,
/// non-comment line (e.g. `\w\-'`, optionally bracketed as `[\w\-']`). When the
/// file is absent, `None` is returned and the built-in default applies.
///
/// # Errors
///
/// Returns an I/O error when the file exists but cannot be read.
pub fn load_word_definition<P: AsRef<Path>>(path: P) -> std::io::Result<Option<WordDefinition>> {
    let path = path.as_ref();
    if !path.exists() {
        return Ok(None);
    }
    let content = std::fs::read_to_string(path)?;
    Ok(content
        .lines()
        .map(str::trim)
        .find(|line| !line.is_empty() && !line.starts_with('#'))
        .map(WordDefinition::new))
}

fn build_token_class_lookup(token_class_list: Option<&TokenClassList>) -> HashMap<String, String> {
    let mut temp_lookup: HashMap<String, Vec<String>> = HashMap::new();
    if let Some(class_list) = token_class_list {
        for (class_name, values) in class_list {
            for value in values {
                temp_lookup
                    .entry(value.clone())
                    .or_default()
                    .push(class_name.clone());
            }
        }
    }

    temp_lookup
        .into_iter()
        .map(|(value, classes)| (value, classes.join("|")))
        .collect()
}

fn extract_tag_value(line: &str, tag: &str) -> Option<String> {
    let opening = format!("<{tag}>");
    let closing = format!("</{tag}>");
    let start = line.find(&opening)? + opening.len();
    let end = line[start..].find(&closing)? + start;
    Some(line[start..end].to_string())
}

fn compile_token_pattern(pattern: &str) -> Pcre2Regex {
    Pcre2RegexBuilder::new()
        .utf(true)
        .ucp(true)
        .jit_if_available(true)
        .build(pattern)
        .expect("valid token definition regex")
}

#[cfg(test)]
mod tests {
    use super::*;
    use std::path::Path;

    #[test]
    fn test_token_model_loads_fixture_model() {
        let base_path = Path::new(env!("CARGO_MANIFEST_DIR")).join("tests/fixtures/model_1");

        let model = TokenModel::load(base_path).expect("fixture model should load");
        assert!(!model.token_definitions().is_empty());
        assert!(!model.available_names().is_empty());
        assert!(!model.compiled_patterns().is_empty());
    }

    #[test]
    fn test_load_word_definition_missing_is_none() {
        let missing = Path::new(env!("CARGO_MANIFEST_DIR"))
            .join("tests/fixtures/model_1/WORDDEFINITION.param");
        assert_eq!(load_word_definition(missing).expect("io"), None);
    }

    #[test]
    fn test_load_word_definition_reads_first_value_line() {
        let dir = std::env::temp_dir().join(format!("tokmat_worddef_{}", std::process::id()));
        std::fs::create_dir_all(&dir).expect("mkdir");
        let path = dir.join("WORDDEFINITION.param");
        std::fs::write(&path, "# the word character class\n[\\w\\-'.]\n").expect("write");

        let definition = load_word_definition(&path).expect("io").expect("present");
        // brackets stripped; comment skipped; derived forms follow the spec
        assert_eq!(definition.chars(), r"\w\-'.");
        assert_eq!(definition.word_regex(), r"[\w\-'.]+");

        std::fs::remove_dir_all(&dir).ok();
    }
}