hamelin_lib 0.22.2

Core library for Hamelin query language
Documentation
use std::collections::BTreeSet;
use std::error::Error;
use std::fs::File;
use std::io::Write;
use std::path::{Path, PathBuf};
use std::process::Command;
use std::{env, fs};

fn main() -> Result<(), Box<dyn Error>> {
    // resolve the current working antlr into a fully qualified path
    let cwd = env::current_dir().unwrap();
    let grammars = vec!["Hamelin", "TrinoType"];
    let antlr_path = cwd.join("bin").join("antlr4-4.8-2-SNAPSHOT-complete.jar");

    for grammar in grammars.into_iter() {
        //ignoring error because we do not need to run anything when deploying to crates.io
        gen_for_grammar(grammar, &antlr_path)?;
    }

    println!("cargo:rerun-if-changed=build.rs");

    println!("cargo:rerun-if-changed={}", antlr_path.display());
    Ok(())
}

fn gen_for_grammar(grammar_file_name: &str, antlr_path: &PathBuf) -> Result<(), Box<dyn Error>> {
    let out_dir = env::var("OUT_DIR")?;
    let dest_path = Path::new(&out_dir);
    // Print the path to easily find it in the Cargo output
    // uncomment this if you're looking to find the stupid shit this thing generates
    // println!("cargo:warning=OUT_DIR is {}", out_dir);

    let input = env::current_dir()?.join("grammars");
    let file_name = grammar_file_name.to_owned() + ".g4";
    let grammar_source = input.join(&file_name);

    // Read the grammar file and apply the transformation
    let grammar_content = fs::read_to_string(&grammar_source)?;
    // ANTLR Rust has panic related challenges with hidden channels. Fuck off, then.
    let transformed_content = grammar_content.replace("-> channel(HIDDEN)", "-> skip");

    // Write the transformed grammar to a temp file in OUT_DIR
    let temp_grammar = dest_path.join(&file_name);
    fs::write(&temp_grammar, transformed_content)?;

    if grammar_file_name == "Hamelin" {
        write_reserved_simple_identifier_tokens(dest_path, &grammar_content)?;
    }

    let out_files = vec![
        dest_path.join(grammar_file_name.to_lowercase() + "lexer.rs"),
        dest_path.join(grammar_file_name.to_lowercase() + "parser.rs"),
        dest_path.join(grammar_file_name.to_lowercase() + "visitor.rs"),
        dest_path.join(grammar_file_name.to_lowercase() + "listener.rs"),
    ];

    let mut c = Command::new("java");
    // What the fuck, Rust
    c.current_dir(dest_path)
        .arg("-jar")
        .arg(antlr_path)
        .arg("-Dlanguage=Rust")
        .arg("-o")
        .arg(dest_path)
        .arg(&file_name)
        .arg("-visitor");

    eprintln!("Running command: {:?}", c);

    c.spawn()?.wait_with_output()?;

    for file in out_files {
        eprintln!("reading {}", file.display());
        let content = fs::read_to_string(file.clone())?;
        eprintln!("opening for write {}", file.display());
        let mut out = File::create(file)?;
        for line in content.lines() {
            if !line.starts_with("#![allow") {
                writeln!(out, "{}", line)?;
            }
        }
    }

    println!("cargo:rerun-if-changed=grammars/{}", file_name);

    Ok(())
}

/// Emit sorted reserved unquoted-identifier spellings from Hamelin.g4.
///
/// ANTLR's generated `_LITERAL_NAMES` only covers single-alternative tokens
/// (mostly operators). Keyword rules like `'SELECT' | 'select'` never appear
/// there, so we scrape exclusive string-literal lexer alternatives from the
/// grammar itself. Interval/trunc rules that concatenate a number with a unit
/// are ignored because a bare unit still lexes as `IDENTIFIER`.
///
/// Build fails loudly when a keyword-shaped rule cannot be classified, or when
/// such a rule is declared after `IDENTIFIER` (ANTLR equal-length tie-break).
fn write_reserved_simple_identifier_tokens(
    dest_path: &Path,
    grammar_content: &str,
) -> Result<(), Box<dyn Error>> {
    let tokens = reserved_simple_identifier_tokens(grammar_content)?;
    if tokens.is_empty() {
        return Err("Hamelin.g4 produced an empty reserved simple-identifier token list".into());
    }

    let mut out = File::create(dest_path.join("reserved_simple_identifiers.rs"))?;
    writeln!(
        out,
        "// @generated by hamelin_lib/build.rs from grammars/Hamelin.g4 — do not edit"
    )?;
    writeln!(out, "const RESERVED_SIMPLE_IDENTIFIER_TOKENS: &[&str] = &[")?;
    for token in &tokens {
        writeln!(out, "    \"{token}\",")?;
    }
    writeln!(out, "];")?;
    Ok(())
}

fn reserved_simple_identifier_tokens(grammar_content: &str) -> Result<Vec<String>, Box<dyn Error>> {
    let rules = uppercase_lexer_rules(grammar_content);
    let identifier_index = rules
        .iter()
        .position(|(name, _)| name == "IDENTIFIER")
        .ok_or("Hamelin.g4 is missing an IDENTIFIER lexer rule")?;

    let mut tokens = BTreeSet::new();
    for (index, (name, body)) in rules.iter().enumerate() {
        if name == "IDENTIFIER" || name == "BACKQUOTED_IDENTIFIER" {
            continue;
        }

        let trimmed = body.trim();
        if !trimmed.starts_with('\'') {
            // Concatenation / fragment-led rules (intervals, truncs, etc.).
            continue;
        }

        let Some(literals) = exclusive_string_literal_alternatives(trimmed) else {
            if looks_like_keyword_literal_alternatives(trimmed) {
                return Err(format!(
                    "Hamelin.g4 lexer rule {name} looks like keyword string alternatives but could not be classified: {trimmed}"
                )
                .into());
            }
            continue;
        };

        let identifier_literals = literals
            .into_iter()
            .filter(|literal| is_identifier_like(literal))
            .collect::<Vec<_>>();
        if identifier_literals.is_empty() {
            continue;
        }

        if index > identifier_index {
            return Err(format!(
                "Hamelin.g4 lexer rule {name} declares identifier-like literals after IDENTIFIER; ANTLR would prefer IDENTIFIER on equal-length ties"
            )
            .into());
        }

        tokens.extend(identifier_literals);
    }

    Ok(tokens.into_iter().collect())
}

fn uppercase_lexer_rules(grammar_content: &str) -> Vec<(String, String)> {
    let mut rules = Vec::new();
    let mut pending_name: Option<String> = None;
    let mut current: Option<(String, String)> = None;

    for raw_line in grammar_content.lines() {
        let line = strip_antlr_line_comment(raw_line).trim().to_owned();
        if line.is_empty() {
            continue;
        }

        if current.is_none() && pending_name.is_none() {
            if is_uppercase_lexer_rule_name(&line) {
                pending_name = Some(line);
                continue;
            }
        }

        if let Some(name) = pending_name.take() {
            if let Some(rest) = line.strip_prefix(':') {
                start_or_finish_rule(&mut rules, &mut current, name, rest);
                continue;
            }
            // Not a multiline rule header after all.
            pending_name = None;
        }

        if let Some((name, rest)) = split_uppercase_lexer_rule_start(&line) {
            start_or_finish_rule(&mut rules, &mut current, name, &rest);
            continue;
        }

        let Some((name, body)) = current.as_mut() else {
            continue;
        };
        if let Some(stripped) = line.strip_suffix(';') {
            if !body.is_empty() {
                body.push(' ');
            }
            body.push_str(stripped.trim());
            rules.push((name.clone(), body.clone()));
            current = None;
        } else {
            if !body.is_empty() {
                body.push(' ');
            }
            body.push_str(&line);
        }
    }

    rules
}

fn start_or_finish_rule(
    rules: &mut Vec<(String, String)>,
    current: &mut Option<(String, String)>,
    name: String,
    rest: &str,
) {
    if let Some(body) = rest.strip_suffix(';').map(str::trim) {
        rules.push((name, body.to_owned()));
        *current = None;
    } else {
        *current = Some((name, rest.trim().to_owned()));
    }
}

fn strip_antlr_line_comment(line: &str) -> String {
    let mut result = String::with_capacity(line.len());
    let mut chars = line.chars().peekable();
    let mut in_string = false;
    while let Some(ch) = chars.next() {
        if in_string {
            result.push(ch);
            if ch == '\\' {
                if let Some(next) = chars.next() {
                    result.push(next);
                }
            } else if ch == '\'' {
                in_string = false;
            }
            continue;
        }
        if ch == '\'' {
            in_string = true;
            result.push(ch);
            continue;
        }
        if ch == '/' && chars.peek() == Some(&'/') {
            break;
        }
        result.push(ch);
    }
    result
}

fn is_uppercase_lexer_rule_name(name: &str) -> bool {
    !name.is_empty()
        && name
            .chars()
            .all(|c| c.is_ascii_uppercase() || c.is_ascii_digit() || c == '_')
        && name
            .chars()
            .next()
            .is_some_and(|c| c.is_ascii_uppercase() || c == '_')
}

fn split_uppercase_lexer_rule_start(line: &str) -> Option<(String, String)> {
    let (name, rest) = line.split_once(':')?;
    let name = name.trim();
    if !is_uppercase_lexer_rule_name(name) {
        return None;
    }
    Some((name.to_owned(), rest.to_owned()))
}

fn exclusive_string_literal_alternatives(body: &str) -> Option<Vec<String>> {
    let mut literals = Vec::new();
    let mut rest = body.trim();
    loop {
        rest = rest.trim_start();
        if !rest.starts_with('\'') {
            return None;
        }
        let after_open = &rest[1..];
        let mut end = None;
        let mut chars = after_open.char_indices().peekable();
        while let Some((index, ch)) = chars.next() {
            if ch == '\\' {
                chars.next();
                continue;
            }
            if ch == '\'' {
                end = Some(index);
                break;
            }
        }
        let end = end?;
        literals.push(after_open[..end].replace("\\'", "'"));
        rest = after_open[end + 1..].trim_start();
        if rest.is_empty() {
            return Some(literals);
        }
        if let Some(next) = rest.strip_prefix('|') {
            rest = next;
            continue;
        }
        return None;
    }
}

fn is_identifier_like(literal: &str) -> bool {
    let mut chars = literal.chars();
    let Some(first) = chars.next() else {
        return false;
    };
    if !(first.is_ascii_alphabetic() || first == '_') {
        return false;
    }
    chars.all(|c| c.is_ascii_alphanumeric() || c == '_')
}

/// True when a failed exclusive-literal parse still looks like `'word' | 'WORD'`.
///
/// Complex lexer patterns (strings, intervals) contain regex metacharacters and
/// are skipped. Ambiguous keyword-shaped bodies must fail the build instead.
fn looks_like_keyword_literal_alternatives(body: &str) -> bool {
    !body.is_empty()
        && body
            .chars()
            .all(|c| c.is_ascii_alphanumeric() || matches!(c, '\'' | '|' | '_' | ' ' | '\t'))
}