use std::collections::BTreeSet;
use std::error::Error;
use std::fs::File;
use std::io::Write;
use std::path::{Path, PathBuf};
use std::process::Command;
use std::{env, fs};
fn main() -> Result<(), Box<dyn Error>> {
let cwd = env::current_dir().unwrap();
let grammars = vec!["Hamelin", "TrinoType"];
let antlr_path = cwd.join("bin").join("antlr4-4.8-2-SNAPSHOT-complete.jar");
for grammar in grammars.into_iter() {
gen_for_grammar(grammar, &antlr_path)?;
}
println!("cargo:rerun-if-changed=build.rs");
println!("cargo:rerun-if-changed={}", antlr_path.display());
Ok(())
}
fn gen_for_grammar(grammar_file_name: &str, antlr_path: &PathBuf) -> Result<(), Box<dyn Error>> {
let out_dir = env::var("OUT_DIR")?;
let dest_path = Path::new(&out_dir);
let input = env::current_dir()?.join("grammars");
let file_name = grammar_file_name.to_owned() + ".g4";
let grammar_source = input.join(&file_name);
let grammar_content = fs::read_to_string(&grammar_source)?;
let transformed_content = grammar_content.replace("-> channel(HIDDEN)", "-> skip");
let temp_grammar = dest_path.join(&file_name);
fs::write(&temp_grammar, transformed_content)?;
if grammar_file_name == "Hamelin" {
write_reserved_simple_identifier_tokens(dest_path, &grammar_content)?;
}
let out_files = vec![
dest_path.join(grammar_file_name.to_lowercase() + "lexer.rs"),
dest_path.join(grammar_file_name.to_lowercase() + "parser.rs"),
dest_path.join(grammar_file_name.to_lowercase() + "visitor.rs"),
dest_path.join(grammar_file_name.to_lowercase() + "listener.rs"),
];
let mut c = Command::new("java");
c.current_dir(dest_path)
.arg("-jar")
.arg(antlr_path)
.arg("-Dlanguage=Rust")
.arg("-o")
.arg(dest_path)
.arg(&file_name)
.arg("-visitor");
eprintln!("Running command: {:?}", c);
c.spawn()?.wait_with_output()?;
for file in out_files {
eprintln!("reading {}", file.display());
let content = fs::read_to_string(file.clone())?;
eprintln!("opening for write {}", file.display());
let mut out = File::create(file)?;
for line in content.lines() {
if !line.starts_with("#![allow") {
writeln!(out, "{}", line)?;
}
}
}
println!("cargo:rerun-if-changed=grammars/{}", file_name);
Ok(())
}
fn write_reserved_simple_identifier_tokens(
dest_path: &Path,
grammar_content: &str,
) -> Result<(), Box<dyn Error>> {
let tokens = reserved_simple_identifier_tokens(grammar_content)?;
if tokens.is_empty() {
return Err("Hamelin.g4 produced an empty reserved simple-identifier token list".into());
}
let mut out = File::create(dest_path.join("reserved_simple_identifiers.rs"))?;
writeln!(
out,
"// @generated by hamelin_lib/build.rs from grammars/Hamelin.g4 — do not edit"
)?;
writeln!(out, "const RESERVED_SIMPLE_IDENTIFIER_TOKENS: &[&str] = &[")?;
for token in &tokens {
writeln!(out, " \"{token}\",")?;
}
writeln!(out, "];")?;
Ok(())
}
fn reserved_simple_identifier_tokens(grammar_content: &str) -> Result<Vec<String>, Box<dyn Error>> {
let rules = uppercase_lexer_rules(grammar_content);
let identifier_index = rules
.iter()
.position(|(name, _)| name == "IDENTIFIER")
.ok_or("Hamelin.g4 is missing an IDENTIFIER lexer rule")?;
let mut tokens = BTreeSet::new();
for (index, (name, body)) in rules.iter().enumerate() {
if name == "IDENTIFIER" || name == "BACKQUOTED_IDENTIFIER" {
continue;
}
let trimmed = body.trim();
if !trimmed.starts_with('\'') {
continue;
}
let Some(literals) = exclusive_string_literal_alternatives(trimmed) else {
if looks_like_keyword_literal_alternatives(trimmed) {
return Err(format!(
"Hamelin.g4 lexer rule {name} looks like keyword string alternatives but could not be classified: {trimmed}"
)
.into());
}
continue;
};
let identifier_literals = literals
.into_iter()
.filter(|literal| is_identifier_like(literal))
.collect::<Vec<_>>();
if identifier_literals.is_empty() {
continue;
}
if index > identifier_index {
return Err(format!(
"Hamelin.g4 lexer rule {name} declares identifier-like literals after IDENTIFIER; ANTLR would prefer IDENTIFIER on equal-length ties"
)
.into());
}
tokens.extend(identifier_literals);
}
Ok(tokens.into_iter().collect())
}
fn uppercase_lexer_rules(grammar_content: &str) -> Vec<(String, String)> {
let mut rules = Vec::new();
let mut pending_name: Option<String> = None;
let mut current: Option<(String, String)> = None;
for raw_line in grammar_content.lines() {
let line = strip_antlr_line_comment(raw_line).trim().to_owned();
if line.is_empty() {
continue;
}
if current.is_none() && pending_name.is_none() {
if is_uppercase_lexer_rule_name(&line) {
pending_name = Some(line);
continue;
}
}
if let Some(name) = pending_name.take() {
if let Some(rest) = line.strip_prefix(':') {
start_or_finish_rule(&mut rules, &mut current, name, rest);
continue;
}
pending_name = None;
}
if let Some((name, rest)) = split_uppercase_lexer_rule_start(&line) {
start_or_finish_rule(&mut rules, &mut current, name, &rest);
continue;
}
let Some((name, body)) = current.as_mut() else {
continue;
};
if let Some(stripped) = line.strip_suffix(';') {
if !body.is_empty() {
body.push(' ');
}
body.push_str(stripped.trim());
rules.push((name.clone(), body.clone()));
current = None;
} else {
if !body.is_empty() {
body.push(' ');
}
body.push_str(&line);
}
}
rules
}
fn start_or_finish_rule(
rules: &mut Vec<(String, String)>,
current: &mut Option<(String, String)>,
name: String,
rest: &str,
) {
if let Some(body) = rest.strip_suffix(';').map(str::trim) {
rules.push((name, body.to_owned()));
*current = None;
} else {
*current = Some((name, rest.trim().to_owned()));
}
}
fn strip_antlr_line_comment(line: &str) -> String {
let mut result = String::with_capacity(line.len());
let mut chars = line.chars().peekable();
let mut in_string = false;
while let Some(ch) = chars.next() {
if in_string {
result.push(ch);
if ch == '\\' {
if let Some(next) = chars.next() {
result.push(next);
}
} else if ch == '\'' {
in_string = false;
}
continue;
}
if ch == '\'' {
in_string = true;
result.push(ch);
continue;
}
if ch == '/' && chars.peek() == Some(&'/') {
break;
}
result.push(ch);
}
result
}
fn is_uppercase_lexer_rule_name(name: &str) -> bool {
!name.is_empty()
&& name
.chars()
.all(|c| c.is_ascii_uppercase() || c.is_ascii_digit() || c == '_')
&& name
.chars()
.next()
.is_some_and(|c| c.is_ascii_uppercase() || c == '_')
}
fn split_uppercase_lexer_rule_start(line: &str) -> Option<(String, String)> {
let (name, rest) = line.split_once(':')?;
let name = name.trim();
if !is_uppercase_lexer_rule_name(name) {
return None;
}
Some((name.to_owned(), rest.to_owned()))
}
fn exclusive_string_literal_alternatives(body: &str) -> Option<Vec<String>> {
let mut literals = Vec::new();
let mut rest = body.trim();
loop {
rest = rest.trim_start();
if !rest.starts_with('\'') {
return None;
}
let after_open = &rest[1..];
let mut end = None;
let mut chars = after_open.char_indices().peekable();
while let Some((index, ch)) = chars.next() {
if ch == '\\' {
chars.next();
continue;
}
if ch == '\'' {
end = Some(index);
break;
}
}
let end = end?;
literals.push(after_open[..end].replace("\\'", "'"));
rest = after_open[end + 1..].trim_start();
if rest.is_empty() {
return Some(literals);
}
if let Some(next) = rest.strip_prefix('|') {
rest = next;
continue;
}
return None;
}
}
fn is_identifier_like(literal: &str) -> bool {
let mut chars = literal.chars();
let Some(first) = chars.next() else {
return false;
};
if !(first.is_ascii_alphabetic() || first == '_') {
return false;
}
chars.all(|c| c.is_ascii_alphanumeric() || c == '_')
}
fn looks_like_keyword_literal_alternatives(body: &str) -> bool {
!body.is_empty()
&& body
.chars()
.all(|c| c.is_ascii_alphanumeric() || matches!(c, '\'' | '|' | '_' | ' ' | '\t'))
}