use anyhow::Result;
use std::collections::HashMap;
use std::fs;
use std::path::Path;
#[derive(Debug, Clone, Default)]
pub struct Glossary {
pub entries: HashMap<String, String>,
pub sorted_terms: Vec<String>,
}
impl Glossary {
pub fn load(book_dir: &Path) -> Result<Self> {
let glossary_path = book_dir.join("GLOSSARY.md");
if !glossary_path.exists() {
return Ok(Self::default());
}
let content = fs::read_to_string(&glossary_path)?;
Self::parse(&content)
}
pub fn parse(content: &str) -> Result<Self> {
let mut entries = HashMap::new();
let mut current_term: Option<String> = None;
let mut current_definition = String::new();
for line in content.lines() {
let trimmed = line.trim();
if trimmed.starts_with("# ") {
continue;
}
if let Some(heading) = trimmed.strip_prefix("## ") {
if let Some(term) = current_term.take() {
let definition = current_definition.trim().to_string();
if !definition.is_empty() {
entries.insert(term, definition);
}
}
current_term = Some(heading.trim().to_string());
current_definition.clear();
continue;
}
if current_term.is_some() && !trimmed.is_empty() {
if !current_definition.is_empty() {
current_definition.push(' ');
}
current_definition.push_str(trimmed);
}
}
if let Some(term) = current_term {
let definition = current_definition.trim().to_string();
if !definition.is_empty() {
entries.insert(term, definition);
}
}
let mut sorted_terms: Vec<String> = entries.keys().cloned().collect();
sorted_terms.sort_by_key(|b| std::cmp::Reverse(b.len()));
Ok(Self {
entries,
sorted_terms,
})
}
pub fn is_empty(&self) -> bool {
self.entries.is_empty()
}
pub fn get(&self, term: &str) -> Option<&String> {
self.entries.get(term)
}
}
pub fn apply_glossary(html: &str, glossary: &Glossary) -> String {
if glossary.is_empty() {
return html.to_string();
}
let mut result = html.to_string();
for term in &glossary.sorted_terms {
if let Some(definition) = glossary.get(term) {
result = replace_term_in_html(&result, term, definition);
}
}
result
}
fn replace_term_in_html(html: &str, term: &str, definition: &str) -> String {
let mut result = String::new();
let mut chars = html.char_indices().peekable();
let mut in_tag = false;
let mut in_code = false;
let mut in_glossary_span = false;
let mut in_anchor = false;
let mut in_heading = false;
let mut in_script = false;
let mut no_glossary_stack: Vec<String> = Vec::new(); let mut tag_content = String::new();
let mut attr_quote: Option<char> = None;
while let Some((i, c)) = chars.next() {
if c == '<' && !in_tag {
in_tag = true;
tag_content.clear();
attr_quote = None;
result.push(c);
continue;
}
if in_tag {
match attr_quote {
Some(q) if c == q => attr_quote = None,
None if c == '"' || c == '\'' => attr_quote = Some(c),
_ => {}
}
}
if c == '>' && in_tag && attr_quote.is_none() {
in_tag = false;
result.push(c);
let tag_lower = tag_content.to_lowercase();
if tag_lower.starts_with("code") || tag_lower.starts_with("pre") {
in_code = true;
} else if tag_lower.starts_with("/code") || tag_lower.starts_with("/pre") {
in_code = false;
}
else if tag_lower.starts_with("span") && tag_lower.contains("glossary-term") {
in_glossary_span = true;
} else if tag_lower.starts_with("/span") && in_glossary_span {
in_glossary_span = false;
}
else if tag_lower.starts_with("a ") || tag_lower == "a" {
in_anchor = true;
} else if tag_lower.starts_with("/a") {
in_anchor = false;
}
else if tag_lower.starts_with('h') && tag_lower.len() >= 2 {
let second_char = tag_lower.chars().nth(1);
if matches!(second_char, Some('1'..='6')) {
let third_char = tag_lower.chars().nth(2);
if third_char.is_none() || !third_char.unwrap().is_alphabetic() {
in_heading = true;
}
}
} else if tag_lower.starts_with("/h") && tag_lower.len() >= 3 {
let third_char = tag_lower.chars().nth(2);
if matches!(third_char, Some('1'..='6')) {
in_heading = false;
}
}
else if tag_lower.starts_with("script") {
in_script = true;
} else if tag_lower.starts_with("/script") {
in_script = false;
}
if !tag_lower.starts_with('/')
&& tag_lower.contains("class=")
&& tag_lower.contains("no-glossary")
&& !tag_lower.trim_end().ends_with('/')
{
let tag_name = tag_lower
.split_whitespace()
.next()
.unwrap_or("")
.to_string();
if !tag_name.is_empty() && !is_void_element(&tag_name) {
no_glossary_stack.push(tag_name);
}
}
if tag_lower.starts_with('/') && !no_glossary_stack.is_empty() {
let closing_tag = tag_lower
.trim_start_matches('/')
.split_whitespace()
.next()
.unwrap_or("");
if let Some(last) = no_glossary_stack.last() {
if last == closing_tag {
no_glossary_stack.pop();
}
}
}
continue;
}
if in_tag {
tag_content.push(c);
result.push(c);
continue;
}
if in_code
|| in_glossary_span
|| in_anchor
|| in_heading
|| in_script
|| !no_glossary_stack.is_empty()
{
result.push(c);
continue;
}
if html[i..].starts_with(term) {
let term_first = term.chars().next().unwrap_or(' ');
let term_last = term.chars().last().unwrap_or(' ');
let before_ok =
i == 0 || !is_same_word_run(result.chars().last().unwrap_or(' '), term_first);
let after_idx = i + term.len();
let after_ok = after_idx >= html.len()
|| !is_same_word_run(term_last, html[after_idx..].chars().next().unwrap_or(' '));
if before_ok && after_ok {
let escaped_def = html_escape_attribute(definition);
result.push_str(&format!(
r#"<span class="glossary-term" data-definition="{}">{}</span>"#,
escaped_def, term
));
for _ in 0..term.chars().count() - 1 {
chars.next();
}
continue;
}
}
result.push(c);
}
result
}
fn is_void_element(tag_name: &str) -> bool {
matches!(
tag_name,
"area"
| "base"
| "br"
| "col"
| "embed"
| "hr"
| "img"
| "input"
| "link"
| "meta"
| "param"
| "source"
| "track"
| "wbr"
)
}
#[derive(PartialEq)]
enum CharClass {
None,
Alnum,
Hiragana,
Katakana,
Kanji,
}
fn char_class(c: char) -> CharClass {
match c {
'\u{3041}'..='\u{309F}' => CharClass::Hiragana,
'\u{30A0}'..='\u{30FF}' | '\u{31F0}'..='\u{31FF}' | '\u{FF66}'..='\u{FF9F}' => {
CharClass::Katakana
}
'\u{3400}'..='\u{4DBF}' | '\u{4E00}'..='\u{9FFF}' | '\u{F900}'..='\u{FAFF}' => {
CharClass::Kanji
}
_ if c.is_alphanumeric() => CharClass::Alnum,
_ => CharClass::None,
}
}
fn is_same_word_run(a: char, b: char) -> bool {
let ca = char_class(a);
if ca == CharClass::None {
return false;
}
ca == char_class(b)
}
fn html_escape_attribute(s: &str) -> String {
s.replace('&', "&")
.replace('"', """)
.replace('<', "<")
.replace('>', ">")
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn test_parse_glossary() {
let content = r#"# GLOSSARY
## API
Application Programming Interface の略
## SDK
Software Development Kit の略
"#;
let glossary = Glossary::parse(content).unwrap();
assert_eq!(glossary.entries.len(), 2);
assert_eq!(
glossary.get("API"),
Some(&"Application Programming Interface の略".to_string())
);
assert_eq!(
glossary.get("SDK"),
Some(&"Software Development Kit の略".to_string())
);
}
#[test]
fn test_parse_multiline_definition() {
let content = r#"# GLOSSARY
## REST
Representational State Transfer の略。
Web APIの設計スタイルの一つ。
"#;
let glossary = Glossary::parse(content).unwrap();
assert_eq!(
glossary.get("REST"),
Some(
&"Representational State Transfer の略。 Web APIの設計スタイルの一つ。".to_string()
)
);
}
#[test]
fn test_apply_glossary() {
let glossary = Glossary::parse("## API\nInterface").unwrap();
let html = "<p>This is an API example.</p>";
let result = apply_glossary(html, &glossary);
assert!(result
.contains(r#"<span class="glossary-term" data-definition="Interface">API</span>"#));
}
#[test]
fn test_apply_glossary_in_code() {
let glossary = Glossary::parse("## API\nInterface").unwrap();
let html = "<p>Use the <code>API</code> endpoint.</p>";
let result = apply_glossary(html, &glossary);
assert!(result.contains("<code>API</code>"));
assert!(!result.contains("glossary-term"));
}
#[test]
fn test_apply_glossary_word_boundary() {
let glossary = Glossary::parse("## API\nInterface").unwrap();
let html = "<p>The APIARY tool is different from API.</p>";
let result = apply_glossary(html, &glossary);
assert!(result.contains("APIARY"));
assert!(result.contains("glossary-term"));
}
#[test]
fn test_empty_glossary() {
let glossary = Glossary::default();
assert!(glossary.is_empty());
}
#[test]
fn test_sorted_terms_longest_first() {
let content = r#"# GLOSSARY
## API
Application Programming Interface
## REST API
RESTful API
## REST
Representational State Transfer
"#;
let glossary = Glossary::parse(content).unwrap();
assert_eq!(glossary.sorted_terms[0], "REST API");
}
#[test]
fn test_apply_glossary_in_anchor() {
let glossary = Glossary::parse("## API\nInterface").unwrap();
let html = r#"<p>See <a href="/api">API documentation</a> for more info about API.</p>"#;
let result = apply_glossary(html, &glossary);
assert!(result.contains(">API documentation</a>"));
assert!(result.contains(
r#"<span class="glossary-term" data-definition="Interface">API</span>.</p>"#
));
}
#[test]
fn test_apply_glossary_in_heading() {
let glossary = Glossary::parse("## API\nInterface").unwrap();
let html = "<h1>API Overview</h1><p>Learn about API.</p>";
let result = apply_glossary(html, &glossary);
assert!(result.contains("<h1>API Overview</h1>"));
assert!(result
.contains(r#"<span class="glossary-term" data-definition="Interface">API</span>"#));
}
#[test]
fn test_apply_glossary_in_all_headings() {
let glossary = Glossary::parse("## API\nInterface").unwrap();
for level in 1..=6 {
let html = format!("<h{}>API</h{}>", level, level);
let result = apply_glossary(&html, &glossary);
assert!(
!result.contains("glossary-term"),
"h{} should exclude glossary",
level
);
assert!(result.contains(&format!("<h{}>API</h{}>", level, level)));
}
}
#[test]
fn test_apply_glossary_in_script() {
let glossary = Glossary::parse("## API\nInterface").unwrap();
let html = r#"<script>const API = "test";</script><p>Use the API.</p>"#;
let result = apply_glossary(html, &glossary);
assert!(result.contains(r#"<script>const API = "test";</script>"#));
assert!(result
.contains(r#"<span class="glossary-term" data-definition="Interface">API</span>"#));
}
#[test]
fn test_apply_glossary_no_glossary_class() {
let glossary = Glossary::parse("## API\nInterface").unwrap();
let html = r#"<p>About API.</p><div class="no-glossary">API is excluded here.</div><p>API again.</p>"#;
let result = apply_glossary(html, &glossary);
assert!(result.contains(r#"<div class="no-glossary">API is excluded here.</div>"#));
let glossary_count = result.matches("glossary-term").count();
assert_eq!(
glossary_count, 2,
"Should have 2 glossary terms (before and after no-glossary)"
);
}
#[test]
fn test_apply_glossary_no_glossary_nested() {
let glossary = Glossary::parse("## API\nInterface").unwrap();
let html = r#"<div class="no-glossary"><p>API in <span>nested API</span> element.</p></div><p>API outside.</p>"#;
let result = apply_glossary(html, &glossary);
assert!(result.contains(r#"<div class="no-glossary"><p>API in <span>nested API</span>"#));
assert!(result.contains(
r#"<span class="glossary-term" data-definition="Interface">API</span> outside"#
));
}
#[test]
fn test_apply_glossary_header_not_matched() {
let glossary = Glossary::parse("## API\nInterface").unwrap();
let html = "<header>API in header</header><p>API in p.</p>";
let result = apply_glossary(html, &glossary);
let glossary_count = result.matches("glossary-term").count();
assert_eq!(glossary_count, 2, "Both API occurrences should be wrapped");
}
#[test]
fn test_apply_glossary_anchor_with_attributes() {
let glossary = Glossary::parse("## API\nInterface").unwrap();
let html = r#"<a href="/doc" class="link" target="_blank">API Guide</a> and API."#;
let result = apply_glossary(html, &glossary);
assert!(result.contains(">API Guide</a>"));
assert!(result
.contains(r#"<span class="glossary-term" data-definition="Interface">API</span>."#));
}
#[test]
fn test_fuzz_empty_glossary_content() {
let result = Glossary::parse("");
assert!(result.is_ok());
assert!(result.unwrap().is_empty());
}
#[test]
fn test_fuzz_heading_only_no_definition() {
let result = Glossary::parse("## TermWithNoDefinition\n");
assert!(result.is_ok());
}
#[test]
fn test_fuzz_special_chars_in_term() {
let inputs = vec![
"## C++\nLanguage",
"## C#\nLanguage",
"## .NET\nFramework",
"## $variable\nShell variable",
"## term<script>\nXSS attempt",
];
for input in inputs {
let result = Glossary::parse(input);
assert!(result.is_ok(), "Should not panic on: {}", input);
}
}
#[test]
fn test_fuzz_apply_glossary_empty_html() {
let glossary = Glossary::parse("## API\nInterface").unwrap();
let result = apply_glossary("", &glossary);
assert_eq!(result, "");
}
#[test]
fn test_fuzz_apply_glossary_malformed_html() {
let glossary = Glossary::parse("## API\nInterface").unwrap();
let inputs = vec![
"<p>Unclosed tag with API",
"<<>>API<<>>",
"<p>API</p><p>API</p><p>API</p>",
"<div style='color:red'>API</div>",
];
for input in inputs {
let result = apply_glossary(input, &glossary);
let _ = result; }
}
#[test]
fn test_apply_glossary_japanese_term_no_text_loss() {
let glossary = Glossary::parse("## 用語\n説明文です。").unwrap();
let html = "<p>用語 とは何か。</p>";
let result = apply_glossary(html, &glossary);
assert!(
result.contains("とは何か。"),
"text after the term must be preserved: {}",
result
);
assert!(result
.contains(r#"<span class="glossary-term" data-definition="説明文です。">用語</span>"#));
}
#[test]
fn test_apply_glossary_japanese_term_in_sentence() {
let glossary = Glossary::parse("## 用語\n説明").unwrap();
let html = "<p>この用語について説明します。</p>";
let result = apply_glossary(html, &glossary);
assert!(
result.contains("glossary-term"),
"kanji term adjacent to hiragana should match: {}",
result
);
}
#[test]
fn test_apply_glossary_japanese_compound_not_matched() {
let glossary = Glossary::parse("## 用語\n説明").unwrap();
let html = "<p>専門用語集を参照。</p>";
let result = apply_glossary(html, &glossary);
assert!(
!result.contains("glossary-term"),
"term inside kanji compound should not match: {}",
result
);
}
#[test]
fn test_gt_inside_attribute_value_does_not_close_tag() {
let glossary = Glossary::parse("## API\nInterface").unwrap();
let html = r#"<div title="see notes>API here">text about API</div>"#;
let result = apply_glossary(html, &glossary);
assert!(
result.contains(r#"title="see notes>API here""#),
"attribute value must stay intact: {}",
result
);
assert!(
result.contains(r#"about <span class="glossary-term""#),
"body text API must still be wrapped: {}",
result
);
}
#[test]
fn test_no_glossary_on_void_element_does_not_stick() {
let glossary = Glossary::parse("## API\nInterface").unwrap();
let html =
r#"<img src="pic.png" class="no-glossary" alt="d"/><p>This API should be wrapped.</p>"#;
let result = apply_glossary(html, &glossary);
assert!(
result.contains("glossary-term"),
"glossary must not be disabled after a void no-glossary element: {}",
result
);
}
#[test]
fn test_no_glossary_self_closing_div_does_not_stick() {
let glossary = Glossary::parse("## API\nInterface").unwrap();
let html = r#"<div class="no-glossary"/><p>API here.</p>"#;
let result = apply_glossary(html, &glossary);
assert!(result.contains("glossary-term"), "{}", result);
}
#[test]
fn test_fuzz_very_long_term() {
let term = "A".repeat(10000);
let content = format!("## {}\nDefinition", term);
let result = Glossary::parse(&content);
assert!(result.is_ok());
}
#[test]
fn test_fuzz_many_terms() {
let mut content = String::from("# Glossary\n\n");
for i in 0..500 {
content.push_str(&format!("## Term{}\nDefinition {}\n\n", i, i));
}
let result = Glossary::parse(&content);
assert!(result.is_ok());
assert_eq!(result.unwrap().entries.len(), 500);
}
}