use crate::error::{Result, TemplateError};
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub(crate) enum TagSyntax {
Angle,
Square,
}
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub(crate) enum ExprCtx {
Tag { square: bool },
Interp,
}
#[derive(Debug, Clone, PartialEq)]
pub(crate) enum Tok {
Ident(String),
Number(String),
Str(String),
RawStr(String),
True,
False,
In,
As,
Using,
Lt,
Lte,
Gt,
Gte,
Plus,
Minus,
Times,
DoubleStar,
Divide,
Percent,
PlusEq,
MinusEq,
TimesEq,
DivEq,
ModEq,
PlusPlus,
MinusMinus,
Eq,
NotEq,
Exclam,
Exists,
Builtin,
And,
Or,
LambdaArrow,
Dot,
DotDot,
DotDotLess,
DotDotStar,
Ellipsis,
Comma,
Semicolon,
Colon,
OpenParen,
CloseParen,
OpenBracket,
CloseBracket,
OpenCurly,
CloseCurly,
TagEnd,
EmptyTagEnd,
InterpEnd,
Eof,
}
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub(crate) enum TextStop {
Eof,
Tag,
Interp,
}
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub(crate) enum TagOpen {
Dir { square: bool },
Call { square: bool },
EndDir { square: bool },
EndCall { square: bool },
TerseComment { square: bool },
}
#[derive(Debug, Clone, Copy)]
pub(crate) struct LexerPos {
pos: usize,
line: u32,
col: u32,
paren_depth: u32,
bracket_depth: u32,
curly_depth: u32,
}
pub(crate) struct Lexer {
chars: Vec<char>,
pos: usize,
line: u32,
col: u32,
pub(crate) paren_depth: u32,
pub(crate) bracket_depth: u32,
pub(crate) curly_depth: u32,
pub(crate) strict_syntax: bool,
pub(crate) tag_syntax: Option<TagSyntax>,
name: String,
}
impl Lexer {
pub(crate) fn new(name: &str, text: &str, strict_syntax: bool) -> Self {
Lexer {
chars: text.chars().collect(),
pos: 0,
line: 1,
col: 1,
paren_depth: 0,
bracket_depth: 0,
curly_depth: 0,
strict_syntax,
tag_syntax: None,
name: name.to_string(),
}
}
pub(crate) fn save(&self) -> LexerPos {
LexerPos {
pos: self.pos,
line: self.line,
col: self.col,
paren_depth: self.paren_depth,
bracket_depth: self.bracket_depth,
curly_depth: self.curly_depth,
}
}
pub(crate) fn restore(&mut self, p: &LexerPos) {
self.pos = p.pos;
self.line = p.line;
self.col = p.col;
self.paren_depth = p.paren_depth;
self.bracket_depth = p.bracket_depth;
self.curly_depth = p.curly_depth;
}
pub(crate) fn peek(&self) -> Option<char> {
self.chars.get(self.pos).copied()
}
pub(crate) fn peek_at(&self, offset: usize) -> Option<char> {
self.chars.get(self.pos + offset).copied()
}
pub(crate) fn bump(&mut self) -> Option<char> {
let c = self.chars.get(self.pos).copied()?;
self.pos += 1;
match c {
'\n' => {
self.line += 1;
self.col = 1;
}
'\r' => {
self.col = 1;
}
_ => {
self.col += 1;
}
}
Some(c)
}
pub(crate) fn line_col(&self) -> (u32, u32) {
(self.line, self.col)
}
pub(crate) fn err(&self, line: u32, col: u32, details: impl Into<String>) -> TemplateError {
TemplateError::Parse {
template: self.name.clone(),
line,
col,
message: details.into(),
}
}
pub(crate) fn skip_ws(&mut self) {
while let Some(c) = self.peek() {
if c == ' ' || c == '\t' || c == '\n' || c == '\r' {
self.bump();
} else {
break;
}
}
}
pub(crate) fn starts_tag(&mut self) -> bool {
let save = self.save();
let r = self.starts_tag_inner();
self.restore(&save);
r
}
fn starts_tag_inner(&mut self) -> bool {
match self.peek() {
Some('<') => match self.peek_at(1) {
Some('#') => match self.peek_at(2) {
Some(c) if c.is_ascii_alphabetic() || c == '_' => true,
Some('-') => self.peek_at(3) == Some('-'),
_ => false,
},
Some('@') => true,
Some('/') => match self.peek_at(2) {
Some('#') | Some('@') => true,
_ => {
if !self.strict_syntax {
self.bump();
self.bump();
return self.non_strict_tag_name(true);
}
false
}
},
Some(c) if c.is_ascii_alphabetic() || c == '_' => {
if self.strict_syntax {
false
} else {
self.bump();
self.non_strict_tag_name(false)
}
}
_ => false,
},
Some('[') => match self.peek_at(1) {
Some('#') => match self.peek_at(2) {
Some(c) if c.is_ascii_alphabetic() || c == '_' => true,
Some('-') => self.peek_at(3) == Some('-'),
_ => false,
},
Some('@') => true,
Some('/') => matches!(self.peek_at(2), Some('#') | Some('@')),
_ => false,
},
_ => false,
}
}
fn non_strict_tag_name(&mut self, is_end: bool) -> bool {
let name = self.read_name();
let Some(name) = name else { return false };
if !DIRECTIVE_NAMES.contains(&name.as_str()) || name == "ftl" {
return false;
}
if is_end {
self.skip_ws();
return matches!(self.peek(), Some('>') | Some(']'))
|| (self.peek() == Some('/') && matches!(self.peek_at(1), Some('>') | Some(']')));
}
let param_required = PARAM_DIRECTIVES.contains(&name.as_str());
if param_required {
matches!(self.peek(), Some(c) if c == ' ' || c == '\t' || c == '\n' || c == '\r')
} else {
self.skip_ws();
match self.peek() {
Some('>') | Some(']') => true,
Some('/') => matches!(self.peek_at(1), Some('>') | Some(']')),
_ => false,
}
}
}
pub(crate) fn read_name(&mut self) -> Option<String> {
let mut s = String::new();
while let Some(c) = self.peek() {
if c.is_ascii_alphabetic() || c == '_' {
s.push(c);
self.bump();
} else {
break;
}
}
if s.is_empty() {
None
} else {
Some(s)
}
}
pub(crate) fn read_tag_open(&mut self) -> TagOpen {
let square = self.peek() == Some('[');
self.bump(); match self.peek() {
Some('#') => {
self.bump();
if self.peek() == Some('-') && self.peek_at(1) == Some('-') {
self.bump();
self.bump();
TagOpen::TerseComment { square }
} else {
TagOpen::Dir { square }
}
}
Some('@') => {
self.bump();
TagOpen::Call { square }
}
Some('/') => {
self.bump();
match self.peek() {
Some('@') => {
self.bump();
TagOpen::EndCall { square }
}
Some('#') => {
self.bump();
TagOpen::EndDir { square }
}
_ => TagOpen::EndDir { square },
}
}
Some(c) if c.is_ascii_alphabetic() || c == '_' => TagOpen::Dir { square },
_ => unreachable!("starts_tag 已保证标签开头"),
}
}
pub(crate) fn try_read_tag_end(&mut self) -> Option<bool> {
while let Some(c) = self.peek() {
if c == ' ' || c == '\t' || c == '\n' || c == '\r' {
self.bump();
} else {
break;
}
}
match self.peek() {
Some('>') => {
self.bump();
Some(false)
}
Some(']') => {
self.bump();
Some(false)
}
Some('/') => match self.peek_at(1) {
Some('>') | Some(']') => {
self.bump();
self.bump();
Some(true)
}
_ => None,
},
_ => None,
}
}
pub(crate) fn scan_text_chunk(&mut self) -> Result<(String, TextStop)> {
let mut text = String::new();
loop {
match self.peek() {
None => return Ok((text, TextStop::Eof)),
Some('<') | Some('[') => {
if self.starts_tag() {
return Ok((text, TextStop::Tag));
}
text.push(self.bump().unwrap());
}
Some('$') => {
if self.peek_at(1) == Some('{') {
return Ok((text, TextStop::Interp));
}
if self.peek_at(1) == Some('$') && self.peek_at(2) == Some('{') {
text.push(self.bump().unwrap());
continue;
}
text.push(self.bump().unwrap());
}
Some('#') => {
if self.peek_at(1) == Some('{') {
return Ok((text, TextStop::Interp));
}
text.push(self.bump().unwrap());
}
Some(c) => {
text.push(c);
self.bump();
}
}
}
}
pub(crate) fn next_expr_token(&mut self, ctx: ExprCtx) -> Result<(Tok, u32, u32, u32, u32)> {
loop {
self.skip_ws();
let (line, col) = self.line_col();
let c = match self.peek() {
None => return Ok((Tok::Eof, line, col, line, col)),
Some(c) => c,
};
if c == '<' || c == '[' {
let n1 = self.peek_at(1);
let n2 = self.peek_at(2);
if matches!(n1, Some('#') | Some('!'))
&& n2 == Some('-')
&& self.peek_at(3) == Some('-')
{
self.bump();
self.bump();
self.bump();
self.bump();
self.skip_expr_comment()?;
continue;
}
}
let (tok, _, _) = self.scan_expr_token_after_ws(ctx, line, col)?;
let (el, ec) = self.line_col();
return Ok((tok, line, col, el, ec));
}
}
fn skip_expr_comment(&mut self) -> Result<()> {
loop {
match self.peek() {
None => {
return Err(self.err(self.line, self.col, "Unclosed comment in expression."))
}
Some('-') if self.peek_at(1) == Some('-') => match self.peek_at(2) {
Some('>') | Some(']') => {
self.bump();
self.bump();
self.bump();
return Ok(());
}
_ => {
self.bump();
}
},
_ => {
self.bump();
}
}
}
}
fn scan_expr_token_after_ws(
&mut self,
ctx: ExprCtx,
line: u32,
col: u32,
) -> Result<(Tok, u32, u32)> {
let c = self.peek().unwrap();
if (c == '$' || c == '#') && self.peek_at(1) == Some('{')
|| c == '[' && self.peek_at(1) == Some('=')
{
let img = if c == '[' {
"[="
} else if c == '$' {
"${"
} else {
"#{"
};
let closer = if c == '[' { "]" } else { "}" };
self.bump();
self.bump();
return Err(self.err(
line,
col,
format!(
"You can't use {img}...{closer} (an interpolation) here as you are \
already in FreeMarker-expression-mode. Thus, instead of {img}myExpression{closer}, \
just write myExpression. ({img}...{closer} is only used where otherwise static \
text is expected, i.e., outside FreeMarker tags and interpolations, or inside \
string literals.)"
),
));
}
let tok = match c {
'<' => {
self.bump();
if self.peek() == Some('=') {
self.bump();
Tok::Lte
} else {
Tok::Lt
}
}
'>' => {
if self.paren_depth > 0 {
self.bump();
if self.peek() == Some('=') {
self.bump();
Tok::Gte
} else {
Tok::Gt
}
} else {
match ctx {
ExprCtx::Tag { .. } => {
self.bump();
Tok::TagEnd
}
ExprCtx::Interp => {
self.bump();
if self.peek() == Some('=') {
self.bump();
Tok::Gte
} else {
Tok::Gt
}
}
}
}
}
']' => {
if self.bracket_depth > 0 {
self.bump();
self.bracket_depth -= 1;
Tok::CloseBracket
} else {
match ctx {
ExprCtx::Tag { square: true } => {
self.bump();
Tok::TagEnd
}
_ => {
return Err(self.err(
line,
col,
"You can't have a \"]\" here, as there's nothing open that it could close.",
));
}
}
}
}
'}' => {
if self.curly_depth > 0 {
self.bump();
self.curly_depth -= 1;
Tok::CloseCurly
} else {
match ctx {
ExprCtx::Interp => {
self.bump();
Tok::InterpEnd
}
ExprCtx::Tag { .. } => {
return Err(self.err(
line,
col,
"You can't have a \"}\" here, as there's nothing open that it could close.",
));
}
}
}
}
')' => {
self.bump();
self.paren_depth = self.paren_depth.saturating_sub(1);
Tok::CloseParen
}
'(' => {
self.bump();
self.paren_depth += 1;
Tok::OpenParen
}
'[' => {
self.bump();
self.bracket_depth += 1;
Tok::OpenBracket
}
'{' => {
self.bump();
self.curly_depth += 1;
Tok::OpenCurly
}
'=' => {
self.bump();
if self.peek() == Some('=') {
self.bump();
}
Tok::Eq
}
'!' => {
self.bump();
if self.peek() == Some('=') {
self.bump();
Tok::NotEq
} else {
Tok::Exclam
}
}
'?' => {
self.bump();
if self.peek() == Some('?') {
self.bump();
Tok::Exists
} else {
Tok::Builtin
}
}
'+' => {
self.bump();
match self.peek() {
Some('+') => {
self.bump();
Tok::PlusPlus
}
Some('=') => {
self.bump();
Tok::PlusEq
}
_ => Tok::Plus,
}
}
'-' => {
self.bump();
match self.peek() {
Some('-') => {
self.bump();
Tok::MinusMinus
}
Some('=') => {
self.bump();
Tok::MinusEq
}
Some('>') => {
self.bump();
Tok::LambdaArrow
}
Some('&')
if self.peek_at(1) == Some('g')
&& self.peek_at(2) == Some('t')
&& self.peek_at(3) == Some(';') =>
{
self.bump();
self.bump();
self.bump();
self.bump();
Tok::LambdaArrow
}
_ => Tok::Minus,
}
}
'*' => {
self.bump();
match self.peek() {
Some('*') => {
self.bump();
Tok::DoubleStar
}
Some('=') => {
self.bump();
Tok::TimesEq
}
_ => Tok::Times,
}
}
'/' => {
self.bump();
if self.peek() == Some('=') {
self.bump();
Tok::DivEq
} else if matches!(ctx, ExprCtx::Tag { .. })
&& matches!(self.peek(), Some('>') | Some(']'))
{
self.bump();
Tok::EmptyTagEnd
} else {
Tok::Divide
}
}
'%' => {
self.bump();
if self.peek() == Some('=') {
self.bump();
Tok::ModEq
} else {
Tok::Percent
}
}
'&' => {
if self.peek_at(1) == Some('l')
&& self.peek_at(2) == Some('t')
&& self.peek_at(3) == Some(';')
{
self.bump();
self.bump();
self.bump();
self.bump();
if self.peek() == Some('=') {
self.bump();
Tok::Lte
} else {
Tok::Lt
}
} else if self.peek_at(1) == Some('g')
&& self.peek_at(2) == Some('t')
&& self.peek_at(3) == Some(';')
{
self.bump();
self.bump();
self.bump();
self.bump();
if self.peek() == Some('=') {
self.bump();
Tok::Gte
} else {
Tok::Gt
}
} else if self.peek_at(1) == Some('a')
&& self.peek_at(2) == Some('m')
&& self.peek_at(3) == Some('p')
&& self.peek_at(4) == Some(';')
&& self.peek_at(5) == Some('&')
&& self.peek_at(6) == Some('a')
&& self.peek_at(7) == Some('m')
&& self.peek_at(8) == Some('p')
&& self.peek_at(9) == Some(';')
{
for _ in 0..10 {
self.bump();
}
Tok::And
} else {
self.bump();
if self.peek() == Some('&') {
self.bump();
}
Tok::And
}
}
'|' => {
self.bump();
if self.peek() == Some('|') {
self.bump();
}
Tok::Or
}
',' => {
self.bump();
Tok::Comma
}
';' => {
self.bump();
Tok::Semicolon
}
':' => {
self.bump();
Tok::Colon
}
'.' => {
self.bump();
if self.peek() == Some('.') {
self.bump();
if self.peek() == Some('.') {
self.bump();
Tok::Ellipsis
} else if matches!(self.peek(), Some('<') | Some('!')) {
self.bump();
Tok::DotDotLess
} else if self.peek() == Some('*') {
self.bump();
Tok::DotDotStar
} else {
Tok::DotDot
}
} else {
Tok::Dot
}
}
'\\' => {
if matches!(
self.peek_at(1),
Some('-') | Some('.') | Some(':') | Some('#')
) {
Tok::Ident(self.scan_ident())
} else {
let n1 = self.peek_at(1);
match n1 {
Some('a')
if self.peek_at(2) == Some('n') && self.peek_at(3) == Some('d') =>
{
for _ in 0..4 {
self.bump();
}
Tok::And
}
Some('l') => {
if self.peek_at(2) == Some('t') {
self.bump();
self.bump();
if self.peek() == Some('e') {
self.bump();
Tok::Lte
} else {
Tok::Lt
}
} else {
return Err(self.err(
line,
col,
format!("Unexpected character \"\\\\{n1:?}\"."),
));
}
}
Some('g') if self.peek_at(2) == Some('t') => {
self.bump();
self.bump();
if self.peek() == Some('e') {
self.bump();
Tok::Gte
} else {
Tok::Gt
}
}
_ => {
return Err(self.err(
line,
col,
format!("Unexpected character \"\\\\{n1:?}\"."),
));
}
}
}
}
'"' | '\'' => {
let (tok, _) = self.scan_string_token()?;
tok
}
'r' if matches!(self.peek_at(1), Some('"') | Some('\'')) => {
self.bump(); let quote = self.peek().unwrap(); self.bump();
let mut s = String::new();
loop {
match self.peek() {
None => {
return Err(self.err(line, col, "Unclosed raw string literal."));
}
Some(q) if q == quote => {
self.bump();
break;
}
Some(q) => {
s.push(q);
self.bump();
}
}
}
Tok::RawStr(s)
}
c if c.is_ascii_digit() => {
let raw = self.scan_number_raw();
Tok::Number(raw)
}
c if is_ident_start(c) => {
let name = self.scan_ident();
match name.as_str() {
"true" => Tok::True,
"false" => Tok::False,
"in" => Tok::In,
"as" => Tok::As,
"using" => Tok::Using,
"lt" => Tok::Lt,
"lte" => Tok::Lte,
"gt" => Tok::Gt,
"gte" => Tok::Gte,
_ => Tok::Ident(name),
}
}
c => {
return Err(self.err(line, col, format!("Unexpected character \"{c}\".")));
}
};
Ok((tok, line, col))
}
fn scan_number_raw(&mut self) -> String {
let mut s = String::new();
while matches!(self.peek(), Some(c) if c.is_ascii_digit()) {
s.push(self.bump().unwrap());
}
if s == "0" && matches!(self.peek(), Some('x') | Some('X')) {
self.bump();
s.push('x');
while matches!(self.peek(), Some(c) if c.is_ascii_hexdigit()) {
s.push(self.bump().unwrap());
}
} else {
if self.peek() == Some('.') && matches!(self.peek_at(1), Some(c) if c.is_ascii_digit())
{
s.push(self.bump().unwrap());
while matches!(self.peek(), Some(c) if c.is_ascii_digit()) {
s.push(self.bump().unwrap());
}
}
if matches!(self.peek(), Some('e') | Some('E')) {
let sign_ok = matches!(self.peek_at(1), Some('+') | Some('-'))
&& matches!(self.peek_at(2), Some(c) if c.is_ascii_digit());
let digit_ok = matches!(self.peek_at(1), Some(c) if c.is_ascii_digit());
if sign_ok || digit_ok {
s.push(self.bump().unwrap());
if matches!(self.peek(), Some('+') | Some('-')) {
s.push(self.bump().unwrap());
}
while matches!(self.peek(), Some(c) if c.is_ascii_digit()) {
s.push(self.bump().unwrap());
}
}
}
}
if matches!(
self.peek(),
Some('l')
| Some('L')
| Some('f')
| Some('F')
| Some('d')
| Some('D')
| Some('b')
| Some('B')
) {
s.push(self.bump().unwrap());
}
s
}
fn add_escapes(s: &str) -> String {
let mut out = String::new();
for c in s.chars() {
match c {
'\0' => {}
'\u{08}' => out.push_str("\\b"),
'\t' => out.push_str("\\t"),
'\n' => out.push_str("\\n"),
'\u{0c}' => out.push_str("\\f"),
'\r' => out.push_str("\\r"),
'"' => out.push_str("\\\""),
'\'' => out.push_str("\\'"),
'\\' => out.push_str("\\\\"),
c if (c as u32) < 0x20 || (c as u32) > 0x7e => {
out.push_str(&format!("\\u{:04x}", c as u32));
}
c => out.push(c),
}
}
out
}
fn scan_string_token(&mut self) -> Result<(Tok, u32)> {
let (_line, col) = self.line_col();
let quote = self.bump().unwrap();
let mut s = String::new();
loop {
match self.peek() {
None => {
return Err(self.err(
self.line,
self.col,
format!(
"Lexical error: encountered <EOF> after \"{}\".",
Self::add_escapes(&format!("{quote}{s}"))
),
));
}
Some(q) if q == quote => {
self.bump();
return Ok((Tok::Str(s), col));
}
Some('\\') => {
s.push('\\');
self.bump();
match self.peek() {
None => {
return Err(self.err(
self.line,
self.col,
format!(
"Lexical error: encountered <EOF> after \"{}\".",
Self::add_escapes(&format!("{quote}{s}\\"))
),
));
}
Some(c) => {
s.push(c);
self.bump();
}
}
}
Some(c) => {
s.push(c);
self.bump();
}
}
}
}
pub(crate) fn scan_ident(&mut self) -> String {
let mut s = String::new();
while let Some(c) = self.peek() {
if is_ident_continue(c) {
s.push(c);
self.bump();
} else if c == '\\'
&& matches!(
self.peek_at(1),
Some('-') | Some('.') | Some(':') | Some('#')
)
{
self.bump();
s.push(self.bump().unwrap());
} else {
break;
}
}
s
}
pub(crate) fn scan_comment(&mut self, square: bool) -> Result<(String, u32, u32)> {
let (line, col) = self.line_col();
let term: [char; 3] = if square {
['-', '-', ']']
} else {
['-', '-', '>']
};
let mut s = String::new();
loop {
match self.peek() {
None => {
let open = if square { "[#--" } else { "<#--" };
return Err(self.err(
line,
col.saturating_sub(4).max(1),
format!("Unclosed \"{open}\""),
));
}
Some('-') if self.peek_at(1) == Some('-') && self.peek_at(2) == Some(term[2]) => {
self.bump();
self.bump();
self.bump();
return Ok((s, line, col));
}
Some(c) => {
s.push(c);
self.bump();
}
}
}
}
pub(crate) fn scan_unparsed(&mut self, end_name: &str) -> Result<(String, u32, u32)> {
let (line, col) = self.line_col();
let mut s = String::new();
loop {
match self.peek() {
None => {
return Err(self.err(
line,
col,
format!("Unclosed \"<#{end_name}>\" (missing \"</#{end_name}>\" or \"</{end_name}>\")."),
));
}
Some('<') | Some('[') => {
let save = self.save();
let n1 = self.peek_at(1);
if n1 == Some('/') {
self.bump();
self.bump();
if self.peek() == Some('#') {
self.bump();
}
if let Some(name) = self.read_name() {
if name.eq_ignore_ascii_case(end_name) {
let mut ok = false;
self.skip_ws();
match self.peek() {
Some('>') | Some(']') => {
self.bump();
ok = true;
}
_ => {}
}
if ok {
return Ok((s, line, col));
}
}
}
self.restore(&save);
s.push(self.bump().unwrap());
} else {
s.push(self.bump().unwrap());
}
}
Some(c) => {
s.push(c);
self.bump();
}
}
}
}
}
fn is_ident_start(c: char) -> bool {
c.is_alphabetic() || c == '$' || c == '_' || c == '@'
}
fn is_ident_continue(c: char) -> bool {
is_ident_start(c) || c.is_ascii_digit()
}
pub(crate) const DIRECTIVE_NAMES: &[&str] = &[
"attempt",
"recover",
"if",
"elseif",
"else",
"list",
"items",
"sep",
"switch",
"case",
"default",
"assign",
"global",
"local",
"include",
"import",
"macro",
"function",
"stop",
"return",
"break",
"continue",
"nested",
"flush",
"t",
"lt",
"rt",
"nt",
"compress",
"comment",
"noparse",
"escape",
"noescape",
"trim",
"autoesc",
"noautoesc",
"outputformat",
"setting",
"call",
"foreach",
"transform",
"visit",
"recurse",
"on",
"fallback",
"ftl",
];
pub(crate) const PARAM_DIRECTIVES: &[&str] = &[
"if",
"elseif",
"list",
"items",
"switch",
"case",
"assign",
"global",
"local",
"include",
"import",
"macro",
"function",
"stop",
"return",
"nested",
"escape",
"setting",
"call",
"foreach",
"transform",
"visit",
"recurse",
"on",
"outputformat",
"ftl",
];
#[cfg(test)]
mod tests {
use super::*;
fn lex(name: &str, text: &str, strict: bool) -> Lexer {
Lexer::new(name, text, strict)
}
fn tokens(l: &mut Lexer, ctx: ExprCtx) -> Vec<Tok> {
let mut out = Vec::new();
loop {
let (t, _, _, _, _) = l.next_expr_token(ctx).unwrap();
let done = t == Tok::Eof;
out.push(t);
if done {
break;
}
}
out
}
#[test]
fn expr_tokens_basic() {
let mut l = lex("t", "a + b*2 != (x??) ?name", true);
let ts = tokens(&mut l, ExprCtx::Tag { square: false });
assert_eq!(
ts,
vec![
Tok::Ident("a".into()),
Tok::Plus,
Tok::Ident("b".into()),
Tok::Times,
Tok::Number("2".into()),
Tok::NotEq,
Tok::OpenParen,
Tok::Ident("x".into()),
Tok::Exists,
Tok::CloseParen,
Tok::Builtin,
Tok::Ident("name".into()),
Tok::Eof,
]
);
}
#[test]
fn word_operators_and_keywords() {
let mut l = lex("t", "a lt b gt c and d or e", true);
let ts = tokens(&mut l, ExprCtx::Tag { square: false });
assert!(ts.contains(&Tok::Lt));
assert!(ts.contains(&Tok::Gt));
assert!(!ts.contains(&Tok::And));
assert!(!ts.contains(&Tok::Or));
let mut l = lex("t", "a \\and b || c && d && e", true);
let ts = tokens(&mut l, ExprCtx::Tag { square: false });
assert_eq!(
ts.iter().filter(|t| **t == Tok::And).count(),
3,
"\\and、&&、&& 是 And(|| 是 Or)"
);
assert_eq!(ts.iter().filter(|t| **t == Tok::Or).count(), 1);
}
#[test]
fn number_forms() {
let mut l = lex("t", "1 1L 1F 1D 1.5 1e3 0x1A 1..5", true);
let ts = tokens(&mut l, ExprCtx::Tag { square: false });
assert_eq!(
ts,
vec![
Tok::Number("1".into()),
Tok::Number("1L".into()),
Tok::Number("1F".into()),
Tok::Number("1D".into()),
Tok::Number("1.5".into()),
Tok::Number("1e3".into()),
Tok::Number("0x1A".into()),
Tok::Number("1".into()),
Tok::DotDot,
Tok::Number("5".into()),
Tok::Eof,
]
);
}
#[test]
fn string_and_raw_string() {
let mut l = lex("t", r#""a\n\t\"\\" 'x' r"raw\ny" "#, true);
let ts = tokens(&mut l, ExprCtx::Tag { square: false });
assert_eq!(
ts,
vec![
Tok::Str("a\\n\\t\\\"\\\\".into()),
Tok::Str("x".into()),
Tok::RawStr("raw\\ny".into()),
Tok::Eof,
]
);
}
#[test]
fn gt_ends_tag_outside_parens() {
let mut l = lex("t", "a > b", true);
let ts = tokens(&mut l, ExprCtx::Tag { square: false });
assert_eq!(
ts,
vec![
Tok::Ident("a".into()),
Tok::TagEnd,
Tok::Ident("b".into()),
Tok::Eof
]
);
let mut l = lex("t", "(a > b)", true);
let ts = tokens(&mut l, ExprCtx::Tag { square: false });
assert!(ts.contains(&Tok::Gt));
let mut l = lex("t", "a >= b", true);
let ts = tokens(&mut l, ExprCtx::Tag { square: false });
assert!(!ts.contains(&Tok::Gte));
assert_eq!(ts[1], Tok::TagEnd);
}
#[test]
fn interp_ctx_closes_with_brace() {
let mut l = lex("t", "a + }", true);
let ts = tokens(&mut l, ExprCtx::Interp);
assert_eq!(
ts,
vec![Tok::Ident("a".into()), Tok::Plus, Tok::InterpEnd, Tok::Eof]
);
}
#[test]
fn curly_bracket_hash_vs_interp_end() {
let mut l = lex("t", r#"{"a": 1} x"#, true);
let ts = tokens(&mut l, ExprCtx::Interp);
assert_eq!(
ts,
vec![
Tok::OpenCurly,
Tok::Str("a".into()),
Tok::Colon,
Tok::Number("1".into()),
Tok::CloseCurly,
Tok::Ident("x".into()),
Tok::Eof,
]
);
}
#[test]
fn bracket_list_depth() {
let mut l = lex("t", "[1, 2] x", true);
let ts = tokens(&mut l, ExprCtx::Tag { square: true });
assert_eq!(
ts,
vec![
Tok::OpenBracket,
Tok::Number("1".into()),
Tok::Comma,
Tok::Number("2".into()),
Tok::CloseBracket,
Tok::Ident("x".into()),
Tok::Eof,
]
);
let mut l = lex("t", "[1, 2]]", true);
let ts = tokens(&mut l, ExprCtx::Tag { square: true });
assert_eq!(
ts,
vec![
Tok::OpenBracket,
Tok::Number("1".into()),
Tok::Comma,
Tok::Number("2".into()),
Tok::CloseBracket,
Tok::TagEnd,
Tok::Eof,
]
);
let mut l = lex("t", "[1, 2]]", true);
for _ in 0..5 {
l.next_expr_token(ExprCtx::Tag { square: false }).unwrap();
}
let r = l.next_expr_token(ExprCtx::Tag { square: false });
assert!(r.is_err());
}
#[test]
fn text_scanning_rules() {
let mut l = lex("t", "a < b", true);
let (t, s) = l.scan_text_chunk().unwrap();
assert_eq!(t, "a < b");
assert_eq!(s, TextStop::Eof);
let mut l = lex("t", "x <if y>", false);
let (t, s) = l.scan_text_chunk().unwrap();
assert_eq!(t, "x ");
assert_eq!(s, TextStop::Tag);
let mut l = lex("t", "a <b> c", false);
let (t, _s) = l.scan_text_chunk().unwrap();
assert_eq!(t, "a <b> c");
let mut l = lex("t", "$${x}", true);
let (t, s) = l.scan_text_chunk().unwrap();
assert_eq!(t, "$");
assert_eq!(s, TextStop::Interp);
let mut l = lex("t", "ab${x}", true);
let (t, s) = l.scan_text_chunk().unwrap();
assert_eq!(t, "ab");
assert_eq!(s, TextStop::Interp);
let mut l = lex("t", "ab#{x}", true);
let (t, s) = l.scan_text_chunk().unwrap();
assert_eq!(t, "ab");
assert_eq!(s, TextStop::Interp);
let mut l = lex("t", "ab<#-- c -->", true);
let (t, s) = l.scan_text_chunk().unwrap();
assert_eq!(t, "ab");
assert_eq!(s, TextStop::Tag);
}
#[test]
fn comment_scanning() {
let mut l = lex("t", "hello --] world -->", true);
let (s, _, _) = l.scan_comment(false).unwrap();
assert_eq!(s, "hello --] world ");
let mut l = lex("t", "hello --> world --]", true);
let (s, _, _) = l.scan_comment(true).unwrap();
assert_eq!(s, "hello --> world ");
let mut l = lex("t", "unclosed", true);
assert!(l.scan_comment(false).is_err());
}
#[test]
fn unparsed_scanning() {
let mut l = lex("t", "a</#noparse>", true);
let (s, _, _) = l.scan_unparsed("noparse").unwrap();
assert_eq!(s, "a");
let mut l = lex("t", "a</noparse>", true);
let (s, _, _) = l.scan_unparsed("noparse").unwrap();
assert_eq!(s, "a");
let mut l = lex("t", "a</#comment>", true);
let (s, _, _) = l.scan_unparsed("comment").unwrap();
assert_eq!(s, "a");
let mut l = lex("t", "a</#noparse x>rest</#noparse>", true);
let (s, _, _) = l.scan_unparsed("noparse").unwrap();
assert_eq!(s, "a</#noparse x>rest");
}
#[test]
fn non_strict_tag_detection() {
let mut l = lex("t", "x <if y>", false);
let (_, stop) = l.scan_text_chunk().unwrap();
assert_eq!(stop, TextStop::Tag);
assert!(l.starts_tag());
let mut l = lex("t", "x <if>", false);
let (t, _) = l.scan_text_chunk().unwrap();
assert_eq!(t, "x <if>");
let mut l = lex("t", "x <else>", false);
let (t, _s) = l.scan_text_chunk().unwrap();
assert_eq!(t, "x ");
let mut l = lex("t", "x <else y>", false);
let (t, _) = l.scan_text_chunk().unwrap();
assert_eq!(t, "x <else y>");
let mut l = lex("t", "x <if y>", true);
let (t, _) = l.scan_text_chunk().unwrap();
assert_eq!(t, "x <if y>");
}
#[test]
fn escaped_identifiers() {
let mut l = lex("t", r"a\-b\.c\:d\#e", true);
let ts = tokens(&mut l, ExprCtx::Tag { square: false });
assert_eq!(ts[0], Tok::Ident("a-b.c:d#e".into()));
}
#[test]
fn ident_special_chars() {
let mut l = lex("t", "$foo _bar français x2", true);
let ts = tokens(&mut l, ExprCtx::Tag { square: false });
assert_eq!(
ts,
vec![
Tok::Ident("$foo".into()),
Tok::Ident("_bar".into()),
Tok::Ident("français".into()),
Tok::Ident("x2".into()),
Tok::Eof,
]
);
}
}