Skip to main content

ironwork_syntax/
lexer.rs

1use crate::messages::{IWS0099, IWX0015};
2use crate::source::{FreeSpan, Source};
3use crate::{Error, Pos};
4use numeric::Compliance;
5
6#[derive(Clone, Debug, PartialEq, Eq)]
7pub enum Tok {
8    /// A COBOL word, uppercased: a name, a reserved word or a level number with letters in it.
9    Word(String),
10    /// A numeric literal as written: optional sign, digits, optional decimal point.
11    Number(String),
12    Alnum(String),
13    Hex(Vec<u8>),
14    National(String),
15    /// A DBCS literal's characters, its shift-out and shift-in removed.
16    Dbcs(String),
17    Pic(String),
18    /// An EXEC ... END-EXEC block, as written: SQL, CICS or DLI for a precompiler.
19    Exec(String),
20    Period,
21    LParen,
22    RParen,
23    Colon,
24    Plus,
25    Minus,
26    Star,
27    Slash,
28    Power,
29    Eq,
30    Lt,
31    Gt,
32    Le,
33    Ge,
34    /// `&` after a literal or a word, under `--compliance extended`: [`crate::extended`] joins the
35    /// literals on either side before the parser sees it.
36    Ampersand,
37}
38
39/// The warning for `<>`, which the lexer reads as NOT =.
40pub const NOT_EQUAL: &str = "<> (Micro Focus and GnuCOBOL; Enterprise COBOL writes NOT =) is read as NOT =";
41
42#[derive(Clone, Debug, PartialEq, Eq)]
43pub struct Token {
44    pub tok: Tok,
45    pub pos: Pos,
46    /// Starts in area A (columns 8 to 11), where division, section and paragraph headers go.
47    pub area_a: bool,
48    /// A word as the source spells it, where that is not all capitals.
49    pub spelled: Option<String>,
50    /// A separator comma or semicolon comes between this token and the one before it.
51    pub after_comma: bool,
52    /// Messages about the token that do not stop the parse.
53    pub messages: Vec<Error>,
54}
55
56struct Lexer<'a> {
57    chars: Vec<char>,
58    positions: &'a [Pos],
59    at: usize,
60    tokens: Vec<Token>,
61    /// DECIMAL-POINT IS COMMA is in force: a comma between digits is the decimal point.
62    decimal_comma: bool,
63    /// The currency symbols CURRENCY SIGN clauses have named so far, whose case a PICTURE keeps.
64    currency: Vec<char>,
65    /// For each program begun and not yet ended, whether the comma was the decimal point before
66    /// it: a contained program has its container's, and a program after it starts afresh.
67    outer: Vec<bool>,
68    comma_pending: bool,
69    /// Messages for the next token emitted.
70    pending: Vec<Error>,
71    extended: bool,
72    /// The lines read in free form, and whether a token from each has carried its warning.
73    free: &'a [FreeSpan],
74    warned: Vec<bool>,
75    /// NSYMBOL(DBCS) is on the cards: N'...' is a DBCS literal.
76    n_is_dbcs: bool,
77}
78
79/// The most characters a DBCS literal holds (Language Reference SC27-8713-03, p. 42).
80const DBCS_LITERAL_MAX: usize = 28;
81
82/// The most characters a user-defined word has.
83pub(crate) const USER_WORD_MAX: usize = 30;
84
85pub fn lex(source: &Source) -> Result<Vec<Token>, Error> {
86    lex_under(source, Compliance::Strict)
87}
88
89/// Lexes `source` under `compliance`: `--compliance extended` reads `<>` as NOT = and keeps `&`
90/// after a literal or a word for [`crate::extended`].
91pub fn lex_under(source: &Source, compliance: Compliance) -> Result<Vec<Token>, Error> {
92    let mut options = numeric::Options::default();
93    for card in &source.options {
94        options.apply(card).ok();
95    }
96    let n_is_dbcs = options.nsymbol == numeric::Nsymbol::Dbcs;
97    let mut lx = Lexer {
98        chars: source.text.chars().collect(),
99        positions: &source.positions,
100        at: 0,
101        tokens: Vec::new(),
102        decimal_comma: false,
103        currency: Vec::new(),
104        outer: Vec::new(),
105        comma_pending: false,
106        pending: source.notes.clone(),
107        extended: compliance == Compliance::Extended,
108        free: &source.free,
109        warned: vec![false; source.free.len()],
110        n_is_dbcs,
111    };
112    while lx.at < lx.chars.len() {
113        lx.next_token()?;
114    }
115    Ok(lx.tokens)
116}
117
118fn is_word_char(c: char) -> bool {
119    c.is_ascii_alphanumeric() || c == '-' || c == '_'
120}
121
122/// A single-byte character outside IBM's basic COBOL character set (Language Reference
123/// SC27-8713-03, Table 1, pp. 3-6), which IBM accepts with an error as part of the text around it
124/// (assumption C123). `$` and `&`, which ironwork reads only where COBOL puts them, are left out.
125/// The lowercase letter an EBCDIC byte is, the same in every code page ironwork carries.
126fn ebcdic_lowercase(byte: u8) -> Option<char> {
127    let (start, first) = match byte {
128        0x81..=0x89 => (0x81, b'a'),
129        0x91..=0x99 => (0x91, b'j'),
130        0xA2..=0xA9 => (0xA2, b's'),
131        _ => return None,
132    };
133    Some(char::from(first + (byte - start)))
134}
135
136fn non_cobol(c: char) -> bool {
137    u32::from(c) <= 0xFF && !c.is_ascii_alphanumeric() && !" \n+-*/=$,;.\"'()><:_&".contains(c)
138}
139
140impl Lexer<'_> {
141    fn peek(&self, ahead: usize) -> Option<char> {
142        self.chars.get(self.at + ahead).copied()
143    }
144
145    fn pos(&self) -> Pos {
146        self.positions.get(self.at).copied().unwrap_or_default()
147    }
148
149    fn separator_follows(&self, ahead: usize) -> bool {
150        self.peek(ahead).is_none_or(|c| c == ' ' || c == '\n')
151    }
152
153    fn emit(&mut self, tok: Tok, pos: Pos) {
154        let before = |back: usize| match self.tokens.len().checked_sub(back).map(|i| &self.tokens[i].tok) {
155            Some(Tok::Word(p)) => p.as_str(),
156            _ => "",
157        };
158        match &tok {
159            Tok::Word(w) => match w.as_str() {
160                "PROGRAM-ID" => self.outer.push(self.decimal_comma),
161                "PROGRAM" if before(1) == "END" => self.decimal_comma = self.outer.pop().unwrap_or(false),
162                "COMMA" if before(1) == "DECIMAL-POINT" || before(1) == "IS" && before(2) == "DECIMAL-POINT" => self.decimal_comma = true,
163                _ => {}
164            },
165            Tok::Alnum(a) => {
166                let mut chars = a.chars();
167                let names_symbol = before(1) == "SYMBOL" || (1..=3).any(|back| before(back) == "CURRENCY" && (1..back).all(|b| matches!(before(b), "SIGN" | "IS")));
168                if let (Some(c), None, true) = (chars.next(), chars.next(), names_symbol) {
169                    self.currency.push(c);
170                }
171            }
172            Tok::Hex(bytes) => {
173                let names_symbol = (1..=3).any(|back| before(back) == "CURRENCY" && (1..back).all(|b| matches!(before(b), "SIGN" | "IS")));
174                if let ([byte], true) = (bytes.as_slice(), names_symbol)
175                    && let Some(c) = ebcdic_lowercase(*byte)
176                {
177                    self.currency.push(c);
178                }
179            }
180            _ => {}
181        }
182        let spelled = match &tok {
183            Tok::Word(w) => self
184                .at
185                .checked_sub(w.chars().count())
186                .map(|start| &self.chars[start..self.at])
187                .filter(|raw| raw.iter().any(char::is_ascii_lowercase))
188                .map(|raw| raw.iter().collect::<String>())
189                .filter(|raw| raw.eq_ignore_ascii_case(w)),
190            _ => None,
191        };
192        let (tok, spelled) = match tok {
193            Tok::Word(w) if w.chars().count() > USER_WORD_MAX => self.long_word(w, spelled, pos),
194            tok => (tok, spelled),
195        };
196        let after_comma = std::mem::take(&mut self.comma_pending);
197        let span = self.free.iter().position(|s| s.holds(pos));
198        let area_a = match span {
199            Some(k) => {
200                if !std::mem::replace(&mut self.warned[k], true)
201                    && let Some(warning) = &self.free[k].warning
202                {
203                    self.pending.insert(0, warning.clone());
204                }
205                self.tokens.last().is_none_or(|t| t.tok == Tok::Period) && !matches!(&tok, Tok::Word(w) if rt::reserved_words::is_reserved(w))
206            }
207            None => (8..=11).contains(&pos.col),
208        };
209        self.tokens.push(Token { tok, pos, area_a, spelled, after_comma, messages: std::mem::take(&mut self.pending) });
210    }
211
212    /// A word longer than a user-defined word can be (Language Reference SC27-8713-03, p. 13).
213    /// Enterprise COBOL reads its first 30 characters, with an error (IGYDS0023-E); Micro Focus
214    /// and GnuCOBOL read it whole.
215    fn long_word(&mut self, word: String, spelled: Option<String>, pos: Pos) -> (Tok, Option<String>) {
216        if self.extended {
217            self.pending.push(IWX0015.at(pos, format!("a user-defined word of more than 30 characters (Micro Focus and GnuCOBOL; Enterprise COBOL reads its first 30): {word} is read whole")));
218            return (Tok::Word(word), spelled);
219        }
220        let first: String = word.chars().take(USER_WORD_MAX).collect();
221        let count = word.chars().count();
222        self.pending.push(IWS0099.at(pos, format!("{word}: a user-defined word has at most 30 characters, and this one has {count}; it is read as its first 30, {first}")));
223        (Tok::Word(first), spelled.map(|s| s.chars().take(USER_WORD_MAX).collect()))
224    }
225
226    /// The character that is a numeric literal's decimal point.
227    fn point(&self) -> char {
228        if self.decimal_comma { ',' } else { '.' }
229    }
230
231    fn expecting_picture(&self) -> bool {
232        let words: Vec<&str> =
233            self.tokens.iter().rev().take(2).map(|t| if let Tok::Word(w) = &t.tok { w.as_str() } else { "" }).collect();
234        matches!(words.as_slice(), ["PIC" | "PICTURE", ..] | ["IS", "PIC" | "PICTURE"])
235    }
236
237    fn next_token(&mut self) -> Result<(), Error> {
238        let c = self.chars[self.at];
239        let pos = self.pos();
240        let point = self.point();
241        let leading_point = c == ',' && point == ',' && self.peek(1).is_some_and(|d| d.is_ascii_digit()) && !self.at.checked_sub(1).is_some_and(|i| is_word_char(self.chars[i]));
242        if leading_point && self.expecting_picture() {
243            return self.picture(pos);
244        }
245        if leading_point {
246            let tok = self.number_or_word(pos)?;
247            self.emit(tok, pos);
248            return Ok(());
249        }
250        if c == ' ' || c == '\n' || c == ',' || c == ';' {
251            self.comma_pending |= c == ',' || c == ';';
252            self.at += 1;
253            return Ok(());
254        }
255        if self.expecting_picture() {
256            return self.picture(pos);
257        }
258        let next = self.peek(1);
259        let quote_next = matches!(next, Some('\'' | '"'));
260        match c {
261            '\'' | '"' => {
262                let text = self.quoted(pos)?;
263                self.emit(Tok::Alnum(text), pos);
264            }
265            'X' | 'x' if quote_next => {
266                self.at += 1;
267                let text = self.quoted(pos)?;
268                let bytes = unhex(&text).ok_or_else(|| crate::messages::IWS0017.at(pos, format!("X'{text}' is not an even number of hex digits")))?;
269                self.emit(Tok::Hex(bytes), pos);
270            }
271            'N' | 'n' if matches!(next, Some('X' | 'x')) && matches!(self.peek(2), Some('\'' | '"')) => {
272                self.at += 2;
273                let text = self.quoted(pos)?;
274                let units = unhex(&text).filter(|b| !b.is_empty() && b.len().is_multiple_of(2) && b.len() <= 160);
275                let units = units.map(|b| b.chunks(2).map(|u| u16::from_be_bytes([u[0], u[1]])).collect::<Vec<_>>());
276                let national = units.and_then(|u| String::from_utf16(&u).ok());
277                let national = national.ok_or_else(|| crate::messages::IWS0018.at(pos, format!("NX'{text}': a national hexadecimal literal is 4 to 320 hex digits, four to each UTF-16 code unit")))?;
278                self.emit(Tok::National(national), pos);
279            }
280            'N' | 'n' if quote_next && self.n_is_dbcs => self.dbcs_literal(pos)?,
281            'G' | 'g' if quote_next => self.dbcs_literal(pos)?,
282            'N' | 'n' if quote_next => {
283                self.at += 1;
284                let text = self.quoted(pos)?;
285                self.emit(Tok::National(text), pos);
286            }
287            'Z' | 'z' if quote_next => {
288                self.at += 1;
289                let text = self.quoted(pos)?;
290                self.emit(Tok::Alnum(format!("{text}\0")), pos);
291            }
292            '.' if self.separator_follows(1) => {
293                self.at += 1;
294                self.emit(Tok::Period, pos);
295            }
296            '.' if self.extended && next == Some('.') && self.separator_follows(self.periods()) => {
297                self.pending.push(crate::messages::IWX0041.at(pos, "periods after a period (GnuCOBOL and Micro Focus; Enterprise COBOL ends a sentence with one): the periods after the first are ignored"));
298                self.at += self.periods();
299                self.emit(Tok::Period, pos);
300            }
301            '+' | '-' if next.is_some_and(|n| n.is_ascii_digit() || (n == point && self.peek(2).is_some_and(|d| d.is_ascii_digit()))) => {
302                self.at += 1;
303                let digits = self.number_or_word(pos)?;
304                match digits {
305                    Tok::Number(n) => self.emit(Tok::Number(format!("{c}{n}")), pos),
306                    _ => return Err(crate::messages::IWS0019.at(pos, "a sign must be followed by a number")),
307                }
308            }
309            _ if c.is_ascii_alphanumeric() || c == '.' || non_cobol(c) => {
310                let tok = self.number_or_word(pos)?;
311                match tok {
312                    Tok::Word(w) if w == "EXEC" || w == "EXECUTE" => {
313                        let tok = self.exec_block(pos)?;
314                        self.emit(tok, pos);
315                    }
316                    tok => self.emit(tok, pos),
317                }
318            }
319            '<' if self.extended && next == Some('>') => {
320                self.pending.push(crate::messages::IWX0003.at(pos, NOT_EQUAL));
321                self.at += 2;
322                self.emit(Tok::Word("NOT".into()), pos);
323                let at = self.positions.get(self.at - 1).copied().unwrap_or(pos);
324                self.emit(Tok::Eq, at);
325            }
326            '&' if self.extended && matches!(self.tokens.last().map(|t| &t.tok), Some(Tok::Alnum(_) | Tok::Hex(_) | Tok::National(_) | Tok::Word(_))) => {
327                self.at += 1;
328                self.emit(Tok::Ampersand, pos);
329            }
330            _ => {
331                let (tok, len) = match (c, next) {
332                    ('*', Some('*')) => (Tok::Power, 2),
333                    ('<', Some('=')) => (Tok::Le, 2),
334                    ('>', Some('=')) => (Tok::Ge, 2),
335                    ('*', _) => (Tok::Star, 1),
336                    ('/', _) => (Tok::Slash, 1),
337                    ('+', _) => (Tok::Plus, 1),
338                    ('-', _) => (Tok::Minus, 1),
339                    ('=', _) => (Tok::Eq, 1),
340                    ('<', _) => (Tok::Lt, 1),
341                    ('>', _) => (Tok::Gt, 1),
342                    ('(', _) => (Tok::LParen, 1),
343                    (')', _) => (Tok::RParen, 1),
344                    (':', _) => (Tok::Colon, 1),
345                    ('&', _) if matches!(self.tokens.last().map(|t| &t.tok), Some(Tok::Alnum(_) | Tok::Hex(_) | Tok::National(_))) => {
346                        return Err(crate::messages::IWS0020.at(pos, "literal concatenation with & is not Enterprise COBOL's"));
347                    }
348                    _ => return Err(crate::messages::IWS0021.at(pos, format!("unexpected character {c:?}"))),
349                };
350                self.at += len;
351                self.emit(tok, pos);
352            }
353        }
354        Ok(())
355    }
356
357    /// The text of an EXEC block up to END-EXEC, which is consumed; quotes inside are skipped whole.
358    fn exec_block(&mut self, pos: Pos) -> Result<Tok, Error> {
359        let start = self.at;
360        let mut quote: Option<char> = None;
361        while let Some(c) = self.peek(0) {
362            match quote {
363                Some(q) if c == q => quote = None,
364                Some(_) => {}
365                None if c == '\'' || c == '"' => quote = Some(c),
366                None if c.eq_ignore_ascii_case(&'E') => {
367                    let ahead: String = self.chars[self.at..].iter().take(8).collect();
368                    let boundary = self.chars.get(self.at + 8).is_none_or(|d| !is_word_char(*d)) && (self.at == 0 || !is_word_char(self.chars[self.at - 1]));
369                    if ahead.eq_ignore_ascii_case("END-EXEC") && boundary {
370                        let text: String = self.chars[start..self.at].iter().collect();
371                        self.at += 8;
372                        return Ok(Tok::Exec(text.split_whitespace().collect::<Vec<_>>().join(" ")));
373                    }
374                }
375                None => {}
376            }
377            self.at += 1;
378        }
379        Err(crate::messages::IWS0022.at(pos, "EXEC with no END-EXEC"))
380    }
381
382    /// Reads a quoted literal starting at the opening quote; a doubled quote stands for one.
383    /// G'...', or N'...' under NSYMBOL(DBCS): the characters, the shift-out after the opening
384    /// delimiter and the shift-in before the closing one, which a source in Unicode drops, removed
385    /// where present.
386    fn dbcs_literal(&mut self, pos: Pos) -> Result<(), Error> {
387        self.at += 1;
388        let text = self.quoted(pos)?;
389        let text = text.strip_prefix('\u{E}').unwrap_or(&text);
390        let text = text.strip_suffix('\u{F}').unwrap_or(text).to_owned();
391        let count = text.chars().count();
392        if count == 0 || count > DBCS_LITERAL_MAX {
393            return Err(crate::messages::IWS0023.at(pos, format!("a DBCS literal holds 1 to {DBCS_LITERAL_MAX} characters, not {count}")));
394        }
395        self.emit(Tok::Dbcs(text), pos);
396        Ok(())
397    }
398
399    fn quoted(&mut self, pos: Pos) -> Result<String, Error> {
400        let quote = self.chars[self.at];
401        self.at += 1;
402        let mut text = String::new();
403        loop {
404            match self.peek(0) {
405                None | Some('\n') => return Err(crate::messages::IWS0024.at(pos, "an unterminated literal")),
406                Some(c) if c == quote && self.peek(1) == Some(quote) => {
407                    text.push(quote);
408                    self.at += 2;
409                }
410                Some(c) if c == quote => {
411                    self.at += 1;
412                    return Ok(text);
413                }
414                Some(c) => {
415                    text.push(c);
416                    self.at += 1;
417                }
418            }
419        }
420    }
421
422    /// How many periods follow one another from here.
423    fn periods(&self) -> usize {
424        (0..).take_while(|&k| self.peek(k) == Some('.')).count()
425    }
426
427    fn number_or_word(&mut self, pos: Pos) -> Result<Tok, Error> {
428        let start = self.at;
429        while let Some(c) = self.peek(0).filter(|&c| is_word_char(c) || non_cobol(c)) {
430            if non_cobol(c) {
431                self.pending.push(crate::messages::IWS0025.at(self.pos(), format!("non-COBOL character {c:?}: the character was accepted")).graded(crate::Severity::Error));
432            }
433            self.at += 1;
434        }
435        let run: String = self.chars[start..self.at].iter().collect();
436        let all_digits = run.chars().all(|c| c.is_ascii_digit());
437        if all_digits && self.peek(0) == Some(self.point()) && self.peek(1).is_some_and(|c| c.is_ascii_digit()) {
438            self.at += 1;
439            let frac_start = self.at;
440            while self.peek(0).is_some_and(|c| c.is_ascii_digit()) {
441                self.at += 1;
442            }
443            let frac: String = self.chars[frac_start..self.at].iter().collect();
444            return Ok(Tok::Number(format!("{run}.{frac}")));
445        }
446        if run.is_empty() {
447            return Err(crate::messages::IWS0026.at(pos, "unexpected '.'"));
448        }
449        Ok(if all_digits { Tok::Number(run) } else { Tok::Word(run.to_ascii_uppercase()) })
450    }
451
452    /// Only the period, comma or semicolon just before the space is a separator (assumption C195).
453    fn picture(&mut self, pos: Pos) -> Result<(), Error> {
454        let start = self.at;
455        while self.peek(0).is_some_and(|c| c != ' ' && c != '\n') {
456            self.at += 1;
457        }
458        let mut end = self.at;
459        if end > start && matches!(self.chars[end - 1], '.' | ',' | ';') {
460            end -= 1;
461        }
462        let text: String = self.chars[start..end].iter().collect();
463        if text.is_empty() {
464            return Err(crate::messages::IWS0027.at(pos, "PICTURE with no character-string"));
465        }
466        self.at = end;
467        let text: String = text.chars().map(|c| if self.currency.contains(&c) { c } else { c.to_ascii_uppercase() }).collect();
468        let tok = if matches!(text.as_str(), "IS" | "SYMBOL") && self.tokens.last().is_some_and(|t| matches!(&t.tok, Tok::Word(w) if w == "PIC" || w == "PICTURE")) {
469            Tok::Word(text)
470        } else {
471            Tok::Pic(text)
472        };
473        self.emit(tok, pos);
474        Ok(())
475    }
476}
477
478fn unhex(text: &str) -> Option<Vec<u8>> {
479    if !text.len().is_multiple_of(2) || !text.is_ascii() {
480        return None;
481    }
482    (0..text.len()).step_by(2).map(|i| u8::from_str_radix(&text[i..i + 2], 16).ok()).collect()
483}
484
485#[cfg(test)]
486mod tests {
487    use super::*;
488    use crate::source;
489
490    fn toks(text: &str) -> Vec<Tok> {
491        lex(&source::read(text).unwrap()).unwrap().into_iter().map(|t| t.tok).collect()
492    }
493
494    fn w(s: &str) -> Tok {
495        Tok::Word(s.into())
496    }
497
498    #[test]
499    fn numbers_words_and_the_separator_period() {
500        assert_eq!(toks("           05 A-1 VALUE 0.1."), [Tok::Number("05".into()), w("A-1"), w("VALUE"), Tok::Number("0.1".into()), Tok::Period]);
501        assert_eq!(toks("           VALUE -12345."), [w("VALUE"), Tok::Number("-12345".into()), Tok::Period]);
502        assert_eq!(toks("       100-MAIN."), [w("100-MAIN"), Tok::Period]);
503    }
504
505    #[test]
506    fn operators_need_spaces_and_signed_literals_do_not() {
507        assert_eq!(toks("           A - 1 ** 2"), [w("A"), Tok::Minus, Tok::Number("1".into()), Tok::Power, Tok::Number("2".into())]);
508        assert_eq!(toks("           >= <= ("), [Tok::Ge, Tok::Le, Tok::LParen]);
509    }
510
511    #[test]
512    fn literals() {
513        assert_eq!(toks("           'IT''S' X'F1C1' N'AB'"), [Tok::Alnum("IT'S".into()), Tok::Hex(vec![0xF1, 0xC1]), Tok::National("AB".into())]);
514        assert_eq!(toks("           NX'00410042' nx\"265ED83DDE00\""), [Tok::National("AB".into()), Tok::National("\u{265E}\u{1F600}".into())]);
515        assert_eq!(toks("           G'\u{E}AB\u{F}' g\"日本\""), [Tok::Dbcs("AB".into()), Tok::Dbcs("日本".into())]);
516        assert_eq!(toks("       CBL NSYMBOL(DBCS)\n           N'AB' NX'0041'"), [Tok::Dbcs("AB".into()), Tok::National("A".into())]);
517        let long = format!("           G'{}'", "A".repeat(29));
518        assert!(lex(&source::read(&long).unwrap()).unwrap_err().message.contains("1 to 28 characters, not 29"));
519        for bad in ["NX'GH'", "NX'1'", "NX'004'", "NX'D83D'"] {
520            let e = lex(&source::read(&format!("           {bad}")).unwrap()).unwrap_err();
521            assert!(e.message.contains("a national hexadecimal literal is 4 to 320 hex digits"), "{bad}: {}", e.message);
522        }
523    }
524
525    #[test]
526    fn a_null_terminated_literal_ends_with_x00() {
527        assert_eq!(toks("           Z'ABC' z\"(I)V\""), [Tok::Alnum("ABC\0".into()), Tok::Alnum("(I)V\0".into())]);
528    }
529
530    #[test]
531    fn a_picture_is_one_token_and_keeps_its_own_periods() {
532        assert_eq!(toks("           PIC S9(3)V99 COMP-3."), [w("PIC"), Tok::Pic("S9(3)V99".into()), w("COMP-3"), Tok::Period]);
533        assert_eq!(toks("           PICTURE IS ZZ,ZZ9.99."), [w("PICTURE"), w("IS"), Tok::Pic("ZZ,ZZ9.99".into()), Tok::Period]);
534    }
535
536    #[test]
537    fn only_the_last_period_or_comma_before_the_space_is_a_separator() {
538        assert_eq!(toks("           PIC 9,9,9,."), [w("PIC"), Tok::Pic("9,9,9,".into()), Tok::Period]);
539        assert_eq!(toks("           PIC 999999999999.."), [w("PIC"), Tok::Pic("999999999999.".into()), Tok::Period]);
540        assert_eq!(toks("           PIC 99, VALUE 1."), [w("PIC"), Tok::Pic("99".into()), w("VALUE"), Tok::Number("1".into()), Tok::Period]);
541        assert_eq!(toks("           PIC 9.9,; VALUE 1."), [w("PIC"), Tok::Pic("9.9,".into()), w("VALUE"), Tok::Number("1".into()), Tok::Period]);
542        assert_eq!(toks("           PIC 999., VALUE 1."), [w("PIC"), Tok::Pic("999.".into()), w("VALUE"), Tok::Number("1".into()), Tok::Period]);
543        let comma = "           DECIMAL-POINT IS COMMA.\n           PIC 9.9.9,. PIC 999,,\n";
544        let pics: Vec<Tok> = toks(comma).into_iter().filter(|t| matches!(t, Tok::Pic(_))).collect();
545        assert_eq!(pics, [Tok::Pic("9.9.9,".into()), Tok::Pic("999,".into())]);
546    }
547
548    #[test]
549    fn commas_between_operands_are_separators() {
550        assert_eq!(toks("           F(A, 1)"), [w("F"), Tok::LParen, w("A"), Tok::Number("1".into()), Tok::RParen]);
551        assert_eq!(toks("           T(I,J)"), [w("T"), Tok::LParen, w("I"), w("J"), Tok::RParen]);
552    }
553
554    #[test]
555    fn an_exec_block_is_one_token() {
556        assert_eq!(
557            toks("           EXEC SQL SELECT A.B INTO :X FROM T WHERE C = 'END-EXEC'\n               END-EXEC."),
558            [Tok::Exec("SQL SELECT A.B INTO :X FROM T WHERE C = 'END-EXEC'".into()), Tok::Period]
559        );
560    }
561
562    #[test]
563    fn a_comment_entry_holds_any_character() {
564        let program = concat!(
565            "       IDENTIFICATION DIVISION.\n",
566            "       PROGRAM-ID. CE3.\n",
567            "       AUTHOR. Smith & Jones @ ACME #1, Café O'Grady.\n",
568            "       INSTALLATION. \"HQ\n",
569            "           ~ ^ ` { } | \\ ?\n",
570            "       DATE-WRITTEN. 01/01/99.\n",
571            "       PROCEDURE DIVISION.\n",
572            "           GOBACK.\n",
573        );
574        let words = [w("IDENTIFICATION"), w("DIVISION"), Tok::Period, w("PROGRAM-ID"), Tok::Period, w("CE3"), Tok::Period];
575        let paragraphs = [w("AUTHOR"), Tok::Period, w("INSTALLATION"), Tok::Period, w("DATE-WRITTEN"), Tok::Period];
576        let procedure = [w("PROCEDURE"), w("DIVISION"), Tok::Period, w("GOBACK"), Tok::Period];
577        assert_eq!(toks(program), [&words[..], &paragraphs[..], &procedure[..]].concat());
578    }
579
580    #[test]
581    fn a_literal_continued_between_the_quotes_of_a_doubled_quote() {
582        let head = "           \"A+0B-1C*2D";
583        let text = format!("{head}{}\"\n      -    \"\"9K(L)M>N<O\".\n", "=".repeat(72 - head.len() - 1));
584        assert_eq!(toks(&text), [Tok::Alnum(format!("A+0B-1C*2D{}\"9K(L)M>N<O", "=".repeat(72 - head.len() - 1))), Tok::Period]);
585    }
586
587    #[test]
588    fn a_continuation_that_opens_a_quote_after_a_closed_literal_is_a_second_literal() {
589        assert_eq!(toks("           VALUE 'ABC'\n      -    'DEF'."), [w("VALUE"), Tok::Alnum("ABC".into()), Tok::Alnum("DEF".into()), Tok::Period]);
590    }
591
592    #[test]
593    fn an_ampersand_after_a_literal_is_named_as_concatenation() {
594        let error = |text: &str| lex(&source::read(text).unwrap()).unwrap_err().message;
595        assert!(error("           'A' & 'B'").contains("literal concatenation with & is not Enterprise COBOL's"));
596        assert!(error("           NOTIFY=&SYSUID").contains("unexpected character '&'"));
597    }
598
599    #[test]
600    fn a_non_cobol_character_is_accepted_into_its_word_with_an_error() {
601        let lexed = lex(&source::read("           MOVE WS#1 TO %\u{1b} 'A@B'.").unwrap()).unwrap();
602        assert_eq!(lexed.iter().map(|t| t.tok.clone()).collect::<Vec<_>>(), [w("MOVE"), w("WS#1"), w("TO"), w("%\u{1b}"), Tok::Alnum("A@B".into()), Tok::Period]);
603        let messages: Vec<(usize, u32, &str, crate::Severity)> =
604            lexed.iter().enumerate().flat_map(|(i, t)| t.messages.iter().map(move |m| (i, m.pos.col, m.message.as_str(), m.severity))).collect();
605        assert_eq!(
606            messages,
607            [
608                (1, 19, "non-COBOL character '#': the character was accepted", crate::Severity::Error),
609                (3, 25, "non-COBOL character '%': the character was accepted", crate::Severity::Error),
610                (3, 26, "non-COBOL character '\\u{1b}': the character was accepted", crate::Severity::Error),
611            ]
612        );
613        let error = |text: &str| lex(&source::read(text).unwrap()).unwrap_err().message;
614        assert_eq!(error("           MOVE $X"), "unexpected character '$'");
615        assert_eq!(error("           MOVE \u{3042}"), "unexpected character '\u{3042}'");
616    }
617
618    #[test]
619    fn under_extended_not_equal_is_not_and_equals_and_an_ampersand_is_kept() {
620        let lexed = lex_under(&source::read("           IF A <> 'B' & X'C1' MOVE C & D").unwrap(), Compliance::Extended).unwrap();
621        let toks: Vec<Tok> = lexed.iter().map(|t| t.tok.clone()).collect();
622        assert_eq!(toks, [w("IF"), w("A"), w("NOT"), Tok::Eq, Tok::Alnum("B".into()), Tok::Ampersand, Tok::Hex(vec![0xC1]), w("MOVE"), w("C"), Tok::Ampersand, w("D")]);
623        assert_eq!(lexed[2].messages.iter().map(|m| (m.pos.col, m.message.as_str(), m.severity)).collect::<Vec<_>>(), [(17, NOT_EQUAL, crate::Severity::Warning)]);
624        assert_eq!(lexed[3].pos.col, 18);
625        let strict = |text: &str| lex(&source::read(text).unwrap()).map(|t| t.into_iter().map(|t| t.tok).collect::<Vec<_>>());
626        assert_eq!(strict("           A <> B"), Ok(vec![w("A"), Tok::Lt, Tok::Gt, w("B")]));
627        assert!(lex_under(&source::read("           NOTIFY=&SYSUID").unwrap(), Compliance::Extended).unwrap_err().message.contains("unexpected character '&'"));
628    }
629
630    #[test]
631    fn in_free_form_area_a_is_a_word_after_a_period_that_is_not_reserved() {
632        let text = "IDENTIFICATION DIVISION.\nPROCEDURE DIVISION.\nMAIN-P.\n    MOVE A TO\n  B.\n    GOBACK.\n0100.\n";
633        let free = source::read_under(text, 0, false, Compliance::Extended).unwrap();
634        let lexed = lex_under(&free, Compliance::Extended).unwrap();
635        let marked: Vec<Tok> = lexed.iter().filter(|t| t.area_a).map(|t| t.tok.clone()).collect();
636        assert_eq!(marked, [w("MAIN-P"), Tok::Number("0100".into())]);
637        assert_eq!(lexed[0].messages[0].id, Some("IWX0001"), "{:?}", lexed[0].messages);
638        assert!(lexed[1..].iter().all(|t| t.messages.is_empty()));
639    }
640
641    #[test]
642    fn area_a_is_marked() {
643        let t = lex(&source::read("       PARA.\n           MOVE").unwrap()).unwrap();
644        assert!(t[0].area_a);
645        assert!(!t[2].area_a);
646    }
647
648    fn numbers(text: &str) -> Vec<String> {
649        toks(text).into_iter().filter_map(|t| if let Tok::Number(n) = t { Some(n) } else { None }).collect()
650    }
651
652    #[test]
653    fn under_decimal_point_is_comma_a_comma_between_digits_is_the_point() {
654        let text = "           DECIMAL-POINT IS COMMA.\n           1,5 -,25 +3,0 ,75 T(1, 2) A,B 1.\n";
655        assert_eq!(numbers(text), ["1.5", "-.25", "+3.0", ".75", "1", "2", "1"]);
656        assert!(toks(text).contains(&w("B")));
657        assert!(toks("           DECIMAL-POINT IS COMMA.\n           PIC ,99.").contains(&Tok::Pic(",99".into())));
658        assert!(lex(&source::read("           DECIMAL-POINT COMMA.\n           MOVE 1.5").unwrap()).is_err());
659    }
660
661    #[test]
662    fn a_contained_program_keeps_the_decimal_comma_and_the_next_program_does_not() {
663        let text = concat!(
664            "       PROGRAM-ID. A.\n           DECIMAL-POINT IS COMMA.\n           1,5\n",
665            "       PROGRAM-ID. B.\n           2,5\n       END PROGRAM B.\n           3,5\n       END PROGRAM A.\n",
666            "       PROGRAM-ID. C.\n           4,5\n",
667        );
668        assert_eq!(numbers(text), ["1.5", "2.5", "3.5", "4", "5"]);
669    }
670}