Skip to main content

ironwork_syntax/
lexer.rs

1use crate::source::Source;
2use crate::{Error, Pos};
3
4#[derive(Clone, Debug, PartialEq, Eq)]
5pub enum Tok {
6    /// A COBOL word, uppercased: a name, a reserved word or a level number with letters in it.
7    Word(String),
8    /// A numeric literal as written: optional sign, digits, optional decimal point.
9    Number(String),
10    Alnum(String),
11    Hex(Vec<u8>),
12    National(String),
13    Pic(String),
14    /// An EXEC ... END-EXEC block, as written: SQL, CICS or DLI for a precompiler.
15    Exec(String),
16    Period,
17    LParen,
18    RParen,
19    Colon,
20    Plus,
21    Minus,
22    Star,
23    Slash,
24    Power,
25    Eq,
26    Lt,
27    Gt,
28    Le,
29    Ge,
30}
31
32#[derive(Clone, Debug, PartialEq, Eq)]
33pub struct Token {
34    pub tok: Tok,
35    pub pos: Pos,
36    /// Starts in area A (columns 8 to 11), where division, section and paragraph headers go.
37    pub area_a: bool,
38    /// A word as the source spells it, where that is not all capitals.
39    pub spelled: Option<String>,
40    /// A separator comma or semicolon comes between this token and the one before it.
41    pub after_comma: bool,
42    /// Messages about the token that do not stop the parse.
43    pub messages: Vec<Error>,
44}
45
46struct Lexer<'a> {
47    chars: Vec<char>,
48    positions: &'a [Pos],
49    at: usize,
50    tokens: Vec<Token>,
51    /// DECIMAL-POINT IS COMMA is in force: a comma between digits is the decimal point.
52    decimal_comma: bool,
53    /// The currency symbols CURRENCY SIGN clauses have named so far, whose case a PICTURE keeps.
54    currency: Vec<char>,
55    /// For each program begun and not yet ended, whether the comma was the decimal point before
56    /// it: a contained program has its container's, and a program after it starts afresh.
57    outer: Vec<bool>,
58    comma_pending: bool,
59    /// Messages for the next token emitted.
60    pending: Vec<Error>,
61}
62
63pub fn lex(source: &Source) -> Result<Vec<Token>, Error> {
64    let mut lx = Lexer { chars: source.text.chars().collect(), positions: &source.positions, at: 0, tokens: Vec::new(), decimal_comma: false, currency: Vec::new(), outer: Vec::new(), comma_pending: false, pending: Vec::new() };
65    while lx.at < lx.chars.len() {
66        lx.next_token()?;
67    }
68    Ok(lx.tokens)
69}
70
71fn is_word_char(c: char) -> bool {
72    c.is_ascii_alphanumeric() || c == '-' || c == '_'
73}
74
75/// A single-byte character outside IBM's basic COBOL character set (Language Reference
76/// SC27-8713-03, Table 1, pp. 3-6), which IBM accepts with an error as part of the text around it
77/// (assumption C123). `$` and `&`, which ironwork reads only where COBOL puts them, are left out.
78/// The lowercase letter an EBCDIC byte is, the same in every code page ironwork carries.
79fn ebcdic_lowercase(byte: u8) -> Option<char> {
80    let (start, first) = match byte {
81        0x81..=0x89 => (0x81, b'a'),
82        0x91..=0x99 => (0x91, b'j'),
83        0xA2..=0xA9 => (0xA2, b's'),
84        _ => return None,
85    };
86    Some(char::from(first + (byte - start)))
87}
88
89fn non_cobol(c: char) -> bool {
90    u32::from(c) <= 0xFF && !c.is_ascii_alphanumeric() && !" \n+-*/=$,;.\"'()><:_&".contains(c)
91}
92
93impl Lexer<'_> {
94    fn peek(&self, ahead: usize) -> Option<char> {
95        self.chars.get(self.at + ahead).copied()
96    }
97
98    fn pos(&self) -> Pos {
99        self.positions.get(self.at).copied().unwrap_or_default()
100    }
101
102    fn separator_follows(&self, ahead: usize) -> bool {
103        self.peek(ahead).is_none_or(|c| c == ' ' || c == '\n')
104    }
105
106    fn emit(&mut self, tok: Tok, pos: Pos) {
107        let before = |back: usize| match self.tokens.len().checked_sub(back).map(|i| &self.tokens[i].tok) {
108            Some(Tok::Word(p)) => p.as_str(),
109            _ => "",
110        };
111        match &tok {
112            Tok::Word(w) => match w.as_str() {
113                "PROGRAM-ID" => self.outer.push(self.decimal_comma),
114                "PROGRAM" if before(1) == "END" => self.decimal_comma = self.outer.pop().unwrap_or(false),
115                "COMMA" if before(1) == "DECIMAL-POINT" || before(1) == "IS" && before(2) == "DECIMAL-POINT" => self.decimal_comma = true,
116                _ => {}
117            },
118            Tok::Alnum(a) => {
119                let mut chars = a.chars();
120                let names_symbol = before(1) == "SYMBOL" || (1..=3).any(|back| before(back) == "CURRENCY" && (1..back).all(|b| matches!(before(b), "SIGN" | "IS")));
121                if let (Some(c), None, true) = (chars.next(), chars.next(), names_symbol) {
122                    self.currency.push(c);
123                }
124            }
125            Tok::Hex(bytes) => {
126                let names_symbol = (1..=3).any(|back| before(back) == "CURRENCY" && (1..back).all(|b| matches!(before(b), "SIGN" | "IS")));
127                if let ([byte], true) = (bytes.as_slice(), names_symbol)
128                    && let Some(c) = ebcdic_lowercase(*byte)
129                {
130                    self.currency.push(c);
131                }
132            }
133            _ => {}
134        }
135        let spelled = match &tok {
136            Tok::Word(w) => self
137                .at
138                .checked_sub(w.chars().count())
139                .map(|start| &self.chars[start..self.at])
140                .filter(|raw| raw.iter().any(char::is_ascii_lowercase))
141                .map(|raw| raw.iter().collect::<String>())
142                .filter(|raw| raw.eq_ignore_ascii_case(w)),
143            _ => None,
144        };
145        let after_comma = std::mem::take(&mut self.comma_pending);
146        self.tokens.push(Token { tok, pos, area_a: (8..=11).contains(&pos.col), spelled, after_comma, messages: std::mem::take(&mut self.pending) });
147    }
148
149    /// The character that is a numeric literal's decimal point.
150    fn point(&self) -> char {
151        if self.decimal_comma { ',' } else { '.' }
152    }
153
154    fn expecting_picture(&self) -> bool {
155        let words: Vec<&str> =
156            self.tokens.iter().rev().take(2).map(|t| if let Tok::Word(w) = &t.tok { w.as_str() } else { "" }).collect();
157        matches!(words.as_slice(), ["PIC" | "PICTURE", ..] | ["IS", "PIC" | "PICTURE"])
158    }
159
160    fn next_token(&mut self) -> Result<(), Error> {
161        let c = self.chars[self.at];
162        let pos = self.pos();
163        let point = self.point();
164        let leading_point = c == ',' && point == ',' && self.peek(1).is_some_and(|d| d.is_ascii_digit()) && !self.at.checked_sub(1).is_some_and(|i| is_word_char(self.chars[i]));
165        if leading_point && self.expecting_picture() {
166            return self.picture(pos);
167        }
168        if leading_point {
169            let tok = self.number_or_word(pos)?;
170            self.emit(tok, pos);
171            return Ok(());
172        }
173        if c == ' ' || c == '\n' || c == ',' || c == ';' {
174            self.comma_pending |= c == ',' || c == ';';
175            self.at += 1;
176            return Ok(());
177        }
178        if self.expecting_picture() {
179            return self.picture(pos);
180        }
181        let next = self.peek(1);
182        let quote_next = matches!(next, Some('\'' | '"'));
183        match c {
184            '\'' | '"' => {
185                let text = self.quoted(pos)?;
186                self.emit(Tok::Alnum(text), pos);
187            }
188            'X' | 'x' if quote_next => {
189                self.at += 1;
190                let text = self.quoted(pos)?;
191                let bytes = unhex(&text).ok_or_else(|| Error::at(pos, format!("X'{text}' is not an even number of hex digits")))?;
192                self.emit(Tok::Hex(bytes), pos);
193            }
194            'N' | 'n' if matches!(next, Some('X' | 'x')) && matches!(self.peek(2), Some('\'' | '"')) => {
195                self.at += 2;
196                let text = self.quoted(pos)?;
197                let units = unhex(&text).filter(|b| !b.is_empty() && b.len().is_multiple_of(2) && b.len() <= 160);
198                let units = units.map(|b| b.chunks(2).map(|u| u16::from_be_bytes([u[0], u[1]])).collect::<Vec<_>>());
199                let national = units.and_then(|u| String::from_utf16(&u).ok());
200                let national = national.ok_or_else(|| Error::at(pos, format!("NX'{text}': a national hexadecimal literal is 4 to 320 hex digits, four to each UTF-16 code unit")))?;
201                self.emit(Tok::National(national), pos);
202            }
203            'N' | 'n' if quote_next => {
204                self.at += 1;
205                let text = self.quoted(pos)?;
206                self.emit(Tok::National(text), pos);
207            }
208            'Z' | 'z' if quote_next => {
209                self.at += 1;
210                let text = self.quoted(pos)?;
211                self.emit(Tok::Alnum(format!("{text}\0")), pos);
212            }
213            '.' if self.separator_follows(1) => {
214                self.at += 1;
215                self.emit(Tok::Period, pos);
216            }
217            '+' | '-' if next.is_some_and(|n| n.is_ascii_digit() || (n == point && self.peek(2).is_some_and(|d| d.is_ascii_digit()))) => {
218                self.at += 1;
219                let digits = self.number_or_word(pos)?;
220                match digits {
221                    Tok::Number(n) => self.emit(Tok::Number(format!("{c}{n}")), pos),
222                    _ => return Err(Error::at(pos, "a sign must be followed by a number")),
223                }
224            }
225            _ if c.is_ascii_alphanumeric() || c == '.' || non_cobol(c) => {
226                let tok = self.number_or_word(pos)?;
227                match tok {
228                    Tok::Word(w) if w == "EXEC" || w == "EXECUTE" => {
229                        let tok = self.exec_block(pos)?;
230                        self.emit(tok, pos);
231                    }
232                    tok => self.emit(tok, pos),
233                }
234            }
235            _ => {
236                let (tok, len) = match (c, next) {
237                    ('*', Some('*')) => (Tok::Power, 2),
238                    ('<', Some('=')) => (Tok::Le, 2),
239                    ('>', Some('=')) => (Tok::Ge, 2),
240                    ('*', _) => (Tok::Star, 1),
241                    ('/', _) => (Tok::Slash, 1),
242                    ('+', _) => (Tok::Plus, 1),
243                    ('-', _) => (Tok::Minus, 1),
244                    ('=', _) => (Tok::Eq, 1),
245                    ('<', _) => (Tok::Lt, 1),
246                    ('>', _) => (Tok::Gt, 1),
247                    ('(', _) => (Tok::LParen, 1),
248                    (')', _) => (Tok::RParen, 1),
249                    (':', _) => (Tok::Colon, 1),
250                    ('&', _) if matches!(self.tokens.last().map(|t| &t.tok), Some(Tok::Alnum(_) | Tok::Hex(_) | Tok::National(_))) => {
251                        return Err(Error::at(pos, "literal concatenation with & is not Enterprise COBOL's"));
252                    }
253                    _ => return Err(Error::at(pos, format!("unexpected character {c:?}"))),
254                };
255                self.at += len;
256                self.emit(tok, pos);
257            }
258        }
259        Ok(())
260    }
261
262    /// The text of an EXEC block up to END-EXEC, which is consumed; quotes inside are skipped whole.
263    fn exec_block(&mut self, pos: Pos) -> Result<Tok, Error> {
264        let start = self.at;
265        let mut quote: Option<char> = None;
266        while let Some(c) = self.peek(0) {
267            match quote {
268                Some(q) if c == q => quote = None,
269                Some(_) => {}
270                None if c == '\'' || c == '"' => quote = Some(c),
271                None if c.eq_ignore_ascii_case(&'E') => {
272                    let ahead: String = self.chars[self.at..].iter().take(8).collect();
273                    let boundary = self.chars.get(self.at + 8).is_none_or(|d| !is_word_char(*d)) && (self.at == 0 || !is_word_char(self.chars[self.at - 1]));
274                    if ahead.eq_ignore_ascii_case("END-EXEC") && boundary {
275                        let text: String = self.chars[start..self.at].iter().collect();
276                        self.at += 8;
277                        return Ok(Tok::Exec(text.split_whitespace().collect::<Vec<_>>().join(" ")));
278                    }
279                }
280                None => {}
281            }
282            self.at += 1;
283        }
284        Err(Error::at(pos, "EXEC with no END-EXEC"))
285    }
286
287    /// Reads a quoted literal starting at the opening quote; a doubled quote stands for one.
288    fn quoted(&mut self, pos: Pos) -> Result<String, Error> {
289        let quote = self.chars[self.at];
290        self.at += 1;
291        let mut text = String::new();
292        loop {
293            match self.peek(0) {
294                None | Some('\n') => return Err(Error::at(pos, "an unterminated literal")),
295                Some(c) if c == quote && self.peek(1) == Some(quote) => {
296                    text.push(quote);
297                    self.at += 2;
298                }
299                Some(c) if c == quote => {
300                    self.at += 1;
301                    return Ok(text);
302                }
303                Some(c) => {
304                    text.push(c);
305                    self.at += 1;
306                }
307            }
308        }
309    }
310
311    fn number_or_word(&mut self, pos: Pos) -> Result<Tok, Error> {
312        let start = self.at;
313        while let Some(c) = self.peek(0).filter(|&c| is_word_char(c) || non_cobol(c)) {
314            if non_cobol(c) {
315                self.pending.push(Error::at(self.pos(), format!("non-COBOL character {c:?}: the character was accepted")).graded(crate::Severity::Error));
316            }
317            self.at += 1;
318        }
319        let run: String = self.chars[start..self.at].iter().collect();
320        let all_digits = run.chars().all(|c| c.is_ascii_digit());
321        if all_digits && self.peek(0) == Some(self.point()) && self.peek(1).is_some_and(|c| c.is_ascii_digit()) {
322            self.at += 1;
323            let frac_start = self.at;
324            while self.peek(0).is_some_and(|c| c.is_ascii_digit()) {
325                self.at += 1;
326            }
327            let frac: String = self.chars[frac_start..self.at].iter().collect();
328            return Ok(Tok::Number(format!("{run}.{frac}")));
329        }
330        if run.is_empty() {
331            return Err(Error::at(pos, "unexpected '.'"));
332        }
333        Ok(if all_digits { Tok::Number(run) } else { Tok::Word(run.to_ascii_uppercase()) })
334    }
335
336    /// Only the period, comma or semicolon just before the space is a separator (assumption C195).
337    fn picture(&mut self, pos: Pos) -> Result<(), Error> {
338        let start = self.at;
339        while self.peek(0).is_some_and(|c| c != ' ' && c != '\n') {
340            self.at += 1;
341        }
342        let mut end = self.at;
343        if end > start && matches!(self.chars[end - 1], '.' | ',' | ';') {
344            end -= 1;
345        }
346        let text: String = self.chars[start..end].iter().collect();
347        if text.is_empty() {
348            return Err(Error::at(pos, "PICTURE with no character-string"));
349        }
350        self.at = end;
351        let text: String = text.chars().map(|c| if self.currency.contains(&c) { c } else { c.to_ascii_uppercase() }).collect();
352        let tok = if matches!(text.as_str(), "IS" | "SYMBOL") && self.tokens.last().is_some_and(|t| matches!(&t.tok, Tok::Word(w) if w == "PIC" || w == "PICTURE")) {
353            Tok::Word(text)
354        } else {
355            Tok::Pic(text)
356        };
357        self.emit(tok, pos);
358        Ok(())
359    }
360}
361
362fn unhex(text: &str) -> Option<Vec<u8>> {
363    if !text.len().is_multiple_of(2) || !text.is_ascii() {
364        return None;
365    }
366    (0..text.len()).step_by(2).map(|i| u8::from_str_radix(&text[i..i + 2], 16).ok()).collect()
367}
368
369#[cfg(test)]
370mod tests {
371    use super::*;
372    use crate::source;
373
374    fn toks(text: &str) -> Vec<Tok> {
375        lex(&source::read(text).unwrap()).unwrap().into_iter().map(|t| t.tok).collect()
376    }
377
378    fn w(s: &str) -> Tok {
379        Tok::Word(s.into())
380    }
381
382    #[test]
383    fn numbers_words_and_the_separator_period() {
384        assert_eq!(toks("           05 A-1 VALUE 0.1."), [Tok::Number("05".into()), w("A-1"), w("VALUE"), Tok::Number("0.1".into()), Tok::Period]);
385        assert_eq!(toks("           VALUE -12345."), [w("VALUE"), Tok::Number("-12345".into()), Tok::Period]);
386        assert_eq!(toks("       100-MAIN."), [w("100-MAIN"), Tok::Period]);
387    }
388
389    #[test]
390    fn operators_need_spaces_and_signed_literals_do_not() {
391        assert_eq!(toks("           A - 1 ** 2"), [w("A"), Tok::Minus, Tok::Number("1".into()), Tok::Power, Tok::Number("2".into())]);
392        assert_eq!(toks("           >= <= ("), [Tok::Ge, Tok::Le, Tok::LParen]);
393    }
394
395    #[test]
396    fn literals() {
397        assert_eq!(toks("           'IT''S' X'F1C1' N'AB'"), [Tok::Alnum("IT'S".into()), Tok::Hex(vec![0xF1, 0xC1]), Tok::National("AB".into())]);
398        assert_eq!(toks("           NX'00410042' nx\"265ED83DDE00\""), [Tok::National("AB".into()), Tok::National("\u{265E}\u{1F600}".into())]);
399        for bad in ["NX'GH'", "NX'1'", "NX'004'", "NX'D83D'"] {
400            let e = lex(&source::read(&format!("           {bad}")).unwrap()).unwrap_err();
401            assert!(e.message.contains("a national hexadecimal literal is 4 to 320 hex digits"), "{bad}: {}", e.message);
402        }
403    }
404
405    #[test]
406    fn a_null_terminated_literal_ends_with_x00() {
407        assert_eq!(toks("           Z'ABC' z\"(I)V\""), [Tok::Alnum("ABC\0".into()), Tok::Alnum("(I)V\0".into())]);
408    }
409
410    #[test]
411    fn a_picture_is_one_token_and_keeps_its_own_periods() {
412        assert_eq!(toks("           PIC S9(3)V99 COMP-3."), [w("PIC"), Tok::Pic("S9(3)V99".into()), w("COMP-3"), Tok::Period]);
413        assert_eq!(toks("           PICTURE IS ZZ,ZZ9.99."), [w("PICTURE"), w("IS"), Tok::Pic("ZZ,ZZ9.99".into()), Tok::Period]);
414    }
415
416    #[test]
417    fn only_the_last_period_or_comma_before_the_space_is_a_separator() {
418        assert_eq!(toks("           PIC 9,9,9,."), [w("PIC"), Tok::Pic("9,9,9,".into()), Tok::Period]);
419        assert_eq!(toks("           PIC 999999999999.."), [w("PIC"), Tok::Pic("999999999999.".into()), Tok::Period]);
420        assert_eq!(toks("           PIC 99, VALUE 1."), [w("PIC"), Tok::Pic("99".into()), w("VALUE"), Tok::Number("1".into()), Tok::Period]);
421        assert_eq!(toks("           PIC 9.9,; VALUE 1."), [w("PIC"), Tok::Pic("9.9,".into()), w("VALUE"), Tok::Number("1".into()), Tok::Period]);
422        assert_eq!(toks("           PIC 999., VALUE 1."), [w("PIC"), Tok::Pic("999.".into()), w("VALUE"), Tok::Number("1".into()), Tok::Period]);
423        let comma = "           DECIMAL-POINT IS COMMA.\n           PIC 9.9.9,. PIC 999,,\n";
424        let pics: Vec<Tok> = toks(comma).into_iter().filter(|t| matches!(t, Tok::Pic(_))).collect();
425        assert_eq!(pics, [Tok::Pic("9.9.9,".into()), Tok::Pic("999,".into())]);
426    }
427
428    #[test]
429    fn commas_between_operands_are_separators() {
430        assert_eq!(toks("           F(A, 1)"), [w("F"), Tok::LParen, w("A"), Tok::Number("1".into()), Tok::RParen]);
431        assert_eq!(toks("           T(I,J)"), [w("T"), Tok::LParen, w("I"), w("J"), Tok::RParen]);
432    }
433
434    #[test]
435    fn an_exec_block_is_one_token() {
436        assert_eq!(
437            toks("           EXEC SQL SELECT A.B INTO :X FROM T WHERE C = 'END-EXEC'\n               END-EXEC."),
438            [Tok::Exec("SQL SELECT A.B INTO :X FROM T WHERE C = 'END-EXEC'".into()), Tok::Period]
439        );
440    }
441
442    #[test]
443    fn a_comment_entry_holds_any_character() {
444        let program = concat!(
445            "       IDENTIFICATION DIVISION.\n",
446            "       PROGRAM-ID. CE3.\n",
447            "       AUTHOR. Smith & Jones @ ACME #1, Café O'Grady.\n",
448            "       INSTALLATION. \"HQ\n",
449            "           ~ ^ ` { } | \\ ?\n",
450            "       DATE-WRITTEN. 01/01/99.\n",
451            "       PROCEDURE DIVISION.\n",
452            "           GOBACK.\n",
453        );
454        let words = [w("IDENTIFICATION"), w("DIVISION"), Tok::Period, w("PROGRAM-ID"), Tok::Period, w("CE3"), Tok::Period];
455        let paragraphs = [w("AUTHOR"), Tok::Period, w("INSTALLATION"), Tok::Period, w("DATE-WRITTEN"), Tok::Period];
456        let procedure = [w("PROCEDURE"), w("DIVISION"), Tok::Period, w("GOBACK"), Tok::Period];
457        assert_eq!(toks(program), [&words[..], &paragraphs[..], &procedure[..]].concat());
458    }
459
460    #[test]
461    fn a_literal_continued_between_the_quotes_of_a_doubled_quote() {
462        let head = "           \"A+0B-1C*2D";
463        let text = format!("{head}{}\"\n      -    \"\"9K(L)M>N<O\".\n", "=".repeat(72 - head.len() - 1));
464        assert_eq!(toks(&text), [Tok::Alnum(format!("A+0B-1C*2D{}\"9K(L)M>N<O", "=".repeat(72 - head.len() - 1))), Tok::Period]);
465    }
466
467    #[test]
468    fn a_continuation_that_opens_a_quote_after_a_closed_literal_is_a_second_literal() {
469        assert_eq!(toks("           VALUE 'ABC'\n      -    'DEF'."), [w("VALUE"), Tok::Alnum("ABC".into()), Tok::Alnum("DEF".into()), Tok::Period]);
470    }
471
472    #[test]
473    fn an_ampersand_after_a_literal_is_named_as_concatenation() {
474        let error = |text: &str| lex(&source::read(text).unwrap()).unwrap_err().message;
475        assert!(error("           'A' & 'B'").contains("literal concatenation with & is not Enterprise COBOL's"));
476        assert!(error("           NOTIFY=&SYSUID").contains("unexpected character '&'"));
477    }
478
479    #[test]
480    fn a_non_cobol_character_is_accepted_into_its_word_with_an_error() {
481        let lexed = lex(&source::read("           MOVE WS#1 TO %\u{1b} 'A@B'.").unwrap()).unwrap();
482        assert_eq!(lexed.iter().map(|t| t.tok.clone()).collect::<Vec<_>>(), [w("MOVE"), w("WS#1"), w("TO"), w("%\u{1b}"), Tok::Alnum("A@B".into()), Tok::Period]);
483        let messages: Vec<(usize, u32, &str, crate::Severity)> =
484            lexed.iter().enumerate().flat_map(|(i, t)| t.messages.iter().map(move |m| (i, m.pos.col, m.message.as_str(), m.severity))).collect();
485        assert_eq!(
486            messages,
487            [
488                (1, 19, "non-COBOL character '#': the character was accepted", crate::Severity::Error),
489                (3, 25, "non-COBOL character '%': the character was accepted", crate::Severity::Error),
490                (3, 26, "non-COBOL character '\\u{1b}': the character was accepted", crate::Severity::Error),
491            ]
492        );
493        let error = |text: &str| lex(&source::read(text).unwrap()).unwrap_err().message;
494        assert_eq!(error("           MOVE $X"), "unexpected character '$'");
495        assert_eq!(error("           MOVE \u{3042}"), "unexpected character '\u{3042}'");
496    }
497
498    #[test]
499    fn area_a_is_marked() {
500        let t = lex(&source::read("       PARA.\n           MOVE").unwrap()).unwrap();
501        assert!(t[0].area_a);
502        assert!(!t[2].area_a);
503    }
504
505    fn numbers(text: &str) -> Vec<String> {
506        toks(text).into_iter().filter_map(|t| if let Tok::Number(n) = t { Some(n) } else { None }).collect()
507    }
508
509    #[test]
510    fn under_decimal_point_is_comma_a_comma_between_digits_is_the_point() {
511        let text = "           DECIMAL-POINT IS COMMA.\n           1,5 -,25 +3,0 ,75 T(1, 2) A,B 1.\n";
512        assert_eq!(numbers(text), ["1.5", "-.25", "+3.0", ".75", "1", "2", "1"]);
513        assert!(toks(text).contains(&w("B")));
514        assert!(toks("           DECIMAL-POINT IS COMMA.\n           PIC ,99.").contains(&Tok::Pic(",99".into())));
515        assert!(lex(&source::read("           DECIMAL-POINT COMMA.\n           MOVE 1.5").unwrap()).is_err());
516    }
517
518    #[test]
519    fn a_contained_program_keeps_the_decimal_comma_and_the_next_program_does_not() {
520        let text = concat!(
521            "       PROGRAM-ID. A.\n           DECIMAL-POINT IS COMMA.\n           1,5\n",
522            "       PROGRAM-ID. B.\n           2,5\n       END PROGRAM B.\n           3,5\n       END PROGRAM A.\n",
523            "       PROGRAM-ID. C.\n           4,5\n",
524        );
525        assert_eq!(numbers(text), ["1.5", "2.5", "3.5", "4", "5"]);
526    }
527}