Skip to main content

ironwork_syntax/
lexer.rs

1use crate::source::Source;
2use crate::{Error, Pos};
3
4#[derive(Clone, Debug, PartialEq, Eq)]
5pub enum Tok {
6    /// A COBOL word, uppercased: a name, a reserved word or a level number with letters in it.
7    Word(String),
8    /// A numeric literal as written: optional sign, digits, optional decimal point.
9    Number(String),
10    Alnum(String),
11    Hex(Vec<u8>),
12    National(String),
13    Pic(String),
14    /// An EXEC ... END-EXEC block, as written: SQL, CICS or DLI for a precompiler.
15    Exec(String),
16    Period,
17    LParen,
18    RParen,
19    Colon,
20    Plus,
21    Minus,
22    Star,
23    Slash,
24    Power,
25    Eq,
26    Lt,
27    Gt,
28    Le,
29    Ge,
30}
31
32#[derive(Clone, Debug, PartialEq, Eq)]
33pub struct Token {
34    pub tok: Tok,
35    pub pos: Pos,
36    /// Starts in area A (columns 8 to 11), where division, section and paragraph headers go.
37    pub area_a: bool,
38}
39
40struct Lexer<'a> {
41    chars: Vec<char>,
42    positions: &'a [Pos],
43    at: usize,
44    tokens: Vec<Token>,
45    /// DECIMAL-POINT IS COMMA is in force: a comma between digits is the decimal point.
46    decimal_comma: bool,
47    /// For each program begun and not yet ended, whether the comma was the decimal point before
48    /// it: a contained program has its container's, and a program after it starts afresh.
49    outer: Vec<bool>,
50}
51
52pub fn lex(source: &Source) -> Result<Vec<Token>, Error> {
53    let mut lx = Lexer { chars: source.text.chars().collect(), positions: &source.positions, at: 0, tokens: Vec::new(), decimal_comma: false, outer: Vec::new() };
54    while lx.at < lx.chars.len() {
55        lx.next_token()?;
56    }
57    Ok(lx.tokens)
58}
59
60fn is_word_char(c: char) -> bool {
61    c.is_ascii_alphanumeric() || c == '-' || c == '_'
62}
63
64impl Lexer<'_> {
65    fn peek(&self, ahead: usize) -> Option<char> {
66        self.chars.get(self.at + ahead).copied()
67    }
68
69    fn pos(&self) -> Pos {
70        self.positions.get(self.at).copied().unwrap_or_default()
71    }
72
73    fn separator_follows(&self, ahead: usize) -> bool {
74        self.peek(ahead).is_none_or(|c| c == ' ' || c == '\n')
75    }
76
77    fn emit(&mut self, tok: Tok, pos: Pos) {
78        if let Tok::Word(w) = &tok {
79            let before = |back: usize| match self.tokens.len().checked_sub(back).map(|i| &self.tokens[i].tok) {
80                Some(Tok::Word(p)) => p.as_str(),
81                _ => "",
82            };
83            match w.as_str() {
84                "PROGRAM-ID" => self.outer.push(self.decimal_comma),
85                "PROGRAM" if before(1) == "END" => self.decimal_comma = self.outer.pop().unwrap_or(false),
86                "COMMA" if before(1) == "DECIMAL-POINT" || before(1) == "IS" && before(2) == "DECIMAL-POINT" => self.decimal_comma = true,
87                _ => {}
88            }
89        }
90        self.tokens.push(Token { tok, pos, area_a: (8..=11).contains(&pos.col) });
91    }
92
93    /// The character that is a numeric literal's decimal point.
94    fn point(&self) -> char {
95        if self.decimal_comma { ',' } else { '.' }
96    }
97
98    fn expecting_picture(&self) -> bool {
99        let words: Vec<&str> =
100            self.tokens.iter().rev().take(2).map(|t| if let Tok::Word(w) = &t.tok { w.as_str() } else { "" }).collect();
101        matches!(words.as_slice(), ["PIC" | "PICTURE", ..] | ["IS", "PIC" | "PICTURE"])
102    }
103
104    fn next_token(&mut self) -> Result<(), Error> {
105        let c = self.chars[self.at];
106        let pos = self.pos();
107        let point = self.point();
108        let leading_point = c == ',' && point == ',' && self.peek(1).is_some_and(|d| d.is_ascii_digit()) && !self.at.checked_sub(1).is_some_and(|i| is_word_char(self.chars[i]));
109        if leading_point && self.expecting_picture() {
110            return self.picture(pos);
111        }
112        if leading_point {
113            let tok = self.number_or_word(pos)?;
114            self.emit(tok, pos);
115            return Ok(());
116        }
117        if c == ' ' || c == '\n' || c == ',' || c == ';' {
118            self.at += 1;
119            return Ok(());
120        }
121        if self.expecting_picture() {
122            return self.picture(pos);
123        }
124        let next = self.peek(1);
125        let quote_next = matches!(next, Some('\'' | '"'));
126        match c {
127            '\'' | '"' => {
128                let text = self.quoted(pos)?;
129                self.emit(Tok::Alnum(text), pos);
130            }
131            'X' | 'x' if quote_next => {
132                self.at += 1;
133                let text = self.quoted(pos)?;
134                let bytes = unhex(&text).ok_or_else(|| Error::at(pos, format!("X'{text}' is not an even number of hex digits")))?;
135                self.emit(Tok::Hex(bytes), pos);
136            }
137            'N' | 'n' if quote_next => {
138                self.at += 1;
139                let text = self.quoted(pos)?;
140                self.emit(Tok::National(text), pos);
141            }
142            'Z' | 'z' if quote_next => {
143                self.at += 1;
144                let text = self.quoted(pos)?;
145                self.emit(Tok::Alnum(format!("{text}\0")), pos);
146            }
147            '.' if self.separator_follows(1) => {
148                self.at += 1;
149                self.emit(Tok::Period, pos);
150            }
151            '+' | '-' if next.is_some_and(|n| n.is_ascii_digit() || (n == point && self.peek(2).is_some_and(|d| d.is_ascii_digit()))) => {
152                self.at += 1;
153                let digits = self.number_or_word(pos)?;
154                match digits {
155                    Tok::Number(n) => self.emit(Tok::Number(format!("{c}{n}")), pos),
156                    _ => return Err(Error::at(pos, "a sign must be followed by a number")),
157                }
158            }
159            _ if c.is_ascii_alphanumeric() || c == '.' => {
160                let tok = self.number_or_word(pos)?;
161                match tok {
162                    Tok::Word(w) if w == "EXEC" || w == "EXECUTE" => {
163                        let tok = self.exec_block(pos)?;
164                        self.emit(tok, pos);
165                    }
166                    tok => self.emit(tok, pos),
167                }
168            }
169            _ => {
170                let (tok, len) = match (c, next) {
171                    ('*', Some('*')) => (Tok::Power, 2),
172                    ('<', Some('=')) => (Tok::Le, 2),
173                    ('>', Some('=')) => (Tok::Ge, 2),
174                    ('*', _) => (Tok::Star, 1),
175                    ('/', _) => (Tok::Slash, 1),
176                    ('+', _) => (Tok::Plus, 1),
177                    ('-', _) => (Tok::Minus, 1),
178                    ('=', _) => (Tok::Eq, 1),
179                    ('<', _) => (Tok::Lt, 1),
180                    ('>', _) => (Tok::Gt, 1),
181                    ('(', _) => (Tok::LParen, 1),
182                    (')', _) => (Tok::RParen, 1),
183                    (':', _) => (Tok::Colon, 1),
184                    ('&', _) if matches!(self.tokens.last().map(|t| &t.tok), Some(Tok::Alnum(_) | Tok::Hex(_) | Tok::National(_))) => {
185                        return Err(Error::at(pos, "literal concatenation with & is not Enterprise COBOL's"));
186                    }
187                    _ => return Err(Error::at(pos, format!("unexpected character {c:?}"))),
188                };
189                self.at += len;
190                self.emit(tok, pos);
191            }
192        }
193        Ok(())
194    }
195
196    /// The text of an EXEC block up to END-EXEC, which is consumed; quotes inside are skipped whole.
197    fn exec_block(&mut self, pos: Pos) -> Result<Tok, Error> {
198        let start = self.at;
199        let mut quote: Option<char> = None;
200        while let Some(c) = self.peek(0) {
201            match quote {
202                Some(q) if c == q => quote = None,
203                Some(_) => {}
204                None if c == '\'' || c == '"' => quote = Some(c),
205                None if c.eq_ignore_ascii_case(&'E') => {
206                    let ahead: String = self.chars[self.at..].iter().take(8).collect();
207                    let boundary = self.chars.get(self.at + 8).is_none_or(|d| !is_word_char(*d)) && (self.at == 0 || !is_word_char(self.chars[self.at - 1]));
208                    if ahead.eq_ignore_ascii_case("END-EXEC") && boundary {
209                        let text: String = self.chars[start..self.at].iter().collect();
210                        self.at += 8;
211                        return Ok(Tok::Exec(text.split_whitespace().collect::<Vec<_>>().join(" ")));
212                    }
213                }
214                None => {}
215            }
216            self.at += 1;
217        }
218        Err(Error::at(pos, "EXEC with no END-EXEC"))
219    }
220
221    /// Reads a quoted literal starting at the opening quote; a doubled quote stands for one.
222    fn quoted(&mut self, pos: Pos) -> Result<String, Error> {
223        let quote = self.chars[self.at];
224        self.at += 1;
225        let mut text = String::new();
226        loop {
227            match self.peek(0) {
228                None | Some('\n') => return Err(Error::at(pos, "an unterminated literal")),
229                Some(c) if c == quote && self.peek(1) == Some(quote) => {
230                    text.push(quote);
231                    self.at += 2;
232                }
233                Some(c) if c == quote => {
234                    self.at += 1;
235                    return Ok(text);
236                }
237                Some(c) => {
238                    text.push(c);
239                    self.at += 1;
240                }
241            }
242        }
243    }
244
245    fn number_or_word(&mut self, pos: Pos) -> Result<Tok, Error> {
246        let start = self.at;
247        while self.peek(0).is_some_and(is_word_char) {
248            self.at += 1;
249        }
250        let run: String = self.chars[start..self.at].iter().collect();
251        let all_digits = run.chars().all(|c| c.is_ascii_digit());
252        if all_digits && self.peek(0) == Some(self.point()) && self.peek(1).is_some_and(|c| c.is_ascii_digit()) {
253            self.at += 1;
254            let frac_start = self.at;
255            while self.peek(0).is_some_and(|c| c.is_ascii_digit()) {
256                self.at += 1;
257            }
258            let frac: String = self.chars[frac_start..self.at].iter().collect();
259            return Ok(Tok::Number(format!("{run}.{frac}")));
260        }
261        if run.is_empty() {
262            return Err(Error::at(pos, "unexpected '.'"));
263        }
264        Ok(if all_digits { Tok::Number(run) } else { Tok::Word(run.to_ascii_uppercase()) })
265    }
266
267    fn picture(&mut self, pos: Pos) -> Result<(), Error> {
268        let start = self.at;
269        while self.peek(0).is_some_and(|c| c != ' ' && c != '\n') {
270            self.at += 1;
271        }
272        let mut end = self.at;
273        while end > start && matches!(self.chars[end - 1], '.' | ',' | ';') {
274            end -= 1;
275        }
276        let text: String = self.chars[start..end].iter().collect();
277        if text.is_empty() {
278            return Err(Error::at(pos, "PICTURE with no character-string"));
279        }
280        self.at = end;
281        let text = text.to_ascii_uppercase();
282        let tok = if text == "IS" && self.tokens.last().is_some_and(|t| matches!(&t.tok, Tok::Word(w) if w == "PIC" || w == "PICTURE")) {
283            Tok::Word(text)
284        } else {
285            Tok::Pic(text)
286        };
287        self.emit(tok, pos);
288        Ok(())
289    }
290}
291
292fn unhex(text: &str) -> Option<Vec<u8>> {
293    if !text.len().is_multiple_of(2) || !text.is_ascii() {
294        return None;
295    }
296    (0..text.len()).step_by(2).map(|i| u8::from_str_radix(&text[i..i + 2], 16).ok()).collect()
297}
298
299#[cfg(test)]
300mod tests {
301    use super::*;
302    use crate::source;
303
304    fn toks(text: &str) -> Vec<Tok> {
305        lex(&source::read(text).unwrap()).unwrap().into_iter().map(|t| t.tok).collect()
306    }
307
308    fn w(s: &str) -> Tok {
309        Tok::Word(s.into())
310    }
311
312    #[test]
313    fn numbers_words_and_the_separator_period() {
314        assert_eq!(toks("           05 A-1 VALUE 0.1."), [Tok::Number("05".into()), w("A-1"), w("VALUE"), Tok::Number("0.1".into()), Tok::Period]);
315        assert_eq!(toks("           VALUE -12345."), [w("VALUE"), Tok::Number("-12345".into()), Tok::Period]);
316        assert_eq!(toks("       100-MAIN."), [w("100-MAIN"), Tok::Period]);
317    }
318
319    #[test]
320    fn operators_need_spaces_and_signed_literals_do_not() {
321        assert_eq!(toks("           A - 1 ** 2"), [w("A"), Tok::Minus, Tok::Number("1".into()), Tok::Power, Tok::Number("2".into())]);
322        assert_eq!(toks("           >= <= ("), [Tok::Ge, Tok::Le, Tok::LParen]);
323    }
324
325    #[test]
326    fn literals() {
327        assert_eq!(toks("           'IT''S' X'F1C1' N'AB'"), [Tok::Alnum("IT'S".into()), Tok::Hex(vec![0xF1, 0xC1]), Tok::National("AB".into())]);
328    }
329
330    #[test]
331    fn a_null_terminated_literal_ends_with_x00() {
332        assert_eq!(toks("           Z'ABC' z\"(I)V\""), [Tok::Alnum("ABC\0".into()), Tok::Alnum("(I)V\0".into())]);
333    }
334
335    #[test]
336    fn a_picture_is_one_token_and_keeps_its_own_periods() {
337        assert_eq!(toks("           PIC S9(3)V99 COMP-3."), [w("PIC"), Tok::Pic("S9(3)V99".into()), w("COMP-3"), Tok::Period]);
338        assert_eq!(toks("           PICTURE IS ZZ,ZZ9.99."), [w("PICTURE"), w("IS"), Tok::Pic("ZZ,ZZ9.99".into()), Tok::Period]);
339    }
340
341    #[test]
342    fn commas_between_operands_are_separators() {
343        assert_eq!(toks("           F(A, 1)"), [w("F"), Tok::LParen, w("A"), Tok::Number("1".into()), Tok::RParen]);
344        assert_eq!(toks("           T(I,J)"), [w("T"), Tok::LParen, w("I"), w("J"), Tok::RParen]);
345    }
346
347    #[test]
348    fn an_exec_block_is_one_token() {
349        assert_eq!(
350            toks("           EXEC SQL SELECT A.B INTO :X FROM T WHERE C = 'END-EXEC'\n               END-EXEC."),
351            [Tok::Exec("SQL SELECT A.B INTO :X FROM T WHERE C = 'END-EXEC'".into()), Tok::Period]
352        );
353    }
354
355    #[test]
356    fn a_comment_entry_holds_any_character() {
357        let program = concat!(
358            "       IDENTIFICATION DIVISION.\n",
359            "       PROGRAM-ID. CE3.\n",
360            "       AUTHOR. Smith & Jones @ ACME #1, Café O'Grady.\n",
361            "       INSTALLATION. \"HQ\n",
362            "           ~ ^ ` { } | \\ ?\n",
363            "       DATE-WRITTEN. 01/01/99.\n",
364            "       PROCEDURE DIVISION.\n",
365            "           GOBACK.\n",
366        );
367        let words = [w("IDENTIFICATION"), w("DIVISION"), Tok::Period, w("PROGRAM-ID"), Tok::Period, w("CE3"), Tok::Period];
368        let paragraphs = [w("AUTHOR"), Tok::Period, w("INSTALLATION"), Tok::Period, w("DATE-WRITTEN"), Tok::Period];
369        let procedure = [w("PROCEDURE"), w("DIVISION"), Tok::Period, w("GOBACK"), Tok::Period];
370        assert_eq!(toks(program), [&words[..], &paragraphs[..], &procedure[..]].concat());
371    }
372
373    #[test]
374    fn a_literal_continued_between_the_quotes_of_a_doubled_quote() {
375        let head = "           \"A+0B-1C*2D";
376        let text = format!("{head}{}\"\n      -    \"\"9K(L)M>N<O\".\n", "=".repeat(72 - head.len() - 1));
377        assert_eq!(toks(&text), [Tok::Alnum(format!("A+0B-1C*2D{}\"9K(L)M>N<O", "=".repeat(72 - head.len() - 1))), Tok::Period]);
378    }
379
380    #[test]
381    fn a_continuation_that_opens_a_quote_after_a_closed_literal_is_a_second_literal() {
382        assert_eq!(toks("           VALUE 'ABC'\n      -    'DEF'."), [w("VALUE"), Tok::Alnum("ABC".into()), Tok::Alnum("DEF".into()), Tok::Period]);
383    }
384
385    #[test]
386    fn an_ampersand_after_a_literal_is_named_as_concatenation() {
387        let error = |text: &str| lex(&source::read(text).unwrap()).unwrap_err().message;
388        assert!(error("           'A' & 'B'").contains("literal concatenation with & is not Enterprise COBOL's"));
389        assert!(error("           NOTIFY=&SYSUID").contains("unexpected character '&'"));
390    }
391
392    #[test]
393    fn area_a_is_marked() {
394        let t = lex(&source::read("       PARA.\n           MOVE").unwrap()).unwrap();
395        assert!(t[0].area_a);
396        assert!(!t[2].area_a);
397    }
398
399    fn numbers(text: &str) -> Vec<String> {
400        toks(text).into_iter().filter_map(|t| if let Tok::Number(n) = t { Some(n) } else { None }).collect()
401    }
402
403    #[test]
404    fn under_decimal_point_is_comma_a_comma_between_digits_is_the_point() {
405        let text = "           DECIMAL-POINT IS COMMA.\n           1,5 -,25 +3,0 ,75 T(1, 2) A,B 1.\n";
406        assert_eq!(numbers(text), ["1.5", "-.25", "+3.0", ".75", "1", "2", "1"]);
407        assert!(toks(text).contains(&w("B")));
408        assert!(toks("           DECIMAL-POINT IS COMMA.\n           PIC ,99.").contains(&Tok::Pic(",99".into())));
409        assert!(lex(&source::read("           DECIMAL-POINT COMMA.\n           MOVE 1.5").unwrap()).is_err());
410    }
411
412    #[test]
413    fn a_contained_program_keeps_the_decimal_comma_and_the_next_program_does_not() {
414        let text = concat!(
415            "       PROGRAM-ID. A.\n           DECIMAL-POINT IS COMMA.\n           1,5\n",
416            "       PROGRAM-ID. B.\n           2,5\n       END PROGRAM B.\n           3,5\n       END PROGRAM A.\n",
417            "       PROGRAM-ID. C.\n           4,5\n",
418        );
419        assert_eq!(numbers(text), ["1.5", "2.5", "3.5", "4", "5"]);
420    }
421}