1use crate::source::Source;
2use crate::{Error, Pos};
3
4#[derive(Clone, Debug, PartialEq, Eq)]
5pub enum Tok {
6 Word(String),
8 Number(String),
10 Alnum(String),
11 Hex(Vec<u8>),
12 National(String),
13 Pic(String),
14 Exec(String),
16 Period,
17 LParen,
18 RParen,
19 Colon,
20 Plus,
21 Minus,
22 Star,
23 Slash,
24 Power,
25 Eq,
26 Lt,
27 Gt,
28 Le,
29 Ge,
30}
31
32#[derive(Clone, Debug, PartialEq, Eq)]
33pub struct Token {
34 pub tok: Tok,
35 pub pos: Pos,
36 pub area_a: bool,
38 pub spelled: Option<String>,
40 pub after_comma: bool,
42 pub messages: Vec<Error>,
44}
45
46struct Lexer<'a> {
47 chars: Vec<char>,
48 positions: &'a [Pos],
49 at: usize,
50 tokens: Vec<Token>,
51 decimal_comma: bool,
53 currency: Vec<char>,
55 outer: Vec<bool>,
58 comma_pending: bool,
59 pending: Vec<Error>,
61}
62
63pub fn lex(source: &Source) -> Result<Vec<Token>, Error> {
64 let mut lx = Lexer { chars: source.text.chars().collect(), positions: &source.positions, at: 0, tokens: Vec::new(), decimal_comma: false, currency: Vec::new(), outer: Vec::new(), comma_pending: false, pending: Vec::new() };
65 while lx.at < lx.chars.len() {
66 lx.next_token()?;
67 }
68 Ok(lx.tokens)
69}
70
71fn is_word_char(c: char) -> bool {
72 c.is_ascii_alphanumeric() || c == '-' || c == '_'
73}
74
75fn non_cobol(c: char) -> bool {
79 u32::from(c) <= 0xFF && !c.is_ascii_alphanumeric() && !" \n+-*/=$,;.\"'()><:_&".contains(c)
80}
81
82impl Lexer<'_> {
83 fn peek(&self, ahead: usize) -> Option<char> {
84 self.chars.get(self.at + ahead).copied()
85 }
86
87 fn pos(&self) -> Pos {
88 self.positions.get(self.at).copied().unwrap_or_default()
89 }
90
91 fn separator_follows(&self, ahead: usize) -> bool {
92 self.peek(ahead).is_none_or(|c| c == ' ' || c == '\n')
93 }
94
95 fn emit(&mut self, tok: Tok, pos: Pos) {
96 let before = |back: usize| match self.tokens.len().checked_sub(back).map(|i| &self.tokens[i].tok) {
97 Some(Tok::Word(p)) => p.as_str(),
98 _ => "",
99 };
100 match &tok {
101 Tok::Word(w) => match w.as_str() {
102 "PROGRAM-ID" => self.outer.push(self.decimal_comma),
103 "PROGRAM" if before(1) == "END" => self.decimal_comma = self.outer.pop().unwrap_or(false),
104 "COMMA" if before(1) == "DECIMAL-POINT" || before(1) == "IS" && before(2) == "DECIMAL-POINT" => self.decimal_comma = true,
105 _ => {}
106 },
107 Tok::Alnum(a) => {
108 let mut chars = a.chars();
109 let names_symbol = before(1) == "SYMBOL" || (1..=3).any(|back| before(back) == "CURRENCY" && (1..back).all(|b| matches!(before(b), "SIGN" | "IS")));
110 if let (Some(c), None, true) = (chars.next(), chars.next(), names_symbol) {
111 self.currency.push(c);
112 }
113 }
114 _ => {}
115 }
116 let spelled = match &tok {
117 Tok::Word(w) => self
118 .at
119 .checked_sub(w.chars().count())
120 .map(|start| &self.chars[start..self.at])
121 .filter(|raw| raw.iter().any(char::is_ascii_lowercase))
122 .map(|raw| raw.iter().collect::<String>())
123 .filter(|raw| raw.eq_ignore_ascii_case(w)),
124 _ => None,
125 };
126 let after_comma = std::mem::take(&mut self.comma_pending);
127 self.tokens.push(Token { tok, pos, area_a: (8..=11).contains(&pos.col), spelled, after_comma, messages: std::mem::take(&mut self.pending) });
128 }
129
130 fn point(&self) -> char {
132 if self.decimal_comma { ',' } else { '.' }
133 }
134
135 fn expecting_picture(&self) -> bool {
136 let words: Vec<&str> =
137 self.tokens.iter().rev().take(2).map(|t| if let Tok::Word(w) = &t.tok { w.as_str() } else { "" }).collect();
138 matches!(words.as_slice(), ["PIC" | "PICTURE", ..] | ["IS", "PIC" | "PICTURE"])
139 }
140
141 fn next_token(&mut self) -> Result<(), Error> {
142 let c = self.chars[self.at];
143 let pos = self.pos();
144 let point = self.point();
145 let leading_point = c == ',' && point == ',' && self.peek(1).is_some_and(|d| d.is_ascii_digit()) && !self.at.checked_sub(1).is_some_and(|i| is_word_char(self.chars[i]));
146 if leading_point && self.expecting_picture() {
147 return self.picture(pos);
148 }
149 if leading_point {
150 let tok = self.number_or_word(pos)?;
151 self.emit(tok, pos);
152 return Ok(());
153 }
154 if c == ' ' || c == '\n' || c == ',' || c == ';' {
155 self.comma_pending |= c == ',' || c == ';';
156 self.at += 1;
157 return Ok(());
158 }
159 if self.expecting_picture() {
160 return self.picture(pos);
161 }
162 let next = self.peek(1);
163 let quote_next = matches!(next, Some('\'' | '"'));
164 match c {
165 '\'' | '"' => {
166 let text = self.quoted(pos)?;
167 self.emit(Tok::Alnum(text), pos);
168 }
169 'X' | 'x' if quote_next => {
170 self.at += 1;
171 let text = self.quoted(pos)?;
172 let bytes = unhex(&text).ok_or_else(|| Error::at(pos, format!("X'{text}' is not an even number of hex digits")))?;
173 self.emit(Tok::Hex(bytes), pos);
174 }
175 'N' | 'n' if quote_next => {
176 self.at += 1;
177 let text = self.quoted(pos)?;
178 self.emit(Tok::National(text), pos);
179 }
180 'Z' | 'z' if quote_next => {
181 self.at += 1;
182 let text = self.quoted(pos)?;
183 self.emit(Tok::Alnum(format!("{text}\0")), pos);
184 }
185 '.' if self.separator_follows(1) => {
186 self.at += 1;
187 self.emit(Tok::Period, pos);
188 }
189 '+' | '-' if next.is_some_and(|n| n.is_ascii_digit() || (n == point && self.peek(2).is_some_and(|d| d.is_ascii_digit()))) => {
190 self.at += 1;
191 let digits = self.number_or_word(pos)?;
192 match digits {
193 Tok::Number(n) => self.emit(Tok::Number(format!("{c}{n}")), pos),
194 _ => return Err(Error::at(pos, "a sign must be followed by a number")),
195 }
196 }
197 _ if c.is_ascii_alphanumeric() || c == '.' || non_cobol(c) => {
198 let tok = self.number_or_word(pos)?;
199 match tok {
200 Tok::Word(w) if w == "EXEC" || w == "EXECUTE" => {
201 let tok = self.exec_block(pos)?;
202 self.emit(tok, pos);
203 }
204 tok => self.emit(tok, pos),
205 }
206 }
207 _ => {
208 let (tok, len) = match (c, next) {
209 ('*', Some('*')) => (Tok::Power, 2),
210 ('<', Some('=')) => (Tok::Le, 2),
211 ('>', Some('=')) => (Tok::Ge, 2),
212 ('*', _) => (Tok::Star, 1),
213 ('/', _) => (Tok::Slash, 1),
214 ('+', _) => (Tok::Plus, 1),
215 ('-', _) => (Tok::Minus, 1),
216 ('=', _) => (Tok::Eq, 1),
217 ('<', _) => (Tok::Lt, 1),
218 ('>', _) => (Tok::Gt, 1),
219 ('(', _) => (Tok::LParen, 1),
220 (')', _) => (Tok::RParen, 1),
221 (':', _) => (Tok::Colon, 1),
222 ('&', _) if matches!(self.tokens.last().map(|t| &t.tok), Some(Tok::Alnum(_) | Tok::Hex(_) | Tok::National(_))) => {
223 return Err(Error::at(pos, "literal concatenation with & is not Enterprise COBOL's"));
224 }
225 _ => return Err(Error::at(pos, format!("unexpected character {c:?}"))),
226 };
227 self.at += len;
228 self.emit(tok, pos);
229 }
230 }
231 Ok(())
232 }
233
234 fn exec_block(&mut self, pos: Pos) -> Result<Tok, Error> {
236 let start = self.at;
237 let mut quote: Option<char> = None;
238 while let Some(c) = self.peek(0) {
239 match quote {
240 Some(q) if c == q => quote = None,
241 Some(_) => {}
242 None if c == '\'' || c == '"' => quote = Some(c),
243 None if c.eq_ignore_ascii_case(&'E') => {
244 let ahead: String = self.chars[self.at..].iter().take(8).collect();
245 let boundary = self.chars.get(self.at + 8).is_none_or(|d| !is_word_char(*d)) && (self.at == 0 || !is_word_char(self.chars[self.at - 1]));
246 if ahead.eq_ignore_ascii_case("END-EXEC") && boundary {
247 let text: String = self.chars[start..self.at].iter().collect();
248 self.at += 8;
249 return Ok(Tok::Exec(text.split_whitespace().collect::<Vec<_>>().join(" ")));
250 }
251 }
252 None => {}
253 }
254 self.at += 1;
255 }
256 Err(Error::at(pos, "EXEC with no END-EXEC"))
257 }
258
259 fn quoted(&mut self, pos: Pos) -> Result<String, Error> {
261 let quote = self.chars[self.at];
262 self.at += 1;
263 let mut text = String::new();
264 loop {
265 match self.peek(0) {
266 None | Some('\n') => return Err(Error::at(pos, "an unterminated literal")),
267 Some(c) if c == quote && self.peek(1) == Some(quote) => {
268 text.push(quote);
269 self.at += 2;
270 }
271 Some(c) if c == quote => {
272 self.at += 1;
273 return Ok(text);
274 }
275 Some(c) => {
276 text.push(c);
277 self.at += 1;
278 }
279 }
280 }
281 }
282
283 fn number_or_word(&mut self, pos: Pos) -> Result<Tok, Error> {
284 let start = self.at;
285 while let Some(c) = self.peek(0).filter(|&c| is_word_char(c) || non_cobol(c)) {
286 if non_cobol(c) {
287 self.pending.push(Error::at(self.pos(), format!("non-COBOL character {c:?}: the character was accepted")).graded(crate::Severity::Error));
288 }
289 self.at += 1;
290 }
291 let run: String = self.chars[start..self.at].iter().collect();
292 let all_digits = run.chars().all(|c| c.is_ascii_digit());
293 if all_digits && self.peek(0) == Some(self.point()) && self.peek(1).is_some_and(|c| c.is_ascii_digit()) {
294 self.at += 1;
295 let frac_start = self.at;
296 while self.peek(0).is_some_and(|c| c.is_ascii_digit()) {
297 self.at += 1;
298 }
299 let frac: String = self.chars[frac_start..self.at].iter().collect();
300 return Ok(Tok::Number(format!("{run}.{frac}")));
301 }
302 if run.is_empty() {
303 return Err(Error::at(pos, "unexpected '.'"));
304 }
305 Ok(if all_digits { Tok::Number(run) } else { Tok::Word(run.to_ascii_uppercase()) })
306 }
307
308 fn picture(&mut self, pos: Pos) -> Result<(), Error> {
310 let start = self.at;
311 while self.peek(0).is_some_and(|c| c != ' ' && c != '\n') {
312 self.at += 1;
313 }
314 let mut end = self.at;
315 if end > start && matches!(self.chars[end - 1], '.' | ',' | ';') {
316 end -= 1;
317 }
318 let text: String = self.chars[start..end].iter().collect();
319 if text.is_empty() {
320 return Err(Error::at(pos, "PICTURE with no character-string"));
321 }
322 self.at = end;
323 let text: String = text.chars().map(|c| if self.currency.contains(&c) { c } else { c.to_ascii_uppercase() }).collect();
324 let tok = if matches!(text.as_str(), "IS" | "SYMBOL") && self.tokens.last().is_some_and(|t| matches!(&t.tok, Tok::Word(w) if w == "PIC" || w == "PICTURE")) {
325 Tok::Word(text)
326 } else {
327 Tok::Pic(text)
328 };
329 self.emit(tok, pos);
330 Ok(())
331 }
332}
333
334fn unhex(text: &str) -> Option<Vec<u8>> {
335 if !text.len().is_multiple_of(2) || !text.is_ascii() {
336 return None;
337 }
338 (0..text.len()).step_by(2).map(|i| u8::from_str_radix(&text[i..i + 2], 16).ok()).collect()
339}
340
341#[cfg(test)]
342mod tests {
343 use super::*;
344 use crate::source;
345
346 fn toks(text: &str) -> Vec<Tok> {
347 lex(&source::read(text).unwrap()).unwrap().into_iter().map(|t| t.tok).collect()
348 }
349
350 fn w(s: &str) -> Tok {
351 Tok::Word(s.into())
352 }
353
354 #[test]
355 fn numbers_words_and_the_separator_period() {
356 assert_eq!(toks(" 05 A-1 VALUE 0.1."), [Tok::Number("05".into()), w("A-1"), w("VALUE"), Tok::Number("0.1".into()), Tok::Period]);
357 assert_eq!(toks(" VALUE -12345."), [w("VALUE"), Tok::Number("-12345".into()), Tok::Period]);
358 assert_eq!(toks(" 100-MAIN."), [w("100-MAIN"), Tok::Period]);
359 }
360
361 #[test]
362 fn operators_need_spaces_and_signed_literals_do_not() {
363 assert_eq!(toks(" A - 1 ** 2"), [w("A"), Tok::Minus, Tok::Number("1".into()), Tok::Power, Tok::Number("2".into())]);
364 assert_eq!(toks(" >= <= ("), [Tok::Ge, Tok::Le, Tok::LParen]);
365 }
366
367 #[test]
368 fn literals() {
369 assert_eq!(toks(" 'IT''S' X'F1C1' N'AB'"), [Tok::Alnum("IT'S".into()), Tok::Hex(vec![0xF1, 0xC1]), Tok::National("AB".into())]);
370 }
371
372 #[test]
373 fn a_null_terminated_literal_ends_with_x00() {
374 assert_eq!(toks(" Z'ABC' z\"(I)V\""), [Tok::Alnum("ABC\0".into()), Tok::Alnum("(I)V\0".into())]);
375 }
376
377 #[test]
378 fn a_picture_is_one_token_and_keeps_its_own_periods() {
379 assert_eq!(toks(" PIC S9(3)V99 COMP-3."), [w("PIC"), Tok::Pic("S9(3)V99".into()), w("COMP-3"), Tok::Period]);
380 assert_eq!(toks(" PICTURE IS ZZ,ZZ9.99."), [w("PICTURE"), w("IS"), Tok::Pic("ZZ,ZZ9.99".into()), Tok::Period]);
381 }
382
383 #[test]
384 fn only_the_last_period_or_comma_before_the_space_is_a_separator() {
385 assert_eq!(toks(" PIC 9,9,9,."), [w("PIC"), Tok::Pic("9,9,9,".into()), Tok::Period]);
386 assert_eq!(toks(" PIC 999999999999.."), [w("PIC"), Tok::Pic("999999999999.".into()), Tok::Period]);
387 assert_eq!(toks(" PIC 99, VALUE 1."), [w("PIC"), Tok::Pic("99".into()), w("VALUE"), Tok::Number("1".into()), Tok::Period]);
388 assert_eq!(toks(" PIC 9.9,; VALUE 1."), [w("PIC"), Tok::Pic("9.9,".into()), w("VALUE"), Tok::Number("1".into()), Tok::Period]);
389 assert_eq!(toks(" PIC 999., VALUE 1."), [w("PIC"), Tok::Pic("999.".into()), w("VALUE"), Tok::Number("1".into()), Tok::Period]);
390 let comma = " DECIMAL-POINT IS COMMA.\n PIC 9.9.9,. PIC 999,,\n";
391 let pics: Vec<Tok> = toks(comma).into_iter().filter(|t| matches!(t, Tok::Pic(_))).collect();
392 assert_eq!(pics, [Tok::Pic("9.9.9,".into()), Tok::Pic("999,".into())]);
393 }
394
395 #[test]
396 fn commas_between_operands_are_separators() {
397 assert_eq!(toks(" F(A, 1)"), [w("F"), Tok::LParen, w("A"), Tok::Number("1".into()), Tok::RParen]);
398 assert_eq!(toks(" T(I,J)"), [w("T"), Tok::LParen, w("I"), w("J"), Tok::RParen]);
399 }
400
401 #[test]
402 fn an_exec_block_is_one_token() {
403 assert_eq!(
404 toks(" EXEC SQL SELECT A.B INTO :X FROM T WHERE C = 'END-EXEC'\n END-EXEC."),
405 [Tok::Exec("SQL SELECT A.B INTO :X FROM T WHERE C = 'END-EXEC'".into()), Tok::Period]
406 );
407 }
408
409 #[test]
410 fn a_comment_entry_holds_any_character() {
411 let program = concat!(
412 " IDENTIFICATION DIVISION.\n",
413 " PROGRAM-ID. CE3.\n",
414 " AUTHOR. Smith & Jones @ ACME #1, Café O'Grady.\n",
415 " INSTALLATION. \"HQ\n",
416 " ~ ^ ` { } | \\ ?\n",
417 " DATE-WRITTEN. 01/01/99.\n",
418 " PROCEDURE DIVISION.\n",
419 " GOBACK.\n",
420 );
421 let words = [w("IDENTIFICATION"), w("DIVISION"), Tok::Period, w("PROGRAM-ID"), Tok::Period, w("CE3"), Tok::Period];
422 let paragraphs = [w("AUTHOR"), Tok::Period, w("INSTALLATION"), Tok::Period, w("DATE-WRITTEN"), Tok::Period];
423 let procedure = [w("PROCEDURE"), w("DIVISION"), Tok::Period, w("GOBACK"), Tok::Period];
424 assert_eq!(toks(program), [&words[..], ¶graphs[..], &procedure[..]].concat());
425 }
426
427 #[test]
428 fn a_literal_continued_between_the_quotes_of_a_doubled_quote() {
429 let head = " \"A+0B-1C*2D";
430 let text = format!("{head}{}\"\n - \"\"9K(L)M>N<O\".\n", "=".repeat(72 - head.len() - 1));
431 assert_eq!(toks(&text), [Tok::Alnum(format!("A+0B-1C*2D{}\"9K(L)M>N<O", "=".repeat(72 - head.len() - 1))), Tok::Period]);
432 }
433
434 #[test]
435 fn a_continuation_that_opens_a_quote_after_a_closed_literal_is_a_second_literal() {
436 assert_eq!(toks(" VALUE 'ABC'\n - 'DEF'."), [w("VALUE"), Tok::Alnum("ABC".into()), Tok::Alnum("DEF".into()), Tok::Period]);
437 }
438
439 #[test]
440 fn an_ampersand_after_a_literal_is_named_as_concatenation() {
441 let error = |text: &str| lex(&source::read(text).unwrap()).unwrap_err().message;
442 assert!(error(" 'A' & 'B'").contains("literal concatenation with & is not Enterprise COBOL's"));
443 assert!(error(" NOTIFY=&SYSUID").contains("unexpected character '&'"));
444 }
445
446 #[test]
447 fn a_non_cobol_character_is_accepted_into_its_word_with_an_error() {
448 let lexed = lex(&source::read(" MOVE WS#1 TO %\u{1b} 'A@B'.").unwrap()).unwrap();
449 assert_eq!(lexed.iter().map(|t| t.tok.clone()).collect::<Vec<_>>(), [w("MOVE"), w("WS#1"), w("TO"), w("%\u{1b}"), Tok::Alnum("A@B".into()), Tok::Period]);
450 let messages: Vec<(usize, u32, &str, crate::Severity)> =
451 lexed.iter().enumerate().flat_map(|(i, t)| t.messages.iter().map(move |m| (i, m.pos.col, m.message.as_str(), m.severity))).collect();
452 assert_eq!(
453 messages,
454 [
455 (1, 19, "non-COBOL character '#': the character was accepted", crate::Severity::Error),
456 (3, 25, "non-COBOL character '%': the character was accepted", crate::Severity::Error),
457 (3, 26, "non-COBOL character '\\u{1b}': the character was accepted", crate::Severity::Error),
458 ]
459 );
460 let error = |text: &str| lex(&source::read(text).unwrap()).unwrap_err().message;
461 assert_eq!(error(" MOVE $X"), "unexpected character '$'");
462 assert_eq!(error(" MOVE \u{3042}"), "unexpected character '\u{3042}'");
463 }
464
465 #[test]
466 fn area_a_is_marked() {
467 let t = lex(&source::read(" PARA.\n MOVE").unwrap()).unwrap();
468 assert!(t[0].area_a);
469 assert!(!t[2].area_a);
470 }
471
472 fn numbers(text: &str) -> Vec<String> {
473 toks(text).into_iter().filter_map(|t| if let Tok::Number(n) = t { Some(n) } else { None }).collect()
474 }
475
476 #[test]
477 fn under_decimal_point_is_comma_a_comma_between_digits_is_the_point() {
478 let text = " DECIMAL-POINT IS COMMA.\n 1,5 -,25 +3,0 ,75 T(1, 2) A,B 1.\n";
479 assert_eq!(numbers(text), ["1.5", "-.25", "+3.0", ".75", "1", "2", "1"]);
480 assert!(toks(text).contains(&w("B")));
481 assert!(toks(" DECIMAL-POINT IS COMMA.\n PIC ,99.").contains(&Tok::Pic(",99".into())));
482 assert!(lex(&source::read(" DECIMAL-POINT COMMA.\n MOVE 1.5").unwrap()).is_err());
483 }
484
485 #[test]
486 fn a_contained_program_keeps_the_decimal_comma_and_the_next_program_does_not() {
487 let text = concat!(
488 " PROGRAM-ID. A.\n DECIMAL-POINT IS COMMA.\n 1,5\n",
489 " PROGRAM-ID. B.\n 2,5\n END PROGRAM B.\n 3,5\n END PROGRAM A.\n",
490 " PROGRAM-ID. C.\n 4,5\n",
491 );
492 assert_eq!(numbers(text), ["1.5", "2.5", "3.5", "4", "5"]);
493 }
494}