#include "tree_sitter/alloc.h"
#include "tree_sitter/parser.h"
#include <stdio.h>
enum TokenType {
AUTOMATIC_SEMICOLON,
TEMPLATE_CHARS,
TERNARY_QMARK,
HTML_COMMENT,
LOGICAL_OR,
ESCAPE_SEQUENCE,
REGEX_PATTERN,
JSX_TEXT,
ARROW_FUNCTION_BLOCK_END,
ARROW_FUNCTION_BLOCK_CONTINUATION,
LINE_BREAK_ENDS_STATEMENT,
LINE_BREAK_AFTER_BINDING,
LINE_BREAK_AFTER_FIELD,
LINE_BREAK_AFTER_MODIFIER,
LINE_BREAK_BEFORE_ATTRIBUTES,
};
typedef struct {
bool automatic_semicolon_pending;
} Scanner;
void *tree_sitter_javascript_external_scanner_create() { return ts_calloc(1, sizeof(Scanner)); }
void tree_sitter_javascript_external_scanner_destroy(void *payload) { ts_free(payload); }
unsigned tree_sitter_javascript_external_scanner_serialize(void *payload, char *buffer) {
Scanner *scanner = (Scanner *)payload;
buffer[0] = (char)scanner->automatic_semicolon_pending;
return 1;
}
void tree_sitter_javascript_external_scanner_deserialize(void *payload, const char *buffer, unsigned length) {
Scanner *scanner = (Scanner *)payload;
scanner->automatic_semicolon_pending = length > 0 && buffer[0];
}
static inline void advance(TSLexer *lexer) { lexer->advance(lexer, false); }
static inline void skip(TSLexer *lexer) { lexer->advance(lexer, true); }
static inline bool is_line_terminator(int32_t c) { return c == '\n' || c == '\r' || c == 0x2028 || c == 0x2029; }
static inline bool is_whitespace(int32_t c) {
switch (c) {
case '\t':
case '\n':
case '\v':
case '\f':
case '\r':
case ' ':
case 0xA0:
case 0x1680:
case 0x200B:
case 0x2028:
case 0x2029:
case 0x202F:
case 0x205F:
case 0x2060:
case 0x3000:
case 0xFEFF:
return true;
default:
return c >= 0x2000 && c <= 0x200A;
}
}
static inline bool is_ascii_letter(int32_t c) { return (c >= 'a' && c <= 'z') || (c >= 'A' && c <= 'Z'); }
static inline bool is_ascii_digit(int32_t c) { return c >= '0' && c <= '9'; }
static inline bool is_identifier_part(int32_t c) {
return is_ascii_letter(c) || is_ascii_digit(c) || c == '_' || c == '$' || c == '\\' ||
(c >= 0x7F && !is_whitespace(c));
}
static bool scan_word(TSLexer *lexer, const char *word) {
for (; *word; word++) {
if (lexer->lookahead != *word) {
return false;
}
skip(lexer);
}
return !is_identifier_part(lexer->lookahead);
}
static bool scan_template_chars(TSLexer *lexer) {
lexer->result_symbol = TEMPLATE_CHARS;
for (bool has_content = false;; has_content = true) {
lexer->mark_end(lexer);
if (lexer->eof(lexer)) {
return false;
}
switch (lexer->lookahead) {
case '`':
return has_content;
case '$':
advance(lexer);
if (lexer->lookahead == '{') {
return has_content;
}
break;
case '\\':
return has_content;
default:
advance(lexer);
}
}
}
typedef enum {
REJECT, NO_NEWLINE, ACCEPT, ACCEPT_IN_BLOCK_COMMENT,
} WhitespaceResult;
static WhitespaceResult scan_whitespace_and_comments(TSLexer *lexer, bool *scanned_comment, bool consume) {
bool saw_block_newline = false;
for (;;) {
while (is_whitespace(lexer->lookahead)) {
skip(lexer);
}
if (lexer->lookahead == '/') {
skip(lexer);
if (lexer->lookahead == '/') {
skip(lexer);
while (!lexer->eof(lexer) && !is_line_terminator(lexer->lookahead)) {
skip(lexer);
}
*scanned_comment = true;
} else if (lexer->lookahead == '*') {
skip(lexer);
while (!lexer->eof(lexer)) {
if (lexer->lookahead == '*') {
skip(lexer);
if (lexer->lookahead == '/') {
skip(lexer);
*scanned_comment = true;
if (lexer->lookahead != '/' && !consume) {
return saw_block_newline ? ACCEPT_IN_BLOCK_COMMENT : NO_NEWLINE;
}
break;
}
} else if (is_line_terminator(lexer->lookahead)) {
saw_block_newline = true;
skip(lexer);
} else {
skip(lexer);
}
}
} else {
return REJECT;
}
} else {
return ACCEPT;
}
}
}
static bool ends_statement_after_block_arrow(TSLexer *lexer, bool *scanned_comment) {
if (scan_whitespace_and_comments(lexer, scanned_comment, true) == REJECT) {
return true;
}
return lexer->lookahead != ',' && lexer->lookahead != ';' && lexer->lookahead != '?';
}
typedef enum {
LINE_BREAK_BY_NEXT_TOKEN,
LINE_BREAK_ENDS,
LINE_BREAK_AFTER_BINDING_NAME,
LINE_BREAK_AFTER_FIELD_NAME,
LINE_BREAK_AFTER_MODIFIER_WORD,
LINE_BREAK_AFTER_ACCESSOR_WORD,
LINE_BREAK_BEFORE_IMPORT_ATTRIBUTES,
} LineBreakRule;
static bool scan_after_line_break(TSLexer *lexer, bool after_block_arrow, LineBreakRule rule, bool *scanned_comment);
static bool scan_automatic_semicolon(TSLexer *lexer, bool comment_condition, bool after_block_arrow,
LineBreakRule rule, bool *scanned_comment) {
lexer->result_symbol = AUTOMATIC_SEMICOLON;
lexer->mark_end(lexer);
bool line_break_in_block_comment = false;
for (;;) {
if (lexer->eof(lexer)) {
return true;
}
if (lexer->lookahead == '/') {
WhitespaceResult result = scan_whitespace_and_comments(lexer, scanned_comment, false);
if (result == REJECT) {
return false;
}
if (result == ACCEPT || result == ACCEPT_IN_BLOCK_COMMENT) {
if (after_block_arrow || rule != LINE_BREAK_BY_NEXT_TOKEN) {
return scan_after_line_break(lexer, after_block_arrow, rule, scanned_comment);
}
if (comment_condition && lexer->lookahead != ',' && lexer->lookahead != '=') {
return true;
}
line_break_in_block_comment = result == ACCEPT_IN_BLOCK_COMMENT;
}
}
if (lexer->lookahead == '}') {
return true;
}
if (lexer->is_at_included_range_start(lexer)) {
return true;
}
if (is_line_terminator(lexer->lookahead)) {
break;
}
if (!is_whitespace(lexer->lookahead)) {
return line_break_in_block_comment && scan_after_line_break(lexer, false, rule, scanned_comment);
}
skip(lexer);
}
skip(lexer);
return scan_after_line_break(lexer, after_block_arrow, rule, scanned_comment);
}
static bool scan_after_line_break(TSLexer *lexer, bool after_block_arrow, LineBreakRule rule,
bool *scanned_comment) {
if (after_block_arrow) {
return ends_statement_after_block_arrow(lexer, scanned_comment);
}
bool before_slash = scan_whitespace_and_comments(lexer, scanned_comment, true) == REJECT;
if (!before_slash && lexer->lookahead == ';') {
return false;
}
switch (rule) {
case LINE_BREAK_ENDS:
switch (lexer->lookahead) {
case '`':
case '[':
case '(':
case '+':
case '-':
case '<':
return true;
default:
if (before_slash) {
return true;
}
break;
}
break;
case LINE_BREAK_AFTER_BINDING_NAME:
return before_slash || (lexer->lookahead != '=' && lexer->lookahead != ',');
case LINE_BREAK_AFTER_FIELD_NAME:
return before_slash || (lexer->lookahead != '=' && lexer->lookahead != '(');
case LINE_BREAK_AFTER_MODIFIER_WORD:
return !before_slash && (lexer->lookahead == '}' || lexer->eof(lexer));
case LINE_BREAK_BEFORE_IMPORT_ATTRIBUTES:
return before_slash || !scan_word(lexer, "with");
case LINE_BREAK_AFTER_ACCESSOR_WORD:
return !before_slash && (lexer->lookahead == '}' || lexer->lookahead == '*' || lexer->eof(lexer));
default:
break;
}
if (before_slash) {
return false;
}
switch (lexer->lookahead) {
case '`':
case ',':
case ':':
case ';':
case '*':
case '%':
case '>':
case '<':
case '=':
case '[':
case '(':
case '?':
case '^':
case '|':
case '&':
case '/':
return false;
case '.':
skip(lexer);
return is_ascii_digit(lexer->lookahead);
case '+':
skip(lexer);
return lexer->lookahead == '+';
case '-':
skip(lexer);
return lexer->lookahead == '-';
case '!':
skip(lexer);
return lexer->lookahead != '=';
case 'i':
skip(lexer);
if (lexer->lookahead != 'n') {
return true;
}
skip(lexer);
if (!is_identifier_part(lexer->lookahead)) {
return false;
}
for (unsigned i = 0; i < 8; i++) {
if (lexer->lookahead != "stanceof"[i]) {
return true;
}
skip(lexer);
}
if (!is_identifier_part(lexer->lookahead)) {
return false;
}
break;
default:
break;
}
return true;
}
static bool scan_ternary_qmark(TSLexer *lexer) {
for (;;) {
if (!is_whitespace(lexer->lookahead)) {
break;
}
skip(lexer);
}
if (lexer->lookahead == '?') {
advance(lexer);
if (lexer->lookahead == '?') {
return false;
}
lexer->mark_end(lexer);
lexer->result_symbol = TERNARY_QMARK;
if (lexer->lookahead == '.') {
advance(lexer);
if (is_ascii_digit(lexer->lookahead)) {
return true;
}
return false;
}
return true;
}
return false;
}
static bool scan_html_comment(TSLexer *lexer) {
while (is_whitespace(lexer->lookahead)) {
skip(lexer);
}
const char *comment_start = "<!--";
const char *comment_end = "-->";
if (lexer->lookahead == '<') {
for (unsigned i = 0; i < 4; i++) {
if (lexer->lookahead != comment_start[i]) {
return false;
}
advance(lexer);
}
} else if (lexer->lookahead == '-') {
for (unsigned i = 0; i < 3; i++) {
if (lexer->lookahead != comment_end[i]) {
return false;
}
advance(lexer);
}
} else {
return false;
}
while (!lexer->eof(lexer) && !is_line_terminator(lexer->lookahead)) {
advance(lexer);
}
lexer->result_symbol = HTML_COMMENT;
lexer->mark_end(lexer);
return true;
}
static inline bool is_hex_digit(int32_t c) { return is_ascii_digit(c) || (c >= 'a' && c <= 'f') || (c >= 'A' && c <= 'F'); }
static bool scan_character_reference(TSLexer *lexer) {
advance(lexer);
unsigned length = 0;
if (lexer->lookahead == '#') {
advance(lexer);
bool hex = lexer->lookahead == 'x' || lexer->lookahead == 'X';
if (hex) {
advance(lexer);
}
unsigned max_length = hex ? 6 : 5;
while (length < max_length && (hex ? is_hex_digit(lexer->lookahead) : is_ascii_digit(lexer->lookahead))) {
advance(lexer);
length++;
}
} else {
while (length < 30 && is_ascii_letter(lexer->lookahead)) {
advance(lexer);
length++;
}
}
return length > 0 && lexer->lookahead == ';';
}
static bool scan_jsx_text(TSLexer *lexer) {
bool saw_text = false;
bool at_newline = false;
lexer->result_symbol = JSX_TEXT;
for (;;) {
lexer->mark_end(lexer);
if (lexer->eof(lexer)) {
return saw_text;
}
switch (lexer->lookahead) {
case '<':
case '>':
case '{':
case '}':
return saw_text;
case '&':
if (scan_character_reference(lexer)) {
return saw_text;
}
saw_text = true;
at_newline = false;
continue;
default:
break;
}
bool is_wspace = (lexer->lookahead >= '\t' && lexer->lookahead <= '\r') || lexer->lookahead == ' ';
if (lexer->lookahead == '\n' || lexer->lookahead == '\r') {
at_newline = true;
} else {
at_newline &= is_wspace;
if (!at_newline) {
saw_text = true;
}
}
advance(lexer);
}
}
bool tree_sitter_javascript_external_scanner_scan(void *payload, TSLexer *lexer, const bool *valid_symbols) {
Scanner *scanner = (Scanner *)payload;
if (valid_symbols[TEMPLATE_CHARS]) {
if (valid_symbols[AUTOMATIC_SEMICOLON]) {
return false;
}
return scan_template_chars(lexer);
}
if (scanner->automatic_semicolon_pending &&
(valid_symbols[AUTOMATIC_SEMICOLON] || valid_symbols[ARROW_FUNCTION_BLOCK_CONTINUATION])) {
scanner->automatic_semicolon_pending = false;
lexer->result_symbol =
valid_symbols[AUTOMATIC_SEMICOLON] ? AUTOMATIC_SEMICOLON : ARROW_FUNCTION_BLOCK_CONTINUATION;
lexer->mark_end(lexer);
return true;
}
if (valid_symbols[JSX_TEXT] && scan_jsx_text(lexer)) {
return true;
}
if (valid_symbols[AUTOMATIC_SEMICOLON] || valid_symbols[ARROW_FUNCTION_BLOCK_END]) {
bool after_block_arrow = valid_symbols[ARROW_FUNCTION_BLOCK_END];
bool scanned_comment = false;
LineBreakRule rule = LINE_BREAK_BY_NEXT_TOKEN;
if (valid_symbols[LINE_BREAK_ENDS_STATEMENT]) {
rule = LINE_BREAK_ENDS;
} else if (valid_symbols[LINE_BREAK_AFTER_BINDING]) {
rule = LINE_BREAK_AFTER_BINDING_NAME;
} else if (valid_symbols[LINE_BREAK_AFTER_MODIFIER]) {
rule = valid_symbols[LINE_BREAK_AFTER_FIELD] ? LINE_BREAK_AFTER_ACCESSOR_WORD : LINE_BREAK_AFTER_MODIFIER_WORD;
} else if (valid_symbols[LINE_BREAK_AFTER_FIELD]) {
rule = LINE_BREAK_AFTER_FIELD_NAME;
} else if (valid_symbols[LINE_BREAK_BEFORE_ATTRIBUTES]) {
rule = LINE_BREAK_BEFORE_IMPORT_ATTRIBUTES;
}
bool ret = scan_automatic_semicolon(lexer, !valid_symbols[LOGICAL_OR], after_block_arrow, rule, &scanned_comment);
if (ret && after_block_arrow) {
lexer->result_symbol = ARROW_FUNCTION_BLOCK_END;
scanner->automatic_semicolon_pending = true;
}
if (!ret && !scanned_comment && valid_symbols[TERNARY_QMARK] && lexer->lookahead == '?') {
return scan_ternary_qmark(lexer);
}
return ret;
}
if (valid_symbols[TERNARY_QMARK]) {
return scan_ternary_qmark(lexer);
}
if (valid_symbols[HTML_COMMENT] && !valid_symbols[LOGICAL_OR] && !valid_symbols[ESCAPE_SEQUENCE] &&
!valid_symbols[REGEX_PATTERN]) {
return scan_html_comment(lexer);
}
return false;
}