krz/orgstar

A native macOS editor for org-mode files. editor org-mode swift

Sources/TreeSitterScanners/javascript/scanner.c

3a3dedba062790e1580c66d38606eda4b02cb501
orgstar/Sources/TreeSitterScanners/javascript/scanner.c history · blame · raw

364 lines · 10576 bytes

  1#include "tree_sitter/parser.h"
  2
  3#include <stdio.h>
  4#include <wctype.h>
  5
  6enum TokenType {
  7    AUTOMATIC_SEMICOLON,
  8    TEMPLATE_CHARS,
  9    TERNARY_QMARK,
 10    HTML_COMMENT,
 11    LOGICAL_OR,
 12    ESCAPE_SEQUENCE,
 13    REGEX_PATTERN,
 14    JSX_TEXT,
 15};
 16
 17void *tree_sitter_javascript_external_scanner_create() { return NULL; }
 18
 19void tree_sitter_javascript_external_scanner_destroy(void *p) {}
 20
 21unsigned tree_sitter_javascript_external_scanner_serialize(void *payload, char *buffer) { return 0; }
 22
 23void tree_sitter_javascript_external_scanner_deserialize(void *p, const char *b, unsigned n) {}
 24
 25static inline void advance(TSLexer *lexer) { lexer->advance(lexer, false); }
 26
 27static inline void skip(TSLexer *lexer) { lexer->advance(lexer, true); }
 28
 29static bool scan_template_chars(TSLexer *lexer) {
 30    lexer->result_symbol = TEMPLATE_CHARS;
 31    for (bool has_content = false;; has_content = true) {
 32        lexer->mark_end(lexer);
 33        switch (lexer->lookahead) {
 34            case '`':
 35                return has_content;
 36            case '\0':
 37                return false;
 38            case '$':
 39                advance(lexer);
 40                if (lexer->lookahead == '{') {
 41                    return has_content;
 42                }
 43                break;
 44            case '\\':
 45                return has_content;
 46            default:
 47                advance(lexer);
 48        }
 49    }
 50}
 51
 52typedef enum {
 53    REJECT,     // Semicolon is illegal, ie a syntax error occurred
 54    NO_NEWLINE, // Unclear if semicolon will be legal, continue
 55    ACCEPT,     // Semicolon is legal, assuming a comment was encountered
 56} WhitespaceResult;
 57
 58/**
 59 * @param consume If false, only consume enough to check if comment indicates semicolon-legality
 60 */
 61static WhitespaceResult scan_whitespace_and_comments(TSLexer *lexer, bool *scanned_comment, bool consume) {
 62    bool saw_block_newline = false;
 63
 64    for (;;) {
 65        while (iswspace(lexer->lookahead)) {
 66            skip(lexer);
 67        }
 68
 69        if (lexer->lookahead == '/') {
 70            skip(lexer);
 71
 72            if (lexer->lookahead == '/') {
 73                skip(lexer);
 74                while (lexer->lookahead != 0 && lexer->lookahead != '\n' && lexer->lookahead != 0x2028 &&
 75                       lexer->lookahead != 0x2029) {
 76                    skip(lexer);
 77                }
 78                *scanned_comment = true;
 79            } else if (lexer->lookahead == '*') {
 80                skip(lexer);
 81                while (lexer->lookahead != 0) {
 82                    if (lexer->lookahead == '*') {
 83                        skip(lexer);
 84                        if (lexer->lookahead == '/') {
 85                            skip(lexer);
 86                            *scanned_comment = true;
 87
 88                            if (lexer->lookahead != '/' && !consume) {
 89                                return saw_block_newline ? ACCEPT : NO_NEWLINE;
 90                            }
 91
 92                            break;
 93                        }
 94                    } else if (lexer->lookahead == '\n' || lexer->lookahead == 0x2028 || lexer->lookahead == 0x2029) {
 95                        saw_block_newline = true;
 96                        skip(lexer);
 97                    } else {
 98                        skip(lexer);
 99                    }
100                }
101            } else {
102                return REJECT;
103            }
104        } else {
105            return ACCEPT;
106        }
107    }
108}
109
110static bool scan_automatic_semicolon(TSLexer *lexer, bool comment_condition, bool *scanned_comment) {
111    lexer->result_symbol = AUTOMATIC_SEMICOLON;
112    lexer->mark_end(lexer);
113
114    for (;;) {
115        if (lexer->lookahead == 0) {
116            return true;
117        }
118
119        if (lexer->lookahead == '/') {
120            WhitespaceResult result = scan_whitespace_and_comments(lexer, scanned_comment, false);
121            if (result == REJECT) {
122                return false;
123            }
124
125            if (result == ACCEPT && comment_condition && lexer->lookahead != ',' && lexer->lookahead != '=') {
126                return true;
127            }
128        }
129
130        if (lexer->lookahead == '}') {
131            return true;
132        }
133
134        if (lexer->is_at_included_range_start(lexer)) {
135            return true;
136        }
137
138        if (lexer->lookahead == '\n' || lexer->lookahead == 0x2028 || lexer->lookahead == 0x2029) {
139            break;
140        }
141
142        if (!iswspace(lexer->lookahead)) {
143            return false;
144        }
145
146        skip(lexer);
147    }
148
149    skip(lexer);
150
151    if (scan_whitespace_and_comments(lexer, scanned_comment, true) == REJECT) {
152        return false;
153    }
154
155    switch (lexer->lookahead) {
156        case '`':
157        case ',':
158        case ':':
159        case ';':
160        case '*':
161        case '%':
162        case '>':
163        case '<':
164        case '=':
165        case '[':
166        case '(':
167        case '?':
168        case '^':
169        case '|':
170        case '&':
171        case '/':
172            return false;
173
174        // Insert a semicolon before decimals literals but not otherwise.
175        case '.':
176            skip(lexer);
177            return iswdigit(lexer->lookahead);
178
179        // Insert a semicolon before `--` and `++`, but not before binary `+` or `-`.
180        case '+':
181            skip(lexer);
182            return lexer->lookahead == '+';
183        case '-':
184            skip(lexer);
185            return lexer->lookahead == '-';
186
187        // Don't insert a semicolon before `!=`, but do insert one before a unary `!`.
188        case '!':
189            skip(lexer);
190            return lexer->lookahead != '=';
191
192        // Don't insert a semicolon before `in` or `instanceof`, but do insert one
193        // before an identifier.
194        case 'i':
195            skip(lexer);
196
197            if (lexer->lookahead != 'n') {
198                return true;
199            }
200            skip(lexer);
201
202            if (!iswalpha(lexer->lookahead)) {
203                return false;
204            }
205
206            for (unsigned i = 0; i < 8; i++) {
207                if (lexer->lookahead != "stanceof"[i]) {
208                    return true;
209                }
210                skip(lexer);
211            }
212
213            if (!iswalpha(lexer->lookahead)) {
214                return false;
215            }
216            break;
217
218        default:
219            break;
220    }
221
222    return true;
223}
224
225static bool scan_ternary_qmark(TSLexer *lexer) {
226    for (;;) {
227        if (!iswspace(lexer->lookahead)) {
228            break;
229        }
230        skip(lexer);
231    }
232
233    if (lexer->lookahead == '?') {
234        advance(lexer);
235
236        if (lexer->lookahead == '?') {
237            return false;
238        }
239
240        lexer->mark_end(lexer);
241        lexer->result_symbol = TERNARY_QMARK;
242
243        if (lexer->lookahead == '.') {
244            advance(lexer);
245            if (iswdigit(lexer->lookahead)) {
246                return true;
247            }
248            return false;
249        }
250        return true;
251    }
252    return false;
253}
254
255static bool scan_html_comment(TSLexer *lexer) {
256    while (iswspace(lexer->lookahead) || lexer->lookahead == 0x2028 || lexer->lookahead == 0x2029) {
257        skip(lexer);
258    }
259
260    const char *comment_start = "<!--";
261    const char *comment_end = "-->";
262
263    if (lexer->lookahead == '<') {
264        for (unsigned i = 0; i < 4; i++) {
265            if (lexer->lookahead != comment_start[i]) {
266                return false;
267            }
268            advance(lexer);
269        }
270    } else if (lexer->lookahead == '-') {
271        for (unsigned i = 0; i < 3; i++) {
272            if (lexer->lookahead != comment_end[i]) {
273                return false;
274            }
275            advance(lexer);
276        }
277    } else {
278        return false;
279    }
280
281    while (lexer->lookahead != 0 && lexer->lookahead != '\n' && lexer->lookahead != 0x2028 &&
282           lexer->lookahead != 0x2029) {
283        advance(lexer);
284    }
285
286    lexer->result_symbol = HTML_COMMENT;
287    lexer->mark_end(lexer);
288
289    return true;
290}
291
292static bool scan_jsx_text(TSLexer *lexer) {
293    // saw_text will be true if we see any non-whitespace content, or any whitespace content that is not a newline and
294    // does not immediately follow a newline.
295    bool saw_text = false;
296    // at_newline will be true if we are currently at a newline, or if we are at whitespace that is not a newline but
297    // immediately follows a newline.
298    bool at_newline = false;
299
300    while (lexer->lookahead != 0 && lexer->lookahead != '<' && lexer->lookahead != '>' && lexer->lookahead != '{' &&
301           lexer->lookahead != '}' && lexer->lookahead != '&') {
302        bool is_wspace = iswspace(lexer->lookahead);
303        if (lexer->lookahead == '\n') {
304            at_newline = true;
305        } else {
306            // If at_newline is already true, and we see some whitespace, then it must stay true.
307            // Otherwise, it should be false.
308            //
309            // See the table below to determine the logic for computing `saw_text`.
310            //
311            // |------------------------------------|
312            // | at_newline | is_wspace | saw_text  |
313            // |------------|-----------|-----------|
314            // | false (0)  | false (0) | true  (1) |
315            // | false (0)  | true  (1) | true  (1) |
316            // | true  (1)  | false (0) | true  (1) |
317            // | true  (1)  | true  (1) | false (0) |
318            // |------------------------------------|
319
320            at_newline &= is_wspace;
321            if (!at_newline) {
322                saw_text = true;
323            }
324        }
325
326        advance(lexer);
327    }
328
329    lexer->result_symbol = JSX_TEXT;
330    return saw_text;
331}
332
333bool tree_sitter_javascript_external_scanner_scan(void *payload, TSLexer *lexer, const bool *valid_symbols) {
334    if (valid_symbols[TEMPLATE_CHARS]) {
335        if (valid_symbols[AUTOMATIC_SEMICOLON]) {
336            return false;
337        }
338        return scan_template_chars(lexer);
339    }
340
341    if (valid_symbols[JSX_TEXT] && scan_jsx_text(lexer)) {
342        return true;
343    }
344
345    if (valid_symbols[AUTOMATIC_SEMICOLON]) {
346        bool scanned_comment = false;
347        bool ret = scan_automatic_semicolon(lexer, !valid_symbols[LOGICAL_OR], &scanned_comment);
348        if (!ret && !scanned_comment && valid_symbols[TERNARY_QMARK] && lexer->lookahead == '?') {
349            return scan_ternary_qmark(lexer);
350        }
351        return ret;
352    }
353
354    if (valid_symbols[TERNARY_QMARK]) {
355        return scan_ternary_qmark(lexer);
356    }
357
358    if (valid_symbols[HTML_COMMENT] && !valid_symbols[LOGICAL_OR] && !valid_symbols[ESCAPE_SEQUENCE] &&
359        !valid_symbols[REGEX_PATTERN]) {
360        return scan_html_comment(lexer);
361    }
362
363    return false;
364}