Sources/TreeSitterScanners/javascript/scanner.c
364 lines · 10576 bytes
1#include "tree_sitter/parser.h"
2
3#include <stdio.h>
4#include <wctype.h>
5
6enum TokenType {
7 AUTOMATIC_SEMICOLON,
8 TEMPLATE_CHARS,
9 TERNARY_QMARK,
10 HTML_COMMENT,
11 LOGICAL_OR,
12 ESCAPE_SEQUENCE,
13 REGEX_PATTERN,
14 JSX_TEXT,
15};
16
17void *tree_sitter_javascript_external_scanner_create() { return NULL; }
18
19void tree_sitter_javascript_external_scanner_destroy(void *p) {}
20
21unsigned tree_sitter_javascript_external_scanner_serialize(void *payload, char *buffer) { return 0; }
22
23void tree_sitter_javascript_external_scanner_deserialize(void *p, const char *b, unsigned n) {}
24
25static inline void advance(TSLexer *lexer) { lexer->advance(lexer, false); }
26
27static inline void skip(TSLexer *lexer) { lexer->advance(lexer, true); }
28
29static bool scan_template_chars(TSLexer *lexer) {
30 lexer->result_symbol = TEMPLATE_CHARS;
31 for (bool has_content = false;; has_content = true) {
32 lexer->mark_end(lexer);
33 switch (lexer->lookahead) {
34 case '`':
35 return has_content;
36 case '\0':
37 return false;
38 case '$':
39 advance(lexer);
40 if (lexer->lookahead == '{') {
41 return has_content;
42 }
43 break;
44 case '\\':
45 return has_content;
46 default:
47 advance(lexer);
48 }
49 }
50}
51
52typedef enum {
53 REJECT, // Semicolon is illegal, ie a syntax error occurred
54 NO_NEWLINE, // Unclear if semicolon will be legal, continue
55 ACCEPT, // Semicolon is legal, assuming a comment was encountered
56} WhitespaceResult;
57
58/**
59 * @param consume If false, only consume enough to check if comment indicates semicolon-legality
60 */
61static WhitespaceResult scan_whitespace_and_comments(TSLexer *lexer, bool *scanned_comment, bool consume) {
62 bool saw_block_newline = false;
63
64 for (;;) {
65 while (iswspace(lexer->lookahead)) {
66 skip(lexer);
67 }
68
69 if (lexer->lookahead == '/') {
70 skip(lexer);
71
72 if (lexer->lookahead == '/') {
73 skip(lexer);
74 while (lexer->lookahead != 0 && lexer->lookahead != '\n' && lexer->lookahead != 0x2028 &&
75 lexer->lookahead != 0x2029) {
76 skip(lexer);
77 }
78 *scanned_comment = true;
79 } else if (lexer->lookahead == '*') {
80 skip(lexer);
81 while (lexer->lookahead != 0) {
82 if (lexer->lookahead == '*') {
83 skip(lexer);
84 if (lexer->lookahead == '/') {
85 skip(lexer);
86 *scanned_comment = true;
87
88 if (lexer->lookahead != '/' && !consume) {
89 return saw_block_newline ? ACCEPT : NO_NEWLINE;
90 }
91
92 break;
93 }
94 } else if (lexer->lookahead == '\n' || lexer->lookahead == 0x2028 || lexer->lookahead == 0x2029) {
95 saw_block_newline = true;
96 skip(lexer);
97 } else {
98 skip(lexer);
99 }
100 }
101 } else {
102 return REJECT;
103 }
104 } else {
105 return ACCEPT;
106 }
107 }
108}
109
110static bool scan_automatic_semicolon(TSLexer *lexer, bool comment_condition, bool *scanned_comment) {
111 lexer->result_symbol = AUTOMATIC_SEMICOLON;
112 lexer->mark_end(lexer);
113
114 for (;;) {
115 if (lexer->lookahead == 0) {
116 return true;
117 }
118
119 if (lexer->lookahead == '/') {
120 WhitespaceResult result = scan_whitespace_and_comments(lexer, scanned_comment, false);
121 if (result == REJECT) {
122 return false;
123 }
124
125 if (result == ACCEPT && comment_condition && lexer->lookahead != ',' && lexer->lookahead != '=') {
126 return true;
127 }
128 }
129
130 if (lexer->lookahead == '}') {
131 return true;
132 }
133
134 if (lexer->is_at_included_range_start(lexer)) {
135 return true;
136 }
137
138 if (lexer->lookahead == '\n' || lexer->lookahead == 0x2028 || lexer->lookahead == 0x2029) {
139 break;
140 }
141
142 if (!iswspace(lexer->lookahead)) {
143 return false;
144 }
145
146 skip(lexer);
147 }
148
149 skip(lexer);
150
151 if (scan_whitespace_and_comments(lexer, scanned_comment, true) == REJECT) {
152 return false;
153 }
154
155 switch (lexer->lookahead) {
156 case '`':
157 case ',':
158 case ':':
159 case ';':
160 case '*':
161 case '%':
162 case '>':
163 case '<':
164 case '=':
165 case '[':
166 case '(':
167 case '?':
168 case '^':
169 case '|':
170 case '&':
171 case '/':
172 return false;
173
174 // Insert a semicolon before decimals literals but not otherwise.
175 case '.':
176 skip(lexer);
177 return iswdigit(lexer->lookahead);
178
179 // Insert a semicolon before `--` and `++`, but not before binary `+` or `-`.
180 case '+':
181 skip(lexer);
182 return lexer->lookahead == '+';
183 case '-':
184 skip(lexer);
185 return lexer->lookahead == '-';
186
187 // Don't insert a semicolon before `!=`, but do insert one before a unary `!`.
188 case '!':
189 skip(lexer);
190 return lexer->lookahead != '=';
191
192 // Don't insert a semicolon before `in` or `instanceof`, but do insert one
193 // before an identifier.
194 case 'i':
195 skip(lexer);
196
197 if (lexer->lookahead != 'n') {
198 return true;
199 }
200 skip(lexer);
201
202 if (!iswalpha(lexer->lookahead)) {
203 return false;
204 }
205
206 for (unsigned i = 0; i < 8; i++) {
207 if (lexer->lookahead != "stanceof"[i]) {
208 return true;
209 }
210 skip(lexer);
211 }
212
213 if (!iswalpha(lexer->lookahead)) {
214 return false;
215 }
216 break;
217
218 default:
219 break;
220 }
221
222 return true;
223}
224
225static bool scan_ternary_qmark(TSLexer *lexer) {
226 for (;;) {
227 if (!iswspace(lexer->lookahead)) {
228 break;
229 }
230 skip(lexer);
231 }
232
233 if (lexer->lookahead == '?') {
234 advance(lexer);
235
236 if (lexer->lookahead == '?') {
237 return false;
238 }
239
240 lexer->mark_end(lexer);
241 lexer->result_symbol = TERNARY_QMARK;
242
243 if (lexer->lookahead == '.') {
244 advance(lexer);
245 if (iswdigit(lexer->lookahead)) {
246 return true;
247 }
248 return false;
249 }
250 return true;
251 }
252 return false;
253}
254
255static bool scan_html_comment(TSLexer *lexer) {
256 while (iswspace(lexer->lookahead) || lexer->lookahead == 0x2028 || lexer->lookahead == 0x2029) {
257 skip(lexer);
258 }
259
260 const char *comment_start = "<!--";
261 const char *comment_end = "-->";
262
263 if (lexer->lookahead == '<') {
264 for (unsigned i = 0; i < 4; i++) {
265 if (lexer->lookahead != comment_start[i]) {
266 return false;
267 }
268 advance(lexer);
269 }
270 } else if (lexer->lookahead == '-') {
271 for (unsigned i = 0; i < 3; i++) {
272 if (lexer->lookahead != comment_end[i]) {
273 return false;
274 }
275 advance(lexer);
276 }
277 } else {
278 return false;
279 }
280
281 while (lexer->lookahead != 0 && lexer->lookahead != '\n' && lexer->lookahead != 0x2028 &&
282 lexer->lookahead != 0x2029) {
283 advance(lexer);
284 }
285
286 lexer->result_symbol = HTML_COMMENT;
287 lexer->mark_end(lexer);
288
289 return true;
290}
291
292static bool scan_jsx_text(TSLexer *lexer) {
293 // saw_text will be true if we see any non-whitespace content, or any whitespace content that is not a newline and
294 // does not immediately follow a newline.
295 bool saw_text = false;
296 // at_newline will be true if we are currently at a newline, or if we are at whitespace that is not a newline but
297 // immediately follows a newline.
298 bool at_newline = false;
299
300 while (lexer->lookahead != 0 && lexer->lookahead != '<' && lexer->lookahead != '>' && lexer->lookahead != '{' &&
301 lexer->lookahead != '}' && lexer->lookahead != '&') {
302 bool is_wspace = iswspace(lexer->lookahead);
303 if (lexer->lookahead == '\n') {
304 at_newline = true;
305 } else {
306 // If at_newline is already true, and we see some whitespace, then it must stay true.
307 // Otherwise, it should be false.
308 //
309 // See the table below to determine the logic for computing `saw_text`.
310 //
311 // |------------------------------------|
312 // | at_newline | is_wspace | saw_text |
313 // |------------|-----------|-----------|
314 // | false (0) | false (0) | true (1) |
315 // | false (0) | true (1) | true (1) |
316 // | true (1) | false (0) | true (1) |
317 // | true (1) | true (1) | false (0) |
318 // |------------------------------------|
319
320 at_newline &= is_wspace;
321 if (!at_newline) {
322 saw_text = true;
323 }
324 }
325
326 advance(lexer);
327 }
328
329 lexer->result_symbol = JSX_TEXT;
330 return saw_text;
331}
332
333bool tree_sitter_javascript_external_scanner_scan(void *payload, TSLexer *lexer, const bool *valid_symbols) {
334 if (valid_symbols[TEMPLATE_CHARS]) {
335 if (valid_symbols[AUTOMATIC_SEMICOLON]) {
336 return false;
337 }
338 return scan_template_chars(lexer);
339 }
340
341 if (valid_symbols[JSX_TEXT] && scan_jsx_text(lexer)) {
342 return true;
343 }
344
345 if (valid_symbols[AUTOMATIC_SEMICOLON]) {
346 bool scanned_comment = false;
347 bool ret = scan_automatic_semicolon(lexer, !valid_symbols[LOGICAL_OR], &scanned_comment);
348 if (!ret && !scanned_comment && valid_symbols[TERNARY_QMARK] && lexer->lookahead == '?') {
349 return scan_ternary_qmark(lexer);
350 }
351 return ret;
352 }
353
354 if (valid_symbols[TERNARY_QMARK]) {
355 return scan_ternary_qmark(lexer);
356 }
357
358 if (valid_symbols[HTML_COMMENT] && !valid_symbols[LOGICAL_OR] && !valid_symbols[ESCAPE_SEQUENCE] &&
359 !valid_symbols[REGEX_PATTERN]) {
360 return scan_html_comment(lexer);
361 }
362
363 return false;
364}