@willbooster/tree-sitter-c 1.5.2 → 2.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +5 -1
- package/grammar.js +5 -5
- package/package.json +1 -1
- package/src/pragma.h +55 -8
- package/src/scanner.c +16 -6
- package/tree-sitter-c.wasm +0 -0
- package/tree-sitter.json +1 -1
package/README.md
CHANGED
|
@@ -59,6 +59,10 @@ parser.setLanguage(await Language.load(c));
|
|
|
59
59
|
|
|
60
60
|
The package also ships `grammar.js`, the scanner sources `src/scanner.c`, `src/pragma.h` and `src/identifier.h`, the queries in `queries/`,
|
|
61
61
|
and the node types in `src/node-types.json` for grammars extending C (such as C++).
|
|
62
|
+
Derived grammars must handle every inherited external token in their scanner, including `_preproc_function_name`.
|
|
63
|
+
Port both the `_preproc_function_name` dispatch and the `_preproc_lparen` splice-skipping loop from `src/scanner.c`,
|
|
64
|
+
using `scan_function_macro_name` from `src/pragma.h`. The macro-name token ends before the splices used to determine
|
|
65
|
+
adjacency, so the parameter scanner must skip them again to reach `(`.
|
|
62
66
|
|
|
63
67
|
In Rust, depend on the [crate](https://crates.io/crates/willbooster-tree-sitter-c) and on
|
|
64
68
|
[willbooster-tree-sitter](https://crates.io/crates/willbooster-tree-sitter), the runtime this package is tested and
|
|
@@ -68,7 +72,7 @@ malformed input):
|
|
|
68
72
|
```toml
|
|
69
73
|
[dependencies]
|
|
70
74
|
tree-sitter = { package = "willbooster-tree-sitter", version = "1" }
|
|
71
|
-
tree-sitter-c = { package = "willbooster-tree-sitter-c", version = "
|
|
75
|
+
tree-sitter-c = { package = "willbooster-tree-sitter-c", version = "2" }
|
|
72
76
|
```
|
|
73
77
|
|
|
74
78
|
```rust
|
package/grammar.js
CHANGED
|
@@ -36,7 +36,7 @@ const PREC = {
|
|
|
36
36
|
};
|
|
37
37
|
|
|
38
38
|
const LINE_COMMENT = seq('//', /(\\+(.|\r?\n)|[^\\\n])*/);
|
|
39
|
-
const PRAGMA_SPACING = repeat(choice(/\s/,
|
|
39
|
+
const PRAGMA_SPACING = repeat(choice(/\s/, /\\(?:\r\n?|\n\r?)/));
|
|
40
40
|
const PREPROC_ARGUMENT = /\S([^/\n]|\/[^*]|\\\r?\n)*/;
|
|
41
41
|
const VA_ARG_KEYWORDS = choice('va_arg', '__builtin_va_arg');
|
|
42
42
|
|
|
@@ -92,9 +92,10 @@ module.exports = Object.assign(
|
|
|
92
92
|
sym('_preproc_newline'),
|
|
93
93
|
sym('_preproc_lparen'),
|
|
94
94
|
sym('_preproc_directive_arg'),
|
|
95
|
+
sym('_preproc_function_name'),
|
|
95
96
|
],
|
|
96
97
|
|
|
97
|
-
extras: ($) => [$.pragma_operator, /\s
|
|
98
|
+
extras: ($) => [$.pragma_operator, /\s|\\(?:\r\n?|\n\r?)/, $.comment],
|
|
98
99
|
|
|
99
100
|
inline: ($) => [
|
|
100
101
|
$._non_identifier_type_specifier,
|
|
@@ -169,7 +170,7 @@ module.exports = Object.assign(
|
|
|
169
170
|
PRAGMA_SPACING,
|
|
170
171
|
optional(choice('L', 'u8', 'u', 'U')),
|
|
171
172
|
'"',
|
|
172
|
-
repeat(choice(/[^\\"\n]/, seq('\\', choice(
|
|
173
|
+
repeat(choice(/[^\\"\r\n]/, seq('\\', choice(/[^\r\n]/, /\r\n?|\n\r?/)))),
|
|
173
174
|
'"',
|
|
174
175
|
PRAGMA_SPACING,
|
|
175
176
|
')'
|
|
@@ -203,7 +204,7 @@ module.exports = Object.assign(
|
|
|
203
204
|
preproc_function_def: ($) =>
|
|
204
205
|
seq(
|
|
205
206
|
preprocessor('define'),
|
|
206
|
-
field('name', $.identifier),
|
|
207
|
+
field('name', alias(sym('_preproc_function_name'), $.identifier)),
|
|
207
208
|
field('parameters', $.preproc_params),
|
|
208
209
|
field('value', optional($.preproc_arg)),
|
|
209
210
|
sym('_preproc_newline')
|
|
@@ -1430,7 +1431,6 @@ module.exports = Object.assign(
|
|
|
1430
1431
|
|
|
1431
1432
|
identifier: () =>
|
|
1432
1433
|
/(\p{XID_Start}|\$|_|\\u[0-9A-Fa-f]{4}|\\U[0-9A-Fa-f]{8})(\p{XID_Continue}|\$|\\u[0-9A-Fa-f]{4}|\\U[0-9A-Fa-f]{8})*/u,
|
|
1433
|
-
|
|
1434
1434
|
_type_identifier: ($) => alias(choice($.identifier, VA_ARG_KEYWORDS), sym('type_identifier')),
|
|
1435
1435
|
_field_identifier: ($) => alias(choice($.identifier, VA_ARG_KEYWORDS), sym('field_identifier')),
|
|
1436
1436
|
_statement_identifier: ($) => alias(choice($.identifier, VA_ARG_KEYWORDS), sym('statement_identifier')),
|
package/package.json
CHANGED
package/src/pragma.h
CHANGED
|
@@ -4,7 +4,10 @@
|
|
|
4
4
|
#include "tree_sitter/parser.h"
|
|
5
5
|
#include "identifier.h"
|
|
6
6
|
|
|
7
|
+
static const char pragma_word[] = "_Pragma";
|
|
8
|
+
|
|
7
9
|
static bool scan_pragma_spacing(TSLexer *lexer);
|
|
10
|
+
static bool scan_pragma_suffix(TSLexer *lexer);
|
|
8
11
|
static bool pragma_space(int32_t c);
|
|
9
12
|
static bool scan_pragma_word(TSLexer *lexer);
|
|
10
13
|
static bool scan_preproc_newline(TSLexer *lexer, bool skip);
|
|
@@ -18,6 +21,10 @@ static bool scan_pragma(TSLexer *lexer) {
|
|
|
18
21
|
lexer->advance(lexer, true);
|
|
19
22
|
}
|
|
20
23
|
if (!scan_pragma_word(lexer)) return false;
|
|
24
|
+
return scan_pragma_suffix(lexer);
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
static bool scan_pragma_suffix(TSLexer *lexer) {
|
|
21
28
|
if (!scan_pragma_spacing(lexer) || lexer->lookahead != '(') return false;
|
|
22
29
|
lexer->advance(lexer, false);
|
|
23
30
|
if (!scan_pragma_spacing(lexer)) return false;
|
|
@@ -33,10 +40,7 @@ static bool scan_pragma(TSLexer *lexer) {
|
|
|
33
40
|
if (lexer->lookahead == '\\') {
|
|
34
41
|
lexer->advance(lexer, false);
|
|
35
42
|
if (lexer->eof(lexer)) return false;
|
|
36
|
-
if (lexer
|
|
37
|
-
lexer->advance(lexer, false);
|
|
38
|
-
if (lexer->lookahead != '\n') return false;
|
|
39
|
-
}
|
|
43
|
+
if (scan_preproc_newline(lexer, false)) continue;
|
|
40
44
|
}
|
|
41
45
|
lexer->advance(lexer, false);
|
|
42
46
|
}
|
|
@@ -53,9 +57,7 @@ static bool scan_pragma_spacing(TSLexer *lexer) {
|
|
|
53
57
|
lexer->advance(lexer, false);
|
|
54
58
|
} else if (lexer->lookahead == '\\') {
|
|
55
59
|
lexer->advance(lexer, false);
|
|
56
|
-
if (lexer
|
|
57
|
-
if (lexer->lookahead != '\n') return false;
|
|
58
|
-
lexer->advance(lexer, false);
|
|
60
|
+
if (!scan_preproc_newline(lexer, false)) return false;
|
|
59
61
|
} else if (lexer->lookahead == '/') {
|
|
60
62
|
lexer->advance(lexer, false);
|
|
61
63
|
if (lexer->lookahead == '/') {
|
|
@@ -298,7 +300,7 @@ static bool preproc_word(int32_t c, bool continuation) {
|
|
|
298
300
|
}
|
|
299
301
|
|
|
300
302
|
static bool scan_pragma_word(TSLexer *lexer) {
|
|
301
|
-
const char *word =
|
|
303
|
+
const char *word = pragma_word;
|
|
302
304
|
for (; *word; word++) {
|
|
303
305
|
if (lexer->lookahead != *word) return false;
|
|
304
306
|
lexer->advance(lexer, false);
|
|
@@ -331,4 +333,49 @@ static bool pragma_space(int32_t c) {
|
|
|
331
333
|
c == 0x2028 || c == 0x2029 || c == 0x202F || c == 0x205F || c == 0x3000;
|
|
332
334
|
}
|
|
333
335
|
|
|
336
|
+
static bool scan_function_macro_name(TSLexer *lexer, bool allow_pragma, TSSymbol pragma_symbol) {
|
|
337
|
+
bool has_name = false;
|
|
338
|
+
bool is_pragma = true;
|
|
339
|
+
unsigned pragma_length = 0;
|
|
340
|
+
for (;;) {
|
|
341
|
+
while (!has_name && pragma_space(lexer->lookahead)) lexer->advance(lexer, true);
|
|
342
|
+
int32_t c = lexer->lookahead;
|
|
343
|
+
if (c == '\\') {
|
|
344
|
+
lexer->advance(lexer, false);
|
|
345
|
+
if (scan_preproc_newline(lexer, !has_name)) {
|
|
346
|
+
if (!has_name) continue;
|
|
347
|
+
if (!scan_preproc_splices(lexer)) return false;
|
|
348
|
+
break;
|
|
349
|
+
}
|
|
350
|
+
is_pragma = false;
|
|
351
|
+
unsigned digits = lexer->lookahead == 'u' ? 4 : lexer->lookahead == 'U' ? 8 : 0;
|
|
352
|
+
if (!digits) return false;
|
|
353
|
+
lexer->advance(lexer, false);
|
|
354
|
+
for (unsigned i = 0; i < digits; i++) {
|
|
355
|
+
c = lexer->lookahead;
|
|
356
|
+
if (!((c >= '0' && c <= '9') || (c >= 'a' && c <= 'f') || (c >= 'A' && c <= 'F'))) return false;
|
|
357
|
+
lexer->advance(lexer, false);
|
|
358
|
+
}
|
|
359
|
+
} else {
|
|
360
|
+
bool identifier = has_name
|
|
361
|
+
? set_contains(preproc_identifier_continue, sizeof(preproc_identifier_continue) / sizeof(TSCharacterRange), c)
|
|
362
|
+
: set_contains(preproc_identifier_start, sizeof(preproc_identifier_start) / sizeof(TSCharacterRange), c);
|
|
363
|
+
if (!identifier) break;
|
|
364
|
+
if (is_pragma) {
|
|
365
|
+
if (pragma_length < sizeof(pragma_word) - 1 && c == pragma_word[pragma_length]) pragma_length++;
|
|
366
|
+
else is_pragma = false;
|
|
367
|
+
}
|
|
368
|
+
lexer->advance(lexer, false);
|
|
369
|
+
}
|
|
370
|
+
has_name = true;
|
|
371
|
+
lexer->mark_end(lexer);
|
|
372
|
+
}
|
|
373
|
+
bool adjacent = has_name && lexer->lookahead == '(';
|
|
374
|
+
if (allow_pragma && is_pragma && pragma_length == sizeof(pragma_word) - 1 && scan_pragma_suffix(lexer)) {
|
|
375
|
+
lexer->result_symbol = pragma_symbol;
|
|
376
|
+
return true;
|
|
377
|
+
}
|
|
378
|
+
return adjacent;
|
|
379
|
+
}
|
|
380
|
+
|
|
334
381
|
#endif
|
package/src/scanner.c
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
#include "pragma.h"
|
|
2
2
|
|
|
3
|
-
enum TokenType { PRAGMA_OPERATOR, PREPROC_ARG, PREPROC_NEWLINE, PREPROC_LPAREN, PREPROC_DIRECTIVE_ARG };
|
|
3
|
+
enum TokenType { PRAGMA_OPERATOR, PREPROC_ARG, PREPROC_NEWLINE, PREPROC_LPAREN, PREPROC_DIRECTIVE_ARG, PREPROC_FUNCTION_NAME };
|
|
4
4
|
|
|
5
5
|
void *tree_sitter_c_external_scanner_create(void) {
|
|
6
6
|
return NULL;
|
|
@@ -24,13 +24,23 @@ void tree_sitter_c_external_scanner_deserialize(void *payload, const char *buffe
|
|
|
24
24
|
|
|
25
25
|
bool tree_sitter_c_external_scanner_scan(void *payload, TSLexer *lexer, const bool *valid_symbols) {
|
|
26
26
|
(void)payload;
|
|
27
|
+
if (valid_symbols[PREPROC_FUNCTION_NAME] && !valid_symbols[PREPROC_LPAREN]) {
|
|
28
|
+
lexer->result_symbol = PREPROC_FUNCTION_NAME;
|
|
29
|
+
return scan_function_macro_name(lexer, valid_symbols[PRAGMA_OPERATOR], PRAGMA_OPERATOR);
|
|
30
|
+
}
|
|
27
31
|
bool directive_text = !valid_symbols[PREPROC_ARG] && valid_symbols[PREPROC_DIRECTIVE_ARG];
|
|
28
32
|
TSSymbol argument_symbol = directive_text ? PREPROC_DIRECTIVE_ARG : PREPROC_ARG;
|
|
29
|
-
if (valid_symbols[PREPROC_LPAREN]
|
|
30
|
-
lexer->
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
33
|
+
if (valid_symbols[PREPROC_LPAREN]) {
|
|
34
|
+
while (lexer->lookahead == '\\') {
|
|
35
|
+
lexer->advance(lexer, true);
|
|
36
|
+
if (!scan_preproc_newline(lexer, true)) return false;
|
|
37
|
+
}
|
|
38
|
+
if (lexer->lookahead == '(') {
|
|
39
|
+
lexer->advance(lexer, false);
|
|
40
|
+
lexer->mark_end(lexer);
|
|
41
|
+
lexer->result_symbol = PREPROC_LPAREN;
|
|
42
|
+
return true;
|
|
43
|
+
}
|
|
34
44
|
}
|
|
35
45
|
if (valid_symbols[PREPROC_NEWLINE]) {
|
|
36
46
|
for (;;) {
|
package/tree-sitter-c.wasm
CHANGED
|
Binary file
|