@willbooster/tree-sitter-c 1.5.1 → 2.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +5 -1
- package/grammar.js +66 -6
- package/package.json +1 -1
- package/src/node-types.json +72 -0
- package/src/pragma.h +55 -8
- package/src/scanner.c +16 -6
- package/tree-sitter-c.wasm +0 -0
- package/tree-sitter.json +1 -1
package/README.md
CHANGED
|
@@ -59,6 +59,10 @@ parser.setLanguage(await Language.load(c));
|
|
|
59
59
|
|
|
60
60
|
The package also ships `grammar.js`, the scanner sources `src/scanner.c`, `src/pragma.h` and `src/identifier.h`, the queries in `queries/`,
|
|
61
61
|
and the node types in `src/node-types.json` for grammars extending C (such as C++).
|
|
62
|
+
Derived grammars must handle every inherited external token in their scanner, including `_preproc_function_name`.
|
|
63
|
+
Port both the `_preproc_function_name` dispatch and the `_preproc_lparen` splice-skipping loop from `src/scanner.c`,
|
|
64
|
+
using `scan_function_macro_name` from `src/pragma.h`. The macro-name token ends before the splices used to determine
|
|
65
|
+
adjacency, so the parameter scanner must skip them again to reach `(`.
|
|
62
66
|
|
|
63
67
|
In Rust, depend on the [crate](https://crates.io/crates/willbooster-tree-sitter-c) and on
|
|
64
68
|
[willbooster-tree-sitter](https://crates.io/crates/willbooster-tree-sitter), the runtime this package is tested and
|
|
@@ -68,7 +72,7 @@ malformed input):
|
|
|
68
72
|
```toml
|
|
69
73
|
[dependencies]
|
|
70
74
|
tree-sitter = { package = "willbooster-tree-sitter", version = "1" }
|
|
71
|
-
tree-sitter-c = { package = "willbooster-tree-sitter-c", version = "
|
|
75
|
+
tree-sitter-c = { package = "willbooster-tree-sitter-c", version = "2" }
|
|
72
76
|
```
|
|
73
77
|
|
|
74
78
|
```rust
|
package/grammar.js
CHANGED
|
@@ -36,7 +36,7 @@ const PREC = {
|
|
|
36
36
|
};
|
|
37
37
|
|
|
38
38
|
const LINE_COMMENT = seq('//', /(\\+(.|\r?\n)|[^\\\n])*/);
|
|
39
|
-
const PRAGMA_SPACING = repeat(choice(/\s/,
|
|
39
|
+
const PRAGMA_SPACING = repeat(choice(/\s/, /\\(?:\r\n?|\n\r?)/));
|
|
40
40
|
const PREPROC_ARGUMENT = /\S([^/\n]|\/[^*]|\\\r?\n)*/;
|
|
41
41
|
const VA_ARG_KEYWORDS = choice('va_arg', '__builtin_va_arg');
|
|
42
42
|
|
|
@@ -71,6 +71,7 @@ module.exports = Object.assign(
|
|
|
71
71
|
[$.type_definition, $.sized_type_specifier],
|
|
72
72
|
[$.type_definition, $._sized_bit_int_specifier],
|
|
73
73
|
[$.attributed_statement],
|
|
74
|
+
[$._single_attributed_statement],
|
|
74
75
|
[$._declaration_modifiers, $.attributed_statement],
|
|
75
76
|
[$.enum_specifier],
|
|
76
77
|
[$.type_specifier, $._old_style_parameter_list],
|
|
@@ -91,9 +92,10 @@ module.exports = Object.assign(
|
|
|
91
92
|
sym('_preproc_newline'),
|
|
92
93
|
sym('_preproc_lparen'),
|
|
93
94
|
sym('_preproc_directive_arg'),
|
|
95
|
+
sym('_preproc_function_name'),
|
|
94
96
|
],
|
|
95
97
|
|
|
96
|
-
extras: ($) => [$.pragma_operator, /\s
|
|
98
|
+
extras: ($) => [$.pragma_operator, /\s|\\(?:\r\n?|\n\r?)/, $.comment],
|
|
97
99
|
|
|
98
100
|
inline: ($) => [
|
|
99
101
|
$._non_identifier_type_specifier,
|
|
@@ -168,7 +170,7 @@ module.exports = Object.assign(
|
|
|
168
170
|
PRAGMA_SPACING,
|
|
169
171
|
optional(choice('L', 'u8', 'u', 'U')),
|
|
170
172
|
'"',
|
|
171
|
-
repeat(choice(/[^\\"\n]/, seq('\\', choice(
|
|
173
|
+
repeat(choice(/[^\\"\r\n]/, seq('\\', choice(/[^\r\n]/, /\r\n?|\n\r?/)))),
|
|
172
174
|
'"',
|
|
173
175
|
PRAGMA_SPACING,
|
|
174
176
|
')'
|
|
@@ -202,7 +204,7 @@ module.exports = Object.assign(
|
|
|
202
204
|
preproc_function_def: ($) =>
|
|
203
205
|
seq(
|
|
204
206
|
preprocessor('define'),
|
|
205
|
-
field('name', $.identifier),
|
|
207
|
+
field('name', alias(sym('_preproc_function_name'), $.identifier)),
|
|
206
208
|
field('parameters', $.preproc_params),
|
|
207
209
|
field('value', optional($.preproc_arg)),
|
|
208
210
|
sym('_preproc_newline')
|
|
@@ -218,6 +220,7 @@ module.exports = Object.assign(
|
|
|
218
220
|
),
|
|
219
221
|
|
|
220
222
|
...preprocIf('', () => sym('_block_item')),
|
|
223
|
+
...preprocIf('_in_single_case', () => sym('_single_case_body'), 0, true),
|
|
221
224
|
...preprocIf('_in_field_declaration_list', () => sym('_field_declaration_list_item')),
|
|
222
225
|
...preprocIf('_in_enumerator_list', () => seq(sym('enumerator'), ',')),
|
|
223
226
|
...preprocIf('_in_enumerator_list_no_comma', () => sym('enumerator'), -1),
|
|
@@ -901,7 +904,65 @@ module.exports = Object.assign(
|
|
|
901
904
|
else_clause: ($) => seq('else', $.statement),
|
|
902
905
|
|
|
903
906
|
switch_statement: ($) =>
|
|
904
|
-
seq('switch', field('condition', $.parenthesized_expression), field('body', $.
|
|
907
|
+
seq('switch', field('condition', $.parenthesized_expression), field('body', $._single_statement)),
|
|
908
|
+
|
|
909
|
+
_single_statement: ($) =>
|
|
910
|
+
choice(
|
|
911
|
+
$.compound_statement,
|
|
912
|
+
$.expression_statement,
|
|
913
|
+
$.switch_statement,
|
|
914
|
+
$.return_statement,
|
|
915
|
+
$.break_statement,
|
|
916
|
+
$.continue_statement,
|
|
917
|
+
$.goto_statement,
|
|
918
|
+
$.seh_try_statement,
|
|
919
|
+
$.seh_leave_statement,
|
|
920
|
+
alias($._single_case_statement, $.case_statement),
|
|
921
|
+
alias($._single_labeled_statement, $.labeled_statement),
|
|
922
|
+
alias($._single_if_statement, $.if_statement),
|
|
923
|
+
alias($._single_while_statement, $.while_statement),
|
|
924
|
+
alias($._single_do_statement, $.do_statement),
|
|
925
|
+
alias($._single_for_statement, $.for_statement),
|
|
926
|
+
alias($._single_attributed_statement, $.attributed_statement)
|
|
927
|
+
),
|
|
928
|
+
|
|
929
|
+
_single_case_statement: ($) =>
|
|
930
|
+
prec.right(
|
|
931
|
+
seq(
|
|
932
|
+
choice(
|
|
933
|
+
seq('case', field('value', $.expression), optional(seq('...', field('end_value', $.expression)))),
|
|
934
|
+
'default'
|
|
935
|
+
),
|
|
936
|
+
':',
|
|
937
|
+
optional($._single_case_body)
|
|
938
|
+
)
|
|
939
|
+
),
|
|
940
|
+
_single_case_body: ($) =>
|
|
941
|
+
choice(
|
|
942
|
+
$._single_statement,
|
|
943
|
+
$.declaration,
|
|
944
|
+
$.type_definition,
|
|
945
|
+
alias(sym('preproc_if_in_single_case'), sym('preproc_if')),
|
|
946
|
+
alias(sym('preproc_ifdef_in_single_case'), sym('preproc_ifdef'))
|
|
947
|
+
),
|
|
948
|
+
_single_labeled_statement: ($) =>
|
|
949
|
+
seq(field('label', $._statement_identifier), ':', choice($._single_statement, $.declaration)),
|
|
950
|
+
_single_if_statement: ($) =>
|
|
951
|
+
prec.right(
|
|
952
|
+
seq(
|
|
953
|
+
'if',
|
|
954
|
+
field('condition', $.parenthesized_expression),
|
|
955
|
+
field('consequence', $._single_statement),
|
|
956
|
+
optional(field('alternative', alias($._single_else_clause, $.else_clause)))
|
|
957
|
+
)
|
|
958
|
+
),
|
|
959
|
+
_single_else_clause: ($) => seq('else', $._single_statement),
|
|
960
|
+
_single_while_statement: ($) =>
|
|
961
|
+
seq('while', field('condition', $.parenthesized_expression), field('body', $._single_statement)),
|
|
962
|
+
_single_do_statement: ($) =>
|
|
963
|
+
seq('do', field('body', $._single_statement), 'while', field('condition', $.parenthesized_expression), ';'),
|
|
964
|
+
_single_for_statement: ($) => seq('for', '(', $._for_statement_body, ')', field('body', $._single_statement)),
|
|
965
|
+
_single_attributed_statement: ($) => seq(repeat1($.attribute_declaration), $._single_statement),
|
|
905
966
|
|
|
906
967
|
case_statement: ($) =>
|
|
907
968
|
prec.right(
|
|
@@ -1370,7 +1431,6 @@ module.exports = Object.assign(
|
|
|
1370
1431
|
|
|
1371
1432
|
identifier: () =>
|
|
1372
1433
|
/(\p{XID_Start}|\$|_|\\u[0-9A-Fa-f]{4}|\\U[0-9A-Fa-f]{8})(\p{XID_Continue}|\$|\\u[0-9A-Fa-f]{4}|\\U[0-9A-Fa-f]{8})*/u,
|
|
1373
|
-
|
|
1374
1434
|
_type_identifier: ($) => alias(choice($.identifier, VA_ARG_KEYWORDS), sym('type_identifier')),
|
|
1375
1435
|
_field_identifier: ($) => alias(choice($.identifier, VA_ARG_KEYWORDS), sym('field_identifier')),
|
|
1376
1436
|
_statement_identifier: ($) => alias(choice($.identifier, VA_ARG_KEYWORDS), sym('statement_identifier')),
|
package/package.json
CHANGED
package/src/node-types.json
CHANGED
|
@@ -1031,6 +1031,10 @@
|
|
|
1031
1031
|
"type": "break_statement",
|
|
1032
1032
|
"named": true
|
|
1033
1033
|
},
|
|
1034
|
+
{
|
|
1035
|
+
"type": "case_statement",
|
|
1036
|
+
"named": true
|
|
1037
|
+
},
|
|
1034
1038
|
{
|
|
1035
1039
|
"type": "compound_statement",
|
|
1036
1040
|
"named": true
|
|
@@ -1067,6 +1071,14 @@
|
|
|
1067
1071
|
"type": "labeled_statement",
|
|
1068
1072
|
"named": true
|
|
1069
1073
|
},
|
|
1074
|
+
{
|
|
1075
|
+
"type": "preproc_if",
|
|
1076
|
+
"named": true
|
|
1077
|
+
},
|
|
1078
|
+
{
|
|
1079
|
+
"type": "preproc_ifdef",
|
|
1080
|
+
"named": true
|
|
1081
|
+
},
|
|
1070
1082
|
{
|
|
1071
1083
|
"type": "return_statement",
|
|
1072
1084
|
"named": true
|
|
@@ -3880,9 +3892,69 @@
|
|
|
3880
3892
|
"multiple": false,
|
|
3881
3893
|
"required": true,
|
|
3882
3894
|
"types": [
|
|
3895
|
+
{
|
|
3896
|
+
"type": "attributed_statement",
|
|
3897
|
+
"named": true
|
|
3898
|
+
},
|
|
3899
|
+
{
|
|
3900
|
+
"type": "break_statement",
|
|
3901
|
+
"named": true
|
|
3902
|
+
},
|
|
3903
|
+
{
|
|
3904
|
+
"type": "case_statement",
|
|
3905
|
+
"named": true
|
|
3906
|
+
},
|
|
3883
3907
|
{
|
|
3884
3908
|
"type": "compound_statement",
|
|
3885
3909
|
"named": true
|
|
3910
|
+
},
|
|
3911
|
+
{
|
|
3912
|
+
"type": "continue_statement",
|
|
3913
|
+
"named": true
|
|
3914
|
+
},
|
|
3915
|
+
{
|
|
3916
|
+
"type": "do_statement",
|
|
3917
|
+
"named": true
|
|
3918
|
+
},
|
|
3919
|
+
{
|
|
3920
|
+
"type": "expression_statement",
|
|
3921
|
+
"named": true
|
|
3922
|
+
},
|
|
3923
|
+
{
|
|
3924
|
+
"type": "for_statement",
|
|
3925
|
+
"named": true
|
|
3926
|
+
},
|
|
3927
|
+
{
|
|
3928
|
+
"type": "goto_statement",
|
|
3929
|
+
"named": true
|
|
3930
|
+
},
|
|
3931
|
+
{
|
|
3932
|
+
"type": "if_statement",
|
|
3933
|
+
"named": true
|
|
3934
|
+
},
|
|
3935
|
+
{
|
|
3936
|
+
"type": "labeled_statement",
|
|
3937
|
+
"named": true
|
|
3938
|
+
},
|
|
3939
|
+
{
|
|
3940
|
+
"type": "return_statement",
|
|
3941
|
+
"named": true
|
|
3942
|
+
},
|
|
3943
|
+
{
|
|
3944
|
+
"type": "seh_leave_statement",
|
|
3945
|
+
"named": true
|
|
3946
|
+
},
|
|
3947
|
+
{
|
|
3948
|
+
"type": "seh_try_statement",
|
|
3949
|
+
"named": true
|
|
3950
|
+
},
|
|
3951
|
+
{
|
|
3952
|
+
"type": "switch_statement",
|
|
3953
|
+
"named": true
|
|
3954
|
+
},
|
|
3955
|
+
{
|
|
3956
|
+
"type": "while_statement",
|
|
3957
|
+
"named": true
|
|
3886
3958
|
}
|
|
3887
3959
|
]
|
|
3888
3960
|
},
|
package/src/pragma.h
CHANGED
|
@@ -4,7 +4,10 @@
|
|
|
4
4
|
#include "tree_sitter/parser.h"
|
|
5
5
|
#include "identifier.h"
|
|
6
6
|
|
|
7
|
+
static const char pragma_word[] = "_Pragma";
|
|
8
|
+
|
|
7
9
|
static bool scan_pragma_spacing(TSLexer *lexer);
|
|
10
|
+
static bool scan_pragma_suffix(TSLexer *lexer);
|
|
8
11
|
static bool pragma_space(int32_t c);
|
|
9
12
|
static bool scan_pragma_word(TSLexer *lexer);
|
|
10
13
|
static bool scan_preproc_newline(TSLexer *lexer, bool skip);
|
|
@@ -18,6 +21,10 @@ static bool scan_pragma(TSLexer *lexer) {
|
|
|
18
21
|
lexer->advance(lexer, true);
|
|
19
22
|
}
|
|
20
23
|
if (!scan_pragma_word(lexer)) return false;
|
|
24
|
+
return scan_pragma_suffix(lexer);
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
static bool scan_pragma_suffix(TSLexer *lexer) {
|
|
21
28
|
if (!scan_pragma_spacing(lexer) || lexer->lookahead != '(') return false;
|
|
22
29
|
lexer->advance(lexer, false);
|
|
23
30
|
if (!scan_pragma_spacing(lexer)) return false;
|
|
@@ -33,10 +40,7 @@ static bool scan_pragma(TSLexer *lexer) {
|
|
|
33
40
|
if (lexer->lookahead == '\\') {
|
|
34
41
|
lexer->advance(lexer, false);
|
|
35
42
|
if (lexer->eof(lexer)) return false;
|
|
36
|
-
if (lexer
|
|
37
|
-
lexer->advance(lexer, false);
|
|
38
|
-
if (lexer->lookahead != '\n') return false;
|
|
39
|
-
}
|
|
43
|
+
if (scan_preproc_newline(lexer, false)) continue;
|
|
40
44
|
}
|
|
41
45
|
lexer->advance(lexer, false);
|
|
42
46
|
}
|
|
@@ -53,9 +57,7 @@ static bool scan_pragma_spacing(TSLexer *lexer) {
|
|
|
53
57
|
lexer->advance(lexer, false);
|
|
54
58
|
} else if (lexer->lookahead == '\\') {
|
|
55
59
|
lexer->advance(lexer, false);
|
|
56
|
-
if (lexer
|
|
57
|
-
if (lexer->lookahead != '\n') return false;
|
|
58
|
-
lexer->advance(lexer, false);
|
|
60
|
+
if (!scan_preproc_newline(lexer, false)) return false;
|
|
59
61
|
} else if (lexer->lookahead == '/') {
|
|
60
62
|
lexer->advance(lexer, false);
|
|
61
63
|
if (lexer->lookahead == '/') {
|
|
@@ -298,7 +300,7 @@ static bool preproc_word(int32_t c, bool continuation) {
|
|
|
298
300
|
}
|
|
299
301
|
|
|
300
302
|
static bool scan_pragma_word(TSLexer *lexer) {
|
|
301
|
-
const char *word =
|
|
303
|
+
const char *word = pragma_word;
|
|
302
304
|
for (; *word; word++) {
|
|
303
305
|
if (lexer->lookahead != *word) return false;
|
|
304
306
|
lexer->advance(lexer, false);
|
|
@@ -331,4 +333,49 @@ static bool pragma_space(int32_t c) {
|
|
|
331
333
|
c == 0x2028 || c == 0x2029 || c == 0x202F || c == 0x205F || c == 0x3000;
|
|
332
334
|
}
|
|
333
335
|
|
|
336
|
+
static bool scan_function_macro_name(TSLexer *lexer, bool allow_pragma, TSSymbol pragma_symbol) {
|
|
337
|
+
bool has_name = false;
|
|
338
|
+
bool is_pragma = true;
|
|
339
|
+
unsigned pragma_length = 0;
|
|
340
|
+
for (;;) {
|
|
341
|
+
while (!has_name && pragma_space(lexer->lookahead)) lexer->advance(lexer, true);
|
|
342
|
+
int32_t c = lexer->lookahead;
|
|
343
|
+
if (c == '\\') {
|
|
344
|
+
lexer->advance(lexer, false);
|
|
345
|
+
if (scan_preproc_newline(lexer, !has_name)) {
|
|
346
|
+
if (!has_name) continue;
|
|
347
|
+
if (!scan_preproc_splices(lexer)) return false;
|
|
348
|
+
break;
|
|
349
|
+
}
|
|
350
|
+
is_pragma = false;
|
|
351
|
+
unsigned digits = lexer->lookahead == 'u' ? 4 : lexer->lookahead == 'U' ? 8 : 0;
|
|
352
|
+
if (!digits) return false;
|
|
353
|
+
lexer->advance(lexer, false);
|
|
354
|
+
for (unsigned i = 0; i < digits; i++) {
|
|
355
|
+
c = lexer->lookahead;
|
|
356
|
+
if (!((c >= '0' && c <= '9') || (c >= 'a' && c <= 'f') || (c >= 'A' && c <= 'F'))) return false;
|
|
357
|
+
lexer->advance(lexer, false);
|
|
358
|
+
}
|
|
359
|
+
} else {
|
|
360
|
+
bool identifier = has_name
|
|
361
|
+
? set_contains(preproc_identifier_continue, sizeof(preproc_identifier_continue) / sizeof(TSCharacterRange), c)
|
|
362
|
+
: set_contains(preproc_identifier_start, sizeof(preproc_identifier_start) / sizeof(TSCharacterRange), c);
|
|
363
|
+
if (!identifier) break;
|
|
364
|
+
if (is_pragma) {
|
|
365
|
+
if (pragma_length < sizeof(pragma_word) - 1 && c == pragma_word[pragma_length]) pragma_length++;
|
|
366
|
+
else is_pragma = false;
|
|
367
|
+
}
|
|
368
|
+
lexer->advance(lexer, false);
|
|
369
|
+
}
|
|
370
|
+
has_name = true;
|
|
371
|
+
lexer->mark_end(lexer);
|
|
372
|
+
}
|
|
373
|
+
bool adjacent = has_name && lexer->lookahead == '(';
|
|
374
|
+
if (allow_pragma && is_pragma && pragma_length == sizeof(pragma_word) - 1 && scan_pragma_suffix(lexer)) {
|
|
375
|
+
lexer->result_symbol = pragma_symbol;
|
|
376
|
+
return true;
|
|
377
|
+
}
|
|
378
|
+
return adjacent;
|
|
379
|
+
}
|
|
380
|
+
|
|
334
381
|
#endif
|
package/src/scanner.c
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
#include "pragma.h"
|
|
2
2
|
|
|
3
|
-
enum TokenType { PRAGMA_OPERATOR, PREPROC_ARG, PREPROC_NEWLINE, PREPROC_LPAREN, PREPROC_DIRECTIVE_ARG };
|
|
3
|
+
enum TokenType { PRAGMA_OPERATOR, PREPROC_ARG, PREPROC_NEWLINE, PREPROC_LPAREN, PREPROC_DIRECTIVE_ARG, PREPROC_FUNCTION_NAME };
|
|
4
4
|
|
|
5
5
|
void *tree_sitter_c_external_scanner_create(void) {
|
|
6
6
|
return NULL;
|
|
@@ -24,13 +24,23 @@ void tree_sitter_c_external_scanner_deserialize(void *payload, const char *buffe
|
|
|
24
24
|
|
|
25
25
|
bool tree_sitter_c_external_scanner_scan(void *payload, TSLexer *lexer, const bool *valid_symbols) {
|
|
26
26
|
(void)payload;
|
|
27
|
+
if (valid_symbols[PREPROC_FUNCTION_NAME] && !valid_symbols[PREPROC_LPAREN]) {
|
|
28
|
+
lexer->result_symbol = PREPROC_FUNCTION_NAME;
|
|
29
|
+
return scan_function_macro_name(lexer, valid_symbols[PRAGMA_OPERATOR], PRAGMA_OPERATOR);
|
|
30
|
+
}
|
|
27
31
|
bool directive_text = !valid_symbols[PREPROC_ARG] && valid_symbols[PREPROC_DIRECTIVE_ARG];
|
|
28
32
|
TSSymbol argument_symbol = directive_text ? PREPROC_DIRECTIVE_ARG : PREPROC_ARG;
|
|
29
|
-
if (valid_symbols[PREPROC_LPAREN]
|
|
30
|
-
lexer->
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
33
|
+
if (valid_symbols[PREPROC_LPAREN]) {
|
|
34
|
+
while (lexer->lookahead == '\\') {
|
|
35
|
+
lexer->advance(lexer, true);
|
|
36
|
+
if (!scan_preproc_newline(lexer, true)) return false;
|
|
37
|
+
}
|
|
38
|
+
if (lexer->lookahead == '(') {
|
|
39
|
+
lexer->advance(lexer, false);
|
|
40
|
+
lexer->mark_end(lexer);
|
|
41
|
+
lexer->result_symbol = PREPROC_LPAREN;
|
|
42
|
+
return true;
|
|
43
|
+
}
|
|
34
44
|
}
|
|
35
45
|
if (valid_symbols[PREPROC_NEWLINE]) {
|
|
36
46
|
for (;;) {
|
package/tree-sitter-c.wasm
CHANGED
|
Binary file
|