@willbooster/tree-sitter-c 1.5.1 → 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -59,6 +59,10 @@ parser.setLanguage(await Language.load(c));
59
59
 
60
60
  The package also ships `grammar.js`, the scanner sources `src/scanner.c`, `src/pragma.h` and `src/identifier.h`, the queries in `queries/`,
61
61
  and the node types in `src/node-types.json` for grammars extending C (such as C++).
62
+ Derived grammars must handle every inherited external token in their scanner, including `_preproc_function_name`.
63
+ Port both the `_preproc_function_name` dispatch and the `_preproc_lparen` splice-skipping loop from `src/scanner.c`,
64
+ using `scan_function_macro_name` from `src/pragma.h`. The macro-name token ends before the splices used to determine
65
+ adjacency, so the parameter scanner must skip them again to reach `(`.
62
66
 
63
67
  In Rust, depend on the [crate](https://crates.io/crates/willbooster-tree-sitter-c) and on
64
68
  [willbooster-tree-sitter](https://crates.io/crates/willbooster-tree-sitter), the runtime this package is tested and
@@ -68,7 +72,7 @@ malformed input):
68
72
  ```toml
69
73
  [dependencies]
70
74
  tree-sitter = { package = "willbooster-tree-sitter", version = "1" }
71
- tree-sitter-c = { package = "willbooster-tree-sitter-c", version = "1" }
75
+ tree-sitter-c = { package = "willbooster-tree-sitter-c", version = "2" }
72
76
  ```
73
77
 
74
78
  ```rust
package/grammar.js CHANGED
@@ -36,7 +36,7 @@ const PREC = {
36
36
  };
37
37
 
38
38
  const LINE_COMMENT = seq('//', /(\\+(.|\r?\n)|[^\\\n])*/);
39
- const PRAGMA_SPACING = repeat(choice(/\s/, /\\\r?\n/));
39
+ const PRAGMA_SPACING = repeat(choice(/\s/, /\\(?:\r\n?|\n\r?)/));
40
40
  const PREPROC_ARGUMENT = /\S([^/\n]|\/[^*]|\\\r?\n)*/;
41
41
  const VA_ARG_KEYWORDS = choice('va_arg', '__builtin_va_arg');
42
42
 
@@ -71,6 +71,7 @@ module.exports = Object.assign(
71
71
  [$.type_definition, $.sized_type_specifier],
72
72
  [$.type_definition, $._sized_bit_int_specifier],
73
73
  [$.attributed_statement],
74
+ [$._single_attributed_statement],
74
75
  [$._declaration_modifiers, $.attributed_statement],
75
76
  [$.enum_specifier],
76
77
  [$.type_specifier, $._old_style_parameter_list],
@@ -91,9 +92,10 @@ module.exports = Object.assign(
91
92
  sym('_preproc_newline'),
92
93
  sym('_preproc_lparen'),
93
94
  sym('_preproc_directive_arg'),
95
+ sym('_preproc_function_name'),
94
96
  ],
95
97
 
96
- extras: ($) => [$.pragma_operator, /\s|\\\r?\n/, $.comment],
98
+ extras: ($) => [$.pragma_operator, /\s|\\(?:\r\n?|\n\r?)/, $.comment],
97
99
 
98
100
  inline: ($) => [
99
101
  $._non_identifier_type_specifier,
@@ -168,7 +170,7 @@ module.exports = Object.assign(
168
170
  PRAGMA_SPACING,
169
171
  optional(choice('L', 'u8', 'u', 'U')),
170
172
  '"',
171
- repeat(choice(/[^\\"\n]/, seq('\\', choice(/./, /\r?\n/)))),
173
+ repeat(choice(/[^\\"\r\n]/, seq('\\', choice(/[^\r\n]/, /\r\n?|\n\r?/)))),
172
174
  '"',
173
175
  PRAGMA_SPACING,
174
176
  ')'
@@ -202,7 +204,7 @@ module.exports = Object.assign(
202
204
  preproc_function_def: ($) =>
203
205
  seq(
204
206
  preprocessor('define'),
205
- field('name', $.identifier),
207
+ field('name', alias(sym('_preproc_function_name'), $.identifier)),
206
208
  field('parameters', $.preproc_params),
207
209
  field('value', optional($.preproc_arg)),
208
210
  sym('_preproc_newline')
@@ -218,6 +220,7 @@ module.exports = Object.assign(
218
220
  ),
219
221
 
220
222
  ...preprocIf('', () => sym('_block_item')),
223
+ ...preprocIf('_in_single_case', () => sym('_single_case_body'), 0, true),
221
224
  ...preprocIf('_in_field_declaration_list', () => sym('_field_declaration_list_item')),
222
225
  ...preprocIf('_in_enumerator_list', () => seq(sym('enumerator'), ',')),
223
226
  ...preprocIf('_in_enumerator_list_no_comma', () => sym('enumerator'), -1),
@@ -901,7 +904,65 @@ module.exports = Object.assign(
901
904
  else_clause: ($) => seq('else', $.statement),
902
905
 
903
906
  switch_statement: ($) =>
904
- seq('switch', field('condition', $.parenthesized_expression), field('body', $.compound_statement)),
907
+ seq('switch', field('condition', $.parenthesized_expression), field('body', $._single_statement)),
908
+
909
+ _single_statement: ($) =>
910
+ choice(
911
+ $.compound_statement,
912
+ $.expression_statement,
913
+ $.switch_statement,
914
+ $.return_statement,
915
+ $.break_statement,
916
+ $.continue_statement,
917
+ $.goto_statement,
918
+ $.seh_try_statement,
919
+ $.seh_leave_statement,
920
+ alias($._single_case_statement, $.case_statement),
921
+ alias($._single_labeled_statement, $.labeled_statement),
922
+ alias($._single_if_statement, $.if_statement),
923
+ alias($._single_while_statement, $.while_statement),
924
+ alias($._single_do_statement, $.do_statement),
925
+ alias($._single_for_statement, $.for_statement),
926
+ alias($._single_attributed_statement, $.attributed_statement)
927
+ ),
928
+
929
+ _single_case_statement: ($) =>
930
+ prec.right(
931
+ seq(
932
+ choice(
933
+ seq('case', field('value', $.expression), optional(seq('...', field('end_value', $.expression)))),
934
+ 'default'
935
+ ),
936
+ ':',
937
+ optional($._single_case_body)
938
+ )
939
+ ),
940
+ _single_case_body: ($) =>
941
+ choice(
942
+ $._single_statement,
943
+ $.declaration,
944
+ $.type_definition,
945
+ alias(sym('preproc_if_in_single_case'), sym('preproc_if')),
946
+ alias(sym('preproc_ifdef_in_single_case'), sym('preproc_ifdef'))
947
+ ),
948
+ _single_labeled_statement: ($) =>
949
+ seq(field('label', $._statement_identifier), ':', choice($._single_statement, $.declaration)),
950
+ _single_if_statement: ($) =>
951
+ prec.right(
952
+ seq(
953
+ 'if',
954
+ field('condition', $.parenthesized_expression),
955
+ field('consequence', $._single_statement),
956
+ optional(field('alternative', alias($._single_else_clause, $.else_clause)))
957
+ )
958
+ ),
959
+ _single_else_clause: ($) => seq('else', $._single_statement),
960
+ _single_while_statement: ($) =>
961
+ seq('while', field('condition', $.parenthesized_expression), field('body', $._single_statement)),
962
+ _single_do_statement: ($) =>
963
+ seq('do', field('body', $._single_statement), 'while', field('condition', $.parenthesized_expression), ';'),
964
+ _single_for_statement: ($) => seq('for', '(', $._for_statement_body, ')', field('body', $._single_statement)),
965
+ _single_attributed_statement: ($) => seq(repeat1($.attribute_declaration), $._single_statement),
905
966
 
906
967
  case_statement: ($) =>
907
968
  prec.right(
@@ -1370,7 +1431,6 @@ module.exports = Object.assign(
1370
1431
 
1371
1432
  identifier: () =>
1372
1433
  /(\p{XID_Start}|\$|_|\\u[0-9A-Fa-f]{4}|\\U[0-9A-Fa-f]{8})(\p{XID_Continue}|\$|\\u[0-9A-Fa-f]{4}|\\U[0-9A-Fa-f]{8})*/u,
1373
-
1374
1434
  _type_identifier: ($) => alias(choice($.identifier, VA_ARG_KEYWORDS), sym('type_identifier')),
1375
1435
  _field_identifier: ($) => alias(choice($.identifier, VA_ARG_KEYWORDS), sym('field_identifier')),
1376
1436
  _statement_identifier: ($) => alias(choice($.identifier, VA_ARG_KEYWORDS), sym('statement_identifier')),
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@willbooster/tree-sitter-c",
3
- "version": "1.5.1",
3
+ "version": "2.0.0",
4
4
  "description": "C grammar for tree-sitter",
5
5
  "keywords": [
6
6
  "incremental",
@@ -1031,6 +1031,10 @@
1031
1031
  "type": "break_statement",
1032
1032
  "named": true
1033
1033
  },
1034
+ {
1035
+ "type": "case_statement",
1036
+ "named": true
1037
+ },
1034
1038
  {
1035
1039
  "type": "compound_statement",
1036
1040
  "named": true
@@ -1067,6 +1071,14 @@
1067
1071
  "type": "labeled_statement",
1068
1072
  "named": true
1069
1073
  },
1074
+ {
1075
+ "type": "preproc_if",
1076
+ "named": true
1077
+ },
1078
+ {
1079
+ "type": "preproc_ifdef",
1080
+ "named": true
1081
+ },
1070
1082
  {
1071
1083
  "type": "return_statement",
1072
1084
  "named": true
@@ -3880,9 +3892,69 @@
3880
3892
  "multiple": false,
3881
3893
  "required": true,
3882
3894
  "types": [
3895
+ {
3896
+ "type": "attributed_statement",
3897
+ "named": true
3898
+ },
3899
+ {
3900
+ "type": "break_statement",
3901
+ "named": true
3902
+ },
3903
+ {
3904
+ "type": "case_statement",
3905
+ "named": true
3906
+ },
3883
3907
  {
3884
3908
  "type": "compound_statement",
3885
3909
  "named": true
3910
+ },
3911
+ {
3912
+ "type": "continue_statement",
3913
+ "named": true
3914
+ },
3915
+ {
3916
+ "type": "do_statement",
3917
+ "named": true
3918
+ },
3919
+ {
3920
+ "type": "expression_statement",
3921
+ "named": true
3922
+ },
3923
+ {
3924
+ "type": "for_statement",
3925
+ "named": true
3926
+ },
3927
+ {
3928
+ "type": "goto_statement",
3929
+ "named": true
3930
+ },
3931
+ {
3932
+ "type": "if_statement",
3933
+ "named": true
3934
+ },
3935
+ {
3936
+ "type": "labeled_statement",
3937
+ "named": true
3938
+ },
3939
+ {
3940
+ "type": "return_statement",
3941
+ "named": true
3942
+ },
3943
+ {
3944
+ "type": "seh_leave_statement",
3945
+ "named": true
3946
+ },
3947
+ {
3948
+ "type": "seh_try_statement",
3949
+ "named": true
3950
+ },
3951
+ {
3952
+ "type": "switch_statement",
3953
+ "named": true
3954
+ },
3955
+ {
3956
+ "type": "while_statement",
3957
+ "named": true
3886
3958
  }
3887
3959
  ]
3888
3960
  },
package/src/pragma.h CHANGED
@@ -4,7 +4,10 @@
4
4
  #include "tree_sitter/parser.h"
5
5
  #include "identifier.h"
6
6
 
7
+ static const char pragma_word[] = "_Pragma";
8
+
7
9
  static bool scan_pragma_spacing(TSLexer *lexer);
10
+ static bool scan_pragma_suffix(TSLexer *lexer);
8
11
  static bool pragma_space(int32_t c);
9
12
  static bool scan_pragma_word(TSLexer *lexer);
10
13
  static bool scan_preproc_newline(TSLexer *lexer, bool skip);
@@ -18,6 +21,10 @@ static bool scan_pragma(TSLexer *lexer) {
18
21
  lexer->advance(lexer, true);
19
22
  }
20
23
  if (!scan_pragma_word(lexer)) return false;
24
+ return scan_pragma_suffix(lexer);
25
+ }
26
+
27
+ static bool scan_pragma_suffix(TSLexer *lexer) {
21
28
  if (!scan_pragma_spacing(lexer) || lexer->lookahead != '(') return false;
22
29
  lexer->advance(lexer, false);
23
30
  if (!scan_pragma_spacing(lexer)) return false;
@@ -33,10 +40,7 @@ static bool scan_pragma(TSLexer *lexer) {
33
40
  if (lexer->lookahead == '\\') {
34
41
  lexer->advance(lexer, false);
35
42
  if (lexer->eof(lexer)) return false;
36
- if (lexer->lookahead == '\r') {
37
- lexer->advance(lexer, false);
38
- if (lexer->lookahead != '\n') return false;
39
- }
43
+ if (scan_preproc_newline(lexer, false)) continue;
40
44
  }
41
45
  lexer->advance(lexer, false);
42
46
  }
@@ -53,9 +57,7 @@ static bool scan_pragma_spacing(TSLexer *lexer) {
53
57
  lexer->advance(lexer, false);
54
58
  } else if (lexer->lookahead == '\\') {
55
59
  lexer->advance(lexer, false);
56
- if (lexer->lookahead == '\r') lexer->advance(lexer, false);
57
- if (lexer->lookahead != '\n') return false;
58
- lexer->advance(lexer, false);
60
+ if (!scan_preproc_newline(lexer, false)) return false;
59
61
  } else if (lexer->lookahead == '/') {
60
62
  lexer->advance(lexer, false);
61
63
  if (lexer->lookahead == '/') {
@@ -298,7 +300,7 @@ static bool preproc_word(int32_t c, bool continuation) {
298
300
  }
299
301
 
300
302
  static bool scan_pragma_word(TSLexer *lexer) {
301
- const char *word = "_Pragma";
303
+ const char *word = pragma_word;
302
304
  for (; *word; word++) {
303
305
  if (lexer->lookahead != *word) return false;
304
306
  lexer->advance(lexer, false);
@@ -331,4 +333,49 @@ static bool pragma_space(int32_t c) {
331
333
  c == 0x2028 || c == 0x2029 || c == 0x202F || c == 0x205F || c == 0x3000;
332
334
  }
333
335
 
336
+ static bool scan_function_macro_name(TSLexer *lexer, bool allow_pragma, TSSymbol pragma_symbol) {
337
+ bool has_name = false;
338
+ bool is_pragma = true;
339
+ unsigned pragma_length = 0;
340
+ for (;;) {
341
+ while (!has_name && pragma_space(lexer->lookahead)) lexer->advance(lexer, true);
342
+ int32_t c = lexer->lookahead;
343
+ if (c == '\\') {
344
+ lexer->advance(lexer, false);
345
+ if (scan_preproc_newline(lexer, !has_name)) {
346
+ if (!has_name) continue;
347
+ if (!scan_preproc_splices(lexer)) return false;
348
+ break;
349
+ }
350
+ is_pragma = false;
351
+ unsigned digits = lexer->lookahead == 'u' ? 4 : lexer->lookahead == 'U' ? 8 : 0;
352
+ if (!digits) return false;
353
+ lexer->advance(lexer, false);
354
+ for (unsigned i = 0; i < digits; i++) {
355
+ c = lexer->lookahead;
356
+ if (!((c >= '0' && c <= '9') || (c >= 'a' && c <= 'f') || (c >= 'A' && c <= 'F'))) return false;
357
+ lexer->advance(lexer, false);
358
+ }
359
+ } else {
360
+ bool identifier = has_name
361
+ ? set_contains(preproc_identifier_continue, sizeof(preproc_identifier_continue) / sizeof(TSCharacterRange), c)
362
+ : set_contains(preproc_identifier_start, sizeof(preproc_identifier_start) / sizeof(TSCharacterRange), c);
363
+ if (!identifier) break;
364
+ if (is_pragma) {
365
+ if (pragma_length < sizeof(pragma_word) - 1 && c == pragma_word[pragma_length]) pragma_length++;
366
+ else is_pragma = false;
367
+ }
368
+ lexer->advance(lexer, false);
369
+ }
370
+ has_name = true;
371
+ lexer->mark_end(lexer);
372
+ }
373
+ bool adjacent = has_name && lexer->lookahead == '(';
374
+ if (allow_pragma && is_pragma && pragma_length == sizeof(pragma_word) - 1 && scan_pragma_suffix(lexer)) {
375
+ lexer->result_symbol = pragma_symbol;
376
+ return true;
377
+ }
378
+ return adjacent;
379
+ }
380
+
334
381
  #endif
package/src/scanner.c CHANGED
@@ -1,6 +1,6 @@
1
1
  #include "pragma.h"
2
2
 
3
- enum TokenType { PRAGMA_OPERATOR, PREPROC_ARG, PREPROC_NEWLINE, PREPROC_LPAREN, PREPROC_DIRECTIVE_ARG };
3
+ enum TokenType { PRAGMA_OPERATOR, PREPROC_ARG, PREPROC_NEWLINE, PREPROC_LPAREN, PREPROC_DIRECTIVE_ARG, PREPROC_FUNCTION_NAME };
4
4
 
5
5
  void *tree_sitter_c_external_scanner_create(void) {
6
6
  return NULL;
@@ -24,13 +24,23 @@ void tree_sitter_c_external_scanner_deserialize(void *payload, const char *buffe
24
24
 
25
25
  bool tree_sitter_c_external_scanner_scan(void *payload, TSLexer *lexer, const bool *valid_symbols) {
26
26
  (void)payload;
27
+ if (valid_symbols[PREPROC_FUNCTION_NAME] && !valid_symbols[PREPROC_LPAREN]) {
28
+ lexer->result_symbol = PREPROC_FUNCTION_NAME;
29
+ return scan_function_macro_name(lexer, valid_symbols[PRAGMA_OPERATOR], PRAGMA_OPERATOR);
30
+ }
27
31
  bool directive_text = !valid_symbols[PREPROC_ARG] && valid_symbols[PREPROC_DIRECTIVE_ARG];
28
32
  TSSymbol argument_symbol = directive_text ? PREPROC_DIRECTIVE_ARG : PREPROC_ARG;
29
- if (valid_symbols[PREPROC_LPAREN] && lexer->lookahead == '(') {
30
- lexer->advance(lexer, false);
31
- lexer->mark_end(lexer);
32
- lexer->result_symbol = PREPROC_LPAREN;
33
- return true;
33
+ if (valid_symbols[PREPROC_LPAREN]) {
34
+ while (lexer->lookahead == '\\') {
35
+ lexer->advance(lexer, true);
36
+ if (!scan_preproc_newline(lexer, true)) return false;
37
+ }
38
+ if (lexer->lookahead == '(') {
39
+ lexer->advance(lexer, false);
40
+ lexer->mark_end(lexer);
41
+ lexer->result_symbol = PREPROC_LPAREN;
42
+ return true;
43
+ }
34
44
  }
35
45
  if (valid_symbols[PREPROC_NEWLINE]) {
36
46
  for (;;) {
Binary file
package/tree-sitter.json CHANGED
@@ -16,7 +16,7 @@
16
16
  }
17
17
  ],
18
18
  "metadata": {
19
- "version": "1.5.1",
19
+ "version": "2.0.0",
20
20
  "license": "MIT",
21
21
  "description": "C grammar for tree-sitter",
22
22
  "authors": [