@willbooster/tree-sitter-c 1.5.2 → 2.0.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -59,6 +59,10 @@ parser.setLanguage(await Language.load(c));
59
59
 
60
60
  The package also ships `grammar.js`, the scanner sources `src/scanner.c`, `src/pragma.h` and `src/identifier.h`, the queries in `queries/`,
61
61
  and the node types in `src/node-types.json` for grammars extending C (such as C++).
62
+ Derived grammars must handle every inherited external token in their scanner, including `_preproc_function_name`.
63
+ Port both the `_preproc_function_name` dispatch and the `_preproc_lparen` splice-skipping loop from `src/scanner.c`,
64
+ using `scan_function_macro_name` from `src/pragma.h`. The macro-name token ends before the splices used to determine
65
+ adjacency, so the parameter scanner must skip them again to reach `(`.
62
66
 
63
67
  In Rust, depend on the [crate](https://crates.io/crates/willbooster-tree-sitter-c) and on
64
68
  [willbooster-tree-sitter](https://crates.io/crates/willbooster-tree-sitter), the runtime this package is tested and
@@ -68,7 +72,7 @@ malformed input):
68
72
  ```toml
69
73
  [dependencies]
70
74
  tree-sitter = { package = "willbooster-tree-sitter", version = "1" }
71
- tree-sitter-c = { package = "willbooster-tree-sitter-c", version = "1" }
75
+ tree-sitter-c = { package = "willbooster-tree-sitter-c", version = "2" }
72
76
  ```
73
77
 
74
78
  ```rust
package/grammar.js CHANGED
@@ -36,7 +36,7 @@ const PREC = {
36
36
  };
37
37
 
38
38
  const LINE_COMMENT = seq('//', /(\\+(.|\r?\n)|[^\\\n])*/);
39
- const PRAGMA_SPACING = repeat(choice(/\s/, /\\\r?\n/));
39
+ const PRAGMA_SPACING = repeat(choice(/\s/, /\\(?:\r\n?|\n\r?)/));
40
40
  const PREPROC_ARGUMENT = /\S([^/\n]|\/[^*]|\\\r?\n)*/;
41
41
  const VA_ARG_KEYWORDS = choice('va_arg', '__builtin_va_arg');
42
42
 
@@ -92,9 +92,10 @@ module.exports = Object.assign(
92
92
  sym('_preproc_newline'),
93
93
  sym('_preproc_lparen'),
94
94
  sym('_preproc_directive_arg'),
95
+ sym('_preproc_function_name'),
95
96
  ],
96
97
 
97
- extras: ($) => [$.pragma_operator, /\s|\\\r?\n/, $.comment],
98
+ extras: ($) => [$.pragma_operator, /\s|\\(?:\r\n?|\n\r?)/, $.comment],
98
99
 
99
100
  inline: ($) => [
100
101
  $._non_identifier_type_specifier,
@@ -169,7 +170,7 @@ module.exports = Object.assign(
169
170
  PRAGMA_SPACING,
170
171
  optional(choice('L', 'u8', 'u', 'U')),
171
172
  '"',
172
- repeat(choice(/[^\\"\n]/, seq('\\', choice(/./, /\r?\n/)))),
173
+ repeat(choice(/[^\\"\r\n]/, seq('\\', choice(/[^\r\n]/, /\r\n?|\n\r?/)))),
173
174
  '"',
174
175
  PRAGMA_SPACING,
175
176
  ')'
@@ -203,7 +204,7 @@ module.exports = Object.assign(
203
204
  preproc_function_def: ($) =>
204
205
  seq(
205
206
  preprocessor('define'),
206
- field('name', $.identifier),
207
+ field('name', alias(sym('_preproc_function_name'), $.identifier)),
207
208
  field('parameters', $.preproc_params),
208
209
  field('value', optional($.preproc_arg)),
209
210
  sym('_preproc_newline')
@@ -1430,7 +1431,6 @@ module.exports = Object.assign(
1430
1431
 
1431
1432
  identifier: () =>
1432
1433
  /(\p{XID_Start}|\$|_|\\u[0-9A-Fa-f]{4}|\\U[0-9A-Fa-f]{8})(\p{XID_Continue}|\$|\\u[0-9A-Fa-f]{4}|\\U[0-9A-Fa-f]{8})*/u,
1433
-
1434
1434
  _type_identifier: ($) => alias(choice($.identifier, VA_ARG_KEYWORDS), sym('type_identifier')),
1435
1435
  _field_identifier: ($) => alias(choice($.identifier, VA_ARG_KEYWORDS), sym('field_identifier')),
1436
1436
  _statement_identifier: ($) => alias(choice($.identifier, VA_ARG_KEYWORDS), sym('statement_identifier')),
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@willbooster/tree-sitter-c",
3
- "version": "1.5.2",
3
+ "version": "2.0.1",
4
4
  "description": "C grammar for tree-sitter",
5
5
  "keywords": [
6
6
  "incremental",
@@ -30,6 +30,8 @@
30
30
 
31
31
  "#define" @keyword
32
32
  "#elif" @keyword
33
+ "#elifdef" @keyword
34
+ "#elifndef" @keyword
33
35
  "#else" @keyword
34
36
  "#endif" @keyword
35
37
  "#if" @keyword
package/src/pragma.h CHANGED
@@ -4,7 +4,10 @@
4
4
  #include "tree_sitter/parser.h"
5
5
  #include "identifier.h"
6
6
 
7
+ static const char pragma_word[] = "_Pragma";
8
+
7
9
  static bool scan_pragma_spacing(TSLexer *lexer);
10
+ static bool scan_pragma_suffix(TSLexer *lexer);
8
11
  static bool pragma_space(int32_t c);
9
12
  static bool scan_pragma_word(TSLexer *lexer);
10
13
  static bool scan_preproc_newline(TSLexer *lexer, bool skip);
@@ -18,6 +21,10 @@ static bool scan_pragma(TSLexer *lexer) {
18
21
  lexer->advance(lexer, true);
19
22
  }
20
23
  if (!scan_pragma_word(lexer)) return false;
24
+ return scan_pragma_suffix(lexer);
25
+ }
26
+
27
+ static bool scan_pragma_suffix(TSLexer *lexer) {
21
28
  if (!scan_pragma_spacing(lexer) || lexer->lookahead != '(') return false;
22
29
  lexer->advance(lexer, false);
23
30
  if (!scan_pragma_spacing(lexer)) return false;
@@ -33,10 +40,7 @@ static bool scan_pragma(TSLexer *lexer) {
33
40
  if (lexer->lookahead == '\\') {
34
41
  lexer->advance(lexer, false);
35
42
  if (lexer->eof(lexer)) return false;
36
- if (lexer->lookahead == '\r') {
37
- lexer->advance(lexer, false);
38
- if (lexer->lookahead != '\n') return false;
39
- }
43
+ if (scan_preproc_newline(lexer, false)) continue;
40
44
  }
41
45
  lexer->advance(lexer, false);
42
46
  }
@@ -53,9 +57,7 @@ static bool scan_pragma_spacing(TSLexer *lexer) {
53
57
  lexer->advance(lexer, false);
54
58
  } else if (lexer->lookahead == '\\') {
55
59
  lexer->advance(lexer, false);
56
- if (lexer->lookahead == '\r') lexer->advance(lexer, false);
57
- if (lexer->lookahead != '\n') return false;
58
- lexer->advance(lexer, false);
60
+ if (!scan_preproc_newline(lexer, false)) return false;
59
61
  } else if (lexer->lookahead == '/') {
60
62
  lexer->advance(lexer, false);
61
63
  if (lexer->lookahead == '/') {
@@ -298,7 +300,7 @@ static bool preproc_word(int32_t c, bool continuation) {
298
300
  }
299
301
 
300
302
  static bool scan_pragma_word(TSLexer *lexer) {
301
- const char *word = "_Pragma";
303
+ const char *word = pragma_word;
302
304
  for (; *word; word++) {
303
305
  if (lexer->lookahead != *word) return false;
304
306
  lexer->advance(lexer, false);
@@ -331,4 +333,49 @@ static bool pragma_space(int32_t c) {
331
333
  c == 0x2028 || c == 0x2029 || c == 0x202F || c == 0x205F || c == 0x3000;
332
334
  }
333
335
 
336
+ static bool scan_function_macro_name(TSLexer *lexer, bool allow_pragma, TSSymbol pragma_symbol) {
337
+ bool has_name = false;
338
+ bool is_pragma = true;
339
+ unsigned pragma_length = 0;
340
+ for (;;) {
341
+ while (!has_name && pragma_space(lexer->lookahead)) lexer->advance(lexer, true);
342
+ int32_t c = lexer->lookahead;
343
+ if (c == '\\') {
344
+ lexer->advance(lexer, false);
345
+ if (scan_preproc_newline(lexer, !has_name)) {
346
+ if (!has_name) continue;
347
+ if (!scan_preproc_splices(lexer)) return false;
348
+ break;
349
+ }
350
+ is_pragma = false;
351
+ unsigned digits = lexer->lookahead == 'u' ? 4 : lexer->lookahead == 'U' ? 8 : 0;
352
+ if (!digits) return false;
353
+ lexer->advance(lexer, false);
354
+ for (unsigned i = 0; i < digits; i++) {
355
+ c = lexer->lookahead;
356
+ if (!((c >= '0' && c <= '9') || (c >= 'a' && c <= 'f') || (c >= 'A' && c <= 'F'))) return false;
357
+ lexer->advance(lexer, false);
358
+ }
359
+ } else {
360
+ bool identifier = has_name
361
+ ? set_contains(preproc_identifier_continue, sizeof(preproc_identifier_continue) / sizeof(TSCharacterRange), c)
362
+ : set_contains(preproc_identifier_start, sizeof(preproc_identifier_start) / sizeof(TSCharacterRange), c);
363
+ if (!identifier) break;
364
+ if (is_pragma) {
365
+ if (pragma_length < sizeof(pragma_word) - 1 && c == pragma_word[pragma_length]) pragma_length++;
366
+ else is_pragma = false;
367
+ }
368
+ lexer->advance(lexer, false);
369
+ }
370
+ has_name = true;
371
+ lexer->mark_end(lexer);
372
+ }
373
+ bool adjacent = has_name && lexer->lookahead == '(';
374
+ if (allow_pragma && is_pragma && pragma_length == sizeof(pragma_word) - 1 && scan_pragma_suffix(lexer)) {
375
+ lexer->result_symbol = pragma_symbol;
376
+ return true;
377
+ }
378
+ return adjacent;
379
+ }
380
+
334
381
  #endif
package/src/scanner.c CHANGED
@@ -1,6 +1,6 @@
1
1
  #include "pragma.h"
2
2
 
3
- enum TokenType { PRAGMA_OPERATOR, PREPROC_ARG, PREPROC_NEWLINE, PREPROC_LPAREN, PREPROC_DIRECTIVE_ARG };
3
+ enum TokenType { PRAGMA_OPERATOR, PREPROC_ARG, PREPROC_NEWLINE, PREPROC_LPAREN, PREPROC_DIRECTIVE_ARG, PREPROC_FUNCTION_NAME };
4
4
 
5
5
  void *tree_sitter_c_external_scanner_create(void) {
6
6
  return NULL;
@@ -24,13 +24,23 @@ void tree_sitter_c_external_scanner_deserialize(void *payload, const char *buffe
24
24
 
25
25
  bool tree_sitter_c_external_scanner_scan(void *payload, TSLexer *lexer, const bool *valid_symbols) {
26
26
  (void)payload;
27
+ if (valid_symbols[PREPROC_FUNCTION_NAME] && !valid_symbols[PREPROC_LPAREN]) {
28
+ lexer->result_symbol = PREPROC_FUNCTION_NAME;
29
+ return scan_function_macro_name(lexer, valid_symbols[PRAGMA_OPERATOR], PRAGMA_OPERATOR);
30
+ }
27
31
  bool directive_text = !valid_symbols[PREPROC_ARG] && valid_symbols[PREPROC_DIRECTIVE_ARG];
28
32
  TSSymbol argument_symbol = directive_text ? PREPROC_DIRECTIVE_ARG : PREPROC_ARG;
29
- if (valid_symbols[PREPROC_LPAREN] && lexer->lookahead == '(') {
30
- lexer->advance(lexer, false);
31
- lexer->mark_end(lexer);
32
- lexer->result_symbol = PREPROC_LPAREN;
33
- return true;
33
+ if (valid_symbols[PREPROC_LPAREN]) {
34
+ while (lexer->lookahead == '\\') {
35
+ lexer->advance(lexer, true);
36
+ if (!scan_preproc_newline(lexer, true)) return false;
37
+ }
38
+ if (lexer->lookahead == '(') {
39
+ lexer->advance(lexer, false);
40
+ lexer->mark_end(lexer);
41
+ lexer->result_symbol = PREPROC_LPAREN;
42
+ return true;
43
+ }
34
44
  }
35
45
  if (valid_symbols[PREPROC_NEWLINE]) {
36
46
  for (;;) {
Binary file
package/tree-sitter.json CHANGED
@@ -16,7 +16,7 @@
16
16
  }
17
17
  ],
18
18
  "metadata": {
19
- "version": "1.5.2",
19
+ "version": "2.0.1",
20
20
  "license": "MIT",
21
21
  "description": "C grammar for tree-sitter",
22
22
  "authors": [