tree-sitter-lispex 0.1.0 → 0.1.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/scanner.c CHANGED
@@ -1,4 +1,5 @@
1
1
  #include "tree_sitter/parser.h"
2
+ #include "unicode.h"
2
3
 
3
4
  #include <stdbool.h>
4
5
  #include <stdint.h>
@@ -8,6 +9,8 @@
8
9
  enum TokenType {
9
10
  BLOCK_COMMENT,
10
11
  CHARACTER,
12
+ MALFORMED_CHARACTER,
13
+ CHARACTER_END,
11
14
  BOOLEAN,
12
15
  BYTE,
13
16
  INTEGER,
@@ -36,8 +39,11 @@ void tree_sitter_lispex_external_scanner_deserialize(void *payload,
36
39
  }
37
40
 
38
41
  static bool is_space(int32_t c) {
39
- return c == ' ' || c == '\t' || c == '\r' || c == '\n' || c == '\f' ||
40
- c == 0xFEFF;
42
+ // Rust char::is_whitespace (Unicode White_Space); U+FEFF is a constituent.
43
+ return (c >= 0x09 && c <= 0x0D) || c == 0x20 || c == 0x85 ||
44
+ c == 0xA0 || c == 0x1680 || (c >= 0x2000 && c <= 0x200A) ||
45
+ c == 0x2028 || c == 0x2029 || c == 0x202F || c == 0x205F ||
46
+ c == 0x3000;
41
47
  }
42
48
 
43
49
  static bool is_delimiter(int32_t c) {
@@ -46,11 +52,6 @@ static bool is_delimiter(int32_t c) {
46
52
  c == '\'' || c == '`' || c == ',';
47
53
  }
48
54
 
49
- static bool is_ascii_alphanumeric(int32_t c) {
50
- return (c >= 'a' && c <= 'z') || (c >= 'A' && c <= 'Z') ||
51
- (c >= '0' && c <= '9');
52
- }
53
-
54
55
  static bool is_hex(int32_t c) {
55
56
  return (c >= '0' && c <= '9') || (c >= 'a' && c <= 'f') ||
56
57
  (c >= 'A' && c <= 'F');
@@ -108,38 +109,32 @@ static bool scan_character_after_hash(TSLexer *lexer) {
108
109
 
109
110
  int32_t first = lexer->lookahead;
110
111
  lexer->advance(lexer, false);
111
- if (!is_ascii_alphanumeric(first) || first >= 128) {
112
+ if (!is_unicode_alphanumeric(first)) {
112
113
  lexer->result_symbol = CHARACTER;
113
114
  return true;
114
115
  }
115
116
 
116
117
  char name[64];
117
118
  size_t length = 0;
118
- name[length++] = (char)first;
119
+ bool ascii_only = first < 128;
120
+ bool hex = first == 'x' || first == 'X';
121
+ bool has_tail = false;
122
+ name[length++] = ascii_only ? (char)first : '\0';
119
123
  while (!is_delimiter(lexer->lookahead) && lexer->lookahead != '|') {
120
- if (lexer->lookahead >= 128 || length + 1 >= sizeof(name)) {
121
- return false;
124
+ has_tail = true;
125
+ ascii_only = ascii_only && lexer->lookahead < 128;
126
+ hex = hex && is_hex(lexer->lookahead);
127
+ // Only named forms need buffering. Consume even arbitrarily long malformed
128
+ // names and hexadecimal spellings as one token, without truncating it.
129
+ if (length + 1 < sizeof(name)) {
130
+ name[length++] = lexer->lookahead < 128 ? (char)lexer->lookahead : '\0';
122
131
  }
123
- name[length++] = (char)lexer->lookahead;
124
132
  lexer->advance(lexer, false);
125
133
  }
126
134
  name[length] = '\0';
127
135
 
128
- bool valid = length == 1;
129
- if (!valid && (name[0] == 'x' || name[0] == 'X')) {
130
- valid = length > 1;
131
- for (size_t index = 1; index < length && valid; index++) {
132
- valid = is_hex(name[index]);
133
- }
134
- }
135
- if (!valid) {
136
- valid = is_named_character(name);
137
- }
138
- if (!valid) {
139
- return false;
140
- }
141
-
142
- lexer->result_symbol = CHARACTER;
136
+ bool valid = !has_tail || hex || (ascii_only && is_named_character(name));
137
+ lexer->result_symbol = valid ? CHARACTER : MALFORMED_CHARACTER;
143
138
  return true;
144
139
  }
145
140
 
@@ -329,7 +324,7 @@ static bool scan_atom(TSLexer *lexer, const bool *valid_symbols) {
329
324
  lexer->result_symbol = INTEGER;
330
325
  matched = true;
331
326
  } else if (valid_symbols[SYMBOL] && text[0] != '#' &&
332
- !(ascii_only && starts_number_like(text)) &&
327
+ !starts_number_like(text) &&
333
328
  !(ascii_only && strcmp(text, ".") == 0)) {
334
329
  lexer->result_symbol = SYMBOL;
335
330
  matched = true;
@@ -349,7 +344,8 @@ bool tree_sitter_lispex_external_scanner_scan(void *payload, TSLexer *lexer,
349
344
  if (lexer->lookahead == '|' && valid_symbols[BLOCK_COMMENT]) {
350
345
  return scan_block_comment_after_hash(lexer);
351
346
  }
352
- if (lexer->lookahead == '\\' && valid_symbols[CHARACTER]) {
347
+ if (lexer->lookahead == '\\' &&
348
+ (valid_symbols[CHARACTER] || valid_symbols[MALFORMED_CHARACTER])) {
353
349
  return scan_character_after_hash(lexer);
354
350
  }
355
351
  return scan_hash_token_after_hash(lexer, valid_symbols);