tree-sitter-lispex 0.1.0 → 0.1.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/grammar.js +6 -4
- package/package.json +2 -1
- package/queries/highlights.scm +1 -2
- package/src/grammar.json +28 -17
- package/src/node-types.json +8 -5
- package/src/parser.c +1734 -1680
- package/src/scanner.c +25 -29
- package/src/unicode.h +864 -0
- package/tree-sitter.json +1 -1
package/src/scanner.c
CHANGED
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
#include "tree_sitter/parser.h"
|
|
2
|
+
#include "unicode.h"
|
|
2
3
|
|
|
3
4
|
#include <stdbool.h>
|
|
4
5
|
#include <stdint.h>
|
|
@@ -8,6 +9,8 @@
|
|
|
8
9
|
enum TokenType {
|
|
9
10
|
BLOCK_COMMENT,
|
|
10
11
|
CHARACTER,
|
|
12
|
+
MALFORMED_CHARACTER,
|
|
13
|
+
CHARACTER_END,
|
|
11
14
|
BOOLEAN,
|
|
12
15
|
BYTE,
|
|
13
16
|
INTEGER,
|
|
@@ -36,8 +39,11 @@ void tree_sitter_lispex_external_scanner_deserialize(void *payload,
|
|
|
36
39
|
}
|
|
37
40
|
|
|
38
41
|
static bool is_space(int32_t c) {
|
|
39
|
-
|
|
40
|
-
|
|
42
|
+
// Rust char::is_whitespace (Unicode White_Space); U+FEFF is a constituent.
|
|
43
|
+
return (c >= 0x09 && c <= 0x0D) || c == 0x20 || c == 0x85 ||
|
|
44
|
+
c == 0xA0 || c == 0x1680 || (c >= 0x2000 && c <= 0x200A) ||
|
|
45
|
+
c == 0x2028 || c == 0x2029 || c == 0x202F || c == 0x205F ||
|
|
46
|
+
c == 0x3000;
|
|
41
47
|
}
|
|
42
48
|
|
|
43
49
|
static bool is_delimiter(int32_t c) {
|
|
@@ -46,11 +52,6 @@ static bool is_delimiter(int32_t c) {
|
|
|
46
52
|
c == '\'' || c == '`' || c == ',';
|
|
47
53
|
}
|
|
48
54
|
|
|
49
|
-
static bool is_ascii_alphanumeric(int32_t c) {
|
|
50
|
-
return (c >= 'a' && c <= 'z') || (c >= 'A' && c <= 'Z') ||
|
|
51
|
-
(c >= '0' && c <= '9');
|
|
52
|
-
}
|
|
53
|
-
|
|
54
55
|
static bool is_hex(int32_t c) {
|
|
55
56
|
return (c >= '0' && c <= '9') || (c >= 'a' && c <= 'f') ||
|
|
56
57
|
(c >= 'A' && c <= 'F');
|
|
@@ -108,38 +109,32 @@ static bool scan_character_after_hash(TSLexer *lexer) {
|
|
|
108
109
|
|
|
109
110
|
int32_t first = lexer->lookahead;
|
|
110
111
|
lexer->advance(lexer, false);
|
|
111
|
-
if (!
|
|
112
|
+
if (!is_unicode_alphanumeric(first)) {
|
|
112
113
|
lexer->result_symbol = CHARACTER;
|
|
113
114
|
return true;
|
|
114
115
|
}
|
|
115
116
|
|
|
116
117
|
char name[64];
|
|
117
118
|
size_t length = 0;
|
|
118
|
-
|
|
119
|
+
bool ascii_only = first < 128;
|
|
120
|
+
bool hex = first == 'x' || first == 'X';
|
|
121
|
+
bool has_tail = false;
|
|
122
|
+
name[length++] = ascii_only ? (char)first : '\0';
|
|
119
123
|
while (!is_delimiter(lexer->lookahead) && lexer->lookahead != '|') {
|
|
120
|
-
|
|
121
|
-
|
|
124
|
+
has_tail = true;
|
|
125
|
+
ascii_only = ascii_only && lexer->lookahead < 128;
|
|
126
|
+
hex = hex && is_hex(lexer->lookahead);
|
|
127
|
+
// Only named forms need buffering. Consume even arbitrarily long malformed
|
|
128
|
+
// names and hexadecimal spellings as one token, without truncating it.
|
|
129
|
+
if (length + 1 < sizeof(name)) {
|
|
130
|
+
name[length++] = lexer->lookahead < 128 ? (char)lexer->lookahead : '\0';
|
|
122
131
|
}
|
|
123
|
-
name[length++] = (char)lexer->lookahead;
|
|
124
132
|
lexer->advance(lexer, false);
|
|
125
133
|
}
|
|
126
134
|
name[length] = '\0';
|
|
127
135
|
|
|
128
|
-
bool valid =
|
|
129
|
-
|
|
130
|
-
valid = length > 1;
|
|
131
|
-
for (size_t index = 1; index < length && valid; index++) {
|
|
132
|
-
valid = is_hex(name[index]);
|
|
133
|
-
}
|
|
134
|
-
}
|
|
135
|
-
if (!valid) {
|
|
136
|
-
valid = is_named_character(name);
|
|
137
|
-
}
|
|
138
|
-
if (!valid) {
|
|
139
|
-
return false;
|
|
140
|
-
}
|
|
141
|
-
|
|
142
|
-
lexer->result_symbol = CHARACTER;
|
|
136
|
+
bool valid = !has_tail || hex || (ascii_only && is_named_character(name));
|
|
137
|
+
lexer->result_symbol = valid ? CHARACTER : MALFORMED_CHARACTER;
|
|
143
138
|
return true;
|
|
144
139
|
}
|
|
145
140
|
|
|
@@ -329,7 +324,7 @@ static bool scan_atom(TSLexer *lexer, const bool *valid_symbols) {
|
|
|
329
324
|
lexer->result_symbol = INTEGER;
|
|
330
325
|
matched = true;
|
|
331
326
|
} else if (valid_symbols[SYMBOL] && text[0] != '#' &&
|
|
332
|
-
!
|
|
327
|
+
!starts_number_like(text) &&
|
|
333
328
|
!(ascii_only && strcmp(text, ".") == 0)) {
|
|
334
329
|
lexer->result_symbol = SYMBOL;
|
|
335
330
|
matched = true;
|
|
@@ -349,7 +344,8 @@ bool tree_sitter_lispex_external_scanner_scan(void *payload, TSLexer *lexer,
|
|
|
349
344
|
if (lexer->lookahead == '|' && valid_symbols[BLOCK_COMMENT]) {
|
|
350
345
|
return scan_block_comment_after_hash(lexer);
|
|
351
346
|
}
|
|
352
|
-
if (lexer->lookahead == '\\' &&
|
|
347
|
+
if (lexer->lookahead == '\\' &&
|
|
348
|
+
(valid_symbols[CHARACTER] || valid_symbols[MALFORMED_CHARACTER])) {
|
|
353
349
|
return scan_character_after_hash(lexer);
|
|
354
350
|
}
|
|
355
351
|
return scan_hash_token_after_hash(lexer, valid_symbols);
|