tree-sitter-ktav 0.2.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +185 -0
- package/README.ru.md +185 -0
- package/README.zh.md +179 -0
- package/binding.gyp +28 -0
- package/bindings/node/binding.cc +20 -0
- package/bindings/node/index.d.ts +28 -0
- package/bindings/node/index.js +7 -0
- package/grammar.js +255 -0
- package/package.json +66 -0
- package/queries/highlights.scm +85 -0
- package/queries/injections.scm +10 -0
- package/queries/locals.scm +12 -0
- package/src/grammar.json +818 -0
- package/src/node-types.json +479 -0
- package/src/parser.c +2874 -0
- package/src/scanner.c +209 -0
- package/src/tree_sitter/alloc.h +54 -0
- package/src/tree_sitter/array.h +330 -0
- package/src/tree_sitter/parser.h +286 -0
package/grammar.js
ADDED
|
@@ -0,0 +1,255 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Tree-sitter grammar for Ktav (כְּתָב) — the Written Configuration Format.
|
|
3
|
+
*
|
|
4
|
+
* Spec: https://github.com/ktav-lang/spec/blob/main/versions/0.1/spec.md
|
|
5
|
+
*
|
|
6
|
+
* Ktav is line-oriented. Every line is one of:
|
|
7
|
+
* - blank
|
|
8
|
+
* - a comment (`# ...`)
|
|
9
|
+
* - a key:value pair (with markers `:`, `::`, `:i`, `:f`)
|
|
10
|
+
* - a structural opener / closer for compounds (`{`, `}`, `[`, `]`,
|
|
11
|
+
* `(`, `((`, `)`, `))`)
|
|
12
|
+
* - an array item (inside an open `[` array)
|
|
13
|
+
* - raw content of a multi-line string
|
|
14
|
+
*
|
|
15
|
+
* Strategy:
|
|
16
|
+
* - Newlines are explicit (`_newline`) and structural openers/closers
|
|
17
|
+
* are tokens that include the trailing whitespace + newline so they
|
|
18
|
+
* can NEVER be confused with a scalar starting with the same byte.
|
|
19
|
+
* - The four pair separators (`:`, `::`, `:i`, `:f`) are recognized
|
|
20
|
+
* by the lexer with longest-match precedence (`::` > `:`, `:i`/`:f`
|
|
21
|
+
* > `:`).
|
|
22
|
+
* - The mandatory-whitespace-after-marker rule (§ 6.10) is enforced
|
|
23
|
+
* by an external scanner token `_marker_ws`, which is a zero-width
|
|
24
|
+
* assertion that only succeeds when the byte right after the
|
|
25
|
+
* separator is space, tab, CR, LF, or EOF. `key:value` (no space)
|
|
26
|
+
* therefore fails to parse.
|
|
27
|
+
* - The closer-on-its-own-line rule for compounds and multi-line
|
|
28
|
+
* strings is enforced via the external scanner's `_strict_eol`
|
|
29
|
+
* token, which only matches `[ \t]*\r?\n` (or EOF) — any non-
|
|
30
|
+
* whitespace text between the closer and the line terminator is
|
|
31
|
+
* a parse error.
|
|
32
|
+
* - Multi-line string content is captured as a sequence of opaque
|
|
33
|
+
* "raw lines" up to the matching terminator.
|
|
34
|
+
* - Indentation is not significant (matches the spec).
|
|
35
|
+
*/
|
|
36
|
+
|
|
37
|
+
module.exports = grammar({
|
|
38
|
+
name: 'ktav',
|
|
39
|
+
|
|
40
|
+
extras: $ => [
|
|
41
|
+
// Inline horizontal whitespace is insignificant between tokens
|
|
42
|
+
// on the same line. Newlines are explicit (`_newline`).
|
|
43
|
+
/[ \t]+/,
|
|
44
|
+
],
|
|
45
|
+
|
|
46
|
+
externals: $ => [
|
|
47
|
+
$._marker_ws, // zero-width assertion after pair separators
|
|
48
|
+
$._strict_eol, // [ \t]*\r?\n (or EOF) — for compound closers
|
|
49
|
+
$._stripped_close, // `)[ \t]*\r?\n` (or EOF) — only valid inside `(...)` body
|
|
50
|
+
$._verbatim_close, // `))[ \t]*\r?\n` (or EOF) — only valid inside `((...))` body
|
|
51
|
+
],
|
|
52
|
+
|
|
53
|
+
conflicts: $ => [],
|
|
54
|
+
|
|
55
|
+
word: $ => $._key_segment,
|
|
56
|
+
|
|
57
|
+
rules: {
|
|
58
|
+
source_file: $ => repeat($._line),
|
|
59
|
+
|
|
60
|
+
// ---- Top-level lines ----
|
|
61
|
+
_line: $ => choice(
|
|
62
|
+
$.comment,
|
|
63
|
+
$.blank_line,
|
|
64
|
+
$.object_pair,
|
|
65
|
+
),
|
|
66
|
+
|
|
67
|
+
blank_line: $ => $._newline,
|
|
68
|
+
|
|
69
|
+
_newline: $ => /\r?\n/,
|
|
70
|
+
|
|
71
|
+
// ---- Comment ----
|
|
72
|
+
comment: $ => seq(
|
|
73
|
+
'#',
|
|
74
|
+
optional(/[^\r\n]*/),
|
|
75
|
+
$._newline,
|
|
76
|
+
),
|
|
77
|
+
|
|
78
|
+
// ---- Object pair ----
|
|
79
|
+
//
|
|
80
|
+
// After the separator, the external `_marker_ws` token asserts
|
|
81
|
+
// that the next byte is whitespace, CR, LF, or EOF. This enforces
|
|
82
|
+
// § 6.10 (mandatory whitespace after marker): `key:value` fails.
|
|
83
|
+
object_pair: $ => choice(
|
|
84
|
+
// After `::`, `:i`, `:f` the body is a literal/typed scalar — § 5.2:
|
|
85
|
+
// it is NOT dispatched through compound-opener / multi-line dispatch.
|
|
86
|
+
// Only `empty_value` (immediate newline) or `scalar` is allowed.
|
|
87
|
+
seq(
|
|
88
|
+
field('key', $.key),
|
|
89
|
+
field('separator', choice($.sep_raw, $.sep_int, $.sep_float)),
|
|
90
|
+
$._marker_ws,
|
|
91
|
+
field('value', choice($.empty_value, $.scalar)),
|
|
92
|
+
),
|
|
93
|
+
// After `:` the body goes through the full § 5.2 dispatch.
|
|
94
|
+
seq(
|
|
95
|
+
field('key', $.key),
|
|
96
|
+
field('separator', $.sep_string),
|
|
97
|
+
$._marker_ws,
|
|
98
|
+
choice(
|
|
99
|
+
field('value', $.empty_value),
|
|
100
|
+
field('value', $._value_line),
|
|
101
|
+
),
|
|
102
|
+
),
|
|
103
|
+
),
|
|
104
|
+
|
|
105
|
+
// Tree-sitter prefers longest token match. To make sure `::` wins
|
|
106
|
+
// over `:`, `::` is given higher precedence; same for `:i`/`:f`.
|
|
107
|
+
sep_raw: $ => token(prec(3, '::')),
|
|
108
|
+
sep_int: $ => token(prec(2, ':i')),
|
|
109
|
+
sep_float: $ => token(prec(2, ':f')),
|
|
110
|
+
sep_string: $ => token(prec(1, ':')),
|
|
111
|
+
|
|
112
|
+
// ---- Keys ----
|
|
113
|
+
key: $ => choice(
|
|
114
|
+
$._key_segment,
|
|
115
|
+
$.dotted_key,
|
|
116
|
+
),
|
|
117
|
+
|
|
118
|
+
dotted_key: $ => prec.left(seq(
|
|
119
|
+
$._key_segment,
|
|
120
|
+
repeat1(seq('.', $._key_segment)),
|
|
121
|
+
)),
|
|
122
|
+
|
|
123
|
+
// Key segment: any chars except whitespace, "[", "]", "{", "}",
|
|
124
|
+
// ":", "#", ".".
|
|
125
|
+
_key_segment: $ => /[^\s\[\]\{\}:#.\r\n]+/,
|
|
126
|
+
|
|
127
|
+
// ---- Value line ----
|
|
128
|
+
_value_line: $ => choice(
|
|
129
|
+
// Compound openers (eat the newline).
|
|
130
|
+
$.compound_object,
|
|
131
|
+
$.compound_array,
|
|
132
|
+
$.multiline_stripped,
|
|
133
|
+
$.multiline_verbatim,
|
|
134
|
+
// Inline empty compound forms (followed by newline).
|
|
135
|
+
$.empty_object,
|
|
136
|
+
$.empty_array,
|
|
137
|
+
$.empty_paren,
|
|
138
|
+
$.empty_double_paren,
|
|
139
|
+
// Keywords (single token followed by newline).
|
|
140
|
+
$.keyword,
|
|
141
|
+
// Scalar — catch-all line content.
|
|
142
|
+
$.scalar,
|
|
143
|
+
),
|
|
144
|
+
|
|
145
|
+
// Empty value = separator immediately followed by newline.
|
|
146
|
+
empty_value: $ => $._newline,
|
|
147
|
+
|
|
148
|
+
// ---- Empty inline compounds (one full line) ----
|
|
149
|
+
empty_object: $ => seq(token(prec(5, '{}')), $._newline),
|
|
150
|
+
empty_array: $ => seq(token(prec(5, '[]')), $._newline),
|
|
151
|
+
empty_paren: $ => seq(token(prec(5, '()')), $._newline),
|
|
152
|
+
empty_double_paren: $ => seq(token(prec(5, '(())')), $._newline),
|
|
153
|
+
|
|
154
|
+
// ---- Multi-line compounds ----
|
|
155
|
+
//
|
|
156
|
+
// Openers are tokens that include the rest of the line up to and
|
|
157
|
+
// including the newline, so they cannot be confused with a scalar
|
|
158
|
+
// starting with the same byte. Closers, by contrast, are split
|
|
159
|
+
// into the bracket character(s) plus the external `_strict_eol`
|
|
160
|
+
// token; this lets the scanner reject pathological lines like
|
|
161
|
+
// `} trailing_text\n` (§ 5.6.1 and the cleanliness rule for
|
|
162
|
+
// object/array closers).
|
|
163
|
+
open_brace: $ => token(prec(4, /\{[ \t]*\r?\n/)),
|
|
164
|
+
close_brace: $ => seq(token(prec(4, '}')), $._strict_eol),
|
|
165
|
+
open_bracket: $ => token(prec(4, /\[[ \t]*\r?\n/)),
|
|
166
|
+
close_bracket: $ => seq(token(prec(4, ']')), $._strict_eol),
|
|
167
|
+
open_paren: $ => token(prec(4, /\([ \t]*\r?\n/)),
|
|
168
|
+
open_dparen: $ => token(prec(5, /\(\([ \t]*\r?\n/)),
|
|
169
|
+
// close_paren / close_dparen are emitted by the external scanner as
|
|
170
|
+
// `_stripped_close` / `_verbatim_close`. They are context-sensitive:
|
|
171
|
+
// inside `(...)` only `)` closes (a `))` line is content); inside
|
|
172
|
+
// `((...))` only `))` closes (a single `)` line is content). The
|
|
173
|
+
// scanner consults `valid_symbols` to decide which form to attempt.
|
|
174
|
+
close_paren: $ => $._stripped_close,
|
|
175
|
+
close_dparen: $ => $._verbatim_close,
|
|
176
|
+
|
|
177
|
+
compound_object: $ => seq(
|
|
178
|
+
$.open_brace,
|
|
179
|
+
repeat(choice($.comment, $.blank_line, $.object_pair)),
|
|
180
|
+
$.close_brace,
|
|
181
|
+
),
|
|
182
|
+
|
|
183
|
+
compound_array: $ => seq(
|
|
184
|
+
$.open_bracket,
|
|
185
|
+
repeat(choice($.comment, $.blank_line, $.array_item)),
|
|
186
|
+
$.close_bracket,
|
|
187
|
+
),
|
|
188
|
+
|
|
189
|
+
// ---- Array items ----
|
|
190
|
+
//
|
|
191
|
+
// Marker-prefixed items must, like pair lines, have whitespace or
|
|
192
|
+
// EOL after the marker (§ 6.10).
|
|
193
|
+
array_item: $ => choice(
|
|
194
|
+
seq(
|
|
195
|
+
field('marker', $.sep_raw),
|
|
196
|
+
$._marker_ws,
|
|
197
|
+
field('value', choice($.empty_value, $.scalar)),
|
|
198
|
+
),
|
|
199
|
+
seq(
|
|
200
|
+
field('marker', $.sep_int),
|
|
201
|
+
$._marker_ws,
|
|
202
|
+
field('value', choice($.empty_value, $.scalar)),
|
|
203
|
+
),
|
|
204
|
+
seq(
|
|
205
|
+
field('marker', $.sep_float),
|
|
206
|
+
$._marker_ws,
|
|
207
|
+
field('value', choice($.empty_value, $.scalar)),
|
|
208
|
+
),
|
|
209
|
+
// Plain value item — same set as object pair value.
|
|
210
|
+
field('value', $._value_line),
|
|
211
|
+
),
|
|
212
|
+
|
|
213
|
+
// ---- Multi-line strings ----
|
|
214
|
+
multiline_stripped: $ => seq(
|
|
215
|
+
$.open_paren,
|
|
216
|
+
repeat($.multiline_content_line),
|
|
217
|
+
$.close_paren,
|
|
218
|
+
),
|
|
219
|
+
|
|
220
|
+
multiline_verbatim: $ => seq(
|
|
221
|
+
$.open_dparen,
|
|
222
|
+
repeat($.multiline_content_line),
|
|
223
|
+
$.close_dparen,
|
|
224
|
+
),
|
|
225
|
+
|
|
226
|
+
// A content line inside a multi-line string. Captured as one
|
|
227
|
+
// token. The closer tokens (`)`, `))` plus _strict_eol) win at
|
|
228
|
+
// the LR(1) boundary because they require their own line; any
|
|
229
|
+
// line whose content includes more than just the terminator
|
|
230
|
+
// falls through to this rule.
|
|
231
|
+
multiline_content_line: $ => token(prec(-1, /[^\r\n]*\r?\n/)),
|
|
232
|
+
|
|
233
|
+
// ---- Scalar (default value body, until end of line) ----
|
|
234
|
+
scalar: $ => seq(
|
|
235
|
+
$._scalar_text,
|
|
236
|
+
$._newline,
|
|
237
|
+
),
|
|
238
|
+
|
|
239
|
+
// Scalar text: any non-whitespace, non-newline content up to end
|
|
240
|
+
// of line. `#` is allowed (`color: #ff00ff` is a valid value);
|
|
241
|
+
// the `#` only opens a comment when it is the first non-whitespace
|
|
242
|
+
// char of a line — tree-sitter's context-aware lexer disambiguates
|
|
243
|
+
// because `comment` is only valid at a line-start parse state.
|
|
244
|
+
_scalar_text: $ => /[^\s\r\n][^\r\n]*/,
|
|
245
|
+
|
|
246
|
+
// ---- Keywords ----
|
|
247
|
+
keyword: $ => seq(
|
|
248
|
+
choice($.kw_null, $.kw_true, $.kw_false),
|
|
249
|
+
$._newline,
|
|
250
|
+
),
|
|
251
|
+
kw_null: $ => token(prec(3, 'null')),
|
|
252
|
+
kw_true: $ => token(prec(3, 'true')),
|
|
253
|
+
kw_false: $ => token(prec(3, 'false')),
|
|
254
|
+
},
|
|
255
|
+
});
|
package/package.json
ADDED
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
{
|
|
2
|
+
"name": "tree-sitter-ktav",
|
|
3
|
+
"version": "0.2.1",
|
|
4
|
+
"description": "Tree-sitter grammar for Ktav (כְּתָב) — the Written Configuration Format",
|
|
5
|
+
"keywords": [
|
|
6
|
+
"parser",
|
|
7
|
+
"lexer",
|
|
8
|
+
"tree-sitter",
|
|
9
|
+
"ktav",
|
|
10
|
+
"config",
|
|
11
|
+
"configuration"
|
|
12
|
+
],
|
|
13
|
+
"author": "Marat K <phpcraftdream@gmail.com>",
|
|
14
|
+
"license": "MIT",
|
|
15
|
+
"homepage": "https://github.com/ktav-lang/tree-sitter-ktav",
|
|
16
|
+
"repository": {
|
|
17
|
+
"type": "git",
|
|
18
|
+
"url": "git+https://github.com/ktav-lang/tree-sitter-ktav.git"
|
|
19
|
+
},
|
|
20
|
+
"bugs": {
|
|
21
|
+
"url": "https://github.com/ktav-lang/tree-sitter-ktav/issues"
|
|
22
|
+
},
|
|
23
|
+
"main": "bindings/node",
|
|
24
|
+
"types": "bindings/node",
|
|
25
|
+
"files": [
|
|
26
|
+
"grammar.js",
|
|
27
|
+
"binding.gyp",
|
|
28
|
+
"bindings/node/*",
|
|
29
|
+
"queries/*",
|
|
30
|
+
"src/**"
|
|
31
|
+
],
|
|
32
|
+
"dependencies": {
|
|
33
|
+
"node-addon-api": "^7.1.0",
|
|
34
|
+
"node-gyp-build": "^4.8.0"
|
|
35
|
+
},
|
|
36
|
+
"devDependencies": {
|
|
37
|
+
"tree-sitter-cli": "^0.26.0",
|
|
38
|
+
"prebuildify": "^6.0.0"
|
|
39
|
+
},
|
|
40
|
+
"peerDependencies": {
|
|
41
|
+
"tree-sitter": "^0.21.0"
|
|
42
|
+
},
|
|
43
|
+
"peerDependenciesMeta": {
|
|
44
|
+
"tree-sitter": {
|
|
45
|
+
"optional": true
|
|
46
|
+
}
|
|
47
|
+
},
|
|
48
|
+
"scripts": {
|
|
49
|
+
"install": "node-gyp-build",
|
|
50
|
+
"prestart": "tree-sitter build --wasm",
|
|
51
|
+
"start": "tree-sitter playground",
|
|
52
|
+
"test": "node --test bindings/node/*_test.js"
|
|
53
|
+
},
|
|
54
|
+
"tree-sitter": [
|
|
55
|
+
{
|
|
56
|
+
"scope": "source.ktav",
|
|
57
|
+
"file-types": [
|
|
58
|
+
"ktav"
|
|
59
|
+
],
|
|
60
|
+
"injection-regex": "^ktav$",
|
|
61
|
+
"highlights": "queries/highlights.scm",
|
|
62
|
+
"locals": "queries/locals.scm",
|
|
63
|
+
"injections": "queries/injections.scm"
|
|
64
|
+
}
|
|
65
|
+
]
|
|
66
|
+
}
|
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
; Tree-sitter highlights for Ktav (כְּתָב).
|
|
2
|
+
; Capture names follow the standard set documented at
|
|
3
|
+
; https://docs.helix-editor.com/themes.html#scopes
|
|
4
|
+
; https://github.com/nvim-treesitter/nvim-treesitter (highlights.scm)
|
|
5
|
+
|
|
6
|
+
; ---- Comments ----
|
|
7
|
+
(comment) @comment
|
|
8
|
+
|
|
9
|
+
; ---- Keys ----
|
|
10
|
+
(key) @property
|
|
11
|
+
(dotted_key) @property
|
|
12
|
+
"." @punctuation.delimiter
|
|
13
|
+
|
|
14
|
+
; ---- Pair separators ----
|
|
15
|
+
(sep_string) @punctuation.delimiter
|
|
16
|
+
(sep_raw) @punctuation.special
|
|
17
|
+
(sep_int) @punctuation.special
|
|
18
|
+
(sep_float) @punctuation.special
|
|
19
|
+
|
|
20
|
+
; ---- Compound brackets ----
|
|
21
|
+
; The structural openers / closers each form their own visible token
|
|
22
|
+
; node (the opener swallows trailing horizontal whitespace + newline,
|
|
23
|
+
; and the closer is the bracket char(s) followed by `_strict_eol`).
|
|
24
|
+
; Capturing them directly leaves the inner content uncoloured so
|
|
25
|
+
; nested highlights work correctly in Helix and nvim-treesitter.
|
|
26
|
+
(open_brace) @punctuation.bracket
|
|
27
|
+
(close_brace) @punctuation.bracket
|
|
28
|
+
(open_bracket) @punctuation.bracket
|
|
29
|
+
(close_bracket) @punctuation.bracket
|
|
30
|
+
(open_paren) @string
|
|
31
|
+
(close_paren) @string
|
|
32
|
+
(open_dparen) @string
|
|
33
|
+
(close_dparen) @string
|
|
34
|
+
|
|
35
|
+
; ---- Empty inline forms ----
|
|
36
|
+
(empty_object) @punctuation.bracket
|
|
37
|
+
(empty_array) @punctuation.bracket
|
|
38
|
+
(empty_paren) @string
|
|
39
|
+
(empty_double_paren) @string
|
|
40
|
+
|
|
41
|
+
; ---- Keywords (null / true / false) ----
|
|
42
|
+
(kw_null) @constant.builtin
|
|
43
|
+
(kw_true) @constant.builtin.boolean
|
|
44
|
+
(kw_false) @constant.builtin.boolean
|
|
45
|
+
|
|
46
|
+
; ---- String values ----
|
|
47
|
+
; Plain scalar after `:` — usually a string. (Numeric coercion happens
|
|
48
|
+
; only after `:i`/`:f`; see below.)
|
|
49
|
+
(object_pair
|
|
50
|
+
separator: (sep_string)
|
|
51
|
+
value: (scalar) @string)
|
|
52
|
+
|
|
53
|
+
; Raw string after `::`
|
|
54
|
+
(object_pair
|
|
55
|
+
separator: (sep_raw)
|
|
56
|
+
value: (scalar) @string.special)
|
|
57
|
+
|
|
58
|
+
; Typed integer / float bodies
|
|
59
|
+
(object_pair
|
|
60
|
+
separator: (sep_int)
|
|
61
|
+
value: (scalar) @number)
|
|
62
|
+
|
|
63
|
+
(object_pair
|
|
64
|
+
separator: (sep_float)
|
|
65
|
+
value: (scalar) @number.float)
|
|
66
|
+
|
|
67
|
+
; ---- Array items ----
|
|
68
|
+
(array_item
|
|
69
|
+
marker: (sep_raw)
|
|
70
|
+
value: (scalar) @string.special)
|
|
71
|
+
|
|
72
|
+
(array_item
|
|
73
|
+
marker: (sep_int)
|
|
74
|
+
value: (scalar) @number)
|
|
75
|
+
|
|
76
|
+
(array_item
|
|
77
|
+
marker: (sep_float)
|
|
78
|
+
value: (scalar) @number.float)
|
|
79
|
+
|
|
80
|
+
(array_item
|
|
81
|
+
value: (scalar) @string)
|
|
82
|
+
|
|
83
|
+
; ---- Multi-line strings ----
|
|
84
|
+
(multiline_stripped) @string
|
|
85
|
+
(multiline_verbatim) @string
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
; Language injections for Ktav multi-line strings.
|
|
2
|
+
;
|
|
3
|
+
; Currently empty. Future versions may inject e.g. JSON / YAML / regex
|
|
4
|
+
; into multi-line string bodies based on the key name (`pattern`,
|
|
5
|
+
; `body`, `template`, …) — see the discussion in the spec rationale.
|
|
6
|
+
;
|
|
7
|
+
; Example (commented out — uncomment once a stable convention emerges):
|
|
8
|
+
;
|
|
9
|
+
; ((multiline_stripped) @injection.content
|
|
10
|
+
; (#set! injection.language "json"))
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
; Local-scope captures for Ktav.
|
|
2
|
+
;
|
|
3
|
+
; Each compound object opens a scope; the key of a pair inside that
|
|
4
|
+
; scope is a "definition" (a property name local to that object).
|
|
5
|
+
; Editors / language servers can use these captures for outline views
|
|
6
|
+
; and rename-within-scope.
|
|
7
|
+
|
|
8
|
+
(source_file) @local.scope
|
|
9
|
+
(compound_object) @local.scope
|
|
10
|
+
|
|
11
|
+
(object_pair
|
|
12
|
+
key: (key) @local.definition.property)
|