tree-sitter-ktav 0.2.1 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -171,7 +171,8 @@ rejects (mostly missing-whitespace-after-marker — § 6.10).
171
171
 
172
172
  ## License
173
173
 
174
- MIT — see [LICENSE](LICENSE).
174
+ Dual-licensed under **MIT OR Apache-2.0** — see
175
+ [LICENSE-MIT](LICENSE-MIT) and [LICENSE-APACHE](LICENSE-APACHE).
175
176
 
176
177
  ## Other Ktav repositories
177
178
 
package/grammar.js CHANGED
@@ -1,24 +1,36 @@
1
1
  /**
2
2
  * Tree-sitter grammar for Ktav (כְּתָב) — the Written Configuration Format.
3
3
  *
4
- * Spec: https://github.com/ktav-lang/spec/blob/main/versions/0.1/spec.md
4
+ * Spec: https://github.com/ktav-lang/spec/blob/main/versions/0.5/spec.md
5
5
  *
6
6
  * Ktav is line-oriented. Every line is one of:
7
7
  * - blank
8
- * - a comment (`# ...`)
9
- * - a key:value pair (with markers `:`, `::`, `:i`, `:f`)
8
+ * - a comment (`## ...`) ← NOTE: double-hash in 0.5.0; single `#` is content
9
+ * - a key:value pair (with markers `:` or `::`)
10
10
  * - a structural opener / closer for compounds (`{`, `}`, `[`, `]`,
11
11
  * `(`, `((`, `)`, `))`)
12
- * - an array item (inside an open `[` array)
12
+ * - an array item (inside an open `[` array, or at the top level)
13
13
  * - raw content of a multi-line string
14
14
  *
15
+ * Changes from 0.3.0 (spec 0.1.1) to 0.5.0:
16
+ * - Comment marker changed from `#` to `##`. Single `#` is now a
17
+ * content byte (allowed in keys and scalar values).
18
+ * - Typed markers `:i` and `:f` removed. Only `:` and `::` remain.
19
+ * - Inline compounds: `{key: value, ...}` and `[v1, v2, ...]` are
20
+ * now valid as pair values or array items (inline_object /
21
+ * inline_array rules).
22
+ * - Number literals: hex (`0x`), octal (`0o`), binary (`0b`), decimal
23
+ * with underscore separators, and floats with `.` or exponent.
24
+ * These are captured as distinct node kinds for highlighting.
25
+ * - Escape sequences inside inline scalars: `\\`, `\,`, `\}`, `\]`,
26
+ * `\{`, `\[`, `\n`, `\r`.
27
+ *
15
28
  * Strategy:
16
29
  * - Newlines are explicit (`_newline`) and structural openers/closers
17
30
  * are tokens that include the trailing whitespace + newline so they
18
31
  * can NEVER be confused with a scalar starting with the same byte.
19
- * - The four pair separators (`:`, `::`, `:i`, `:f`) are recognized
20
- * by the lexer with longest-match precedence (`::` > `:`, `:i`/`:f`
21
- * > `:`).
32
+ * - The two pair separators (`:`, `::`) are recognized by the lexer
33
+ * with longest-match precedence (`::` > `:`).
22
34
  * - The mandatory-whitespace-after-marker rule (§ 6.10) is enforced
23
35
  * by an external scanner token `_marker_ws`, which is a zero-width
24
36
  * assertion that only succeeds when the byte right after the
@@ -31,6 +43,8 @@
31
43
  * a parse error.
32
44
  * - Multi-line string content is captured as a sequence of opaque
33
45
  * "raw lines" up to the matching terminator.
46
+ * - Inline compound values are fully parsed; escape sequences inside
47
+ * inline scalars are captured as `escape_sequence` nodes.
34
48
  * - Indentation is not significant (matches the spec).
35
49
  */
36
50
 
@@ -55,6 +69,10 @@ module.exports = grammar({
55
69
  word: $ => $._key_segment,
56
70
 
57
71
  rules: {
72
+ // The top-level document is a sequence of lines. Per spec § 5.0.1
73
+ // the root may be either an Object or an Array. Tree-sitter accepts
74
+ // both kinds of line anywhere; semantic dispatch is left to the
75
+ // reference parser.
58
76
  source_file: $ => repeat($._line),
59
77
 
60
78
  // ---- Top-level lines ----
@@ -62,6 +80,7 @@ module.exports = grammar({
62
80
  $.comment,
63
81
  $.blank_line,
64
82
  $.object_pair,
83
+ $.top_array_item,
65
84
  ),
66
85
 
67
86
  blank_line: $ => $._newline,
@@ -69,26 +88,26 @@ module.exports = grammar({
69
88
  _newline: $ => /\r?\n/,
70
89
 
71
90
  // ---- Comment ----
72
- comment: $ => seq(
73
- '#',
74
- optional(/[^\r\n]*/),
75
- $._newline,
76
- ),
91
+ //
92
+ // In spec 0.5.0 a comment starts with `##` (two hashes). A single
93
+ // `#` is ordinary content. The token captures the whole line
94
+ // including the trailing newline to beat `_top_scalar_text` at the
95
+ // lexer's longest-match step.
96
+ comment: $ => token(prec(1, /##[^\r\n]*\r?\n/)),
77
97
 
78
98
  // ---- Object pair ----
79
99
  //
80
100
  // After the separator, the external `_marker_ws` token asserts
81
- // that the next byte is whitespace, CR, LF, or EOF. This enforces
82
- // § 6.10 (mandatory whitespace after marker): `key:value` fails.
101
+ // that the next byte is whitespace, CR, LF, or EOF (§ 6.10).
83
102
  object_pair: $ => choice(
84
- // After `::`, `:i`, `:f` the body is a literal/typed scalar — § 5.2:
85
- // it is NOT dispatched through compound-opener / multi-line dispatch.
86
- // Only `empty_value` (immediate newline) or `scalar` is allowed.
103
+ // After `::` the body is a literal scalar (NOT dispatched through
104
+ // compound-opener / multi-line dispatch). `raw_scalar` accepts any
105
+ // line content including `(`, `{`, `[`-starting text.
87
106
  seq(
88
107
  field('key', $.key),
89
- field('separator', choice($.sep_raw, $.sep_int, $.sep_float)),
108
+ field('separator', $.sep_raw),
90
109
  $._marker_ws,
91
- field('value', choice($.empty_value, $.scalar)),
110
+ field('value', choice($.empty_value, $.raw_scalar)),
92
111
  ),
93
112
  // After `:` the body goes through the full § 5.2 dispatch.
94
113
  seq(
@@ -102,27 +121,31 @@ module.exports = grammar({
102
121
  ),
103
122
  ),
104
123
 
105
- // Tree-sitter prefers longest token match. To make sure `::` wins
106
- // over `:`, `::` is given higher precedence; same for `:i`/`:f`.
124
+ // `::` wins over `:` via higher precedence.
107
125
  sep_raw: $ => token(prec(3, '::')),
108
- sep_int: $ => token(prec(2, ':i')),
109
- sep_float: $ => token(prec(2, ':f')),
110
126
  sep_string: $ => token(prec(1, ':')),
111
127
 
112
128
  // ---- Keys ----
113
129
  key: $ => choice(
114
- $._key_segment,
130
+ $._spaced_key,
115
131
  $.dotted_key,
116
132
  ),
117
133
 
134
+ // A key may contain internal whitespace (spec 0.5.0 § 4): the run of
135
+ // space-separated segments up to the separator is one key
136
+ // (`multi word key: value`). The inter-segment spaces are `extras`,
137
+ // so the `key` node still spans the whole text with no named children
138
+ // (renders as `(key)`, same as a single-segment key).
139
+ _spaced_key: $ => prec.left(repeat1($._key_segment)),
140
+
118
141
  dotted_key: $ => prec.left(seq(
119
142
  $._key_segment,
120
143
  repeat1(seq('.', $._key_segment)),
121
144
  )),
122
145
 
123
- // Key segment: any chars except whitespace, "[", "]", "{", "}",
124
- // ":", "#", ".".
125
- _key_segment: $ => /[^\s\[\]\{\}:#.\r\n]+/,
146
+ // Key segment: any chars except whitespace, "[", "]", "{", "}", "(", ")",
147
+ // ":", ",", ".". Note: "#" is now allowed in keys (spec 0.5.0 § 4).
148
+ _key_segment: $ => /[^\s\[\]\{\}\(\):#,.\r\n]+/,
126
149
 
127
150
  // ---- Value line ----
128
151
  _value_line: $ => choice(
@@ -136,8 +159,14 @@ module.exports = grammar({
136
159
  $.empty_array,
137
160
  $.empty_paren,
138
161
  $.empty_double_paren,
162
+ // Inline compounds (new in 0.5.0).
163
+ $.inline_object,
164
+ $.inline_array,
139
165
  // Keywords (single token followed by newline).
140
166
  $.keyword,
167
+ // Number literals (distinct nodes for highlighting).
168
+ $.integer,
169
+ $.float,
141
170
  // Scalar — catch-all line content.
142
171
  $.scalar,
143
172
  ),
@@ -152,25 +181,12 @@ module.exports = grammar({
152
181
  empty_double_paren: $ => seq(token(prec(5, '(())')), $._newline),
153
182
 
154
183
  // ---- Multi-line compounds ----
155
- //
156
- // Openers are tokens that include the rest of the line up to and
157
- // including the newline, so they cannot be confused with a scalar
158
- // starting with the same byte. Closers, by contrast, are split
159
- // into the bracket character(s) plus the external `_strict_eol`
160
- // token; this lets the scanner reject pathological lines like
161
- // `} trailing_text\n` (§ 5.6.1 and the cleanliness rule for
162
- // object/array closers).
163
184
  open_brace: $ => token(prec(4, /\{[ \t]*\r?\n/)),
164
185
  close_brace: $ => seq(token(prec(4, '}')), $._strict_eol),
165
186
  open_bracket: $ => token(prec(4, /\[[ \t]*\r?\n/)),
166
187
  close_bracket: $ => seq(token(prec(4, ']')), $._strict_eol),
167
188
  open_paren: $ => token(prec(4, /\([ \t]*\r?\n/)),
168
189
  open_dparen: $ => token(prec(5, /\(\([ \t]*\r?\n/)),
169
- // close_paren / close_dparen are emitted by the external scanner as
170
- // `_stripped_close` / `_verbatim_close`. They are context-sensitive:
171
- // inside `(...)` only `)` closes (a `))` line is content); inside
172
- // `((...))` only `))` closes (a single `)` line is content). The
173
- // scanner consults `valid_symbols` to decide which form to attempt.
174
190
  close_paren: $ => $._stripped_close,
175
191
  close_dparen: $ => $._verbatim_close,
176
192
 
@@ -186,30 +202,144 @@ module.exports = grammar({
186
202
  $.close_bracket,
187
203
  ),
188
204
 
189
- // ---- Array items ----
205
+ // ---- Inline compounds (new in spec 0.5.0) ----
190
206
  //
191
- // Marker-prefixed items must, like pair lines, have whitespace or
192
- // EOL after the marker (§ 6.10).
207
+ // `{key: value, key2: value2}` and `[v1, v2, v3]` are valid as a
208
+ // value on the right-hand side of a pair or as an array item.
209
+ // Trailing comma is allowed. Nesting is supported.
210
+ //
211
+ // These are followed by a newline (they consume the rest of the line).
212
+ inline_object: $ => seq(
213
+ '{',
214
+ optional($._inline_pair_list),
215
+ '}',
216
+ $._newline,
217
+ ),
218
+
219
+ inline_array: $ => seq(
220
+ '[',
221
+ optional($._inline_item_list),
222
+ ']',
223
+ $._newline,
224
+ ),
225
+
226
+ _inline_pair_list: $ => seq(
227
+ $.inline_pair,
228
+ repeat(seq(',', $.inline_pair)),
229
+ optional(','),
230
+ ),
231
+
232
+ // The value is optional: `{x:, y: 1}` and `{empty:}` are valid —
233
+ // a separator immediately followed by `,` or `}` is an empty value
234
+ // (spec 0.5.0 § 5.8).
235
+ inline_pair: $ => seq(
236
+ field('key', $.key),
237
+ field('separator', choice($.sep_raw, $.sep_string)),
238
+ optional(field('value', $.inline_value)),
239
+ ),
240
+
241
+ _inline_item_list: $ => seq(
242
+ $.inline_value,
243
+ repeat(seq(',', $.inline_value)),
244
+ optional(','),
245
+ ),
246
+
247
+ // An inline value is either a nested inline compound, or an inline
248
+ // scalar (which may contain escape sequences).
249
+ inline_value: $ => choice(
250
+ $.nested_inline_object,
251
+ $.nested_inline_array,
252
+ $.inline_scalar,
253
+ ),
254
+
255
+ nested_inline_object: $ => seq(
256
+ '{',
257
+ optional($._inline_pair_list),
258
+ '}',
259
+ ),
260
+
261
+ nested_inline_array: $ => seq(
262
+ '[',
263
+ optional($._inline_item_list),
264
+ ']',
265
+ ),
266
+
267
+ // An inline scalar is terminated by an unescaped `,`, `}`, or `]`.
268
+ // It may contain escape sequences (§ 3.7).
269
+ //
270
+ // The first chunk must NOT begin with `{` or `[`: a value position
271
+ // that opens with `{`/`[` is a nested compound, not a scalar. After
272
+ // the first character those bytes are ordinary literal content
273
+ // (`hello{world`, `mid[bracket`). This head/rest split keeps
274
+ // `nested_inline_*` vs `inline_scalar` unambiguous without sacrificing
275
+ // mid-value literal braces (§ 5.8).
276
+ inline_scalar: $ => seq(
277
+ choice($.escape_sequence, $._inline_scalar_head),
278
+ repeat(choice($.escape_sequence, $._inline_scalar_text)),
279
+ ),
280
+
281
+ // Escape sequences recognised inside inline scalars (§ 3.7).
282
+ escape_sequence: $ => token(choice(
283
+ '\\\\',
284
+ '\\,',
285
+ '\\}',
286
+ '\\]',
287
+ '\\{',
288
+ '\\[',
289
+ '\\n',
290
+ '\\r',
291
+ )),
292
+
293
+ // Leading chunk of a scalar: first byte excludes whitespace and the
294
+ // openers `{`/`[` (so a value starting with an opener is a nested
295
+ // compound), plus the usual `\` `,` `}` `]` / CR / LF. Subsequent
296
+ // bytes allow `{`/`[` as literal content.
297
+ _inline_scalar_head: $ => token(/[^\s\\,\{\[\}\]\r\n][^\\,\}\]\r\n]*/),
298
+
299
+ // Continuation text after the head (or after an escape): any byte
300
+ // except `\`, `,`, the closers `}` `]`, and CR / LF. Open delimiters
301
+ // `{`/`[` are allowed here as literal content.
302
+ _inline_scalar_text: $ => token(/[^\\,\}\]\r\n]+/),
303
+
304
+ // ---- Array items ----
193
305
  array_item: $ => choice(
194
306
  seq(
195
307
  field('marker', $.sep_raw),
196
308
  $._marker_ws,
197
- field('value', choice($.empty_value, $.scalar)),
198
- ),
199
- seq(
200
- field('marker', $.sep_int),
201
- $._marker_ws,
202
- field('value', choice($.empty_value, $.scalar)),
309
+ field('value', choice($.empty_value, $.raw_scalar)),
203
310
  ),
311
+ // Plain value item — same set as object pair value.
312
+ field('value', $._value_line),
313
+ ),
314
+
315
+ // Top-level array item (§ 5.0.1).
316
+ top_array_item: $ => choice(
204
317
  seq(
205
- field('marker', $.sep_float),
318
+ field('marker', $.sep_raw),
206
319
  $._marker_ws,
207
- field('value', choice($.empty_value, $.scalar)),
320
+ field('value', choice($.empty_value, $.raw_scalar)),
208
321
  ),
209
- // Plain value item — same set as object pair value.
210
- field('value', $._value_line),
322
+ field('value', $.compound_object),
323
+ field('value', $.compound_array),
324
+ field('value', $.multiline_stripped),
325
+ field('value', $.multiline_verbatim),
326
+ field('value', $.empty_object),
327
+ field('value', $.empty_array),
328
+ field('value', $.empty_paren),
329
+ field('value', $.empty_double_paren),
330
+ field('value', $.inline_object),
331
+ field('value', $.inline_array),
332
+ field('value', $.keyword),
333
+ field('value', $.integer),
334
+ field('value', $.float),
335
+ field('value', $.top_scalar),
211
336
  ),
212
337
 
338
+ // `top_scalar` — bare-scalar at the document root. Forbids `:` so
339
+ // that pair-shaped lines always parse as `object_pair`.
340
+ top_scalar: $ => $._top_scalar_text,
341
+ _top_scalar_text: $ => token(/[^\s:\{\[\(\r\n][^:\r\n]*\r?\n/),
342
+
213
343
  // ---- Multi-line strings ----
214
344
  multiline_stripped: $ => seq(
215
345
  $.open_paren,
@@ -223,11 +353,6 @@ module.exports = grammar({
223
353
  $.close_dparen,
224
354
  ),
225
355
 
226
- // A content line inside a multi-line string. Captured as one
227
- // token. The closer tokens (`)`, `))` plus _strict_eol) win at
228
- // the LR(1) boundary because they require their own line; any
229
- // line whose content includes more than just the terminator
230
- // falls through to this rule.
231
356
  multiline_content_line: $ => token(prec(-1, /[^\r\n]*\r?\n/)),
232
357
 
233
358
  // ---- Scalar (default value body, until end of line) ----
@@ -236,12 +361,55 @@ module.exports = grammar({
236
361
  $._newline,
237
362
  ),
238
363
 
364
+ // `raw_scalar` is used exclusively after `::` (raw marker). It accepts
365
+ // any non-empty line content, including `(`, `{`, `[`-starting text.
366
+ // Per spec § 5.2: after `::` the body is NEVER dispatched as a
367
+ // compound opener or multi-line string opener — it is always a literal
368
+ // String value. This is a separate rule (not `scalar`) because
369
+ // `scalar`'s underlying `_scalar_text` deliberately excludes those
370
+ // opening bytes to avoid lexer ambiguity in the non-raw value context.
371
+ raw_scalar: $ => seq(
372
+ $._raw_scalar_text,
373
+ $._newline,
374
+ ),
375
+ _raw_scalar_text: $ => /[^\s\r\n][^\r\n]*/,
376
+
239
377
  // Scalar text: any non-whitespace, non-newline content up to end
240
- // of line. `#` is allowed (`color: #ff00ff` is a valid value);
241
- // the `#` only opens a comment when it is the first non-whitespace
242
- // char of a line — tree-sitter's context-aware lexer disambiguates
243
- // because `comment` is only valid at a line-start parse state.
244
- _scalar_text: $ => /[^\s\r\n][^\r\n]*/,
378
+ // of line. Both `#` and `##` are allowed as content bytes in 0.5.0.
379
+ // Lines starting with `{` or `[` are always handled by structural
380
+ // rules (compound_object, compound_array, empty_object, empty_array,
381
+ // inline_object, inline_array), so `_scalar_text` explicitly excludes
382
+ // those opening bytes at position 0 to avoid the greedy-token
383
+ // ambiguity. Lines starting with `(` are handled by multiline or
384
+ // empty-paren rules likewise.
385
+ _scalar_text: $ => /[^\s\{\[\(\r\n][^\r\n]*/,
386
+
387
+ // ---- Number literals ----
388
+ //
389
+ // Captured as distinct node kinds so syntax highlighters can colour
390
+ // them differently from plain strings.
391
+ //
392
+ // IMPORTANT: These are whole-line tokens (pattern + optional
393
+ // horizontal whitespace + newline). Using a whole-line token (like
394
+ // `comment` and `_top_scalar_text`) avoids the ambiguity where the
395
+ // partial integer token `1` wins over the scalar token `1:2:3` via
396
+ // prec — with a whole-line token the match for `1:2:3\n` is length
397
+ // 5 for scalar and no match for integer (`:` breaks the pattern),
398
+ // so scalar correctly wins on non-numeric lines.
399
+ //
400
+ // Float must be tested before integer because the float pattern
401
+ // (decimal point form) is a strict superset of the integer pattern
402
+ // prefix. Float is given prec(3) so it beats integer on `1.5\n`.
403
+ //
404
+ // Integer forms (§ 3.6): hex, octal, binary, decimal with underscores.
405
+ integer: $ => token(prec(2,
406
+ /[+-]?(0x[0-9a-fA-F]([_]?[0-9a-fA-F])*|0o[0-7]([_]?[0-7])*|0b[01]([_]?[01])*|[0-9]([_]?[0-9])*)[ \t]*\r?\n/
407
+ )),
408
+
409
+ // Float forms (§ 3.6): decimal-point form or exponent-only form.
410
+ float: $ => token(prec(3,
411
+ /([+-]?[0-9]([_]?[0-9])*\.[0-9]([_]?[0-9])*([eE][+-]?[0-9]([_]?[0-9])*)?|[+-]?[0-9]([_]?[0-9])*[eE][+-]?[0-9]([_]?[0-9])*)[ \t]*\r?\n/
412
+ )),
245
413
 
246
414
  // ---- Keywords ----
247
415
  keyword: $ => seq(
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "tree-sitter-ktav",
3
- "version": "0.2.1",
3
+ "version": "0.5.0",
4
4
  "description": "Tree-sitter grammar for Ktav (כְּתָב) — the Written Configuration Format",
5
5
  "keywords": [
6
6
  "parser",
@@ -11,7 +11,7 @@
11
11
  "configuration"
12
12
  ],
13
13
  "author": "Marat K <phpcraftdream@gmail.com>",
14
- "license": "MIT",
14
+ "license": "MIT OR Apache-2.0",
15
15
  "homepage": "https://github.com/ktav-lang/tree-sitter-ktav",
16
16
  "repository": {
17
17
  "type": "git",
@@ -14,8 +14,6 @@
14
14
  ; ---- Pair separators ----
15
15
  (sep_string) @punctuation.delimiter
16
16
  (sep_raw) @punctuation.special
17
- (sep_int) @punctuation.special
18
- (sep_float) @punctuation.special
19
17
 
20
18
  ; ---- Compound brackets ----
21
19
  ; The structural openers / closers each form their own visible token
@@ -43,9 +41,12 @@
43
41
  (kw_true) @constant.builtin.boolean
44
42
  (kw_false) @constant.builtin.boolean
45
43
 
44
+ ; ---- Number literals (new in spec 0.5.0) ----
45
+ (integer) @number
46
+ (float) @number.float
47
+
46
48
  ; ---- String values ----
47
- ; Plain scalar after `:` — usually a string. (Numeric coercion happens
48
- ; only after `:i`/`:f`; see below.)
49
+ ; Plain scalar after `:` — a string.
49
50
  (object_pair
50
51
  separator: (sep_string)
51
52
  value: (scalar) @string)
@@ -53,33 +54,22 @@
53
54
  ; Raw string after `::`
54
55
  (object_pair
55
56
  separator: (sep_raw)
56
- value: (scalar) @string.special)
57
-
58
- ; Typed integer / float bodies
59
- (object_pair
60
- separator: (sep_int)
61
- value: (scalar) @number)
62
-
63
- (object_pair
64
- separator: (sep_float)
65
- value: (scalar) @number.float)
57
+ value: (raw_scalar) @string.special)
66
58
 
67
59
  ; ---- Array items ----
68
60
  (array_item
69
61
  marker: (sep_raw)
70
- value: (scalar) @string.special)
71
-
72
- (array_item
73
- marker: (sep_int)
74
- value: (scalar) @number)
75
-
76
- (array_item
77
- marker: (sep_float)
78
- value: (scalar) @number.float)
62
+ value: (raw_scalar) @string.special)
79
63
 
80
64
  (array_item
81
65
  value: (scalar) @string)
82
66
 
67
+ ; ---- Inline compounds (new in spec 0.5.0) ----
68
+ (inline_object) @punctuation.bracket
69
+ (inline_array) @punctuation.bracket
70
+ (inline_scalar) @string
71
+ (escape_sequence) @string.escape
72
+
83
73
  ; ---- Multi-line strings ----
84
74
  (multiline_stripped) @string
85
75
  (multiline_verbatim) @string