volaro 0.0.2 → 0.1.0-alpha.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. package/README.md +114 -22
  2. package/bin/vl.js +466 -28
  3. package/compiler/SOURCE_INFO.json +6 -0
  4. package/compiler/SOURCE_REV +1 -0
  5. package/compiler/validator/vlcheck/__init__.py +10 -0
  6. package/compiler/validator/vlcheck/__main__.py +128 -0
  7. package/compiler/validator/vlcheck/ast_nodes.py +397 -0
  8. package/compiler/validator/vlcheck/checks.py +1227 -0
  9. package/compiler/validator/vlcheck/diagnostics.py +88 -0
  10. package/compiler/validator/vlcheck/lexer.py +343 -0
  11. package/compiler/validator/vlcheck/parser.py +1638 -0
  12. package/compiler/validator/vlcheck/project_config.py +97 -0
  13. package/compiler/validator/vlcheck/resolve.py +849 -0
  14. package/compiler/validator/vlcheck/test_ids.py +98 -0
  15. package/compiler/vlbuild/styling/README.md +39 -0
  16. package/compiler/vlbuild/styling/build-css.mjs +103 -0
  17. package/compiler/vlbuild/styling/package-lock.json +1254 -0
  18. package/compiler/vlbuild/styling/package.json +15 -0
  19. package/compiler/vlbuild/styling/test-build-css.mjs +69 -0
  20. package/compiler/vlbuild/vlbuild/__init__.py +16 -0
  21. package/compiler/vlbuild/vlbuild/__main__.py +246 -0
  22. package/compiler/vlbuild/vlbuild/assets/vlrt.css +165 -0
  23. package/compiler/vlbuild/vlbuild/assets/vlrt.js +1291 -0
  24. package/compiler/vlbuild/vlbuild/emit.py +2023 -0
  25. package/compiler/vlbuild/vlbuild/server_emit.py +1551 -0
  26. package/compiler/vlbuild/vlbuild/static_assets.py +83 -0
  27. package/compiler/vlbuild/vlbuild/style_config.py +347 -0
  28. package/compiler/vlbuild/vlbuild/styling.py +39 -0
  29. package/examples/station.vl +2 -2
  30. package/language/crib.md +131 -10
  31. package/language/spec.md +209 -7
  32. package/language/supported.md +185 -0
  33. package/lib/env.js +107 -0
  34. package/package.json +19 -2
  35. package/scripts/record-provenance.mjs +51 -0
  36. package/scripts/selftest.mjs +54 -0
  37. package/scripts/sync-compiler.sh +52 -0
@@ -0,0 +1,88 @@
1
+ """Diagnostics: severity, source position mapping, and human / JSON rendering.
2
+
3
+ The JSON shape mirrors spec section 18 (`vl check --json`): a stable `code`, a
4
+ `severity`, a `message`, a byte `range`, and an optional `fix`.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ from dataclasses import dataclass, field
10
+ from enum import Enum
11
+
12
+
13
+ class Severity(str, Enum):
14
+ ERROR = "error"
15
+ WARNING = "warning"
16
+ INFO = "info"
17
+
18
+
19
+ class LineIndex:
20
+ """Maps byte offsets to 1-based (line, column). Built once per file."""
21
+
22
+ def __init__(self, text: str) -> None:
23
+ self.text = text
24
+ self._starts = [0]
25
+ for i, ch in enumerate(text):
26
+ if ch == "\n":
27
+ self._starts.append(i + 1)
28
+ self._lines = text.split("\n")
29
+
30
+ def locate(self, offset: int) -> tuple[int, int]:
31
+ offset = max(0, min(offset, len(self.text)))
32
+ lo, hi = 0, len(self._starts) - 1
33
+ while lo < hi:
34
+ mid = (lo + hi + 1) // 2
35
+ if self._starts[mid] <= offset:
36
+ lo = mid
37
+ else:
38
+ hi = mid - 1
39
+ return lo + 1, offset - self._starts[lo] + 1
40
+
41
+ def line_text(self, line_no: int) -> str:
42
+ if 1 <= line_no <= len(self._lines):
43
+ return self._lines[line_no - 1]
44
+ return ""
45
+
46
+
47
+ @dataclass
48
+ class Diagnostic:
49
+ code: str
50
+ severity: Severity
51
+ message: str
52
+ start: int
53
+ end: int
54
+ fix: str | None = None
55
+ file: str = "<input>"
56
+
57
+ def render_human(self, li: LineIndex) -> str:
58
+ line, col = li.locate(self.start)
59
+ src = li.line_text(line)
60
+ span = max(1, self.end - self.start)
61
+ # keep the caret run on the line
62
+ span = min(span, max(1, len(src) - (col - 1)))
63
+ gutter = f"{line:>4} "
64
+ pad = " " * len(gutter)
65
+ out = [
66
+ f"{self.severity.value}[{self.code}]: {self.message}",
67
+ f"{pad}┌─ {self.file}:{line}:{col}",
68
+ f"{pad}│",
69
+ f"{gutter}│ {src}",
70
+ f"{pad}│ {' ' * (col - 1)}{'^' * span}",
71
+ ]
72
+ if self.fix:
73
+ out.append(f"{pad}= fix: {self.fix}")
74
+ return "\n".join(out)
75
+
76
+ def to_json(self, li: LineIndex) -> dict:
77
+ line, col = li.locate(self.start)
78
+ obj = {
79
+ "code": self.code,
80
+ "severity": self.severity.value,
81
+ "message": self.message,
82
+ "file": self.file,
83
+ "range": {"start": self.start, "end": self.end},
84
+ "position": {"line": line, "col": col},
85
+ }
86
+ if self.fix:
87
+ obj["fix"] = {"description": self.fix}
88
+ return obj
@@ -0,0 +1,343 @@
1
+ """Lexer: UTF-8 source -> token stream with synthesized NEWLINE / INDENT / DEDENT.
2
+
3
+ Follows spec section 3:
4
+ - 2 spaces per indent level; tabs in leading whitespace are a hard error
5
+ - newline ends a statement; indent opens a block; dedent closes it
6
+ - inside an open bracket pair ( [ { indentation produces no INDENT/DEDENT;
7
+ NEWLINE is still emitted, since spec 3.7 makes it an item separator there
8
+ - `#` line comment, `##` doc comment (both produce no tokens in this prototype)
9
+ - duration literals (10s, 500ms, 30d) are their own kind, distinct from int
10
+ The lexer never fails hard: it records a Diagnostic and keeps going.
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ from dataclasses import dataclass, field
16
+
17
+ from .diagnostics import Diagnostic, Severity
18
+
19
+ # The 29 reserved words (spec section 3.5). Never identifiers, in any position.
20
+ KEYWORDS = {
21
+ "and", "as", "auth", "break", "catch", "derive", "else", "err", "false",
22
+ "fn", "for", "if", "in", "let", "match", "mut", "nil", "not",
23
+ "on", "or", "par", "return", "skip", "spawn", "true", "try", "type",
24
+ "use", "while",
25
+ }
26
+ # The 18 contextual words (spec section 3.5). Keywords only as the head of a
27
+ # declaration or a block that expects them; ordinary identifiers everywhere else
28
+ # (a field named `state`, a function named `load`). Recognised by position in the
29
+ # parser via `at_soft`; the lexer emits them as plain NAME tokens.
30
+ SOFT_KEYWORDS = {
31
+ "api", "error", "errors", "index", "key", "list", "load", "model", "none",
32
+ "page", "pending", "prefix", "public", "role", "service", "session", "state",
33
+ "theme", "view", "recipe",
34
+ }
35
+
36
+ PRIMS = {"int", "float", "str", "bool", "bytes", "time", "dur", "uuid", "json"}
37
+
38
+ DURATION_SUFFIXES = ("ms", "ns", "us", "s", "m", "h", "d")
39
+
40
+ _OPS2 = ("..", "|>", "??", "->", "==", "!=", "<=", ">=", "+=", "-=", "*=", "/=")
41
+ _OPS1 = set("+-*/%<>=(){}[].,:?!|")
42
+ _OPEN = set("([{")
43
+ _CLOSE = set(")]}")
44
+
45
+
46
+ @dataclass
47
+ class Token:
48
+ kind: str # NAME TYPENAME PRIM KW OP INT FLOAT STRING DURATION BOOL NIL
49
+ # NEWLINE INDENT DEDENT EOF
50
+ value: str
51
+ start: int
52
+ end: int
53
+ interps: list = field(default_factory=list) # STRING only: [(start, end)] of {expr} spans
54
+
55
+
56
+ class Lexer:
57
+ def __init__(self, src: str, filename: str = "<input>") -> None:
58
+ self.src = src
59
+ self.filename = filename
60
+ self.i = 0
61
+ self.n = len(src)
62
+ self.tokens: list[Token] = []
63
+ self.diags: list[Diagnostic] = []
64
+ self.indents = [0]
65
+ self.bracket_depth = 0
66
+ self.at_line_start = True
67
+
68
+ # -- helpers -----------------------------------------------------------
69
+ def _err(self, code: str, msg: str, s: int, e: int, fix: str | None = None) -> None:
70
+ self.diags.append(Diagnostic(code, Severity.ERROR, msg, s, e, fix, self.filename))
71
+
72
+ def _add(self, kind: str, value: str, s: int, e: int) -> Token:
73
+ t = Token(kind, value, s, e)
74
+ self.tokens.append(t)
75
+ return t
76
+
77
+ def _emit_newline(self, at: int) -> None:
78
+ if not self.tokens:
79
+ return
80
+ if self.tokens[-1].kind in ("NEWLINE", "INDENT", "DEDENT"):
81
+ return
82
+ self._add("NEWLINE", "", at, at)
83
+
84
+ # -- main ------------------------------------------------------------
85
+ def lex(self) -> list[Token]:
86
+ if self.src.startswith(""):
87
+ self._err("E-BOM", "byte-order mark at the start of a file is not allowed",
88
+ 0, 1, "save the file as UTF-8 with no signature")
89
+ self.i = 1
90
+
91
+ while self.i < self.n:
92
+ if self.at_line_start:
93
+ if self.bracket_depth == 0:
94
+ self._handle_line_start()
95
+ if self.i >= self.n:
96
+ break
97
+ else:
98
+ # Inside an open bracket, indentation produces no tokens (by
99
+ # design — see the module docstring). Clear the flag here
100
+ # rather than leaving it set: otherwise, if bracket_depth
101
+ # drops back to 0 mid-line (a closing bracket is not the
102
+ # last thing on its line, e.g. `)?`), this stale flag fires
103
+ # on the next loop iteration at the post-bracket cursor
104
+ # position — not real leading whitespace — and emits a
105
+ # bogus DEDENT computed from width 0.
106
+ self.at_line_start = False
107
+
108
+ c = self.src[self.i]
109
+ if c == "\n":
110
+ self._emit_newline(self.i)
111
+ self.i += 1
112
+ self.at_line_start = True
113
+ continue
114
+ if c in " \r":
115
+ self.i += 1
116
+ continue
117
+ if c == "#":
118
+ self._consume_comment()
119
+ continue
120
+ if c == '"':
121
+ self._lex_string()
122
+ continue
123
+ if c.isdigit():
124
+ self._lex_number()
125
+ continue
126
+ if c.isalpha() or c == "_":
127
+ self._lex_ident()
128
+ continue
129
+ if self._lex_operator():
130
+ continue
131
+ self._err("E-CHAR", f"unexpected character {c!r}", self.i, self.i + 1)
132
+ self.i += 1
133
+
134
+ # flush
135
+ self._emit_newline(self.n)
136
+ while len(self.indents) > 1:
137
+ self.indents.pop()
138
+ self._add("DEDENT", "", self.n, self.n)
139
+ self._add("EOF", "", self.n, self.n)
140
+ return self.tokens
141
+
142
+ # -- indentation ---------------------------------------------------------
143
+ def _handle_line_start(self) -> None:
144
+ self.at_line_start = False
145
+ j = self.i
146
+ width = 0
147
+ saw_tab = False
148
+ while j < self.n and self.src[j] in " \t":
149
+ if self.src[j] == "\t":
150
+ saw_tab = True
151
+ width += 2 # count as one level so recovery keeps a plausible shape
152
+ else:
153
+ width += 1
154
+ j += 1
155
+
156
+ # blank line or comment-only line: do not touch the indent stack
157
+ if j >= self.n or self.src[j] in "\n#":
158
+ self.i = j
159
+ return
160
+
161
+ if saw_tab:
162
+ self._err("E-TAB", "tabs are not allowed in indentation", self.i, j,
163
+ "use two spaces per indent level")
164
+ if width % 2 != 0:
165
+ self._err("E-INDENT-ODD",
166
+ f"indentation of {width} spaces is not a multiple of 2",
167
+ self.i, j, "use two spaces per indent level")
168
+
169
+ self.i = j
170
+ top = self.indents[-1]
171
+ if width > top:
172
+ self.indents.append(width)
173
+ self._add("INDENT", "", self.i, self.i)
174
+ elif width < top:
175
+ while len(self.indents) > 1 and self.indents[-1] > width:
176
+ self.indents.pop()
177
+ self._add("DEDENT", "", self.i, self.i)
178
+ if self.indents[-1] != width:
179
+ self._err("E-INDENT", "dedent does not match any enclosing block",
180
+ self.i, self.i + 1, "align this line with an enclosing block")
181
+ self.indents.append(width) # recover
182
+
183
+ # -- tokens ------------------------------------------------------------
184
+ def _consume_comment(self) -> None:
185
+ while self.i < self.n and self.src[self.i] != "\n":
186
+ self.i += 1
187
+
188
+ def _lex_string(self) -> None:
189
+ s = self.i
190
+ src = self.src
191
+ if src[self.i:self.i + 3] == '"""':
192
+ self.i += 3
193
+ while self.i < self.n and src[self.i:self.i + 3] != '"""':
194
+ if src[self.i] == "\\":
195
+ self._lex_escape()
196
+ else:
197
+ self.i += 1
198
+ if src[self.i:self.i + 3] == '"""':
199
+ self.i += 3
200
+ else:
201
+ self._err("E-STRING", "unterminated multi-line string", s, self.n)
202
+ self._add("STRING", src[s:self.i], s, self.i)
203
+ return
204
+
205
+ self.i += 1 # opening quote
206
+ interps: list[tuple[int, int]] = []
207
+ while self.i < self.n:
208
+ c = src[self.i]
209
+ if c == "\\":
210
+ self._lex_escape()
211
+ continue
212
+ if c == "\n":
213
+ self._err("E-STRING", "unterminated string literal", s, self.i,
214
+ 'use """ for a multi-line string')
215
+ break
216
+ if c == '"':
217
+ self.i += 1
218
+ break
219
+ if c == "{":
220
+ self.i += 1
221
+ estart = self.i
222
+ depth = 1
223
+ while self.i < self.n and depth > 0:
224
+ cc = src[self.i]
225
+ if cc == "\n":
226
+ break
227
+ if cc == "{":
228
+ depth += 1
229
+ elif cc == "}":
230
+ depth -= 1
231
+ if depth == 0:
232
+ break
233
+ self.i += 1
234
+ interps.append((estart, self.i))
235
+ if self.i < self.n and src[self.i] == "}":
236
+ self.i += 1
237
+ continue
238
+ self.i += 1
239
+ tok = self._add("STRING", src[s:self.i], s, self.i)
240
+ tok.interps = interps
241
+
242
+ def _lex_escape(self) -> None:
243
+ """Validate escapes before emission can silently change their meaning."""
244
+ start = self.i
245
+ self.i += 1
246
+ if self.i < self.n and self.src[self.i] in 'nt"\\{}':
247
+ self.i += 1
248
+ return
249
+ if self.src[self.i:self.i + 2] == "u{":
250
+ end = self.src.find("}", self.i + 2)
251
+ digits = self.src[self.i + 2:end] if end != -1 else ""
252
+ if (1 <= len(digits) <= 6
253
+ and all(c in "0123456789abcdefABCDEF" for c in digits)):
254
+ value = int(digits, 16)
255
+ if value <= 0x10ffff and not 0xd800 <= value <= 0xdfff:
256
+ self.i = end + 1
257
+ return
258
+ self._err("E-STRING-ESCAPE", "invalid Unicode string escape", start,
259
+ min(self.n, end + 1 if end != -1 else self.i + 1),
260
+ r"use \u{HEX} with a Unicode scalar value (no surrogate code points)")
261
+ if end != -1 and not any(c in self.src[self.i:end] for c in '\n"'):
262
+ self.i = end + 1
263
+ return
264
+ self._err("E-STRING-ESCAPE", "unknown string escape", start,
265
+ min(self.n, self.i + 1),
266
+ r'double a literal backslash: "\\s" for regex \s; supported escapes: \n \t \" \\ \{ \} \u{HEX}')
267
+ # Keep a closing quote/newline available for normal string recovery.
268
+ if self.i < self.n and self.src[self.i] not in '\n"':
269
+ self.i += 1
270
+
271
+ def _lex_number(self) -> None:
272
+ s = self.i
273
+ src = self.src
274
+ if src[self.i] == "0" and self.i + 1 < self.n and src[self.i + 1] in "xXbB":
275
+ self.i += 2
276
+ while self.i < self.n and (src[self.i].isalnum() or src[self.i] == "_"):
277
+ self.i += 1
278
+ self._add("INT", src[s:self.i], s, self.i)
279
+ return
280
+
281
+ while self.i < self.n and (src[self.i].isdigit() or src[self.i] == "_"):
282
+ self.i += 1
283
+
284
+ is_float = False
285
+ if (self.i + 1 < self.n and src[self.i] == "." and src[self.i + 1].isdigit()):
286
+ is_float = True
287
+ self.i += 1
288
+ while self.i < self.n and (src[self.i].isdigit() or src[self.i] == "_"):
289
+ self.i += 1
290
+
291
+ if not is_float:
292
+ for suf in DURATION_SUFFIXES:
293
+ if src[self.i:self.i + len(suf)] == suf:
294
+ after = self.i + len(suf)
295
+ if after >= self.n or not (src[after].isalnum() or src[after] == "_"):
296
+ self.i = after
297
+ self._add("DURATION", src[s:self.i], s, self.i)
298
+ return
299
+
300
+ self._add("FLOAT" if is_float else "INT", src[s:self.i], s, self.i)
301
+
302
+ def _lex_ident(self) -> None:
303
+ s = self.i
304
+ src = self.src
305
+ while self.i < self.n and (src[self.i].isalnum() or src[self.i] == "_"):
306
+ self.i += 1
307
+ word = src[s:self.i]
308
+ if word in ("true", "false"):
309
+ self._add("BOOL", word, s, self.i)
310
+ elif word == "nil":
311
+ self._add("NIL", word, s, self.i)
312
+ elif word in KEYWORDS:
313
+ self._add("KW", word, s, self.i)
314
+ elif word in PRIMS:
315
+ self._add("PRIM", word, s, self.i)
316
+ elif word[0].isupper():
317
+ self._add("TYPENAME", word, s, self.i)
318
+ else:
319
+ self._add("NAME", word, s, self.i)
320
+
321
+ def _lex_operator(self) -> bool:
322
+ src = self.src
323
+ if src[self.i:self.i + 3] == "..=":
324
+ self._push_op("..=", 3)
325
+ return True
326
+ two = src[self.i:self.i + 2]
327
+ if two in _OPS2:
328
+ self._push_op(two, 2)
329
+ return True
330
+ c = src[self.i]
331
+ if c in _OPS1:
332
+ if c in _OPEN:
333
+ self.bracket_depth += 1
334
+ elif c in _CLOSE:
335
+ self.bracket_depth = max(0, self.bracket_depth - 1)
336
+ self._push_op(c, 1)
337
+ return True
338
+ return False
339
+
340
+ def _push_op(self, text: str, length: int) -> None:
341
+ s = self.i
342
+ self.i += length
343
+ self._add("OP", text, s, self.i)