@pineforge/codegen-pyodide 1.1.0 → 1.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +245 -0
- package/README.md +4 -0
- package/glue.py +4 -0
- package/package.json +6 -4
- package/pineforge_codegen/__init__.py +1 -0
- package/pineforge_codegen/codegen/base.py +38 -1
- package/pineforge_codegen/codegen/capabilities.py +186 -0
- package/pineforge_codegen/codegen/checked_settings.py +16 -3
- package/pineforge_codegen/codegen/emit_top.py +2 -0
- package/pineforge_codegen/codegen/helpers.py +1 -0
- package/pineforge_codegen/diagnostic_codes.py +341 -0
- package/pineforge_codegen/diagnostics_catalog.json +631 -0
- package/pineforge_codegen/errors.py +23 -0
- package/pineforge_codegen/external_requests.py +16 -0
- package/pineforge_codegen-1.2.0.tar.gz +0 -0
- package/release.json +2 -2
- package/tables.json +1 -1
- package/transpile.worker.mjs +4 -0
- package/pineforge_codegen-1.1.0.tar.gz +0 -0
|
@@ -0,0 +1,341 @@
|
|
|
1
|
+
"""Stable codes and named arguments for every transpile diagnostic.
|
|
2
|
+
|
|
3
|
+
Every :class:`~pineforge_codegen.errors.Diagnostic` carries a ``code``
|
|
4
|
+
(``PF-E1203`` for an error, ``PF-W0412`` for a warning) and ``args``, the
|
|
5
|
+
named data its English ``message`` and ``hint`` were built from. The codes,
|
|
6
|
+
their severities, English ICU MessageFormat templates and one-line
|
|
7
|
+
explanations live in ``diagnostics_catalog.json`` beside this module, which
|
|
8
|
+
ships with the package and is returned by :func:`diagnostics_catalog`.
|
|
9
|
+
|
|
10
|
+
The emitters keep spelling their English text as they always have, so the
|
|
11
|
+
``message`` stays byte for byte what it was; a diagnostic is coded by the
|
|
12
|
+
catalog template its text renders from (:func:`classify`). The template's
|
|
13
|
+
literal text must equal the message's around its arguments, so a code is
|
|
14
|
+
never guessed: a text no template renders gets the uncatalogued code of its
|
|
15
|
+
severity (``PF-E0000`` / ``PF-W0000``), which the test suite refuses.
|
|
16
|
+
|
|
17
|
+
The match never backtracks over the text's characters: each literal segment
|
|
18
|
+
of a template goes to its leftmost place after the previous one, the last to
|
|
19
|
+
the text's end, and every argument is the text between its literals -- the
|
|
20
|
+
split a fullmatch of the template with lazy ``(.*?)`` arguments returns. Where
|
|
21
|
+
an argument is named twice (``{receiver}`` in the message and the hint) the
|
|
22
|
+
leftmost split can name it two values; then the later places of a literal are
|
|
23
|
+
tried in order, as the regex's backtracking tries them, within a fixed budget
|
|
24
|
+
of steps. The regex took seconds, then minutes, as a crafted message grew,
|
|
25
|
+
and a user's script spells argument text.
|
|
26
|
+
|
|
27
|
+
Rendering (:func:`render`) is the ICU MessageFormat subset the catalog uses:
|
|
28
|
+
literal text with ICU apostrophe quoting (``''`` is one apostrophe, ``'{'``
|
|
29
|
+
a literal brace) and simple ``{name}`` arguments, a string argument
|
|
30
|
+
inserted as is and a number in plain decimal digits.
|
|
31
|
+
"""
|
|
32
|
+
|
|
33
|
+
from __future__ import annotations
|
|
34
|
+
|
|
35
|
+
import json
|
|
36
|
+
import re
|
|
37
|
+
from functools import lru_cache
|
|
38
|
+
from pathlib import Path
|
|
39
|
+
from typing import Any
|
|
40
|
+
|
|
41
|
+
CATALOG_PATH = Path(__file__).with_name("diagnostics_catalog.json")
|
|
42
|
+
CATALOG_SCHEMA = "pineforge-diagnostics-catalog/v1"
|
|
43
|
+
UNCATALOGUED = {"error": "PF-E0000", "warning": "PF-W0000"}
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
@lru_cache(maxsize=1)
|
|
47
|
+
def _load() -> dict:
|
|
48
|
+
with CATALOG_PATH.open(encoding="utf-8") as handle:
|
|
49
|
+
return json.load(handle)
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def diagnostics_catalog() -> dict:
|
|
53
|
+
"""The diagnostics catalog: ``{"schema", "codes": {code: entry}}``.
|
|
54
|
+
|
|
55
|
+
Each entry gives ``severity`` (``error`` / ``warning``), ``area``, the
|
|
56
|
+
English ICU MessageFormat ``message`` template, the ``hint`` template or
|
|
57
|
+
``None``, a one-line ``explanation`` and ``args``: per argument name its
|
|
58
|
+
``kind`` (``identifier``, ``type``, ``keyword``, ``number``, ``vocab`` or
|
|
59
|
+
``text``) and, for ``vocab``, its closed set of ``values``. A fresh copy
|
|
60
|
+
is returned on every call.
|
|
61
|
+
"""
|
|
62
|
+
return json.loads(json.dumps(_load()))
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
# ---------------------------------------------------------------------------
|
|
66
|
+
# ICU MessageFormat subset
|
|
67
|
+
# ---------------------------------------------------------------------------
|
|
68
|
+
|
|
69
|
+
_SPECIAL = "{}"
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def parse_template(template: str) -> list:
|
|
73
|
+
"""Split a catalog template into literal strings and ``(name,)`` argument
|
|
74
|
+
tuples, applying ICU's apostrophe rules (``ApostropheMode.DOUBLE_OPTIONAL``,
|
|
75
|
+
the ICU and FormatJS default)."""
|
|
76
|
+
parts: list = []
|
|
77
|
+
text: list[str] = []
|
|
78
|
+
i, n = 0, len(template)
|
|
79
|
+
while i < n:
|
|
80
|
+
ch = template[i]
|
|
81
|
+
if ch == "'":
|
|
82
|
+
if i + 1 < n and template[i + 1] == "'":
|
|
83
|
+
text.append("'")
|
|
84
|
+
i += 2
|
|
85
|
+
continue
|
|
86
|
+
if i + 1 < n and template[i + 1] in _SPECIAL:
|
|
87
|
+
# A quoted literal runs to the next lone apostrophe.
|
|
88
|
+
i += 1
|
|
89
|
+
while i < n:
|
|
90
|
+
if template[i] == "'":
|
|
91
|
+
if i + 1 < n and template[i + 1] == "'":
|
|
92
|
+
text.append("'")
|
|
93
|
+
i += 2
|
|
94
|
+
continue
|
|
95
|
+
i += 1
|
|
96
|
+
break
|
|
97
|
+
text.append(template[i])
|
|
98
|
+
i += 1
|
|
99
|
+
continue
|
|
100
|
+
text.append("'")
|
|
101
|
+
i += 1
|
|
102
|
+
continue
|
|
103
|
+
if ch == "{":
|
|
104
|
+
end = template.find("}", i)
|
|
105
|
+
if end < 0:
|
|
106
|
+
raise ValueError(f"unclosed argument in template: {template!r}")
|
|
107
|
+
name = template[i + 1:end].strip()
|
|
108
|
+
if not re.fullmatch(r"[A-Za-z_][A-Za-z0-9_]*", name):
|
|
109
|
+
raise ValueError(f"unsupported argument {name!r} in template: {template!r}")
|
|
110
|
+
if text:
|
|
111
|
+
parts.append("".join(text))
|
|
112
|
+
text = []
|
|
113
|
+
parts.append((name,))
|
|
114
|
+
i = end + 1
|
|
115
|
+
continue
|
|
116
|
+
if ch == "}":
|
|
117
|
+
raise ValueError(f"unbalanced '}}' in template: {template!r}")
|
|
118
|
+
text.append(ch)
|
|
119
|
+
i += 1
|
|
120
|
+
if text:
|
|
121
|
+
parts.append("".join(text))
|
|
122
|
+
return parts
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def escape_literal(text: str) -> str:
|
|
126
|
+
"""ICU-escape literal text: every apostrophe doubled, every brace quoted."""
|
|
127
|
+
out: list[str] = []
|
|
128
|
+
for ch in text:
|
|
129
|
+
if ch == "'":
|
|
130
|
+
out.append("''")
|
|
131
|
+
elif ch in _SPECIAL:
|
|
132
|
+
out.append("'" + ch + "'")
|
|
133
|
+
else:
|
|
134
|
+
out.append(ch)
|
|
135
|
+
return "".join(out)
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
def format_arg(value: Any) -> str:
|
|
139
|
+
"""An argument's text: a string as is, a number in plain decimal digits."""
|
|
140
|
+
if isinstance(value, bool):
|
|
141
|
+
raise TypeError("diagnostic arguments are strings or numbers")
|
|
142
|
+
if isinstance(value, int):
|
|
143
|
+
return str(value)
|
|
144
|
+
if isinstance(value, float):
|
|
145
|
+
return repr(value)
|
|
146
|
+
if isinstance(value, str):
|
|
147
|
+
return value
|
|
148
|
+
raise TypeError(f"diagnostic argument of type {type(value).__name__}")
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def render(template: str | None, args: dict) -> str | None:
|
|
152
|
+
"""Render a catalog template with a diagnostic's ``args``."""
|
|
153
|
+
if template is None:
|
|
154
|
+
return None
|
|
155
|
+
return "".join(
|
|
156
|
+
part if isinstance(part, str) else format_arg(args[part[0]])
|
|
157
|
+
for part in parse_template(template)
|
|
158
|
+
)
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
def render_diagnostic(code: str, args: dict) -> tuple[str, str | None]:
|
|
162
|
+
"""The English ``(message, hint)`` a code renders with ``args``."""
|
|
163
|
+
entry = _load()["codes"][code]
|
|
164
|
+
return render(entry["message"], args), render(entry.get("hint"), args)
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
# ---------------------------------------------------------------------------
|
|
168
|
+
# Classification
|
|
169
|
+
# ---------------------------------------------------------------------------
|
|
170
|
+
|
|
171
|
+
_CANONICAL_INT = re.compile(r"-?(?:0|[1-9][0-9]*)")
|
|
172
|
+
_CANONICAL_FLOAT = re.compile(r"-?(?:0|[1-9][0-9]*)\.[0-9]+")
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
def _segments(parts: list) -> tuple[str, tuple[tuple[str, str], ...]]:
|
|
176
|
+
"""A parsed template as its leading literal and ``(argument, literal after
|
|
177
|
+
it)`` pairs; the literal after an argument may be empty."""
|
|
178
|
+
head = ""
|
|
179
|
+
pairs: list[list[str]] = []
|
|
180
|
+
for part in parts:
|
|
181
|
+
if isinstance(part, str):
|
|
182
|
+
if pairs:
|
|
183
|
+
pairs[-1][1] += part
|
|
184
|
+
else:
|
|
185
|
+
head += part
|
|
186
|
+
else:
|
|
187
|
+
pairs.append([part[0], ""])
|
|
188
|
+
return head, tuple((name, literal) for name, literal in pairs)
|
|
189
|
+
|
|
190
|
+
|
|
191
|
+
# Literal places a template whose argument is named twice may try, per
|
|
192
|
+
# classification: a script's text never needs more than a few, and a crafted
|
|
193
|
+
# one gets the uncatalogued code instead of a long search.
|
|
194
|
+
_SEARCH_BUDGET = 4096
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
def _search(segments: list, texts: list[str], repeats: bool) -> dict[str, str] | None:
|
|
198
|
+
"""Split ``texts`` (the message, then the hint) by their templates'
|
|
199
|
+
``segments``: each argument ends at the leftmost place of the literal after
|
|
200
|
+
it, the last argument of a text at its end -- the split a lazy-regex
|
|
201
|
+
fullmatch returns. Without an argument named twice (``repeats``) that split
|
|
202
|
+
succeeds whenever any does, since a leftmost place leaves the rest the most
|
|
203
|
+
room: it is the only one tried, so the work is linear. With one, a split can
|
|
204
|
+
give the name two values; then the later places of a literal are tried in
|
|
205
|
+
order, as the regex backtracks, within ``_SEARCH_BUDGET`` places."""
|
|
206
|
+
starts: list[int] = []
|
|
207
|
+
ends: list[int] = []
|
|
208
|
+
slots: list[tuple[int, str, str, bool]] = [] # text, argument, literal after it, last
|
|
209
|
+
for index, ((head, pairs), text) in enumerate(zip(segments, texts)):
|
|
210
|
+
if not text.startswith(head):
|
|
211
|
+
return None
|
|
212
|
+
if not pairs:
|
|
213
|
+
if len(text) != len(head):
|
|
214
|
+
return None
|
|
215
|
+
starts.append(len(head))
|
|
216
|
+
ends.append(len(head))
|
|
217
|
+
continue
|
|
218
|
+
tail = pairs[-1][1]
|
|
219
|
+
end = len(text) - len(tail)
|
|
220
|
+
if end < len(head) or not text.endswith(tail):
|
|
221
|
+
return None
|
|
222
|
+
starts.append(len(head))
|
|
223
|
+
ends.append(end)
|
|
224
|
+
for at, (name, literal) in enumerate(pairs):
|
|
225
|
+
slots.append((index, name, literal, at == len(pairs) - 1))
|
|
226
|
+
found: dict[str, str] = {}
|
|
227
|
+
budget = [_SEARCH_BUDGET]
|
|
228
|
+
|
|
229
|
+
def step(slot: int, pos: int) -> bool:
|
|
230
|
+
if slot == len(slots):
|
|
231
|
+
return True
|
|
232
|
+
text_index, name, literal, last = slots[slot]
|
|
233
|
+
text, end = texts[text_index], ends[text_index]
|
|
234
|
+
following = slot + 1
|
|
235
|
+
known = found.get(name)
|
|
236
|
+
if last:
|
|
237
|
+
value = text[pos:end]
|
|
238
|
+
resume = (starts[slots[following][0]] if following < len(slots) else end)
|
|
239
|
+
if known is not None:
|
|
240
|
+
return known == value and step(following, resume)
|
|
241
|
+
found[name] = value
|
|
242
|
+
if step(following, resume):
|
|
243
|
+
return True
|
|
244
|
+
del found[name]
|
|
245
|
+
return False
|
|
246
|
+
if known is not None:
|
|
247
|
+
stop = pos + len(known)
|
|
248
|
+
return (text.startswith(known, pos) and stop + len(literal) <= end
|
|
249
|
+
and text.startswith(literal, stop) and step(following, stop + len(literal)))
|
|
250
|
+
at = text.find(literal, pos, end)
|
|
251
|
+
while at >= 0:
|
|
252
|
+
budget[0] -= 1
|
|
253
|
+
if budget[0] < 0:
|
|
254
|
+
return False
|
|
255
|
+
found[name] = text[pos:at]
|
|
256
|
+
if step(following, at + len(literal)):
|
|
257
|
+
return True
|
|
258
|
+
del found[name]
|
|
259
|
+
if not repeats:
|
|
260
|
+
return False
|
|
261
|
+
at = text.find(literal, at + 1, end) if at < end else -1
|
|
262
|
+
return False
|
|
263
|
+
|
|
264
|
+
first = starts[slots[0][0]] if slots else 0
|
|
265
|
+
return found if step(0, first) else None
|
|
266
|
+
|
|
267
|
+
|
|
268
|
+
class _Matcher:
|
|
269
|
+
__slots__ = ("code", "has_hint", "prefix", "suffix", "specificity",
|
|
270
|
+
"_message", "_hint", "_kinds", "_repeats")
|
|
271
|
+
|
|
272
|
+
def __init__(self, code: str, entry: dict):
|
|
273
|
+
self.code = code
|
|
274
|
+
message = parse_template(entry["message"])
|
|
275
|
+
hint = entry.get("hint")
|
|
276
|
+
hint_parts = parse_template(hint) if hint is not None else []
|
|
277
|
+
self.has_hint = hint is not None
|
|
278
|
+
self._message = _segments(message)
|
|
279
|
+
self._hint = _segments(hint_parts) if hint is not None else None
|
|
280
|
+
self.prefix = message[0] if message and isinstance(message[0], str) else ""
|
|
281
|
+
self.suffix = message[-1] if message and isinstance(message[-1], str) else ""
|
|
282
|
+
# The text a template spells itself; the hint's separator counted as one
|
|
283
|
+
# character, as the order of the catalog's codes has always assumed.
|
|
284
|
+
self.specificity = (sum(len(p) for p in message + hint_parts if isinstance(p, str))
|
|
285
|
+
+ (1 if hint is not None else 0))
|
|
286
|
+
self._kinds = {name: spec.get("kind") for name, spec in entry.get("args", {}).items()}
|
|
287
|
+
names = [part[0] for part in message + hint_parts if isinstance(part, tuple)]
|
|
288
|
+
self._repeats = len(names) != len(set(names))
|
|
289
|
+
|
|
290
|
+
def match(self, message: str, hint: str | None) -> dict | None:
|
|
291
|
+
segments = [self._message] + ([self._hint] if self._hint is not None else [])
|
|
292
|
+
texts = [message] + ([hint] if self._hint is not None else [])
|
|
293
|
+
found = _search(segments, texts, self._repeats)
|
|
294
|
+
if found is None:
|
|
295
|
+
return None
|
|
296
|
+
args: dict[str, Any] = {}
|
|
297
|
+
for name, value in found.items():
|
|
298
|
+
if self._kinds.get(name) == "number":
|
|
299
|
+
if _CANONICAL_INT.fullmatch(value):
|
|
300
|
+
args[name] = int(value)
|
|
301
|
+
continue
|
|
302
|
+
if _CANONICAL_FLOAT.fullmatch(value) and repr(float(value)) == value:
|
|
303
|
+
args[name] = float(value)
|
|
304
|
+
continue
|
|
305
|
+
args[name] = value
|
|
306
|
+
return args
|
|
307
|
+
|
|
308
|
+
|
|
309
|
+
@lru_cache(maxsize=1)
|
|
310
|
+
def _matchers() -> dict[str, list[_Matcher]]:
|
|
311
|
+
by_severity: dict[str, list[_Matcher]] = {"error": [], "warning": []}
|
|
312
|
+
for code, entry in _load()["codes"].items():
|
|
313
|
+
if code in UNCATALOGUED.values():
|
|
314
|
+
continue
|
|
315
|
+
by_severity[entry["severity"]].append(_Matcher(code, entry))
|
|
316
|
+
for matchers in by_severity.values():
|
|
317
|
+
# The template with the most literal text wins: a message two templates
|
|
318
|
+
# render reads as the more specific one (both render it byte for byte).
|
|
319
|
+
matchers.sort(key=lambda m: (-m.specificity, m.code))
|
|
320
|
+
return by_severity
|
|
321
|
+
|
|
322
|
+
|
|
323
|
+
def classify(severity: str, message: str, hint: str | None = None) -> tuple[str, dict]:
|
|
324
|
+
"""The ``(code, args)`` of a diagnostic's English text.
|
|
325
|
+
|
|
326
|
+
``severity`` is ``"error"`` or ``"warning"``. A text no catalog template
|
|
327
|
+
renders gets ``PF-E0000`` / ``PF-W0000`` with its text as ``args``.
|
|
328
|
+
"""
|
|
329
|
+
for matcher in _matchers().get(severity, ()):
|
|
330
|
+
if matcher.has_hint != (hint is not None):
|
|
331
|
+
continue
|
|
332
|
+
if not message.startswith(matcher.prefix) or not message.endswith(matcher.suffix):
|
|
333
|
+
continue
|
|
334
|
+
args = matcher.match(message, hint)
|
|
335
|
+
if args is not None:
|
|
336
|
+
return matcher.code, args
|
|
337
|
+
args = {"message": message}
|
|
338
|
+
if hint is not None:
|
|
339
|
+
args["hint"] = hint
|
|
340
|
+
code = UNCATALOGUED.get(severity, UNCATALOGUED["error"])
|
|
341
|
+
return code, args
|