@pineforge/codegen-pyodide 1.1.0 → 1.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -952,6 +952,7 @@ class ExprVisitor:
952
952
  if node.member == "isfirst":
953
953
  return "(bar_index_ == 0)"
954
954
  if node.member == "islast":
955
+ self._order_shape_host_reads.add("barstate_islast_")
955
956
  return "barstate_islast_"
956
957
  if node.member == "isnew":
957
958
  return "is_first_tick()"
@@ -962,6 +963,7 @@ class ExprVisitor:
962
963
  if node.member == "isrealtime":
963
964
  return "false"
964
965
  if node.member == "islastconfirmedhistory":
966
+ self._order_shape_host_reads.add("barstate_islast_")
965
967
  return "barstate_islast_"
966
968
  return "false"
967
969
  if ns in ("backadjustment", "settlement_as_close"):
@@ -1046,12 +1048,16 @@ class ExprVisitor:
1046
1048
  return ismarket
1047
1049
  return f"(!{ismarket} && {predicate})"
1048
1050
  if node.member == "isfirstbar":
1051
+ self._order_shape_host_reads.add("session_isfirstbar_")
1049
1052
  return "session_isfirstbar_"
1050
1053
  if node.member == "islastbar":
1054
+ self._order_shape_host_reads.add("session_islastbar_")
1051
1055
  return "session_islastbar_"
1052
1056
  if node.member == "isfirstbar_regular":
1057
+ self._order_shape_host_reads.add("session_isfirstbar_regular_")
1053
1058
  return "session_isfirstbar_regular_"
1054
1059
  if node.member == "islastbar_regular":
1060
+ self._order_shape_host_reads.add("session_islastbar_regular_")
1055
1061
  return "session_islastbar_regular_"
1056
1062
  return "false"
1057
1063
  if ns == "syminfo":
@@ -1083,7 +1089,7 @@ class ExprVisitor:
1083
1089
  if ns in SKIP_NAMESPACES:
1084
1090
  return "0"
1085
1091
  if ns == "currency":
1086
- return f'std::string("{node.member}")'
1092
+ return f'std::string("{self._cpp_string_escape(node.member)}")'
1087
1093
  if ns == "order":
1088
1094
  # order.ascending / order.descending. Unknown member -> "ascending".
1089
1095
  return ORDER_DIRECTION_MAP.get(node.member, 'std::string("ascending")')
@@ -1163,7 +1169,7 @@ class ExprVisitor:
1163
1169
  return f"{safe}.{node.member}"
1164
1170
  if name not in self.ctx.series_vars:
1165
1171
  # Unknown identifier — likely an enum value
1166
- return f'std::string("{node.member}")'
1172
+ return f'std::string("{self._cpp_string_escape(node.member)}")'
1167
1173
  return f"{obj}.{node.member}"
1168
1174
 
1169
1175
  def _operand_na_kind(self, node, cpp_type: str) -> str | None:
@@ -1100,10 +1100,8 @@ class StmtVisitor:
1100
1100
  # the backtest runtime. See: pineforge-codegen issue #10.
1101
1101
  if self._is_omitted_udt_field(node.target):
1102
1102
  recv = self._visit_expr(node.target.object)
1103
- lines.append(
1104
- f"{pad}/* drawing field assignment omitted: "
1105
- f"{recv}.{node.target.member} {node.op} ... */"
1106
- )
1103
+ omitted = self._cpp_comment_escape(f"{recv}.{node.target.member} {node.op}")
1104
+ lines.append(f"{pad}/* drawing field assignment omitted: {omitted} ... */")
1107
1105
  return
1108
1106
  # General expression target (e.g., member access)
1109
1107
  target_cpp = self._visit_mutable_expr(node.target)
@@ -0,0 +1,341 @@
1
+ """Stable codes and named arguments for every transpile diagnostic.
2
+
3
+ Every :class:`~pineforge_codegen.errors.Diagnostic` carries a ``code``
4
+ (``PF-E1203`` for an error, ``PF-W0412`` for a warning) and ``args``, the
5
+ named data its English ``message`` and ``hint`` were built from. The codes,
6
+ their severities, English ICU MessageFormat templates and one-line
7
+ explanations live in ``diagnostics_catalog.json`` beside this module, which
8
+ ships with the package and is returned by :func:`diagnostics_catalog`.
9
+
10
+ The emitters keep spelling their English text as they always have, so the
11
+ ``message`` stays byte for byte what it was; a diagnostic is coded by the
12
+ catalog template its text renders from (:func:`classify`). The template's
13
+ literal text must equal the message's around its arguments, so a code is
14
+ never guessed: a text no template renders gets the uncatalogued code of its
15
+ severity (``PF-E0000`` / ``PF-W0000``), which the test suite refuses.
16
+
17
+ The match never backtracks over the text's characters: each literal segment
18
+ of a template goes to its leftmost place after the previous one, the last to
19
+ the text's end, and every argument is the text between its literals -- the
20
+ split a fullmatch of the template with lazy ``(.*?)`` arguments returns. Where
21
+ an argument is named twice (``{receiver}`` in the message and the hint) the
22
+ leftmost split can name it two values; then the later places of a literal are
23
+ tried in order, as the regex's backtracking tries them, within a fixed budget
24
+ of steps. The regex took seconds, then minutes, as a crafted message grew,
25
+ and a user's script spells argument text.
26
+
27
+ Rendering (:func:`render`) is the ICU MessageFormat subset the catalog uses:
28
+ literal text with ICU apostrophe quoting (``''`` is one apostrophe, ``'{'``
29
+ a literal brace) and simple ``{name}`` arguments, a string argument
30
+ inserted as is and a number in plain decimal digits.
31
+ """
32
+
33
+ from __future__ import annotations
34
+
35
+ import json
36
+ import re
37
+ from functools import lru_cache
38
+ from pathlib import Path
39
+ from typing import Any
40
+
41
+ CATALOG_PATH = Path(__file__).with_name("diagnostics_catalog.json")
42
+ CATALOG_SCHEMA = "pineforge-diagnostics-catalog/v1"
43
+ UNCATALOGUED = {"error": "PF-E0000", "warning": "PF-W0000"}
44
+
45
+
46
+ @lru_cache(maxsize=1)
47
+ def _load() -> dict:
48
+ with CATALOG_PATH.open(encoding="utf-8") as handle:
49
+ return json.load(handle)
50
+
51
+
52
+ def diagnostics_catalog() -> dict:
53
+ """The diagnostics catalog: ``{"schema", "codes": {code: entry}}``.
54
+
55
+ Each entry gives ``severity`` (``error`` / ``warning``), ``area``, the
56
+ English ICU MessageFormat ``message`` template, the ``hint`` template or
57
+ ``None``, a one-line ``explanation`` and ``args``: per argument name its
58
+ ``kind`` (``identifier``, ``type``, ``keyword``, ``number``, ``vocab`` or
59
+ ``text``) and, for ``vocab``, its closed set of ``values``. A fresh copy
60
+ is returned on every call.
61
+ """
62
+ return json.loads(json.dumps(_load()))
63
+
64
+
65
+ # ---------------------------------------------------------------------------
66
+ # ICU MessageFormat subset
67
+ # ---------------------------------------------------------------------------
68
+
69
+ _SPECIAL = "{}"
70
+
71
+
72
+ def parse_template(template: str) -> list:
73
+ """Split a catalog template into literal strings and ``(name,)`` argument
74
+ tuples, applying ICU's apostrophe rules (``ApostropheMode.DOUBLE_OPTIONAL``,
75
+ the ICU and FormatJS default)."""
76
+ parts: list = []
77
+ text: list[str] = []
78
+ i, n = 0, len(template)
79
+ while i < n:
80
+ ch = template[i]
81
+ if ch == "'":
82
+ if i + 1 < n and template[i + 1] == "'":
83
+ text.append("'")
84
+ i += 2
85
+ continue
86
+ if i + 1 < n and template[i + 1] in _SPECIAL:
87
+ # A quoted literal runs to the next lone apostrophe.
88
+ i += 1
89
+ while i < n:
90
+ if template[i] == "'":
91
+ if i + 1 < n and template[i + 1] == "'":
92
+ text.append("'")
93
+ i += 2
94
+ continue
95
+ i += 1
96
+ break
97
+ text.append(template[i])
98
+ i += 1
99
+ continue
100
+ text.append("'")
101
+ i += 1
102
+ continue
103
+ if ch == "{":
104
+ end = template.find("}", i)
105
+ if end < 0:
106
+ raise ValueError(f"unclosed argument in template: {template!r}")
107
+ name = template[i + 1:end].strip()
108
+ if not re.fullmatch(r"[A-Za-z_][A-Za-z0-9_]*", name):
109
+ raise ValueError(f"unsupported argument {name!r} in template: {template!r}")
110
+ if text:
111
+ parts.append("".join(text))
112
+ text = []
113
+ parts.append((name,))
114
+ i = end + 1
115
+ continue
116
+ if ch == "}":
117
+ raise ValueError(f"unbalanced '}}' in template: {template!r}")
118
+ text.append(ch)
119
+ i += 1
120
+ if text:
121
+ parts.append("".join(text))
122
+ return parts
123
+
124
+
125
+ def escape_literal(text: str) -> str:
126
+ """ICU-escape literal text: every apostrophe doubled, every brace quoted."""
127
+ out: list[str] = []
128
+ for ch in text:
129
+ if ch == "'":
130
+ out.append("''")
131
+ elif ch in _SPECIAL:
132
+ out.append("'" + ch + "'")
133
+ else:
134
+ out.append(ch)
135
+ return "".join(out)
136
+
137
+
138
+ def format_arg(value: Any) -> str:
139
+ """An argument's text: a string as is, a number in plain decimal digits."""
140
+ if isinstance(value, bool):
141
+ raise TypeError("diagnostic arguments are strings or numbers")
142
+ if isinstance(value, int):
143
+ return str(value)
144
+ if isinstance(value, float):
145
+ return repr(value)
146
+ if isinstance(value, str):
147
+ return value
148
+ raise TypeError(f"diagnostic argument of type {type(value).__name__}")
149
+
150
+
151
+ def render(template: str | None, args: dict) -> str | None:
152
+ """Render a catalog template with a diagnostic's ``args``."""
153
+ if template is None:
154
+ return None
155
+ return "".join(
156
+ part if isinstance(part, str) else format_arg(args[part[0]])
157
+ for part in parse_template(template)
158
+ )
159
+
160
+
161
+ def render_diagnostic(code: str, args: dict) -> tuple[str, str | None]:
162
+ """The English ``(message, hint)`` a code renders with ``args``."""
163
+ entry = _load()["codes"][code]
164
+ return render(entry["message"], args), render(entry.get("hint"), args)
165
+
166
+
167
+ # ---------------------------------------------------------------------------
168
+ # Classification
169
+ # ---------------------------------------------------------------------------
170
+
171
+ _CANONICAL_INT = re.compile(r"-?(?:0|[1-9][0-9]*)")
172
+ _CANONICAL_FLOAT = re.compile(r"-?(?:0|[1-9][0-9]*)\.[0-9]+")
173
+
174
+
175
+ def _segments(parts: list) -> tuple[str, tuple[tuple[str, str], ...]]:
176
+ """A parsed template as its leading literal and ``(argument, literal after
177
+ it)`` pairs; the literal after an argument may be empty."""
178
+ head = ""
179
+ pairs: list[list[str]] = []
180
+ for part in parts:
181
+ if isinstance(part, str):
182
+ if pairs:
183
+ pairs[-1][1] += part
184
+ else:
185
+ head += part
186
+ else:
187
+ pairs.append([part[0], ""])
188
+ return head, tuple((name, literal) for name, literal in pairs)
189
+
190
+
191
+ # Literal places a template whose argument is named twice may try, per
192
+ # classification: a script's text never needs more than a few, and a crafted
193
+ # one gets the uncatalogued code instead of a long search.
194
+ _SEARCH_BUDGET = 4096
195
+
196
+
197
+ def _search(segments: list, texts: list[str], repeats: bool) -> dict[str, str] | None:
198
+ """Split ``texts`` (the message, then the hint) by their templates'
199
+ ``segments``: each argument ends at the leftmost place of the literal after
200
+ it, the last argument of a text at its end -- the split a lazy-regex
201
+ fullmatch returns. Without an argument named twice (``repeats``) that split
202
+ succeeds whenever any does, since a leftmost place leaves the rest the most
203
+ room: it is the only one tried, so the work is linear. With one, a split can
204
+ give the name two values; then the later places of a literal are tried in
205
+ order, as the regex backtracks, within ``_SEARCH_BUDGET`` places."""
206
+ starts: list[int] = []
207
+ ends: list[int] = []
208
+ slots: list[tuple[int, str, str, bool]] = [] # text, argument, literal after it, last
209
+ for index, ((head, pairs), text) in enumerate(zip(segments, texts)):
210
+ if not text.startswith(head):
211
+ return None
212
+ if not pairs:
213
+ if len(text) != len(head):
214
+ return None
215
+ starts.append(len(head))
216
+ ends.append(len(head))
217
+ continue
218
+ tail = pairs[-1][1]
219
+ end = len(text) - len(tail)
220
+ if end < len(head) or not text.endswith(tail):
221
+ return None
222
+ starts.append(len(head))
223
+ ends.append(end)
224
+ for at, (name, literal) in enumerate(pairs):
225
+ slots.append((index, name, literal, at == len(pairs) - 1))
226
+ found: dict[str, str] = {}
227
+ budget = [_SEARCH_BUDGET]
228
+
229
+ def step(slot: int, pos: int) -> bool:
230
+ if slot == len(slots):
231
+ return True
232
+ text_index, name, literal, last = slots[slot]
233
+ text, end = texts[text_index], ends[text_index]
234
+ following = slot + 1
235
+ known = found.get(name)
236
+ if last:
237
+ value = text[pos:end]
238
+ resume = (starts[slots[following][0]] if following < len(slots) else end)
239
+ if known is not None:
240
+ return known == value and step(following, resume)
241
+ found[name] = value
242
+ if step(following, resume):
243
+ return True
244
+ del found[name]
245
+ return False
246
+ if known is not None:
247
+ stop = pos + len(known)
248
+ return (text.startswith(known, pos) and stop + len(literal) <= end
249
+ and text.startswith(literal, stop) and step(following, stop + len(literal)))
250
+ at = text.find(literal, pos, end)
251
+ while at >= 0:
252
+ budget[0] -= 1
253
+ if budget[0] < 0:
254
+ return False
255
+ found[name] = text[pos:at]
256
+ if step(following, at + len(literal)):
257
+ return True
258
+ del found[name]
259
+ if not repeats:
260
+ return False
261
+ at = text.find(literal, at + 1, end) if at < end else -1
262
+ return False
263
+
264
+ first = starts[slots[0][0]] if slots else 0
265
+ return found if step(0, first) else None
266
+
267
+
268
+ class _Matcher:
269
+ __slots__ = ("code", "has_hint", "prefix", "suffix", "specificity",
270
+ "_message", "_hint", "_kinds", "_repeats")
271
+
272
+ def __init__(self, code: str, entry: dict):
273
+ self.code = code
274
+ message = parse_template(entry["message"])
275
+ hint = entry.get("hint")
276
+ hint_parts = parse_template(hint) if hint is not None else []
277
+ self.has_hint = hint is not None
278
+ self._message = _segments(message)
279
+ self._hint = _segments(hint_parts) if hint is not None else None
280
+ self.prefix = message[0] if message and isinstance(message[0], str) else ""
281
+ self.suffix = message[-1] if message and isinstance(message[-1], str) else ""
282
+ # The text a template spells itself; the hint's separator counted as one
283
+ # character, as the order of the catalog's codes has always assumed.
284
+ self.specificity = (sum(len(p) for p in message + hint_parts if isinstance(p, str))
285
+ + (1 if hint is not None else 0))
286
+ self._kinds = {name: spec.get("kind") for name, spec in entry.get("args", {}).items()}
287
+ names = [part[0] for part in message + hint_parts if isinstance(part, tuple)]
288
+ self._repeats = len(names) != len(set(names))
289
+
290
+ def match(self, message: str, hint: str | None) -> dict | None:
291
+ segments = [self._message] + ([self._hint] if self._hint is not None else [])
292
+ texts = [message] + ([hint] if self._hint is not None else [])
293
+ found = _search(segments, texts, self._repeats)
294
+ if found is None:
295
+ return None
296
+ args: dict[str, Any] = {}
297
+ for name, value in found.items():
298
+ if self._kinds.get(name) == "number":
299
+ if _CANONICAL_INT.fullmatch(value):
300
+ args[name] = int(value)
301
+ continue
302
+ if _CANONICAL_FLOAT.fullmatch(value) and repr(float(value)) == value:
303
+ args[name] = float(value)
304
+ continue
305
+ args[name] = value
306
+ return args
307
+
308
+
309
+ @lru_cache(maxsize=1)
310
+ def _matchers() -> dict[str, list[_Matcher]]:
311
+ by_severity: dict[str, list[_Matcher]] = {"error": [], "warning": []}
312
+ for code, entry in _load()["codes"].items():
313
+ if code in UNCATALOGUED.values():
314
+ continue
315
+ by_severity[entry["severity"]].append(_Matcher(code, entry))
316
+ for matchers in by_severity.values():
317
+ # The template with the most literal text wins: a message two templates
318
+ # render reads as the more specific one (both render it byte for byte).
319
+ matchers.sort(key=lambda m: (-m.specificity, m.code))
320
+ return by_severity
321
+
322
+
323
+ def classify(severity: str, message: str, hint: str | None = None) -> tuple[str, dict]:
324
+ """The ``(code, args)`` of a diagnostic's English text.
325
+
326
+ ``severity`` is ``"error"`` or ``"warning"``. A text no catalog template
327
+ renders gets ``PF-E0000`` / ``PF-W0000`` with its text as ``args``.
328
+ """
329
+ for matcher in _matchers().get(severity, ()):
330
+ if matcher.has_hint != (hint is not None):
331
+ continue
332
+ if not message.startswith(matcher.prefix) or not message.endswith(matcher.suffix):
333
+ continue
334
+ args = matcher.match(message, hint)
335
+ if args is not None:
336
+ return matcher.code, args
337
+ args = {"message": message}
338
+ if hint is not None:
339
+ args["hint"] = hint
340
+ code = UNCATALOGUED.get(severity, UNCATALOGUED["error"])
341
+ return code, args