uplox 3.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
uplox/__init__.py ADDED
@@ -0,0 +1,4 @@
1
+ """uplox — compiler front-end generator."""
2
+
3
+ __version__ = "3.0.0"
4
+ UPLOX_SCHEMA_VERSION = "1"
uplox/ast/__init__.py ADDED
@@ -0,0 +1,150 @@
1
+ """AST node schema and the default tree-builder action set.
2
+
3
+ The parser runtime in :mod:`uplox.parse.runtime` produces :class:`uplox.parse.runtime.ParseNode`
4
+ trees by default — a one-to-one mirror of the grammar's parse tree. That is fine
5
+ for prototyping but generally too verbose for downstream tools.
6
+
7
+ This module gives:
8
+
9
+ * A neutral :class:`AstNode` schema that grammar authors can target through
10
+ semantic actions, with one node kind per concept rather than per production.
11
+ * The :func:`build_ast` helper: walks a ParseNode tree and applies user-supplied
12
+ per-rule reducers to flatten chains, drop literal punctuation, and lift
13
+ meaningful children. This is the "default tree-builder" deliverable from the
14
+ Phase-4 plan.
15
+ * JSON ser/deser in :func:`ast_to_json` / :func:`ast_from_json` so AST nodes
16
+ fit cleanly into the bundle's ``ast`` section once a backend wants to
17
+ pre-compute them.
18
+
19
+ What this module deliberately does *not* do:
20
+
21
+ * Type checking, name resolution, scope handling — those are hooks (see
22
+ :mod:`uplox.hooks`), not AST concerns.
23
+ * Pretty-printing — would couple AST shape to host-language style; left to
24
+ callers.
25
+ """
26
+
27
+ from __future__ import annotations
28
+
29
+ from dataclasses import dataclass, field
30
+ from typing import Any, Callable, Iterable, Optional, Union
31
+
32
+ from ..lex.scanner import Token
33
+ from ..parse.runtime import ParseNode
34
+
35
+ AstChild = Union["AstNode", Token, str, int, float, bool, None]
36
+
37
+
38
+ @dataclass
39
+ class AstNode:
40
+ """Neutral AST node. Distinct from :class:`ParseNode` so the AST shape can
41
+ diverge from the grammar's parse tree.
42
+
43
+ ``kind`` is a stable string the user picks (e.g. ``"BinaryOp"``). ``attrs``
44
+ holds named per-node fields (operator, name, value). ``children`` are
45
+ ordered children in source order.
46
+ """
47
+
48
+ kind: str
49
+ attrs: dict[str, Any] = field(default_factory=dict)
50
+ children: list[AstChild] = field(default_factory=list)
51
+ position: Optional[tuple[int, int]] = None # (line, column) of the first token
52
+
53
+ def __repr__(self) -> str: # pragma: no cover - cosmetic
54
+ return f"AstNode({self.kind!r}, attrs={self.attrs!r}, children={self.children!r})"
55
+
56
+
57
+ # A reducer takes a ParseNode (or token leaf) and returns an AstChild.
58
+ # The dispatch is by ParseNode.kind (i.e. the grammar's LHS for that node).
59
+ Reducer = Callable[[ParseNode, list[AstChild]], AstChild]
60
+
61
+
62
+ def build_ast(
63
+ root: ParseNode | Token,
64
+ reducers: dict[str, Reducer],
65
+ *,
66
+ drop_tokens: Iterable[str] = (),
67
+ ) -> AstChild:
68
+ """Walk ``root`` bottom-up; transform each ParseNode through its reducer.
69
+
70
+ For each ParseNode encountered:
71
+ * Recursively transform every child (Token children become themselves
72
+ unless their token name is in ``drop_tokens``, in which case they are
73
+ filtered out — handy for punctuation like LPAREN/SEMI).
74
+ * Look up ``reducers[node.kind]``. If present, call it with the original
75
+ ParseNode and the (already-transformed) child list. The return value
76
+ is what ends up in the parent's child list.
77
+ * If no reducer is registered, fall back to wrapping the node in an
78
+ AstNode with the same ``kind``.
79
+
80
+ Token leaves at the root level pass through unchanged (the caller can
81
+ map them as they please).
82
+ """
83
+ drop_set = set(drop_tokens)
84
+
85
+ def visit(node: ParseNode | Token) -> AstChild:
86
+ if isinstance(node, Token):
87
+ return node
88
+ new_children: list[AstChild] = []
89
+ for c in node.children:
90
+ if isinstance(c, Token) and c.name in drop_set:
91
+ continue
92
+ new_children.append(visit(c) if isinstance(c, (ParseNode, Token)) else c)
93
+ reducer = reducers.get(node.kind)
94
+ if reducer is None:
95
+ return AstNode(kind=node.kind, children=new_children)
96
+ return reducer(node, new_children)
97
+
98
+ return visit(root)
99
+
100
+
101
+ # ---- JSON serialisation ------------------------------------------------------
102
+
103
+
104
+ def ast_to_json(node: AstChild) -> Any:
105
+ """Serialise an AST tree to JSON-compatible primitives.
106
+
107
+ AstNodes become dicts with ``kind``, ``attrs``, ``children`` (and ``position``
108
+ when set). Tokens become ``{"_token": name, "text": ..., "line": ..., "column": ...}``.
109
+ Plain Python primitives pass through untouched. The shape is deliberately
110
+ explicit so a backend in any language can reconstruct it without inferring
111
+ types.
112
+ """
113
+ if isinstance(node, AstNode):
114
+ out: dict[str, Any] = {
115
+ "_kind": node.kind,
116
+ "attrs": dict(node.attrs),
117
+ "children": [ast_to_json(c) for c in node.children],
118
+ }
119
+ if node.position is not None:
120
+ out["position"] = list(node.position)
121
+ return out
122
+ if isinstance(node, Token):
123
+ return {
124
+ "_token": node.name,
125
+ "text": node.text,
126
+ "line": node.line,
127
+ "column": node.column,
128
+ }
129
+ return node
130
+
131
+
132
+ def ast_from_json(value: Any) -> AstChild:
133
+ if isinstance(value, dict):
134
+ if "_token" in value:
135
+ return Token(
136
+ name=value["_token"],
137
+ text=value["text"],
138
+ line=value["line"],
139
+ column=value["column"],
140
+ offset=value.get("offset", -1),
141
+ )
142
+ if "_kind" in value:
143
+ pos = value.get("position")
144
+ return AstNode(
145
+ kind=value["_kind"],
146
+ attrs=dict(value.get("attrs", {})),
147
+ children=[ast_from_json(c) for c in value.get("children", [])],
148
+ position=tuple(pos) if pos else None,
149
+ )
150
+ return value
uplox/cli/__init__.py ADDED
@@ -0,0 +1 @@
1
+ """``uplox`` command-line interface."""
uplox/cli/main.py ADDED
@@ -0,0 +1,312 @@
1
+ """``uplox`` command entry point.
2
+
3
+ Subcommands:
4
+
5
+ * ``uplox version`` — print uplox version and schema version.
6
+ * ``uplox build <grammar.uplox> -o <out.json>``
7
+ — build the JSON bundle. In Phase 2 only the
8
+ lex section is populated; the parse / ast /
9
+ hooks sections are emitted empty so backends
10
+ can already start consuming bundles.
11
+ * ``uplox check <grammar.uplox>`` — parse + lower without emitting JSON; reports
12
+ syntax errors and lex-construction failures.
13
+ * ``uplox emit <bundle.json> --target=c|cpp|py|lua --out=<dir>``
14
+ — drive a backend. Stubbed until Phase 7-8.
15
+ """
16
+
17
+ from __future__ import annotations
18
+
19
+ import argparse
20
+ import sys
21
+
22
+ from .. import UPLOX_SCHEMA_VERSION, __version__
23
+ from ..lex.build import lex_from_ir
24
+ from ..lex.scanner import Scanner
25
+ from ..parse.grammar import GrammarError, compile_grammar
26
+ from ..parse.lr1 import build_lr1
27
+ from ..gen.c import emit_c
28
+ from ..gen.cpp import emit_cpp
29
+ from ..gen.lua import emit_lua
30
+ from ..gen.py import emit_py
31
+ from ..parse.glr import GLRParseError, glr_from_lr, glr_parse
32
+ from ..parse.glr.runtime import AmbiguityNode, GLRNode
33
+ from ..parse.runtime import HookRegistry, ParseError, parse as run_parser
34
+ from ..spec.reader import ReaderError, read_file
35
+ from ..lex.build import balanced_tokens
36
+ from ..tables import (
37
+ balanced_from_json,
38
+ dfa_from_json,
39
+ dfa_to_json,
40
+ dump_bundle,
41
+ empty_bundle,
42
+ table_from_json,
43
+ table_to_json,
44
+ )
45
+
46
+
47
+ def _cmd_version(_args: argparse.Namespace) -> int:
48
+ print(f"uplox {__version__} (schema {UPLOX_SCHEMA_VERSION})")
49
+ return 0
50
+
51
+
52
+ def _cmd_build(args: argparse.Namespace) -> int:
53
+ try:
54
+ ir = read_file(args.source)
55
+ except ReaderError as e:
56
+ print(str(e), file=sys.stderr)
57
+ return 1
58
+ try:
59
+ dfa, tokens, skip = lex_from_ir(ir)
60
+ except ValueError as e:
61
+ print(f"{args.source}: {e}", file=sys.stderr)
62
+ return 1
63
+
64
+ bundle = empty_bundle(ir.name)
65
+ bundle["lex"] = dfa_to_json(
66
+ dfa, tokens=tokens, skip=skip, balanced=balanced_tokens(ir)
67
+ )
68
+
69
+ if not args.lex_only:
70
+ try:
71
+ grammar = compile_grammar(ir)
72
+ table = build_lr1(grammar)
73
+ except GrammarError as e:
74
+ print(f"{args.source}: {e}", file=sys.stderr)
75
+ return 1
76
+ if table.conflicts:
77
+ print(
78
+ f"{args.source}: refusing to build with {len(table.conflicts)} parser conflict(s):",
79
+ file=sys.stderr,
80
+ )
81
+ for c in table.conflicts:
82
+ print(c.describe(table.grammar), file=sys.stderr)
83
+ print("", file=sys.stderr)
84
+ return 1
85
+ bundle["parse"] = table_to_json(table)
86
+
87
+ text = dump_bundle(bundle)
88
+ if args.output == "-":
89
+ sys.stdout.write(text)
90
+ else:
91
+ with open(args.output, "w", encoding="utf-8") as fh:
92
+ fh.write(text)
93
+ return 0
94
+
95
+
96
+ def _cmd_parse(args: argparse.Namespace) -> int:
97
+ """Smoke-parse: load a bundle, run scanner+parser on stdin or a path, dump tree."""
98
+ import json as _json
99
+
100
+ with open(args.bundle, "r", encoding="utf-8") as fh:
101
+ bundle = _json.load(fh)
102
+ if not bundle.get("parse"):
103
+ print(f"{args.bundle}: bundle has no parse section (was it built --lex-only?)", file=sys.stderr)
104
+ return 1
105
+
106
+ dfa, _tokens, skip = dfa_from_json(bundle["lex"])
107
+ scanner = Scanner(
108
+ dfa=dfa,
109
+ skip_tokens=frozenset(skip),
110
+ balanced=balanced_from_json(bundle["lex"]),
111
+ )
112
+ table = table_from_json(bundle["parse"])
113
+
114
+ if args.input == "-":
115
+ text = sys.stdin.read()
116
+ else:
117
+ with open(args.input, "r", encoding="utf-8") as fh:
118
+ text = fh.read()
119
+
120
+ try:
121
+ if args.glr:
122
+ tree = glr_parse(glr_from_lr(table), scanner.scan(text))
123
+ else:
124
+ # The LR runtime is a smoke tool here; we don't try to resolve
125
+ # hooks the way a real host driver would. Unknown names no-op so
126
+ # any grammar builds and parses end-to-end.
127
+ tree = run_parser(
128
+ table,
129
+ scanner.scan(text),
130
+ hooks=HookRegistry(ignore_missing=True),
131
+ )
132
+ except (ParseError, GLRParseError) as e:
133
+ print(f"{args.input}: {e}", file=sys.stderr)
134
+ return 1
135
+
136
+ sys.stdout.write(_render_tree(tree) + "\n")
137
+ return 0
138
+
139
+
140
+ def _render_tree(tree, indent: int = 0) -> str:
141
+ from ..lex.scanner import Token
142
+ from ..parse.runtime import ParseNode
143
+ pad = " " * indent
144
+ if isinstance(tree, Token):
145
+ return f"{pad}{tree.name} {tree.text!r}"
146
+ if isinstance(tree, (ParseNode, GLRNode)):
147
+ lines = [f"{pad}{tree.kind}"]
148
+ for c in tree.children:
149
+ lines.append(_render_tree(c, indent + 1))
150
+ return "\n".join(lines)
151
+ if isinstance(tree, AmbiguityNode):
152
+ lines = [f"{pad}AMBIGUITY[{tree.kind}] ({len(tree.alternatives)} alternatives)"]
153
+ for i, alt in enumerate(tree.alternatives):
154
+ lines.append(f"{pad} alt {i + 1}:")
155
+ lines.append(_render_tree(alt, indent + 2))
156
+ return "\n".join(lines)
157
+ return f"{pad}{tree!r}"
158
+
159
+
160
+ def _cmd_check(args: argparse.Namespace) -> int:
161
+ try:
162
+ ir = read_file(args.source)
163
+ lex_from_ir(ir)
164
+ except (ReaderError, ValueError) as e:
165
+ print(str(e), file=sys.stderr)
166
+ return 1
167
+
168
+ parser_summary = ""
169
+ parser_conflicts = 0
170
+ try:
171
+ grammar = compile_grammar(ir)
172
+ table = build_lr1(grammar)
173
+ parser_conflicts = len(table.conflicts)
174
+ if table.conflicts:
175
+ print(f"{args.source}: {parser_conflicts} parser conflict(s):", file=sys.stderr)
176
+ for c in table.conflicts:
177
+ print(c.describe(table.grammar), file=sys.stderr)
178
+ print("", file=sys.stderr)
179
+ parser_summary = (
180
+ f", {len(grammar.productions)} productions, "
181
+ f"{len(table.states)} states, {parser_conflicts} conflicts"
182
+ )
183
+ except GrammarError as e:
184
+ print(f"{args.source}: {e}", file=sys.stderr)
185
+ return 1
186
+
187
+ print(
188
+ f"{args.source}: {ir.name} — {len(ir.tokens)} tokens, "
189
+ f"{len(ir.hooks)} hooks{parser_summary}"
190
+ )
191
+ return 1 if parser_conflicts else 0
192
+
193
+
194
+ def _cmd_emit(args: argparse.Namespace) -> int:
195
+ import json as _json
196
+ import os as _os
197
+
198
+ with open(args.bundle, "r", encoding="utf-8") as fh:
199
+ bundle = _json.load(fh)
200
+
201
+ grammar = (args.prefix or bundle.get("meta", {}).get("grammar") or "grammar").lower()
202
+ _os.makedirs(args.out, exist_ok=True)
203
+
204
+ try:
205
+ if args.target == "c":
206
+ header, impl = emit_c(bundle, prefix=args.prefix)
207
+ header_path = _os.path.join(args.out, f"uplox_{grammar}.h")
208
+ impl_path = _os.path.join(args.out, f"uplox_{grammar}.c")
209
+ elif args.target == "cpp":
210
+ header, impl = emit_cpp(bundle, prefix=args.prefix)
211
+ header_path = _os.path.join(args.out, f"uplox_{grammar}.hpp")
212
+ impl_path = _os.path.join(args.out, f"uplox_{grammar}.cpp")
213
+ elif args.target == "lua":
214
+ module_text = emit_lua(bundle, prefix=args.prefix)
215
+ module_path = _os.path.join(args.out, f"uplox_{grammar}.lua")
216
+ with open(module_path, "w", encoding="utf-8") as fh:
217
+ fh.write(module_text)
218
+ print(f"wrote {module_path}")
219
+ return 0
220
+ elif args.target == "py":
221
+ module_text = emit_py(bundle, prefix=args.prefix)
222
+ module_path = _os.path.join(args.out, f"uplox_{grammar}.py")
223
+ with open(module_path, "w", encoding="utf-8") as fh:
224
+ fh.write(module_text)
225
+ print(f"wrote {module_path}")
226
+ return 0
227
+ else:
228
+ print(
229
+ f"uplox emit --target={args.target}: unknown target",
230
+ file=sys.stderr,
231
+ )
232
+ return 2
233
+ except ValueError as e:
234
+ print(f"{args.bundle}: {e}", file=sys.stderr)
235
+ return 1
236
+
237
+ with open(header_path, "w", encoding="utf-8") as fh:
238
+ fh.write(header)
239
+ with open(impl_path, "w", encoding="utf-8") as fh:
240
+ fh.write(impl)
241
+ print(f"wrote {header_path}\nwrote {impl_path}")
242
+ return 0
243
+
244
+
245
+ def build_parser() -> argparse.ArgumentParser:
246
+ parser = argparse.ArgumentParser(
247
+ prog="uplox",
248
+ description="Compiler front-end generator (grammar -> JSON tables + drivers)",
249
+ )
250
+ sub = parser.add_subparsers(dest="command", required=True)
251
+
252
+ p_version = sub.add_parser("version", help="print uplox and schema versions")
253
+ p_version.set_defaults(func=_cmd_version)
254
+
255
+ p_build = sub.add_parser("build", help="compile a .uplox grammar to a JSON bundle")
256
+ p_build.add_argument("source", help="path to .uplox source file")
257
+ p_build.add_argument(
258
+ "-o", "--output",
259
+ default="-",
260
+ help="output bundle path; '-' (default) writes to stdout",
261
+ )
262
+ p_build.add_argument(
263
+ "--lex-only",
264
+ action="store_true",
265
+ help="emit only the lex section (skip the LR table)",
266
+ )
267
+ p_build.set_defaults(func=_cmd_build)
268
+
269
+ p_check = sub.add_parser("check", help="parse and validate a .uplox grammar without emitting")
270
+ p_check.add_argument("source", help="path to .uplox source file")
271
+ p_check.set_defaults(func=_cmd_check)
272
+
273
+ p_parse = sub.add_parser(
274
+ "parse",
275
+ help="parse input through a built bundle and pretty-print the parse tree",
276
+ )
277
+ p_parse.add_argument("bundle", help="path to JSON bundle (output of `uplox build`)")
278
+ p_parse.add_argument(
279
+ "input",
280
+ help="path to input file; '-' reads from stdin",
281
+ nargs="?",
282
+ default="-",
283
+ )
284
+ p_parse.add_argument(
285
+ "--glr",
286
+ action="store_true",
287
+ help="parse with the GLR runtime (handles ambiguous grammars; produces a parse forest)",
288
+ )
289
+ p_parse.set_defaults(func=_cmd_parse)
290
+
291
+ p_emit = sub.add_parser("emit", help="emit a C driver from a bundle (--target=c is supported in Phase 7)")
292
+ p_emit.add_argument("bundle", help="path to JSON bundle")
293
+ p_emit.add_argument("--target", required=True, choices=["c", "cpp", "py", "lua"])
294
+ p_emit.add_argument("--out", required=True, help="output directory")
295
+ p_emit.add_argument(
296
+ "--prefix",
297
+ default=None,
298
+ help="override grammar name used as the symbol prefix (default: meta.grammar)",
299
+ )
300
+ p_emit.set_defaults(func=_cmd_emit)
301
+
302
+ return parser
303
+
304
+
305
+ def main(argv: list[str] | None = None) -> int:
306
+ parser = build_parser()
307
+ args = parser.parse_args(argv)
308
+ return args.func(args)
309
+
310
+
311
+ if __name__ == "__main__": # pragma: no cover
312
+ raise SystemExit(main())
uplox/gen/__init__.py ADDED
@@ -0,0 +1,5 @@
1
+ """Backends. Each subpackage emits a driver skeleton in one target language.
2
+
3
+ Backends consume only the JSON bundle described in :mod:`uplox.tables.schema`.
4
+ They never read the grammar source directly — the JSON is the contract.
5
+ """
@@ -0,0 +1,12 @@
1
+ """C backend. Emits a self-contained ``.c`` / ``.h`` pair from a JSON bundle.
2
+
3
+ Re-entrant by construction: all parser state lives on a ``uplox_<grammar>_ctx``
4
+ struct, all symbols are prefixed with the grammar name, no file-scope
5
+ mutable state. Two grammars built with uplox can link into the same binary
6
+ without symbol collisions — the design's acceptance test (per
7
+ ``docs/c_backend.md``).
8
+ """
9
+
10
+ from .emit import emit_c
11
+
12
+ __all__ = ["emit_c"]