cpyte 3.3.0__tar.gz → 3.3.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {cpyte-3.3.0/source/cpyte.egg-info → cpyte-3.3.2}/PKG-INFO +16 -2
- {cpyte-3.3.0 → cpyte-3.3.2}/pyproject.toml +1 -1
- {cpyte-3.3.0 → cpyte-3.3.2}/readme.md +16 -2
- cpyte-3.3.2/source/cpyte/__init__.py +1 -0
- {cpyte-3.3.0 → cpyte-3.3.2}/source/cpyte/astparse.py +153 -7
- {cpyte-3.3.0 → cpyte-3.3.2}/source/cpyte/bytecoding.py +109 -43
- cpyte-3.3.2/source/cpyte/clib.py +1416 -0
- {cpyte-3.3.0 → cpyte-3.3.2}/source/cpyte/compiling.py +333 -155
- cpyte-3.3.2/source/cpyte/extension_hooks.py +1115 -0
- {cpyte-3.3.0 → cpyte-3.3.2}/source/cpyte/gc_runtime.c +89 -54
- {cpyte-3.3.0 → cpyte-3.3.2}/source/cpyte/generate_bc.py +81 -53
- {cpyte-3.3.0 → cpyte-3.3.2}/source/cpyte/lexar.py +2 -0
- cpyte-3.3.2/source/cpyte/mainpie.py +903 -0
- {cpyte-3.3.0 → cpyte-3.3.2}/source/cpyte/semantic_analasis.py +882 -68
- {cpyte-3.3.0 → cpyte-3.3.2/source/cpyte.egg-info}/PKG-INFO +16 -2
- {cpyte-3.3.0 → cpyte-3.3.2}/source/cpyte.egg-info/SOURCES.txt +0 -1
- cpyte-3.3.0/source/cpyte/__init__.py +0 -1
- cpyte-3.3.0/source/cpyte/_gc_bc.py +0 -187
- cpyte-3.3.0/source/cpyte/clib.py +0 -1100
- cpyte-3.3.0/source/cpyte/extension_hooks.py +0 -465
- cpyte-3.3.0/source/cpyte/mainpie.py +0 -725
- {cpyte-3.3.0 → cpyte-3.3.2}/MANIFEST.in +0 -0
- {cpyte-3.3.0 → cpyte-3.3.2}/setup.cfg +0 -0
- {cpyte-3.3.0 → cpyte-3.3.2}/source/cpyte/__main__.py +0 -0
- {cpyte-3.3.0 → cpyte-3.3.2}/source/cpyte/_bignum_bc.py +0 -0
- {cpyte-3.3.0 → cpyte-3.3.2}/source/cpyte/_runtime_bc.py +0 -0
- {cpyte-3.3.0 → cpyte-3.3.2}/source/cpyte/bignum.c +0 -0
- {cpyte-3.3.0 → cpyte-3.3.2}/source/cpyte/formatter.py +0 -0
- {cpyte-3.3.0 → cpyte-3.3.2}/source/cpyte/linker.py +0 -0
- {cpyte-3.3.0 → cpyte-3.3.2}/source/cpyte/lsp_server.py +0 -0
- {cpyte-3.3.0 → cpyte-3.3.2}/source/cpyte/package_manifest.py +0 -0
- {cpyte-3.3.0 → cpyte-3.3.2}/source/cpyte/runtime.c +0 -0
- {cpyte-3.3.0 → cpyte-3.3.2}/source/cpyte/runtime_scorpion.c +0 -0
- {cpyte-3.3.0 → cpyte-3.3.2}/source/cpyte/sef.py +0 -0
- {cpyte-3.3.0 → cpyte-3.3.2}/source/cpyte/ui.py +0 -0
- {cpyte-3.3.0 → cpyte-3.3.2}/source/cpyte/update_check.py +0 -0
- {cpyte-3.3.0 → cpyte-3.3.2}/source/cpyte.egg-info/dependency_links.txt +0 -0
- {cpyte-3.3.0 → cpyte-3.3.2}/source/cpyte.egg-info/entry_points.txt +0 -0
- {cpyte-3.3.0 → cpyte-3.3.2}/source/cpyte.egg-info/requires.txt +0 -0
- {cpyte-3.3.0 → cpyte-3.3.2}/source/cpyte.egg-info/top_level.txt +0 -0
- {cpyte-3.3.0 → cpyte-3.3.2}/test/test_bignum_jit.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: cpyte
|
|
3
|
-
Version: 3.3.
|
|
3
|
+
Version: 3.3.2
|
|
4
4
|
Summary: The Cpyte programming language compiler
|
|
5
5
|
Author: Hoang Duy Tung
|
|
6
6
|
License: MIT
|
|
@@ -16,7 +16,7 @@ Check out the official documentation [here](https://gitea.5gnew.io.vn/Cpyte-Proj
|
|
|
16
16
|
Cpyte supports a package extension system that allows packages to extend the compiler with custom keywords, operators, and compiler hooks. Packages can provide:
|
|
17
17
|
|
|
18
18
|
- **Custom Keywords**: Add new language keywords via `package.json`
|
|
19
|
-
- **Custom Operators**: Define new operators for syntax extensions
|
|
19
|
+
- **Custom Operators**: Define new operators for syntax extensions
|
|
20
20
|
- **Compiler Hooks**: Extend lexing, parsing, semantic analysis, and code generation
|
|
21
21
|
- **Runtime Extensions**: Add runtime code and libraries
|
|
22
22
|
|
|
@@ -83,3 +83,17 @@ Cpyte is experimental software. The compiler is continuously tested with fuzzing
|
|
|
83
83
|
Cpyte uses a **concurrent tri-color garbage collector** for automatic memory management. Heap-allocated objects (via `new`) are managed automatically — no manual `free` needed.
|
|
84
84
|
|
|
85
85
|
**Note:** The collector adds a small runtime overhead (~5-10%) compared to manual `malloc/free`. For performance-critical ecosystem projects or embedded use cases, this trade-off may be significant. The GC is required for safety but will slow down programs that do heavy heap allocation.
|
|
86
|
+
|
|
87
|
+
## Cross-Platform Runtime
|
|
88
|
+
|
|
89
|
+
The compiler is fully cross-platform. On every target OS (macOS, Linux, Windows) the C runtime (`runtime.c`), the GC runtime (`gc_runtime.c`), and any embedded `ccode:` blocks are **compiled from source** at JIT/AOT time for the host platform, rather than linking pre-built, OS-specific bitcode. This keeps native helpers and the garbage collector correct on each system:
|
|
90
|
+
|
|
91
|
+
- Portable threading: `pthread` on POSIX; native Windows threads + `CRITICAL_SECTION` on Windows.
|
|
92
|
+
- Portable stack scanning for the GC that works on macOS, Linux, and Windows.
|
|
93
|
+
- A portable sleep/yield helper replaces the POSIX-only `nanosleep`.
|
|
94
|
+
|
|
95
|
+
The compiler auto-discovers a C compiler (`clang` → `cc` → `gcc`) and system linker on the current platform, so `cpy`, `cpy build`, and `cpy --aot` all work without per-OS configuration.
|
|
96
|
+
|
|
97
|
+
## Continuous Integration
|
|
98
|
+
|
|
99
|
+
GitHub Actions (see `.github/workflows/code_quality.yml`) builds and runs a curated, cross-platform regression corpus on Ubuntu (x86_64 + arm64), macOS (x86_64 + arm64), and Windows. `ci_test.py` AOT-compiles each program with the system linker and executes the result to exercise the full pipeline — lexer, parser, semantic analysis, LLVM codegen, the C and GC runtimes, and linking.
|
|
@@ -5,7 +5,7 @@ Check out the official documentation [here](https://gitea.5gnew.io.vn/Cpyte-Proj
|
|
|
5
5
|
Cpyte supports a package extension system that allows packages to extend the compiler with custom keywords, operators, and compiler hooks. Packages can provide:
|
|
6
6
|
|
|
7
7
|
- **Custom Keywords**: Add new language keywords via `package.json`
|
|
8
|
-
- **Custom Operators**: Define new operators for syntax extensions
|
|
8
|
+
- **Custom Operators**: Define new operators for syntax extensions
|
|
9
9
|
- **Compiler Hooks**: Extend lexing, parsing, semantic analysis, and code generation
|
|
10
10
|
- **Runtime Extensions**: Add runtime code and libraries
|
|
11
11
|
|
|
@@ -71,4 +71,18 @@ Cpyte is experimental software. The compiler is continuously tested with fuzzing
|
|
|
71
71
|
|
|
72
72
|
Cpyte uses a **concurrent tri-color garbage collector** for automatic memory management. Heap-allocated objects (via `new`) are managed automatically — no manual `free` needed.
|
|
73
73
|
|
|
74
|
-
**Note:** The collector adds a small runtime overhead (~5-10%) compared to manual `malloc/free`. For performance-critical ecosystem projects or embedded use cases, this trade-off may be significant. The GC is required for safety but will slow down programs that do heavy heap allocation.
|
|
74
|
+
**Note:** The collector adds a small runtime overhead (~5-10%) compared to manual `malloc/free`. For performance-critical ecosystem projects or embedded use cases, this trade-off may be significant. The GC is required for safety but will slow down programs that do heavy heap allocation.
|
|
75
|
+
|
|
76
|
+
## Cross-Platform Runtime
|
|
77
|
+
|
|
78
|
+
The compiler is fully cross-platform. On every target OS (macOS, Linux, Windows) the C runtime (`runtime.c`), the GC runtime (`gc_runtime.c`), and any embedded `ccode:` blocks are **compiled from source** at JIT/AOT time for the host platform, rather than linking pre-built, OS-specific bitcode. This keeps native helpers and the garbage collector correct on each system:
|
|
79
|
+
|
|
80
|
+
- Portable threading: `pthread` on POSIX; native Windows threads + `CRITICAL_SECTION` on Windows.
|
|
81
|
+
- Portable stack scanning for the GC that works on macOS, Linux, and Windows.
|
|
82
|
+
- A portable sleep/yield helper replaces the POSIX-only `nanosleep`.
|
|
83
|
+
|
|
84
|
+
The compiler auto-discovers a C compiler (`clang` → `cc` → `gcc`) and system linker on the current platform, so `cpy`, `cpy build`, and `cpy --aot` all work without per-OS configuration.
|
|
85
|
+
|
|
86
|
+
## Continuous Integration
|
|
87
|
+
|
|
88
|
+
GitHub Actions (see `.github/workflows/code_quality.yml`) builds and runs a curated, cross-platform regression corpus on Ubuntu (x86_64 + arm64), macOS (x86_64 + arm64), and Windows. `ci_test.py` AOT-compiles each program with the system linker and executes the result to exercise the full pipeline — lexer, parser, semantic analysis, LLVM codegen, the C and GC runtimes, and linking.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
__version__ = "3.3.2"
|
|
@@ -4,6 +4,31 @@ from .lexar import Lexer, Token, TokenType, _unescape_run
|
|
|
4
4
|
|
|
5
5
|
_parser_hooks: list[Any] = []
|
|
6
6
|
|
|
7
|
+
# User-declared type names (structs, type aliases, classes) collected in a
|
|
8
|
+
# pre-scan so bare-name casts like `(MyInt)x` are recognized as casts at parse
|
|
9
|
+
# time. Reset per parse_file().
|
|
10
|
+
_user_type_names: set[str] = set()
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def _pre_scan_user_types(tokens):
|
|
14
|
+
"""Collect identifiers declared as struct/type/class so the parser can
|
|
15
|
+
recognize casts to user-defined types even without a pointer suffix."""
|
|
16
|
+
global _user_type_names
|
|
17
|
+
names = set()
|
|
18
|
+
i = 0
|
|
19
|
+
n = len(tokens)
|
|
20
|
+
while i < n:
|
|
21
|
+
t = tokens[i]
|
|
22
|
+
if t.type == TokenType.KEYWORD and t.value in ("struct", "type", "class"):
|
|
23
|
+
j = i + 1
|
|
24
|
+
if j < n and tokens[j].type == TokenType.IDENTIFIER:
|
|
25
|
+
names.add(tokens[j].value)
|
|
26
|
+
i = j + 1
|
|
27
|
+
continue
|
|
28
|
+
i += 1
|
|
29
|
+
_user_type_names = names
|
|
30
|
+
return names
|
|
31
|
+
|
|
7
32
|
|
|
8
33
|
def register_parser_hook(hook: Any) -> None:
|
|
9
34
|
_parser_hooks.append(hook)
|
|
@@ -21,10 +46,10 @@ def _get_all_parser_hooks(enable_extensions: bool) -> list[Any]:
|
|
|
21
46
|
return []
|
|
22
47
|
hooks = list(_parser_hooks)
|
|
23
48
|
try:
|
|
24
|
-
from .extension_hooks import get_global_hook_registry
|
|
49
|
+
from .extension_hooks import HookStage, ParserHook, get_global_hook_registry
|
|
25
50
|
registry = get_global_hook_registry()
|
|
26
|
-
for hook in registry.
|
|
27
|
-
if hook not in hooks:
|
|
51
|
+
for hook in registry.get(HookStage.PARSER):
|
|
52
|
+
if isinstance(hook, ParserHook) and hook not in hooks:
|
|
28
53
|
hooks.append(hook)
|
|
29
54
|
except ImportError:
|
|
30
55
|
pass
|
|
@@ -296,36 +321,75 @@ def _parse_unary(tokens: list[Token], pos: int):
|
|
|
296
321
|
|
|
297
322
|
def _try_parse_cast_type(tokens, pos):
|
|
298
323
|
"""Try to parse a type name for a C-style cast like (int), (size_t*), (float[]).
|
|
324
|
+
Also accepts user-defined types (structs / type aliases / classes) as cast
|
|
325
|
+
targets, e.g. ``(Point*)p`` or ``(MyInt)x``.
|
|
299
326
|
|
|
300
327
|
Returns (type_string, new_pos) on success, or (None, original_pos) if not a cast.
|
|
328
|
+
The returned type string retains pointer/array/ref suffixes, so ``(int*)x``
|
|
329
|
+
decodes to a pointer cast rather than silently degrading to ``int``.
|
|
301
330
|
"""
|
|
302
331
|
if pos >= len(tokens) or tokens[pos].type != TokenType.IDENTIFIER:
|
|
303
332
|
return None, pos
|
|
304
333
|
name = tokens[pos].value
|
|
305
|
-
|
|
306
|
-
return None, pos
|
|
334
|
+
base = name
|
|
307
335
|
pos += 1
|
|
336
|
+
has_suffix = False
|
|
308
337
|
while pos < len(tokens):
|
|
309
338
|
t = tokens[pos].type
|
|
310
339
|
if t == TokenType.STAR:
|
|
340
|
+
base = _type_to_str(base) + "*"
|
|
341
|
+
has_suffix = True
|
|
311
342
|
pos += 1
|
|
312
343
|
elif t == TokenType.POW:
|
|
344
|
+
base = _type_to_str(base) + "**"
|
|
345
|
+
has_suffix = True
|
|
313
346
|
pos += 1
|
|
314
347
|
elif t == TokenType.LBRACKET:
|
|
315
348
|
if pos + 1 < len(tokens) and tokens[pos + 1].type == TokenType.RBRACKET:
|
|
349
|
+
base = _type_to_str(base) + "[]"
|
|
350
|
+
has_suffix = True
|
|
316
351
|
pos += 2
|
|
317
352
|
else:
|
|
318
353
|
break
|
|
319
354
|
elif t == TokenType.AMPERSAND:
|
|
355
|
+
base = _type_to_str(base) + "&"
|
|
356
|
+
has_suffix = True
|
|
320
357
|
pos += 1
|
|
321
358
|
else:
|
|
322
359
|
break
|
|
323
|
-
|
|
360
|
+
# A bare (no-suffix) name is accepted as a cast target only when it is a
|
|
361
|
+
# known builtin type or a user-declared type. Otherwise leave it to the
|
|
362
|
+
# normal parenthesized-expression/call path to avoid ambiguity.
|
|
363
|
+
if not has_suffix and name not in _TYPE_NAMES and name not in _user_type_names:
|
|
364
|
+
return None, pos
|
|
365
|
+
return base, pos
|
|
324
366
|
|
|
325
367
|
|
|
326
368
|
def _parse_atom(tokens: list[Token], pos: int):
|
|
327
369
|
tok = tokens[pos]
|
|
328
370
|
|
|
371
|
+
if tok.type == TokenType.KEYWORD and tok.value == 'borrow':
|
|
372
|
+
pos += 1
|
|
373
|
+
mutable = False
|
|
374
|
+
if (
|
|
375
|
+
pos < len(tokens)
|
|
376
|
+
and tokens[pos].type == TokenType.KEYWORD
|
|
377
|
+
and tokens[pos].value == 'mut'
|
|
378
|
+
):
|
|
379
|
+
mutable = True
|
|
380
|
+
pos += 1
|
|
381
|
+
if pos >= len(tokens):
|
|
382
|
+
raise ParseError('Expected expression after `borrow`', tok)
|
|
383
|
+
operand, pos = parse_expression(tokens, pos)
|
|
384
|
+
return BorrowExpr(operand, mutable=mutable, token=tok), pos
|
|
385
|
+
|
|
386
|
+
if tok.type == TokenType.KEYWORD and tok.value == 'move':
|
|
387
|
+
pos += 1
|
|
388
|
+
if pos >= len(tokens):
|
|
389
|
+
raise ParseError('Expected expression after `move`', tok)
|
|
390
|
+
operand, pos = parse_expression(tokens, pos)
|
|
391
|
+
return MoveExpr(operand, token=tok), pos
|
|
392
|
+
|
|
329
393
|
if tok.type == TokenType.NUMBER and tok.value == '67':
|
|
330
394
|
pos += 1
|
|
331
395
|
if pos < len(tokens) and tokens[pos].type == TokenType.LPAREN:
|
|
@@ -815,6 +879,7 @@ def _parse_expr_iterative(tokens: list[Token], pos: int, min_prec: int):
|
|
|
815
879
|
|
|
816
880
|
|
|
817
881
|
def parse_file(tokens: list[Token], pos: int = 0, enable_extensions: bool = True):
|
|
882
|
+
_pre_scan_user_types(tokens)
|
|
818
883
|
nodes = []
|
|
819
884
|
while pos < len(tokens) and tokens[pos].type not in (TokenType.EOF, TokenType.DEDENT):
|
|
820
885
|
while pos < len(tokens) and tokens[pos].type == TokenType.NEWLINE:
|
|
@@ -827,11 +892,12 @@ def parse_file(tokens: list[Token], pos: int = 0, enable_extensions: bool = True
|
|
|
827
892
|
|
|
828
893
|
all_hooks = _get_all_parser_hooks(enable_extensions)
|
|
829
894
|
if all_hooks:
|
|
895
|
+
ctx = _make_parser_ctx(tokens, pos)
|
|
830
896
|
handled = False
|
|
831
897
|
for hook in all_hooks:
|
|
832
898
|
try:
|
|
833
899
|
if hasattr(hook, 'should_handle_statement') and hook.should_handle_statement(tokens, pos):
|
|
834
|
-
node, pos = hook.parse_statement(tokens, pos,
|
|
900
|
+
node, pos = hook.parse_statement(tokens, pos, ctx)
|
|
835
901
|
nodes.append(node)
|
|
836
902
|
handled = True
|
|
837
903
|
break
|
|
@@ -847,6 +913,25 @@ def parse_file(tokens: list[Token], pos: int = 0, enable_extensions: bool = True
|
|
|
847
913
|
return nodes, pos
|
|
848
914
|
|
|
849
915
|
|
|
916
|
+
_parser_ctx = None
|
|
917
|
+
|
|
918
|
+
|
|
919
|
+
def _make_parser_ctx(tokens, pos):
|
|
920
|
+
"""Build a lazily-cached CompilerContext for parser hooks."""
|
|
921
|
+
global _parser_ctx
|
|
922
|
+
if _parser_ctx is None:
|
|
923
|
+
from .extension_hooks import CompilerContext
|
|
924
|
+
_parser_ctx = CompilerContext(data={"astparse": _get_module_ref()})
|
|
925
|
+
_parser_ctx.data["tokens"] = tokens
|
|
926
|
+
_parser_ctx.data["pos"] = pos
|
|
927
|
+
return _parser_ctx
|
|
928
|
+
|
|
929
|
+
|
|
930
|
+
def _get_module_ref():
|
|
931
|
+
import sys
|
|
932
|
+
return sys.modules.get(__name__)
|
|
933
|
+
|
|
934
|
+
|
|
850
935
|
def _parse_standard_statement(tokens: list[Token], pos: int):
|
|
851
936
|
"""Parse a statement using standard grammar (non-hooked)."""
|
|
852
937
|
tok = tokens[pos]
|
|
@@ -1430,6 +1515,53 @@ class CastExpr(Node):
|
|
|
1430
1515
|
return f'CastExpr({self.type_expr}, {self.expr})'
|
|
1431
1516
|
|
|
1432
1517
|
|
|
1518
|
+
class BorrowExpr(Node):
|
|
1519
|
+
"""`borrow x` or `borrow mut x` — create a reference to x.
|
|
1520
|
+
|
|
1521
|
+
`borrow x` is an immutable borrow (x must not be mutated while borrowed);
|
|
1522
|
+
`borrow mut x` is a mutable borrow. Both lower to taking x's address, but
|
|
1523
|
+
the semantic/ownership pass enforces borrow rules.
|
|
1524
|
+
"""
|
|
1525
|
+
__slots__ = ('_token', 'operand', 'mutable', 'inferred_type')
|
|
1526
|
+
def __init__(self, operand, mutable: bool = False, token=None):
|
|
1527
|
+
self.operand = operand
|
|
1528
|
+
self.mutable = mutable
|
|
1529
|
+
self._token = token
|
|
1530
|
+
self.inferred_type = None
|
|
1531
|
+
def __repr__(self):
|
|
1532
|
+
return f'BorrowExpr(mut={self.mutable}, {self.operand})'
|
|
1533
|
+
|
|
1534
|
+
|
|
1535
|
+
class MoveExpr(Node):
|
|
1536
|
+
"""`move x` — transfer ownership of x (heap/owned value) to the mover.
|
|
1537
|
+
|
|
1538
|
+
After a move, x is no longer valid to use; the ownership pass errors if x is
|
|
1539
|
+
used afterward unless it is reassigned.
|
|
1540
|
+
"""
|
|
1541
|
+
__slots__ = ('_token', 'operand', 'inferred_type')
|
|
1542
|
+
def __init__(self, operand, token=None):
|
|
1543
|
+
self.operand = operand
|
|
1544
|
+
self._token = token
|
|
1545
|
+
self.inferred_type = None
|
|
1546
|
+
def __repr__(self):
|
|
1547
|
+
return f'MoveExpr({self.operand})'
|
|
1548
|
+
|
|
1549
|
+
|
|
1550
|
+
class DeferStmt(Node):
|
|
1551
|
+
"""`defer <stmt>` — run <stmt> when the enclosing function returns (LIFO).
|
|
1552
|
+
|
|
1553
|
+
Deferred statements are collected per function and emitted (in reverse
|
|
1554
|
+
order) just before every function exit point: return statements and the
|
|
1555
|
+
implicit end-of-function return.
|
|
1556
|
+
"""
|
|
1557
|
+
__slots__ = ('_token', 'body')
|
|
1558
|
+
def __init__(self, body, token=None):
|
|
1559
|
+
self.body = body
|
|
1560
|
+
self._token = token
|
|
1561
|
+
def __repr__(self):
|
|
1562
|
+
return f'DeferStmt({self.body!r})'
|
|
1563
|
+
|
|
1564
|
+
|
|
1433
1565
|
class StructDef(Node):
|
|
1434
1566
|
__slots__ = ('_token', 'fields', 'generic_params', 'name')
|
|
1435
1567
|
def __init__(self, name: str, fields: list, generic_params: list | None = None, token=None):
|
|
@@ -1780,6 +1912,17 @@ def parse_var_decl(tokens: list[Token], pos: int):
|
|
|
1780
1912
|
return VarDecl(name, var_type_str, init, is_const=is_const, token=tok), pos
|
|
1781
1913
|
|
|
1782
1914
|
|
|
1915
|
+
def parse_defer(tokens: list[Token], pos: int):
|
|
1916
|
+
tok = tokens[pos] # 'defer'
|
|
1917
|
+
pos += 1
|
|
1918
|
+
if pos >= len(tokens):
|
|
1919
|
+
raise ParseError('Expected a statement after `defer`', tok)
|
|
1920
|
+
stmt, pos = parse_statement(tokens, pos)
|
|
1921
|
+
if stmt is None:
|
|
1922
|
+
stmt = parse_expr_stmt(tokens, pos)
|
|
1923
|
+
return DeferStmt(stmt, token=tok), pos
|
|
1924
|
+
|
|
1925
|
+
|
|
1783
1926
|
def parse_statement(tokens: list[Token], pos: int):
|
|
1784
1927
|
if pos >= len(tokens):
|
|
1785
1928
|
return None, pos
|
|
@@ -1835,6 +1978,9 @@ def parse_statement(tokens: list[Token], pos: int):
|
|
|
1835
1978
|
if tok.type == TokenType.KEYWORD and tok.value == 'unsafe':
|
|
1836
1979
|
return parse_llvm(tokens, pos, unsafe=True)
|
|
1837
1980
|
|
|
1981
|
+
if tok.type == TokenType.KEYWORD and tok.value == 'defer':
|
|
1982
|
+
return parse_defer(tokens, pos)
|
|
1983
|
+
|
|
1838
1984
|
if tok.type == TokenType.IDENTIFIER:
|
|
1839
1985
|
if tok.value in _TYPE_NAMES or _looks_like_type(tokens, pos):
|
|
1840
1986
|
try:
|
|
@@ -7,7 +7,14 @@ from llvmlite import binding, ir
|
|
|
7
7
|
from llvmlite.ir import instructions
|
|
8
8
|
|
|
9
9
|
from .astparse import *
|
|
10
|
-
from .extension_hooks import
|
|
10
|
+
from .extension_hooks import (
|
|
11
|
+
CodegenHook,
|
|
12
|
+
CompilerContext,
|
|
13
|
+
HookLoadError,
|
|
14
|
+
HookStage,
|
|
15
|
+
RuntimeHook,
|
|
16
|
+
get_global_hook_registry,
|
|
17
|
+
)
|
|
11
18
|
from .lexar import TokenType
|
|
12
19
|
from .ui import *
|
|
13
20
|
|
|
@@ -321,6 +328,17 @@ class LLVM:
|
|
|
321
328
|
q = abs(a) // abs(b)
|
|
322
329
|
return -q if (a < 0) != (b < 0) else q
|
|
323
330
|
|
|
331
|
+
def _hook_context(self, data: dict | None = None) -> CompilerContext:
|
|
332
|
+
"""Build a CompilerContext exposing this codegen instance to hooks."""
|
|
333
|
+
ctx = CompilerContext(llvm_module=self.module)
|
|
334
|
+
ctx.data["llvm"] = self
|
|
335
|
+
ctx.data["module"] = self.module
|
|
336
|
+
if getattr(self, "builder", None) is not None:
|
|
337
|
+
ctx.data["builder"] = self.builder
|
|
338
|
+
if data:
|
|
339
|
+
ctx.data.update(data)
|
|
340
|
+
return ctx
|
|
341
|
+
|
|
324
342
|
def _emit_int_divmod(self, left, right, is_rem):
|
|
325
343
|
"""Signed int division/remainder that stays well-defined in LLVM IR.
|
|
326
344
|
|
|
@@ -469,6 +487,7 @@ class LLVM:
|
|
|
469
487
|
self.ssa_values = {}
|
|
470
488
|
self.ssa_types = {} # Track types of SSA values
|
|
471
489
|
self.scope_stack = []
|
|
490
|
+
self._deferred: list = [] # statements collected by `defer`, run LIFO on function exit
|
|
472
491
|
self.structs = {}
|
|
473
492
|
self.struct_fields = {}
|
|
474
493
|
self.import_src_files = []
|
|
@@ -1007,9 +1026,17 @@ class LLVM:
|
|
|
1007
1026
|
import os
|
|
1008
1027
|
import tempfile
|
|
1009
1028
|
|
|
1010
|
-
|
|
1029
|
+
context = self._hook_context()
|
|
1030
|
+
for hook in self._hook_registry.get(HookStage.RUNTIME):
|
|
1031
|
+
if not isinstance(hook, RuntimeHook):
|
|
1032
|
+
continue
|
|
1011
1033
|
try:
|
|
1012
|
-
|
|
1034
|
+
for rel_file in hook.get_runtime_files(context):
|
|
1035
|
+
base = os.path.dirname(hook.hook_path) if hook.hook_path else ""
|
|
1036
|
+
abs_path = os.path.join(base, rel_file)
|
|
1037
|
+
if os.path.isfile(abs_path):
|
|
1038
|
+
self.import_src_files.append(abs_path)
|
|
1039
|
+
runtime_code = hook.get_runtime_code(context)
|
|
1013
1040
|
if runtime_code:
|
|
1014
1041
|
fd, tmp_path = tempfile.mkstemp(
|
|
1015
1042
|
suffix=".c", prefix="hook_runtime_"
|
|
@@ -1157,7 +1184,9 @@ class LLVM:
|
|
|
1157
1184
|
left_len = self.builder.call(strlen_fn, [left])
|
|
1158
1185
|
right_len = self.builder.call(strlen_fn, [right])
|
|
1159
1186
|
total_len = self.builder.add(left_len, right_len)
|
|
1160
|
-
plus_one = self.builder.add(
|
|
1187
|
+
plus_one = self.builder.add(
|
|
1188
|
+
total_len, ir.Constant(total_len.type, 1)
|
|
1189
|
+
)
|
|
1161
1190
|
new_str = self.builder.call(malloc_fn, [self.builder.zext(plus_one, _i64)])
|
|
1162
1191
|
self.builder.call(memcpy_fn, [new_str, left, left_len])
|
|
1163
1192
|
dest_plus = self.builder.gep(new_str, [left_len], inbounds=True)
|
|
@@ -1170,6 +1199,13 @@ class LLVM:
|
|
|
1170
1199
|
def emit_exprstmt(self, node):
|
|
1171
1200
|
return self.emit(node.expr)
|
|
1172
1201
|
|
|
1202
|
+
@register_emitter(DeferStmt)
|
|
1203
|
+
def emit_deferstmt(self, node: DeferStmt):
|
|
1204
|
+
# Collect the deferred statement; it is emitted (in reverse order) at
|
|
1205
|
+
# the function's exit points, both explicit `return` and implicit end.
|
|
1206
|
+
self._deferred.append(node.body)
|
|
1207
|
+
return None
|
|
1208
|
+
|
|
1173
1209
|
def emit(self, node: Node | dict) -> _IRValue:
|
|
1174
1210
|
key = id(node)
|
|
1175
1211
|
if key in self._emit_memo:
|
|
@@ -1425,17 +1461,15 @@ class LLVM:
|
|
|
1425
1461
|
def _emit_recursive(self, node: Node | dict) -> _IRValue:
|
|
1426
1462
|
# Try codegen hooks if extensions are enabled
|
|
1427
1463
|
if self.enable_extensions:
|
|
1428
|
-
for hook in self._hook_registry.
|
|
1464
|
+
for hook in self._hook_registry.get(HookStage.CODEGEN):
|
|
1465
|
+
if not isinstance(hook, CodegenHook):
|
|
1466
|
+
continue
|
|
1429
1467
|
try:
|
|
1430
1468
|
if hook.should_emit_node(node):
|
|
1431
1469
|
return hook.emit_node(
|
|
1432
1470
|
node,
|
|
1433
1471
|
self.builder,
|
|
1434
|
-
{
|
|
1435
|
-
"llvm": self,
|
|
1436
|
-
"module": self.module,
|
|
1437
|
-
"builder": self.builder,
|
|
1438
|
-
},
|
|
1472
|
+
self._hook_context(data={"node": node}),
|
|
1439
1473
|
)
|
|
1440
1474
|
except Exception as e:
|
|
1441
1475
|
raise HookLoadError(
|
|
@@ -1798,6 +1832,18 @@ class LLVM:
|
|
|
1798
1832
|
raise Exception(f"Undefined variable '{name}'")
|
|
1799
1833
|
raise Exception("Address-of requires a variable")
|
|
1800
1834
|
|
|
1835
|
+
@register_emitter(BorrowExpr)
|
|
1836
|
+
def emit_borrow(self, node: BorrowExpr):
|
|
1837
|
+
# `borrow x` / `borrow mut x` lowers to taking x's address, exactly like
|
|
1838
|
+
# `&x`. Mutable vs immutable is enforced at the semantic stage.
|
|
1839
|
+
return self.emit_addrof(AddrOf(node.operand))
|
|
1840
|
+
|
|
1841
|
+
@register_emitter(MoveExpr)
|
|
1842
|
+
def emit_move(self, node: MoveExpr):
|
|
1843
|
+
# `move x` is a compile-time ownership transfer; at runtime it just
|
|
1844
|
+
# yields the value of x (no copy is made).
|
|
1845
|
+
return self.emit(node.operand)
|
|
1846
|
+
|
|
1801
1847
|
@register_emitter(SizeOf)
|
|
1802
1848
|
def emit_sizeof(self, node: SizeOf):
|
|
1803
1849
|
ty = self.llvm_type(node.type_expr)
|
|
@@ -2086,11 +2132,13 @@ class LLVM:
|
|
|
2086
2132
|
old_ssa = self.ssa_values
|
|
2087
2133
|
old_ssa_types = self.ssa_types
|
|
2088
2134
|
old_scope_stack = self.scope_stack
|
|
2135
|
+
old_deferred = self._deferred
|
|
2089
2136
|
self.locals = {}
|
|
2090
2137
|
self.local_types = {}
|
|
2091
2138
|
self.ssa_values = {}
|
|
2092
2139
|
self.ssa_types = {}
|
|
2093
2140
|
self.scope_stack = [{}]
|
|
2141
|
+
self._deferred = []
|
|
2094
2142
|
for llvm_arg, (name, ptype) in zip(func.args, node.params.items()):
|
|
2095
2143
|
if isinstance(llvm_arg.type, ir.VoidType):
|
|
2096
2144
|
self._codegen_error(
|
|
@@ -2108,6 +2156,7 @@ class LLVM:
|
|
|
2108
2156
|
self.emit(stmt)
|
|
2109
2157
|
|
|
2110
2158
|
if not self._block_terminated():
|
|
2159
|
+
self._run_deferred()
|
|
2111
2160
|
# Shutdown GC before main returns
|
|
2112
2161
|
if node.name == "main" and not self.no_gc:
|
|
2113
2162
|
gc_shutdown_fn = self.functions.get("gc_shutdown")
|
|
@@ -2133,6 +2182,7 @@ class LLVM:
|
|
|
2133
2182
|
self.ssa_values = old_ssa
|
|
2134
2183
|
self.ssa_types = old_ssa_types
|
|
2135
2184
|
self.scope_stack = old_scope_stack
|
|
2185
|
+
self._deferred = old_deferred
|
|
2136
2186
|
|
|
2137
2187
|
def _emit_decorated_funcdef(self, node: FuncDef):
|
|
2138
2188
|
"""Emit a decorated function: original as __name, trampoline, and wrapper."""
|
|
@@ -2247,6 +2297,7 @@ class LLVM:
|
|
|
2247
2297
|
def emit_return(self, node: Return):
|
|
2248
2298
|
if self._block_terminated():
|
|
2249
2299
|
return
|
|
2300
|
+
self._run_deferred()
|
|
2250
2301
|
# Shutdown GC before main returns
|
|
2251
2302
|
fn_name = self.builder.function.name
|
|
2252
2303
|
if fn_name == "main" and not self.no_gc:
|
|
@@ -2503,7 +2554,7 @@ class LLVM:
|
|
|
2503
2554
|
return _DYN_STR
|
|
2504
2555
|
if ty == "big":
|
|
2505
2556
|
return _DYN_BIG
|
|
2506
|
-
if ty.endswith("*"
|
|
2557
|
+
if ty.endswith(("*", "&")):
|
|
2507
2558
|
return _DYN_PTR
|
|
2508
2559
|
return _DYN_INT
|
|
2509
2560
|
|
|
@@ -2554,7 +2605,7 @@ class LLVM:
|
|
|
2554
2605
|
return self.builder.bitcast(bits, _double)
|
|
2555
2606
|
if ty in ("str", "big", "void*"):
|
|
2556
2607
|
return self.builder.inttoptr(bits, _i8ptr)
|
|
2557
|
-
if ty.endswith("*"
|
|
2608
|
+
if ty.endswith(("*", "&")):
|
|
2558
2609
|
ptr_ty = self.llvm_type(ty)
|
|
2559
2610
|
if isinstance(ptr_ty, ir.PointerType):
|
|
2560
2611
|
return self.builder.inttoptr(bits, ptr_ty)
|
|
@@ -2567,9 +2618,7 @@ class LLVM:
|
|
|
2567
2618
|
"""True when a node carries a runtime-typed (dynamic) value."""
|
|
2568
2619
|
if getattr(node, "inferred_type", None) == "dynamic":
|
|
2569
2620
|
return True
|
|
2570
|
-
|
|
2571
|
-
return True
|
|
2572
|
-
return False
|
|
2621
|
+
return bool(isinstance(node, Variable) and getattr(node, "dynamic", False))
|
|
2573
2622
|
|
|
2574
2623
|
def _dyn_local_ptr(self, name):
|
|
2575
2624
|
"""Return (creating if needed) the {i32,i64} slot that backs a dynamic local."""
|
|
@@ -2665,6 +2714,25 @@ class LLVM:
|
|
|
2665
2714
|
return True
|
|
2666
2715
|
return block.is_terminated
|
|
2667
2716
|
|
|
2717
|
+
def _run_deferred(self):
|
|
2718
|
+
"""Emit all accumulated deferred statements in LIFO order.
|
|
2719
|
+
|
|
2720
|
+
Called at every function exit point (explicit `return` statements and the
|
|
2721
|
+
implicit end-of-function return). It does NOT clear the accumulated list,
|
|
2722
|
+
so every static exit site of the function emits the full set of defers;
|
|
2723
|
+
at runtime only the exit that is actually reached executes them. The list
|
|
2724
|
+
is reset when the function's own emission completes (emit_funcdef restores
|
|
2725
|
+
the outer value).
|
|
2726
|
+
"""
|
|
2727
|
+
if not self._deferred:
|
|
2728
|
+
return
|
|
2729
|
+
for stmt in reversed(self._deferred):
|
|
2730
|
+
if self._block_terminated():
|
|
2731
|
+
break
|
|
2732
|
+
self.emit(stmt)
|
|
2733
|
+
if self._block_terminated():
|
|
2734
|
+
break
|
|
2735
|
+
|
|
2668
2736
|
@register_emitter(BinOp)
|
|
2669
2737
|
def emit_binop(self, node):
|
|
2670
2738
|
if node.op == TokenType.PLUS and self._is_string_concat(node):
|
|
@@ -2673,27 +2741,26 @@ class LLVM:
|
|
|
2673
2741
|
# Runtime dispatch when either operand is dynamically typed.
|
|
2674
2742
|
if not self.no_userspace and (
|
|
2675
2743
|
self._is_dynamic_expr(node.left) or self._is_dynamic_expr(node.right)
|
|
2744
|
+
) and node.op in (
|
|
2745
|
+
TokenType.PLUS,
|
|
2746
|
+
TokenType.MINUS,
|
|
2747
|
+
TokenType.STAR,
|
|
2748
|
+
TokenType.SLASH,
|
|
2749
|
+
TokenType.SLASH_SLASH,
|
|
2750
|
+
TokenType.PERCENT,
|
|
2751
|
+
TokenType.EQ_EQ,
|
|
2752
|
+
TokenType.NOT_EQ,
|
|
2753
|
+
TokenType.LESS,
|
|
2754
|
+
TokenType.GREATER,
|
|
2755
|
+
TokenType.LESS_EQ,
|
|
2756
|
+
TokenType.GREATER_EQ,
|
|
2757
|
+
TokenType.AMPERSAND,
|
|
2758
|
+
TokenType.PIPE,
|
|
2759
|
+
TokenType.CARET,
|
|
2760
|
+
TokenType.SHL,
|
|
2761
|
+
TokenType.SHR,
|
|
2676
2762
|
):
|
|
2677
|
-
|
|
2678
|
-
TokenType.PLUS,
|
|
2679
|
-
TokenType.MINUS,
|
|
2680
|
-
TokenType.STAR,
|
|
2681
|
-
TokenType.SLASH,
|
|
2682
|
-
TokenType.SLASH_SLASH,
|
|
2683
|
-
TokenType.PERCENT,
|
|
2684
|
-
TokenType.EQ_EQ,
|
|
2685
|
-
TokenType.NOT_EQ,
|
|
2686
|
-
TokenType.LESS,
|
|
2687
|
-
TokenType.GREATER,
|
|
2688
|
-
TokenType.LESS_EQ,
|
|
2689
|
-
TokenType.GREATER_EQ,
|
|
2690
|
-
TokenType.AMPERSAND,
|
|
2691
|
-
TokenType.PIPE,
|
|
2692
|
-
TokenType.CARET,
|
|
2693
|
-
TokenType.SHL,
|
|
2694
|
-
TokenType.SHR,
|
|
2695
|
-
):
|
|
2696
|
-
return self._emit_dyn_binop(node)
|
|
2763
|
+
return self._emit_dyn_binop(node)
|
|
2697
2764
|
|
|
2698
2765
|
match node.op:
|
|
2699
2766
|
case TokenType.AND:
|
|
@@ -2939,12 +3006,7 @@ class LLVM:
|
|
|
2939
3006
|
and self.local_types.get(node.left.name) == "str"
|
|
2940
3007
|
):
|
|
2941
3008
|
return True
|
|
2942
|
-
|
|
2943
|
-
isinstance(node.right, Variable)
|
|
2944
|
-
and self.local_types.get(node.right.name) == "str"
|
|
2945
|
-
):
|
|
2946
|
-
return True
|
|
2947
|
-
return False
|
|
3009
|
+
return bool(isinstance(node.right, Variable) and self.local_types.get(node.right.name) == "str")
|
|
2948
3010
|
|
|
2949
3011
|
def _emit_dyn_binop(self, node):
|
|
2950
3012
|
dynop_map = {
|
|
@@ -2987,7 +3049,8 @@ class LLVM:
|
|
|
2987
3049
|
if f.name == "strlen":
|
|
2988
3050
|
self._strlen_fn = f
|
|
2989
3051
|
return f
|
|
2990
|
-
|
|
3052
|
+
# strlen returns size_t (i64 on every cpyte target), not int.
|
|
3053
|
+
fnty = ir.FunctionType(_i64, [_i8ptr])
|
|
2991
3054
|
fn = ir.Function(self.module, fnty, "strlen")
|
|
2992
3055
|
self._strlen_fn = fn
|
|
2993
3056
|
return fn
|
|
@@ -3000,7 +3063,8 @@ class LLVM:
|
|
|
3000
3063
|
if f.name == "memcpy":
|
|
3001
3064
|
self._memcpy_fn = f
|
|
3002
3065
|
return f
|
|
3003
|
-
|
|
3066
|
+
# n is size_t (i64 on every cpyte target).
|
|
3067
|
+
fnty = ir.FunctionType(_i8ptr, [_i8ptr, _i8ptr, _i64])
|
|
3004
3068
|
fn = ir.Function(self.module, fnty, "memcpy")
|
|
3005
3069
|
self._memcpy_fn = fn
|
|
3006
3070
|
return fn
|
|
@@ -3927,6 +3991,8 @@ class LLVM:
|
|
|
3927
3991
|
|
|
3928
3992
|
len_fn = self._get_strlen_fn()
|
|
3929
3993
|
length = self.builder.call(len_fn, [iter_ptr])
|
|
3994
|
+
if length.type.width != 32:
|
|
3995
|
+
length = self.builder.trunc(length, _i32)
|
|
3930
3996
|
|
|
3931
3997
|
idx_ptr = self._alloca(_i32, name=f"{var_name}.idx")
|
|
3932
3998
|
self.builder.store(ir.Constant(_i32, 0), idx_ptr)
|