cpyte 3.3.0__tar.gz → 3.3.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (41) hide show
  1. {cpyte-3.3.0/source/cpyte.egg-info → cpyte-3.3.2}/PKG-INFO +16 -2
  2. {cpyte-3.3.0 → cpyte-3.3.2}/pyproject.toml +1 -1
  3. {cpyte-3.3.0 → cpyte-3.3.2}/readme.md +16 -2
  4. cpyte-3.3.2/source/cpyte/__init__.py +1 -0
  5. {cpyte-3.3.0 → cpyte-3.3.2}/source/cpyte/astparse.py +153 -7
  6. {cpyte-3.3.0 → cpyte-3.3.2}/source/cpyte/bytecoding.py +109 -43
  7. cpyte-3.3.2/source/cpyte/clib.py +1416 -0
  8. {cpyte-3.3.0 → cpyte-3.3.2}/source/cpyte/compiling.py +333 -155
  9. cpyte-3.3.2/source/cpyte/extension_hooks.py +1115 -0
  10. {cpyte-3.3.0 → cpyte-3.3.2}/source/cpyte/gc_runtime.c +89 -54
  11. {cpyte-3.3.0 → cpyte-3.3.2}/source/cpyte/generate_bc.py +81 -53
  12. {cpyte-3.3.0 → cpyte-3.3.2}/source/cpyte/lexar.py +2 -0
  13. cpyte-3.3.2/source/cpyte/mainpie.py +903 -0
  14. {cpyte-3.3.0 → cpyte-3.3.2}/source/cpyte/semantic_analasis.py +882 -68
  15. {cpyte-3.3.0 → cpyte-3.3.2/source/cpyte.egg-info}/PKG-INFO +16 -2
  16. {cpyte-3.3.0 → cpyte-3.3.2}/source/cpyte.egg-info/SOURCES.txt +0 -1
  17. cpyte-3.3.0/source/cpyte/__init__.py +0 -1
  18. cpyte-3.3.0/source/cpyte/_gc_bc.py +0 -187
  19. cpyte-3.3.0/source/cpyte/clib.py +0 -1100
  20. cpyte-3.3.0/source/cpyte/extension_hooks.py +0 -465
  21. cpyte-3.3.0/source/cpyte/mainpie.py +0 -725
  22. {cpyte-3.3.0 → cpyte-3.3.2}/MANIFEST.in +0 -0
  23. {cpyte-3.3.0 → cpyte-3.3.2}/setup.cfg +0 -0
  24. {cpyte-3.3.0 → cpyte-3.3.2}/source/cpyte/__main__.py +0 -0
  25. {cpyte-3.3.0 → cpyte-3.3.2}/source/cpyte/_bignum_bc.py +0 -0
  26. {cpyte-3.3.0 → cpyte-3.3.2}/source/cpyte/_runtime_bc.py +0 -0
  27. {cpyte-3.3.0 → cpyte-3.3.2}/source/cpyte/bignum.c +0 -0
  28. {cpyte-3.3.0 → cpyte-3.3.2}/source/cpyte/formatter.py +0 -0
  29. {cpyte-3.3.0 → cpyte-3.3.2}/source/cpyte/linker.py +0 -0
  30. {cpyte-3.3.0 → cpyte-3.3.2}/source/cpyte/lsp_server.py +0 -0
  31. {cpyte-3.3.0 → cpyte-3.3.2}/source/cpyte/package_manifest.py +0 -0
  32. {cpyte-3.3.0 → cpyte-3.3.2}/source/cpyte/runtime.c +0 -0
  33. {cpyte-3.3.0 → cpyte-3.3.2}/source/cpyte/runtime_scorpion.c +0 -0
  34. {cpyte-3.3.0 → cpyte-3.3.2}/source/cpyte/sef.py +0 -0
  35. {cpyte-3.3.0 → cpyte-3.3.2}/source/cpyte/ui.py +0 -0
  36. {cpyte-3.3.0 → cpyte-3.3.2}/source/cpyte/update_check.py +0 -0
  37. {cpyte-3.3.0 → cpyte-3.3.2}/source/cpyte.egg-info/dependency_links.txt +0 -0
  38. {cpyte-3.3.0 → cpyte-3.3.2}/source/cpyte.egg-info/entry_points.txt +0 -0
  39. {cpyte-3.3.0 → cpyte-3.3.2}/source/cpyte.egg-info/requires.txt +0 -0
  40. {cpyte-3.3.0 → cpyte-3.3.2}/source/cpyte.egg-info/top_level.txt +0 -0
  41. {cpyte-3.3.0 → cpyte-3.3.2}/test/test_bignum_jit.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: cpyte
3
- Version: 3.3.0
3
+ Version: 3.3.2
4
4
  Summary: The Cpyte programming language compiler
5
5
  Author: Hoang Duy Tung
6
6
  License: MIT
@@ -16,7 +16,7 @@ Check out the official documentation [here](https://gitea.5gnew.io.vn/Cpyte-Proj
16
16
  Cpyte supports a package extension system that allows packages to extend the compiler with custom keywords, operators, and compiler hooks. Packages can provide:
17
17
 
18
18
  - **Custom Keywords**: Add new language keywords via `package.json`
19
- - **Custom Operators**: Define new operators for syntax extensions
19
+ - **Custom Operators**: Define new operators for syntax extensions
20
20
  - **Compiler Hooks**: Extend lexing, parsing, semantic analysis, and code generation
21
21
  - **Runtime Extensions**: Add runtime code and libraries
22
22
 
@@ -83,3 +83,17 @@ Cpyte is experimental software. The compiler is continuously tested with fuzzing
83
83
  Cpyte uses a **concurrent tri-color garbage collector** for automatic memory management. Heap-allocated objects (via `new`) are managed automatically — no manual `free` needed.
84
84
 
85
85
  **Note:** The collector adds a small runtime overhead (~5-10%) compared to manual `malloc/free`. For performance-critical ecosystem projects or embedded use cases, this trade-off may be significant. The GC is required for safety but will slow down programs that do heavy heap allocation.
86
+
87
+ ## Cross-Platform Runtime
88
+
89
+ The compiler is fully cross-platform. On every target OS (macOS, Linux, Windows) the C runtime (`runtime.c`), the GC runtime (`gc_runtime.c`), and any embedded `ccode:` blocks are **compiled from source** at JIT/AOT time for the host platform, rather than linking pre-built, OS-specific bitcode. This keeps native helpers and the garbage collector correct on each system:
90
+
91
+ - Portable threading: `pthread` on POSIX; native Windows threads + `CRITICAL_SECTION` on Windows.
92
+ - Portable stack scanning for the GC that works on macOS, Linux, and Windows.
93
+ - A portable sleep/yield helper replaces the POSIX-only `nanosleep`.
94
+
95
+ The compiler auto-discovers a C compiler (`clang` → `cc` → `gcc`) and system linker on the current platform, so `cpy`, `cpy build`, and `cpy --aot` all work without per-OS configuration.
96
+
97
+ ## Continuous Integration
98
+
99
+ GitHub Actions (see `.github/workflows/code_quality.yml`) builds and runs a curated, cross-platform regression corpus on Ubuntu (x86_64 + arm64), macOS (x86_64 + arm64), and Windows. `ci_test.py` AOT-compiles each program with the system linker and executes the result to exercise the full pipeline — lexer, parser, semantic analysis, LLVM codegen, the C and GC runtimes, and linking.
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "cpyte"
7
- version = "3.3.0"
7
+ version = "3.3.2"
8
8
  description = "The Cpyte programming language compiler"
9
9
  readme = "readme.md"
10
10
  requires-python = ">=3.11"
@@ -5,7 +5,7 @@ Check out the official documentation [here](https://gitea.5gnew.io.vn/Cpyte-Proj
5
5
  Cpyte supports a package extension system that allows packages to extend the compiler with custom keywords, operators, and compiler hooks. Packages can provide:
6
6
 
7
7
  - **Custom Keywords**: Add new language keywords via `package.json`
8
- - **Custom Operators**: Define new operators for syntax extensions
8
+ - **Custom Operators**: Define new operators for syntax extensions
9
9
  - **Compiler Hooks**: Extend lexing, parsing, semantic analysis, and code generation
10
10
  - **Runtime Extensions**: Add runtime code and libraries
11
11
 
@@ -71,4 +71,18 @@ Cpyte is experimental software. The compiler is continuously tested with fuzzing
71
71
 
72
72
  Cpyte uses a **concurrent tri-color garbage collector** for automatic memory management. Heap-allocated objects (via `new`) are managed automatically — no manual `free` needed.
73
73
 
74
- **Note:** The collector adds a small runtime overhead (~5-10%) compared to manual `malloc/free`. For performance-critical ecosystem projects or embedded use cases, this trade-off may be significant. The GC is required for safety but will slow down programs that do heavy heap allocation.
74
+ **Note:** The collector adds a small runtime overhead (~5-10%) compared to manual `malloc/free`. For performance-critical ecosystem projects or embedded use cases, this trade-off may be significant. The GC is required for safety but will slow down programs that do heavy heap allocation.
75
+
76
+ ## Cross-Platform Runtime
77
+
78
+ The compiler is fully cross-platform. On every target OS (macOS, Linux, Windows) the C runtime (`runtime.c`), the GC runtime (`gc_runtime.c`), and any embedded `ccode:` blocks are **compiled from source** at JIT/AOT time for the host platform, rather than linking pre-built, OS-specific bitcode. This keeps native helpers and the garbage collector correct on each system:
79
+
80
+ - Portable threading: `pthread` on POSIX; native Windows threads + `CRITICAL_SECTION` on Windows.
81
+ - Portable stack scanning for the GC that works on macOS, Linux, and Windows.
82
+ - A portable sleep/yield helper replaces the POSIX-only `nanosleep`.
83
+
84
+ The compiler auto-discovers a C compiler (`clang` → `cc` → `gcc`) and system linker on the current platform, so `cpy`, `cpy build`, and `cpy --aot` all work without per-OS configuration.
85
+
86
+ ## Continuous Integration
87
+
88
+ GitHub Actions (see `.github/workflows/code_quality.yml`) builds and runs a curated, cross-platform regression corpus on Ubuntu (x86_64 + arm64), macOS (x86_64 + arm64), and Windows. `ci_test.py` AOT-compiles each program with the system linker and executes the result to exercise the full pipeline — lexer, parser, semantic analysis, LLVM codegen, the C and GC runtimes, and linking.
@@ -0,0 +1 @@
1
+ __version__ = "3.3.2"
@@ -4,6 +4,31 @@ from .lexar import Lexer, Token, TokenType, _unescape_run
4
4
 
5
5
  _parser_hooks: list[Any] = []
6
6
 
7
+ # User-declared type names (structs, type aliases, classes) collected in a
8
+ # pre-scan so bare-name casts like `(MyInt)x` are recognized as casts at parse
9
+ # time. Reset per parse_file().
10
+ _user_type_names: set[str] = set()
11
+
12
+
13
+ def _pre_scan_user_types(tokens):
14
+ """Collect identifiers declared as struct/type/class so the parser can
15
+ recognize casts to user-defined types even without a pointer suffix."""
16
+ global _user_type_names
17
+ names = set()
18
+ i = 0
19
+ n = len(tokens)
20
+ while i < n:
21
+ t = tokens[i]
22
+ if t.type == TokenType.KEYWORD and t.value in ("struct", "type", "class"):
23
+ j = i + 1
24
+ if j < n and tokens[j].type == TokenType.IDENTIFIER:
25
+ names.add(tokens[j].value)
26
+ i = j + 1
27
+ continue
28
+ i += 1
29
+ _user_type_names = names
30
+ return names
31
+
7
32
 
8
33
  def register_parser_hook(hook: Any) -> None:
9
34
  _parser_hooks.append(hook)
@@ -21,10 +46,10 @@ def _get_all_parser_hooks(enable_extensions: bool) -> list[Any]:
21
46
  return []
22
47
  hooks = list(_parser_hooks)
23
48
  try:
24
- from .extension_hooks import get_global_hook_registry
49
+ from .extension_hooks import HookStage, ParserHook, get_global_hook_registry
25
50
  registry = get_global_hook_registry()
26
- for hook in registry.get_parser_hooks():
27
- if hook not in hooks:
51
+ for hook in registry.get(HookStage.PARSER):
52
+ if isinstance(hook, ParserHook) and hook not in hooks:
28
53
  hooks.append(hook)
29
54
  except ImportError:
30
55
  pass
@@ -296,36 +321,75 @@ def _parse_unary(tokens: list[Token], pos: int):
296
321
 
297
322
  def _try_parse_cast_type(tokens, pos):
298
323
  """Try to parse a type name for a C-style cast like (int), (size_t*), (float[]).
324
+ Also accepts user-defined types (structs / type aliases / classes) as cast
325
+ targets, e.g. ``(Point*)p`` or ``(MyInt)x``.
299
326
 
300
327
  Returns (type_string, new_pos) on success, or (None, original_pos) if not a cast.
328
+ The returned type string retains pointer/array/ref suffixes, so ``(int*)x``
329
+ decodes to a pointer cast rather than silently degrading to ``int``.
301
330
  """
302
331
  if pos >= len(tokens) or tokens[pos].type != TokenType.IDENTIFIER:
303
332
  return None, pos
304
333
  name = tokens[pos].value
305
- if name not in _TYPE_NAMES:
306
- return None, pos
334
+ base = name
307
335
  pos += 1
336
+ has_suffix = False
308
337
  while pos < len(tokens):
309
338
  t = tokens[pos].type
310
339
  if t == TokenType.STAR:
340
+ base = _type_to_str(base) + "*"
341
+ has_suffix = True
311
342
  pos += 1
312
343
  elif t == TokenType.POW:
344
+ base = _type_to_str(base) + "**"
345
+ has_suffix = True
313
346
  pos += 1
314
347
  elif t == TokenType.LBRACKET:
315
348
  if pos + 1 < len(tokens) and tokens[pos + 1].type == TokenType.RBRACKET:
349
+ base = _type_to_str(base) + "[]"
350
+ has_suffix = True
316
351
  pos += 2
317
352
  else:
318
353
  break
319
354
  elif t == TokenType.AMPERSAND:
355
+ base = _type_to_str(base) + "&"
356
+ has_suffix = True
320
357
  pos += 1
321
358
  else:
322
359
  break
323
- return name, pos
360
+ # A bare (no-suffix) name is accepted as a cast target only when it is a
361
+ # known builtin type or a user-declared type. Otherwise leave it to the
362
+ # normal parenthesized-expression/call path to avoid ambiguity.
363
+ if not has_suffix and name not in _TYPE_NAMES and name not in _user_type_names:
364
+ return None, pos
365
+ return base, pos
324
366
 
325
367
 
326
368
  def _parse_atom(tokens: list[Token], pos: int):
327
369
  tok = tokens[pos]
328
370
 
371
+ if tok.type == TokenType.KEYWORD and tok.value == 'borrow':
372
+ pos += 1
373
+ mutable = False
374
+ if (
375
+ pos < len(tokens)
376
+ and tokens[pos].type == TokenType.KEYWORD
377
+ and tokens[pos].value == 'mut'
378
+ ):
379
+ mutable = True
380
+ pos += 1
381
+ if pos >= len(tokens):
382
+ raise ParseError('Expected expression after `borrow`', tok)
383
+ operand, pos = parse_expression(tokens, pos)
384
+ return BorrowExpr(operand, mutable=mutable, token=tok), pos
385
+
386
+ if tok.type == TokenType.KEYWORD and tok.value == 'move':
387
+ pos += 1
388
+ if pos >= len(tokens):
389
+ raise ParseError('Expected expression after `move`', tok)
390
+ operand, pos = parse_expression(tokens, pos)
391
+ return MoveExpr(operand, token=tok), pos
392
+
329
393
  if tok.type == TokenType.NUMBER and tok.value == '67':
330
394
  pos += 1
331
395
  if pos < len(tokens) and tokens[pos].type == TokenType.LPAREN:
@@ -815,6 +879,7 @@ def _parse_expr_iterative(tokens: list[Token], pos: int, min_prec: int):
815
879
 
816
880
 
817
881
  def parse_file(tokens: list[Token], pos: int = 0, enable_extensions: bool = True):
882
+ _pre_scan_user_types(tokens)
818
883
  nodes = []
819
884
  while pos < len(tokens) and tokens[pos].type not in (TokenType.EOF, TokenType.DEDENT):
820
885
  while pos < len(tokens) and tokens[pos].type == TokenType.NEWLINE:
@@ -827,11 +892,12 @@ def parse_file(tokens: list[Token], pos: int = 0, enable_extensions: bool = True
827
892
 
828
893
  all_hooks = _get_all_parser_hooks(enable_extensions)
829
894
  if all_hooks:
895
+ ctx = _make_parser_ctx(tokens, pos)
830
896
  handled = False
831
897
  for hook in all_hooks:
832
898
  try:
833
899
  if hasattr(hook, 'should_handle_statement') and hook.should_handle_statement(tokens, pos):
834
- node, pos = hook.parse_statement(tokens, pos, {'tokens': tokens, 'pos': pos})
900
+ node, pos = hook.parse_statement(tokens, pos, ctx)
835
901
  nodes.append(node)
836
902
  handled = True
837
903
  break
@@ -847,6 +913,25 @@ def parse_file(tokens: list[Token], pos: int = 0, enable_extensions: bool = True
847
913
  return nodes, pos
848
914
 
849
915
 
916
+ _parser_ctx = None
917
+
918
+
919
+ def _make_parser_ctx(tokens, pos):
920
+ """Build a lazily-cached CompilerContext for parser hooks."""
921
+ global _parser_ctx
922
+ if _parser_ctx is None:
923
+ from .extension_hooks import CompilerContext
924
+ _parser_ctx = CompilerContext(data={"astparse": _get_module_ref()})
925
+ _parser_ctx.data["tokens"] = tokens
926
+ _parser_ctx.data["pos"] = pos
927
+ return _parser_ctx
928
+
929
+
930
+ def _get_module_ref():
931
+ import sys
932
+ return sys.modules.get(__name__)
933
+
934
+
850
935
  def _parse_standard_statement(tokens: list[Token], pos: int):
851
936
  """Parse a statement using standard grammar (non-hooked)."""
852
937
  tok = tokens[pos]
@@ -1430,6 +1515,53 @@ class CastExpr(Node):
1430
1515
  return f'CastExpr({self.type_expr}, {self.expr})'
1431
1516
 
1432
1517
 
1518
+ class BorrowExpr(Node):
1519
+ """`borrow x` or `borrow mut x` — create a reference to x.
1520
+
1521
+ `borrow x` is an immutable borrow (x must not be mutated while borrowed);
1522
+ `borrow mut x` is a mutable borrow. Both lower to taking x's address, but
1523
+ the semantic/ownership pass enforces borrow rules.
1524
+ """
1525
+ __slots__ = ('_token', 'operand', 'mutable', 'inferred_type')
1526
+ def __init__(self, operand, mutable: bool = False, token=None):
1527
+ self.operand = operand
1528
+ self.mutable = mutable
1529
+ self._token = token
1530
+ self.inferred_type = None
1531
+ def __repr__(self):
1532
+ return f'BorrowExpr(mut={self.mutable}, {self.operand})'
1533
+
1534
+
1535
+ class MoveExpr(Node):
1536
+ """`move x` — transfer ownership of x (heap/owned value) to the mover.
1537
+
1538
+ After a move, x is no longer valid to use; the ownership pass errors if x is
1539
+ used afterward unless it is reassigned.
1540
+ """
1541
+ __slots__ = ('_token', 'operand', 'inferred_type')
1542
+ def __init__(self, operand, token=None):
1543
+ self.operand = operand
1544
+ self._token = token
1545
+ self.inferred_type = None
1546
+ def __repr__(self):
1547
+ return f'MoveExpr({self.operand})'
1548
+
1549
+
1550
+ class DeferStmt(Node):
1551
+ """`defer <stmt>` — run <stmt> when the enclosing function returns (LIFO).
1552
+
1553
+ Deferred statements are collected per function and emitted (in reverse
1554
+ order) just before every function exit point: return statements and the
1555
+ implicit end-of-function return.
1556
+ """
1557
+ __slots__ = ('_token', 'body')
1558
+ def __init__(self, body, token=None):
1559
+ self.body = body
1560
+ self._token = token
1561
+ def __repr__(self):
1562
+ return f'DeferStmt({self.body!r})'
1563
+
1564
+
1433
1565
  class StructDef(Node):
1434
1566
  __slots__ = ('_token', 'fields', 'generic_params', 'name')
1435
1567
  def __init__(self, name: str, fields: list, generic_params: list | None = None, token=None):
@@ -1780,6 +1912,17 @@ def parse_var_decl(tokens: list[Token], pos: int):
1780
1912
  return VarDecl(name, var_type_str, init, is_const=is_const, token=tok), pos
1781
1913
 
1782
1914
 
1915
+ def parse_defer(tokens: list[Token], pos: int):
1916
+ tok = tokens[pos] # 'defer'
1917
+ pos += 1
1918
+ if pos >= len(tokens):
1919
+ raise ParseError('Expected a statement after `defer`', tok)
1920
+ stmt, pos = parse_statement(tokens, pos)
1921
+ if stmt is None:
1922
+ stmt = parse_expr_stmt(tokens, pos)
1923
+ return DeferStmt(stmt, token=tok), pos
1924
+
1925
+
1783
1926
  def parse_statement(tokens: list[Token], pos: int):
1784
1927
  if pos >= len(tokens):
1785
1928
  return None, pos
@@ -1835,6 +1978,9 @@ def parse_statement(tokens: list[Token], pos: int):
1835
1978
  if tok.type == TokenType.KEYWORD and tok.value == 'unsafe':
1836
1979
  return parse_llvm(tokens, pos, unsafe=True)
1837
1980
 
1981
+ if tok.type == TokenType.KEYWORD and tok.value == 'defer':
1982
+ return parse_defer(tokens, pos)
1983
+
1838
1984
  if tok.type == TokenType.IDENTIFIER:
1839
1985
  if tok.value in _TYPE_NAMES or _looks_like_type(tokens, pos):
1840
1986
  try:
@@ -7,7 +7,14 @@ from llvmlite import binding, ir
7
7
  from llvmlite.ir import instructions
8
8
 
9
9
  from .astparse import *
10
- from .extension_hooks import HookLoadError, get_global_hook_registry
10
+ from .extension_hooks import (
11
+ CodegenHook,
12
+ CompilerContext,
13
+ HookLoadError,
14
+ HookStage,
15
+ RuntimeHook,
16
+ get_global_hook_registry,
17
+ )
11
18
  from .lexar import TokenType
12
19
  from .ui import *
13
20
 
@@ -321,6 +328,17 @@ class LLVM:
321
328
  q = abs(a) // abs(b)
322
329
  return -q if (a < 0) != (b < 0) else q
323
330
 
331
+ def _hook_context(self, data: dict | None = None) -> CompilerContext:
332
+ """Build a CompilerContext exposing this codegen instance to hooks."""
333
+ ctx = CompilerContext(llvm_module=self.module)
334
+ ctx.data["llvm"] = self
335
+ ctx.data["module"] = self.module
336
+ if getattr(self, "builder", None) is not None:
337
+ ctx.data["builder"] = self.builder
338
+ if data:
339
+ ctx.data.update(data)
340
+ return ctx
341
+
324
342
  def _emit_int_divmod(self, left, right, is_rem):
325
343
  """Signed int division/remainder that stays well-defined in LLVM IR.
326
344
 
@@ -469,6 +487,7 @@ class LLVM:
469
487
  self.ssa_values = {}
470
488
  self.ssa_types = {} # Track types of SSA values
471
489
  self.scope_stack = []
490
+ self._deferred: list = [] # statements collected by `defer`, run LIFO on function exit
472
491
  self.structs = {}
473
492
  self.struct_fields = {}
474
493
  self.import_src_files = []
@@ -1007,9 +1026,17 @@ class LLVM:
1007
1026
  import os
1008
1027
  import tempfile
1009
1028
 
1010
- for hook in self._hook_registry.get_runtime_hooks():
1029
+ context = self._hook_context()
1030
+ for hook in self._hook_registry.get(HookStage.RUNTIME):
1031
+ if not isinstance(hook, RuntimeHook):
1032
+ continue
1011
1033
  try:
1012
- runtime_code = hook.get_runtime_code()
1034
+ for rel_file in hook.get_runtime_files(context):
1035
+ base = os.path.dirname(hook.hook_path) if hook.hook_path else ""
1036
+ abs_path = os.path.join(base, rel_file)
1037
+ if os.path.isfile(abs_path):
1038
+ self.import_src_files.append(abs_path)
1039
+ runtime_code = hook.get_runtime_code(context)
1013
1040
  if runtime_code:
1014
1041
  fd, tmp_path = tempfile.mkstemp(
1015
1042
  suffix=".c", prefix="hook_runtime_"
@@ -1157,7 +1184,9 @@ class LLVM:
1157
1184
  left_len = self.builder.call(strlen_fn, [left])
1158
1185
  right_len = self.builder.call(strlen_fn, [right])
1159
1186
  total_len = self.builder.add(left_len, right_len)
1160
- plus_one = self.builder.add(total_len, ir.Constant(_i32, 1))
1187
+ plus_one = self.builder.add(
1188
+ total_len, ir.Constant(total_len.type, 1)
1189
+ )
1161
1190
  new_str = self.builder.call(malloc_fn, [self.builder.zext(plus_one, _i64)])
1162
1191
  self.builder.call(memcpy_fn, [new_str, left, left_len])
1163
1192
  dest_plus = self.builder.gep(new_str, [left_len], inbounds=True)
@@ -1170,6 +1199,13 @@ class LLVM:
1170
1199
  def emit_exprstmt(self, node):
1171
1200
  return self.emit(node.expr)
1172
1201
 
1202
+ @register_emitter(DeferStmt)
1203
+ def emit_deferstmt(self, node: DeferStmt):
1204
+ # Collect the deferred statement; it is emitted (in reverse order) at
1205
+ # the function's exit points, both explicit `return` and implicit end.
1206
+ self._deferred.append(node.body)
1207
+ return None
1208
+
1173
1209
  def emit(self, node: Node | dict) -> _IRValue:
1174
1210
  key = id(node)
1175
1211
  if key in self._emit_memo:
@@ -1425,17 +1461,15 @@ class LLVM:
1425
1461
  def _emit_recursive(self, node: Node | dict) -> _IRValue:
1426
1462
  # Try codegen hooks if extensions are enabled
1427
1463
  if self.enable_extensions:
1428
- for hook in self._hook_registry.get_codegen_hooks():
1464
+ for hook in self._hook_registry.get(HookStage.CODEGEN):
1465
+ if not isinstance(hook, CodegenHook):
1466
+ continue
1429
1467
  try:
1430
1468
  if hook.should_emit_node(node):
1431
1469
  return hook.emit_node(
1432
1470
  node,
1433
1471
  self.builder,
1434
- {
1435
- "llvm": self,
1436
- "module": self.module,
1437
- "builder": self.builder,
1438
- },
1472
+ self._hook_context(data={"node": node}),
1439
1473
  )
1440
1474
  except Exception as e:
1441
1475
  raise HookLoadError(
@@ -1798,6 +1832,18 @@ class LLVM:
1798
1832
  raise Exception(f"Undefined variable '{name}'")
1799
1833
  raise Exception("Address-of requires a variable")
1800
1834
 
1835
+ @register_emitter(BorrowExpr)
1836
+ def emit_borrow(self, node: BorrowExpr):
1837
+ # `borrow x` / `borrow mut x` lowers to taking x's address, exactly like
1838
+ # `&x`. Mutable vs immutable is enforced at the semantic stage.
1839
+ return self.emit_addrof(AddrOf(node.operand))
1840
+
1841
+ @register_emitter(MoveExpr)
1842
+ def emit_move(self, node: MoveExpr):
1843
+ # `move x` is a compile-time ownership transfer; at runtime it just
1844
+ # yields the value of x (no copy is made).
1845
+ return self.emit(node.operand)
1846
+
1801
1847
  @register_emitter(SizeOf)
1802
1848
  def emit_sizeof(self, node: SizeOf):
1803
1849
  ty = self.llvm_type(node.type_expr)
@@ -2086,11 +2132,13 @@ class LLVM:
2086
2132
  old_ssa = self.ssa_values
2087
2133
  old_ssa_types = self.ssa_types
2088
2134
  old_scope_stack = self.scope_stack
2135
+ old_deferred = self._deferred
2089
2136
  self.locals = {}
2090
2137
  self.local_types = {}
2091
2138
  self.ssa_values = {}
2092
2139
  self.ssa_types = {}
2093
2140
  self.scope_stack = [{}]
2141
+ self._deferred = []
2094
2142
  for llvm_arg, (name, ptype) in zip(func.args, node.params.items()):
2095
2143
  if isinstance(llvm_arg.type, ir.VoidType):
2096
2144
  self._codegen_error(
@@ -2108,6 +2156,7 @@ class LLVM:
2108
2156
  self.emit(stmt)
2109
2157
 
2110
2158
  if not self._block_terminated():
2159
+ self._run_deferred()
2111
2160
  # Shutdown GC before main returns
2112
2161
  if node.name == "main" and not self.no_gc:
2113
2162
  gc_shutdown_fn = self.functions.get("gc_shutdown")
@@ -2133,6 +2182,7 @@ class LLVM:
2133
2182
  self.ssa_values = old_ssa
2134
2183
  self.ssa_types = old_ssa_types
2135
2184
  self.scope_stack = old_scope_stack
2185
+ self._deferred = old_deferred
2136
2186
 
2137
2187
  def _emit_decorated_funcdef(self, node: FuncDef):
2138
2188
  """Emit a decorated function: original as __name, trampoline, and wrapper."""
@@ -2247,6 +2297,7 @@ class LLVM:
2247
2297
  def emit_return(self, node: Return):
2248
2298
  if self._block_terminated():
2249
2299
  return
2300
+ self._run_deferred()
2250
2301
  # Shutdown GC before main returns
2251
2302
  fn_name = self.builder.function.name
2252
2303
  if fn_name == "main" and not self.no_gc:
@@ -2503,7 +2554,7 @@ class LLVM:
2503
2554
  return _DYN_STR
2504
2555
  if ty == "big":
2505
2556
  return _DYN_BIG
2506
- if ty.endswith("*") or ty.endswith("&"):
2557
+ if ty.endswith(("*", "&")):
2507
2558
  return _DYN_PTR
2508
2559
  return _DYN_INT
2509
2560
 
@@ -2554,7 +2605,7 @@ class LLVM:
2554
2605
  return self.builder.bitcast(bits, _double)
2555
2606
  if ty in ("str", "big", "void*"):
2556
2607
  return self.builder.inttoptr(bits, _i8ptr)
2557
- if ty.endswith("*") or ty.endswith("&"):
2608
+ if ty.endswith(("*", "&")):
2558
2609
  ptr_ty = self.llvm_type(ty)
2559
2610
  if isinstance(ptr_ty, ir.PointerType):
2560
2611
  return self.builder.inttoptr(bits, ptr_ty)
@@ -2567,9 +2618,7 @@ class LLVM:
2567
2618
  """True when a node carries a runtime-typed (dynamic) value."""
2568
2619
  if getattr(node, "inferred_type", None) == "dynamic":
2569
2620
  return True
2570
- if isinstance(node, Variable) and getattr(node, "dynamic", False):
2571
- return True
2572
- return False
2621
+ return bool(isinstance(node, Variable) and getattr(node, "dynamic", False))
2573
2622
 
2574
2623
  def _dyn_local_ptr(self, name):
2575
2624
  """Return (creating if needed) the {i32,i64} slot that backs a dynamic local."""
@@ -2665,6 +2714,25 @@ class LLVM:
2665
2714
  return True
2666
2715
  return block.is_terminated
2667
2716
 
2717
+ def _run_deferred(self):
2718
+ """Emit all accumulated deferred statements in LIFO order.
2719
+
2720
+ Called at every function exit point (explicit `return` statements and the
2721
+ implicit end-of-function return). It does NOT clear the accumulated list,
2722
+ so every static exit site of the function emits the full set of defers;
2723
+ at runtime only the exit that is actually reached executes them. The list
2724
+ is reset when the function's own emission completes (emit_funcdef restores
2725
+ the outer value).
2726
+ """
2727
+ if not self._deferred:
2728
+ return
2729
+ for stmt in reversed(self._deferred):
2730
+ if self._block_terminated():
2731
+ break
2732
+ self.emit(stmt)
2733
+ if self._block_terminated():
2734
+ break
2735
+
2668
2736
  @register_emitter(BinOp)
2669
2737
  def emit_binop(self, node):
2670
2738
  if node.op == TokenType.PLUS and self._is_string_concat(node):
@@ -2673,27 +2741,26 @@ class LLVM:
2673
2741
  # Runtime dispatch when either operand is dynamically typed.
2674
2742
  if not self.no_userspace and (
2675
2743
  self._is_dynamic_expr(node.left) or self._is_dynamic_expr(node.right)
2744
+ ) and node.op in (
2745
+ TokenType.PLUS,
2746
+ TokenType.MINUS,
2747
+ TokenType.STAR,
2748
+ TokenType.SLASH,
2749
+ TokenType.SLASH_SLASH,
2750
+ TokenType.PERCENT,
2751
+ TokenType.EQ_EQ,
2752
+ TokenType.NOT_EQ,
2753
+ TokenType.LESS,
2754
+ TokenType.GREATER,
2755
+ TokenType.LESS_EQ,
2756
+ TokenType.GREATER_EQ,
2757
+ TokenType.AMPERSAND,
2758
+ TokenType.PIPE,
2759
+ TokenType.CARET,
2760
+ TokenType.SHL,
2761
+ TokenType.SHR,
2676
2762
  ):
2677
- if node.op in (
2678
- TokenType.PLUS,
2679
- TokenType.MINUS,
2680
- TokenType.STAR,
2681
- TokenType.SLASH,
2682
- TokenType.SLASH_SLASH,
2683
- TokenType.PERCENT,
2684
- TokenType.EQ_EQ,
2685
- TokenType.NOT_EQ,
2686
- TokenType.LESS,
2687
- TokenType.GREATER,
2688
- TokenType.LESS_EQ,
2689
- TokenType.GREATER_EQ,
2690
- TokenType.AMPERSAND,
2691
- TokenType.PIPE,
2692
- TokenType.CARET,
2693
- TokenType.SHL,
2694
- TokenType.SHR,
2695
- ):
2696
- return self._emit_dyn_binop(node)
2763
+ return self._emit_dyn_binop(node)
2697
2764
 
2698
2765
  match node.op:
2699
2766
  case TokenType.AND:
@@ -2939,12 +3006,7 @@ class LLVM:
2939
3006
  and self.local_types.get(node.left.name) == "str"
2940
3007
  ):
2941
3008
  return True
2942
- if (
2943
- isinstance(node.right, Variable)
2944
- and self.local_types.get(node.right.name) == "str"
2945
- ):
2946
- return True
2947
- return False
3009
+ return bool(isinstance(node.right, Variable) and self.local_types.get(node.right.name) == "str")
2948
3010
 
2949
3011
  def _emit_dyn_binop(self, node):
2950
3012
  dynop_map = {
@@ -2987,7 +3049,8 @@ class LLVM:
2987
3049
  if f.name == "strlen":
2988
3050
  self._strlen_fn = f
2989
3051
  return f
2990
- fnty = ir.FunctionType(_i32, [_i8ptr])
3052
+ # strlen returns size_t (i64 on every cpyte target), not int.
3053
+ fnty = ir.FunctionType(_i64, [_i8ptr])
2991
3054
  fn = ir.Function(self.module, fnty, "strlen")
2992
3055
  self._strlen_fn = fn
2993
3056
  return fn
@@ -3000,7 +3063,8 @@ class LLVM:
3000
3063
  if f.name == "memcpy":
3001
3064
  self._memcpy_fn = f
3002
3065
  return f
3003
- fnty = ir.FunctionType(_i8ptr, [_i8ptr, _i8ptr, _i32])
3066
+ # n is size_t (i64 on every cpyte target).
3067
+ fnty = ir.FunctionType(_i8ptr, [_i8ptr, _i8ptr, _i64])
3004
3068
  fn = ir.Function(self.module, fnty, "memcpy")
3005
3069
  self._memcpy_fn = fn
3006
3070
  return fn
@@ -3927,6 +3991,8 @@ class LLVM:
3927
3991
 
3928
3992
  len_fn = self._get_strlen_fn()
3929
3993
  length = self.builder.call(len_fn, [iter_ptr])
3994
+ if length.type.width != 32:
3995
+ length = self.builder.trunc(length, _i32)
3930
3996
 
3931
3997
  idx_ptr = self._alloca(_i32, name=f"{var_name}.idx")
3932
3998
  self.builder.store(ir.Constant(_i32, 0), idx_ptr)