cpyte 2.0.2__tar.gz → 2.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (27) hide show
  1. {cpyte-2.0.2 → cpyte-2.2.0}/PKG-INFO +7 -1
  2. {cpyte-2.0.2 → cpyte-2.2.0}/pyproject.toml +1 -1
  3. {cpyte-2.0.2 → cpyte-2.2.0}/readme.md +7 -1
  4. {cpyte-2.0.2 → cpyte-2.2.0}/source/cpyte/astparse.py +82 -4
  5. {cpyte-2.0.2 → cpyte-2.2.0}/source/cpyte/bytecoding.py +62 -19
  6. {cpyte-2.0.2 → cpyte-2.2.0}/source/cpyte/compiling.py +76 -12
  7. {cpyte-2.0.2 → cpyte-2.2.0}/source/cpyte/extension_hooks.py +16 -3
  8. {cpyte-2.0.2 → cpyte-2.2.0}/source/cpyte/lexar.py +1 -0
  9. {cpyte-2.0.2 → cpyte-2.2.0}/source/cpyte/linker.py +9 -5
  10. {cpyte-2.0.2 → cpyte-2.2.0}/source/cpyte/mainpie.py +22 -12
  11. {cpyte-2.0.2 → cpyte-2.2.0}/source/cpyte/semantic_analasis.py +80 -1
  12. {cpyte-2.0.2 → cpyte-2.2.0}/source/cpyte.egg-info/PKG-INFO +7 -1
  13. {cpyte-2.0.2 → cpyte-2.2.0}/setup.cfg +0 -0
  14. {cpyte-2.0.2 → cpyte-2.2.0}/source/cpyte/__init__.py +0 -0
  15. {cpyte-2.0.2 → cpyte-2.2.0}/source/cpyte/__main__.py +0 -0
  16. {cpyte-2.0.2 → cpyte-2.2.0}/source/cpyte/_bignum_bc.py +0 -0
  17. {cpyte-2.0.2 → cpyte-2.2.0}/source/cpyte/_runtime_bc.py +0 -0
  18. {cpyte-2.0.2 → cpyte-2.2.0}/source/cpyte/clib.py +0 -0
  19. {cpyte-2.0.2 → cpyte-2.2.0}/source/cpyte/generate_bc.py +0 -0
  20. {cpyte-2.0.2 → cpyte-2.2.0}/source/cpyte/lsp_server.py +0 -0
  21. {cpyte-2.0.2 → cpyte-2.2.0}/source/cpyte/package_manifest.py +0 -0
  22. {cpyte-2.0.2 → cpyte-2.2.0}/source/cpyte.egg-info/SOURCES.txt +0 -0
  23. {cpyte-2.0.2 → cpyte-2.2.0}/source/cpyte.egg-info/dependency_links.txt +0 -0
  24. {cpyte-2.0.2 → cpyte-2.2.0}/source/cpyte.egg-info/entry_points.txt +0 -0
  25. {cpyte-2.0.2 → cpyte-2.2.0}/source/cpyte.egg-info/requires.txt +0 -0
  26. {cpyte-2.0.2 → cpyte-2.2.0}/source/cpyte.egg-info/top_level.txt +0 -0
  27. {cpyte-2.0.2 → cpyte-2.2.0}/test/test_bignum_jit.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: cpyte
3
- Version: 2.0.2
3
+ Version: 2.2.0
4
4
  Summary: The Cpyte programming language compiler
5
5
  Author: Hoang Duy Tung
6
6
  License: MIT
@@ -81,3 +81,9 @@ python source/mainpie.py --jit examples/<filename>.cpy
81
81
  Cpyte is experimental software. The compiler is continuously tested with fuzzing, and we're still discovering and fixing correctness bugs. While many programs compile and run correctly, I can't make any guarantees about correctness or stability.
82
82
 
83
83
  If you decide to use Cpyte, always use the latest version — older versions have bugs that are pretty easy to run into.
84
+
85
+ ## Memory Management
86
+
87
+ Cpyte uses **Boehm GC** (bdw-gc) for automatic garbage collection. Heap-allocated objects (via `new`) are managed automatically — no manual `free` needed.
88
+
89
+ **Note:** Boehm GC adds a small runtime overhead (~5-10%) compared to manual `malloc/free`. For performance-critical ecosystem projects or embedded use cases, this trade-off may be significant. The GC is required for safety but will slow down programs that do heavy heap allocation.
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "cpyte"
7
- version = "2.0.2"
7
+ version = "2.2.0"
8
8
  description = "The Cpyte programming language compiler"
9
9
  readme = "readme.md"
10
10
  requires-python = ">=3.11"
@@ -69,4 +69,10 @@ python source/mainpie.py --jit examples/<filename>.cpy
69
69
  ## Note
70
70
  Cpyte is experimental software. The compiler is continuously tested with fuzzing, and we're still discovering and fixing correctness bugs. While many programs compile and run correctly, I can't make any guarantees about correctness or stability.
71
71
 
72
- If you decide to use Cpyte, always use the latest version — older versions have bugs that are pretty easy to run into.
72
+ If you decide to use Cpyte, always use the latest version — older versions have bugs that are pretty easy to run into.
73
+
74
+ ## Memory Management
75
+
76
+ Cpyte uses **Boehm GC** (bdw-gc) for automatic garbage collection. Heap-allocated objects (via `new`) are managed automatically — no manual `free` needed.
77
+
78
+ **Note:** Boehm GC adds a small runtime overhead (~5-10%) compared to manual `malloc/free`. For performance-critical ecosystem projects or embedded use cases, this trade-off may be significant. The GC is required for safety but will slow down programs that do heavy heap allocation.
@@ -116,11 +116,12 @@ class Index:
116
116
  return f'Index({self.obj}, {self.index})'
117
117
 
118
118
  class Attr:
119
- __slots__ = ('obj', 'name', '_token')
119
+ __slots__ = ('obj', 'name', '_token', '_enum_member_value')
120
120
  def __init__(self, obj, name: str, token=None):
121
121
  self.obj = obj
122
122
  self.name = name
123
123
  self._token = token
124
+ self._enum_member_value = None
124
125
  def __repr__(self):
125
126
  return f'Attr({self.obj}, {self.name})'
126
127
 
@@ -426,6 +427,10 @@ def _parse_standard_statement(tokens: list[Token], pos: int):
426
427
  node, pos = parse_try(tokens, pos)
427
428
  elif tok.type == TokenType.KEYWORD and tok.value == 'raise':
428
429
  node, pos = parse_raise(tokens, pos)
430
+ elif tok.type == TokenType.KEYWORD and tok.value == 'enum':
431
+ node, pos = parse_enum(tokens, pos)
432
+ elif tok.type == TokenType.KEYWORD and tok.value == 'type':
433
+ node, pos = parse_type_alias(tokens, pos)
429
434
  elif tok.type == TokenType.IDENTIFIER and pos + 1 < len(tokens):
430
435
  if tok.value in _TYPE_NAMES or _looks_like_type(tokens, pos):
431
436
  try:
@@ -1158,8 +1163,8 @@ def parse_statement(tokens: list[Token], pos: int):
1158
1163
  if tok.type == TokenType.KEYWORD and tok.value == 'switch':
1159
1164
  return parse_switch(tokens, pos)
1160
1165
 
1161
- if tok.type == TokenType.KEYWORD and tok.value in ('while', 'for', 'class', 'struct', 'import'):
1162
- handler = {'while': parse_while, 'for': parse_for, 'class': parse_class, 'struct': parse_struct_def, 'import': parse_import}[tok.value]
1166
+ if tok.type == TokenType.KEYWORD and tok.value in ('while', 'for', 'class', 'struct', 'import', 'enum', 'type'):
1167
+ handler = {'while': parse_while, 'for': parse_for, 'class': parse_class, 'struct': parse_struct_def, 'import': parse_import, 'enum': parse_enum, 'type': parse_type_alias}[tok.value]
1163
1168
  return handler(tokens, pos)
1164
1169
 
1165
1170
  if tok.type == TokenType.KEYWORD and tok.value == 'try':
@@ -1268,8 +1273,26 @@ class Switch:
1268
1273
  self._token = token
1269
1274
  def __repr__(self):
1270
1275
  return f'Switch({self.value}, {self.cases})'
1276
+
1277
+
1278
+ class EnumDef:
1279
+ __slots__ = ('name', 'members', '_token')
1280
+ def __init__(self, name: str, members: list, token=None):
1281
+ self.name = name
1282
+ self.members = members
1283
+ self._token = token
1271
1284
  def __repr__(self):
1272
- return f'Switch({self.value}, {self.cases})'
1285
+ return f'EnumDef({self.name}, {self.members})'
1286
+
1287
+
1288
+ class TypeAlias:
1289
+ __slots__ = ('name', 'target_type', '_token')
1290
+ def __init__(self, name: str, target_type, token=None):
1291
+ self.name = name
1292
+ self.target_type = target_type
1293
+ self._token = token
1294
+ def __repr__(self):
1295
+ return f'TypeAlias({self.name}, {self.target_type})'
1273
1296
 
1274
1297
 
1275
1298
  def parse_switch(tokens: list[Token], pos: int):
@@ -1348,6 +1371,61 @@ def parse_generic_params(tokens: list[Token], pos: int):
1348
1371
  return params, pos + 1
1349
1372
 
1350
1373
 
1374
+ def parse_enum(tokens: list[Token], pos: int):
1375
+ tok = tokens[pos]
1376
+ pos += 1
1377
+ if pos >= len(tokens) or tokens[pos].type != TokenType.IDENTIFIER:
1378
+ raise ParseError('Expected enum name', tokens[pos] if pos < len(tokens) else None)
1379
+ name = tokens[pos].value
1380
+ pos += 1
1381
+ if pos >= len(tokens) or tokens[pos].type != TokenType.COLON:
1382
+ raise ParseError('Expected ":" after enum name', tokens[pos] if pos < len(tokens) else None)
1383
+ pos += 1
1384
+ if pos >= len(tokens) or tokens[pos].type != TokenType.NEWLINE:
1385
+ raise ParseError('Expected newline after ":"', tokens[pos] if pos < len(tokens) else None)
1386
+ pos += 1
1387
+ if pos >= len(tokens) or tokens[pos].type != TokenType.INDENT:
1388
+ raise ParseError('Expected indented block', tokens[pos] if pos < len(tokens) else None)
1389
+ pos += 1
1390
+
1391
+ members = []
1392
+ while pos < len(tokens) and tokens[pos].type not in (TokenType.DEDENT, TokenType.EOF):
1393
+ if tokens[pos].type != TokenType.IDENTIFIER:
1394
+ raise ParseError('Expected enum member name', tokens[pos])
1395
+ member_name = tokens[pos].value
1396
+ pos += 1
1397
+ member_value = None
1398
+ if pos < len(tokens) and tokens[pos].type == TokenType.KEYWORD and tokens[pos].value == '=':
1399
+ pos += 1
1400
+ member_value, pos = parse_expression(tokens, pos)
1401
+ members.append({'name': member_name, 'value': member_value, '_token': tokens[pos - 1]})
1402
+ if pos < len(tokens) and tokens[pos].type == TokenType.NEWLINE:
1403
+ pos += 1
1404
+
1405
+ if pos < len(tokens) and tokens[pos].type == TokenType.DEDENT:
1406
+ pos += 1
1407
+
1408
+ return EnumDef(name, members, token=tok), pos
1409
+
1410
+
1411
+ def parse_type_alias(tokens: list[Token], pos: int):
1412
+ tok = tokens[pos]
1413
+ pos += 1
1414
+ if pos >= len(tokens) or tokens[pos].type != TokenType.IDENTIFIER:
1415
+ raise ParseError('Expected type alias name', tokens[pos] if pos < len(tokens) else None)
1416
+ name = tokens[pos].value
1417
+ pos += 1
1418
+ if pos >= len(tokens) or tokens[pos].type != TokenType.EQUAL:
1419
+ raise ParseError('Expected "="', tokens[pos] if pos < len(tokens) else None)
1420
+ pos += 1
1421
+ target_type, pos = parse_type(tokens, pos)
1422
+ target_type_str = _type_to_str(target_type) if isinstance(target_type, tuple) else target_type
1423
+ _expect_newline(tokens, pos, tok)
1424
+ if pos < len(tokens) and tokens[pos].type == TokenType.NEWLINE:
1425
+ pos += 1
1426
+ return TypeAlias(name, target_type_str, token=tok), pos
1427
+
1428
+
1351
1429
  def parse_def(tokens: list[Token], pos: int):
1352
1430
  tok = tokens[pos]
1353
1431
  pos += 1
@@ -107,6 +107,15 @@ class LLVM:
107
107
 
108
108
  def __init__(self, no_userspace=False, enable_extensions=True):
109
109
  self.module = ir.Module("main")
110
+ try:
111
+ import llvmlite.binding as _binding
112
+ _binding.initialize_native_target()
113
+ _target = _binding.Target.from_default_triple()
114
+ _tm = _target.create_target_machine()
115
+ self.module.triple = _tm.triple
116
+ self.module.data_layout = _tm.target_data.get_data_layout()
117
+ except Exception:
118
+ pass
110
119
  self.builder = None
111
120
  self.functions = {}
112
121
  self.global_vars = {}
@@ -122,6 +131,9 @@ class LLVM:
122
131
  self.loop_stack = []
123
132
  self._malloc_fn = None
124
133
  self._free_fn = None
134
+ self._gc_alloc_fn = None
135
+ self._gc_mark_fn = None
136
+ self._gc_root_register_fn = None
125
137
  self._strlen_fn = None
126
138
  self._memcpy_fn = None
127
139
  self.no_userspace = no_userspace
@@ -171,6 +183,24 @@ class LLVM:
171
183
  fn = ir.Function(self.module, ir.FunctionType(ret, args), name=name)
172
184
  self.functions[name] = fn
173
185
 
186
+ # Boehm GC runtime functions
187
+ gc_fns = [
188
+ ('GC_init', _void, []),
189
+ ('GC_malloc', _i8ptr, [_i64]),
190
+ ('GC_realloc', _i8ptr, [_i8ptr, _i64]),
191
+ ('GC_free', _void, [_i8ptr]),
192
+ ('GC_gcollect', _void, []),
193
+ ('GC_get_heap_size', _i64, []),
194
+ ('GC_get_free_bytes', _i64, []),
195
+ ('GC_disable', _void, []),
196
+ ('GC_enable', _void, []),
197
+ ('GC_add_roots', _void, [_i8ptr, _i8ptr]),
198
+ ('GC_remove_roots', _void, [_i8ptr, _i8ptr]),
199
+ ]
200
+ for name, ret, args in gc_fns:
201
+ fn = ir.Function(self.module, ir.FunctionType(ret, args), name=name)
202
+ self.functions[name] = fn
203
+
174
204
  # Exception handling globals
175
205
  self._exc_buf_ptr = ir.GlobalVariable(self.module, _i8ptr, "_exc_buf_ptr")
176
206
  self._exc_buf_ptr.initializer = ir.Constant(_i8ptr, None)
@@ -413,6 +443,10 @@ class LLVM:
413
443
 
414
444
  if wrapper_builder is not None:
415
445
  self.builder = wrapper_builder
446
+ # Initialize Boehm GC at program start
447
+ gc_init_fn = self.functions.get('GC_init')
448
+ if gc_init_fn:
449
+ self.builder.call(gc_init_fn, [])
416
450
  for node in toplevel:
417
451
  self.emit(node)
418
452
  self.builder.ret(ir.Constant(ir.IntType(32), 0))
@@ -547,19 +581,21 @@ class LLVM:
547
581
  return ptr
548
582
 
549
583
  def _get_malloc_fn(self):
550
- fn = self._malloc_fn
584
+ # Use Boehm GC allocator instead of malloc
585
+ fn = self._gc_alloc_fn
551
586
  if fn is not None:
552
587
  return fn
553
588
  for f in self.module.functions:
554
- if f.name == 'malloc':
555
- self._malloc_fn = f
589
+ if f.name == 'GC_malloc':
590
+ self._gc_alloc_fn = f
556
591
  return f
557
592
  fnty = ir.FunctionType(_i8ptr, [_i64])
558
- fn = ir.Function(self.module, fnty, 'malloc')
559
- self._malloc_fn = fn
593
+ fn = ir.Function(self.module, fnty, 'GC_malloc')
594
+ self._gc_alloc_fn = fn
560
595
  return fn
561
596
 
562
597
  def _get_free_fn(self):
598
+ # Boehm GC manages memory - free is a no-op but keep for compatibility
563
599
  fn = self._free_fn
564
600
  if fn is not None:
565
601
  return fn
@@ -674,6 +710,9 @@ class LLVM:
674
710
  return self.builder.gep(obj, [idx], inbounds=True)
675
711
 
676
712
  def emit_attr(self, node: Attr):
713
+ enum_val = getattr(node, '_enum_member_value', None)
714
+ if enum_val is not None:
715
+ return ir.Constant(ir.IntType(32), enum_val)
677
716
  ptr = self._emit_lvalue_attr(node)
678
717
  return self.builder.load(ptr)
679
718
 
@@ -801,6 +840,12 @@ class LLVM:
801
840
  self.builder = ir.IRBuilder(entry)
802
841
  self.builder.position_at_end(entry)
803
842
 
843
+ # Initialize Boehm GC in main function
844
+ if node.name == 'main':
845
+ gc_init_fn = self.functions.get('GC_init')
846
+ if gc_init_fn:
847
+ self.builder.call(gc_init_fn, [])
848
+
804
849
  old_locals = self.locals
805
850
  old_local_types = self.local_types
806
851
  self.locals = {}
@@ -1067,6 +1112,18 @@ class LLVM:
1067
1112
  cmp = self.builder.call(self.functions['bigint_cmp'], [left, right])
1068
1113
  return self.builder.icmp_signed('>=', cmp, ir.Constant(_i64, 0))
1069
1114
 
1115
+ # String equality via strcmp (content, not pointer comparison)
1116
+ if node.op in (TokenType.EQ_EQ, TokenType.NOT_EQ):
1117
+ left_str = getattr(node.left, 'inferred_type', None) == 'str'
1118
+ right_str = getattr(node.right, 'inferred_type', None) == 'str'
1119
+ if left_str and right_str:
1120
+ cmp = self.builder.call(self._strcmp_fn, [left, right])
1121
+ zero = ir.Constant(_i32, 0)
1122
+ if node.op == TokenType.EQ_EQ:
1123
+ return self.builder.icmp_signed('==', cmp, zero)
1124
+ else:
1125
+ return self.builder.icmp_signed('!=', cmp, zero)
1126
+
1070
1127
  left, right = self._promote(left, right)
1071
1128
  is_float = isinstance(left.type, ir.DoubleType) or isinstance(right.type, ir.DoubleType)
1072
1129
  if is_float:
@@ -1096,20 +1153,6 @@ class LLVM:
1096
1153
  case TokenType.NOT_EQ:
1097
1154
  return self.builder.fcmp_ordered('!=', left, right)
1098
1155
 
1099
- # String equality via strcmp (content, not pointer comparison)
1100
- if node.op in (TokenType.EQ_EQ, TokenType.NOT_EQ):
1101
- left_str = getattr(node.left, 'inferred_type', None) == 'str'
1102
- right_str = getattr(node.right, 'inferred_type', None) == 'str'
1103
- if left_str and right_str:
1104
- fnty = ir.FunctionType(_i32, [_i8ptr, _i8ptr])
1105
- fn = ir.Function(self.module, fnty, '__cpy_strcmp')
1106
- cmp = self.builder.call(fn, [left, right])
1107
- zero = ir.Constant(_i32, 0)
1108
- if node.op == TokenType.EQ_EQ:
1109
- return self.builder.icmp_signed('==', cmp, zero)
1110
- else:
1111
- return self.builder.icmp_signed('!=', cmp, zero)
1112
-
1113
1156
  match node.op:
1114
1157
  case TokenType.PLUS:
1115
1158
  return self.builder.add(left, right)
@@ -82,13 +82,15 @@ def optimize(mod, opt_level=3):
82
82
  npm.run(mod, pb)
83
83
 
84
84
 
85
- def _compile_sources_object(src_files, target_triple=None):
85
+ def _compile_sources_object(src_files, target_triple=None, pic=False):
86
86
  objs = []
87
87
  for src in src_files:
88
88
  obj = src.rsplit('.', 1)[0] + '.o'
89
89
  cmd = ['clang', '-c', '-O3', '-o', obj, src]
90
90
  if target_triple:
91
91
  cmd.extend(['-target', target_triple])
92
+ if pic:
93
+ cmd.append('-fPIC')
92
94
  r = subprocess.run(cmd, capture_output=True, text=True)
93
95
  if r.returncode != 0:
94
96
  print(f'error compiling {src}: {r.stderr}', file=__import__('sys').stderr)
@@ -110,7 +112,7 @@ def _maybe_compile(module):
110
112
  return prog, src_files
111
113
  return module, None
112
114
 
113
- def run_jit(module, opt_level=3, src_files=None, no_userspace=False):
115
+ def run_jit(module, opt_level=3, src_files=None, no_userspace=False, pic=False):
114
116
  module, src_files_auto = _maybe_compile(module)
115
117
  if src_files_auto is not None:
116
118
  src_files = src_files_auto
@@ -128,16 +130,15 @@ def run_jit(module, opt_level=3, src_files=None, no_userspace=False):
128
130
  with open(src) as f:
129
131
  src_ir = f.read()
130
132
  else:
131
- ll = src.rsplit('.', 1)[0] + '.ll'
132
133
  r = subprocess.run(
133
134
  ['clang', '-S', '-emit-llvm', '-O0', '-target', target.triple,
134
- '-fno-stack-protector', '-o', ll, src],
135
- capture_output=True, text=True
136
- )
135
+ '-fno-stack-protector', '-o', '-', src],
136
+ capture_output=True, text=True)
137
+
137
138
  if r.returncode != 0:
138
139
  print(f'error compiling {src}: {r.stderr}', file=__import__('sys').stderr)
139
140
  raise SystemExit(1)
140
- src_ir = open(ll).read()
141
+ src_ir = r.stdout
141
142
  src_ir = _remove_probe_stack_ir(src_ir)
142
143
  src_mod = binding.parse_assembly(src_ir)
143
144
  binding.link_modules(mod, src_mod)
@@ -228,10 +229,20 @@ def run_jit(module, opt_level=3, src_files=None, no_userspace=False):
228
229
  pass
229
230
 
230
231
  _map_libc_fn(engine, mod, 'malloc', ctypes.c_size_t, ctypes.c_void_p)
232
+ _map_libc_fn(engine, mod, 'free', None, None, argtypes=[ctypes.c_void_p])
231
233
  _map_libc_fn(engine, mod, 'strlen', ctypes.c_char_p, ctypes.c_int)
232
234
  _map_libc_fn(engine, mod, 'memcpy', None, ctypes.c_void_p,
233
235
  argtypes=[ctypes.c_void_p, ctypes.c_void_p, ctypes.c_int])
234
236
 
237
+ try:
238
+ fn = mod.get_function('strcmp')
239
+ cfunctype = ctypes.CFUNCTYPE(ctypes.c_int, ctypes.c_char_p, ctypes.c_char_p)
240
+ cfn = cfunctype(_libc.strcmp)
241
+ _callbacks.append(cfn)
242
+ engine.add_global_mapping(fn, ctypes.cast(cfn, ctypes.c_void_p).value)
243
+ except NameError:
244
+ pass
245
+
235
246
  try:
236
247
  fn = mod.get_function('__cpy_strcmp')
237
248
  cfunctype = ctypes.CFUNCTYPE(ctypes.c_int, ctypes.c_char_p, ctypes.c_char_p)
@@ -241,6 +252,49 @@ def run_jit(module, opt_level=3, src_files=None, no_userspace=False):
241
252
  except NameError:
242
253
  pass
243
254
 
255
+ # Map Boehm GC functions
256
+ # NOTE: Boehm GC cannot safely scan JIT'd code stacks on ARM64 macOS
257
+ # because JIT'd frames lack DWARF unwind info. For JIT, GC_malloc
258
+ # falls back to malloc (no collection needed - process exits immediately).
259
+ # For AOT, Boehm GC works normally via libgc linkage (-lgc).
260
+ try:
261
+ _libgc = ctypes.CDLL('/opt/homebrew/lib/libgc.dylib')
262
+ _gc_fns = {
263
+ 'GC_init': (None, []),
264
+ 'GC_realloc': (ctypes.c_void_p, [ctypes.c_void_p, ctypes.c_size_t]),
265
+ 'GC_free': (None, [ctypes.c_void_p]),
266
+ 'GC_gcollect': (None, []),
267
+ 'GC_get_heap_size': (ctypes.c_size_t, []),
268
+ 'GC_get_free_bytes': (ctypes.c_size_t, []),
269
+ 'GC_disable': (None, []),
270
+ 'GC_enable': (None, []),
271
+ 'GC_add_roots': (None, [ctypes.c_void_p, ctypes.c_void_p]),
272
+ 'GC_remove_roots': (None, [ctypes.c_void_p, ctypes.c_void_p]),
273
+ }
274
+ for gc_name, (gc_restype, gc_argtypes) in _gc_fns.items():
275
+ try:
276
+ fn = mod.get_function(gc_name)
277
+ if gc_argtypes:
278
+ cfunctype = ctypes.CFUNCTYPE(gc_restype, *gc_argtypes)
279
+ else:
280
+ cfunctype = ctypes.CFUNCTYPE(gc_restype)
281
+ cfn = cfunctype(getattr(_libgc, gc_name))
282
+ _callbacks.append(cfn)
283
+ engine.add_global_mapping(fn, ctypes.cast(cfn, ctypes.c_void_p).value)
284
+ except (NameError, AttributeError):
285
+ pass
286
+ # For JIT: map GC_malloc to malloc (avoids Boehm stack scanning crashes)
287
+ try:
288
+ fn = mod.get_function('GC_malloc')
289
+ cfunctype = ctypes.CFUNCTYPE(ctypes.c_void_p, ctypes.c_size_t)
290
+ cfn = cfunctype(_libc.malloc)
291
+ _callbacks.append(cfn)
292
+ engine.add_global_mapping(fn, ctypes.cast(cfn, ctypes.c_void_p).value)
293
+ except (NameError, AttributeError):
294
+ pass
295
+ except OSError:
296
+ pass
297
+
244
298
  engine.finalize_object()
245
299
  engine.run_static_constructors()
246
300
 
@@ -271,7 +325,7 @@ def _map_libc_fn(engine, mod, name, argtype, restype, argtypes=None):
271
325
  pass
272
326
 
273
327
 
274
- def run_aot(module, output="program.o", opt_level=3, src_files=None, no_userspace=False):
328
+ def run_aot(module, output="program.o", opt_level=3, src_files=None, no_userspace=False, pic=False):
275
329
  llvm_ir = str(module)
276
330
  import llvmlite.binding as binding
277
331
  binding.initialize_native_target()
@@ -285,7 +339,10 @@ def run_aot(module, output="program.o", opt_level=3, src_files=None, no_userspac
285
339
  mod.verify()
286
340
 
287
341
  target = binding.Target.from_default_triple()
288
- target_machine = target.create_target_machine()
342
+ if pic:
343
+ target_machine = target.create_target_machine(reloc='pic')
344
+ else:
345
+ target_machine = target.create_target_machine()
289
346
 
290
347
  obj = target_machine.emit_object(mod)
291
348
 
@@ -295,8 +352,11 @@ def run_aot(module, output="program.o", opt_level=3, src_files=None, no_userspac
295
352
  objs = [output]
296
353
  for src in (src_files or []):
297
354
  src_obj = src.rsplit('.', 1)[0] + '.o'
355
+ cmd = ['clang', '-c', '-O3', '-o', src_obj, src]
356
+ if pic:
357
+ cmd.append('-fPIC')
298
358
  r = subprocess.run(
299
- ['clang', '-c', '-O3', '-o', src_obj, src],
359
+ cmd,
300
360
  capture_output=True, text=True
301
361
  )
302
362
  if r.returncode != 0:
@@ -306,8 +366,11 @@ def run_aot(module, output="program.o", opt_level=3, src_files=None, no_userspac
306
366
 
307
367
  if not no_userspace:
308
368
  runtime_obj = output + '.runtime.o'
369
+ cmd = ['clang', '-c', '-O3', '-o', runtime_obj, _RUNTIME_C]
370
+ if pic:
371
+ cmd.append('-fPIC')
309
372
  r = subprocess.run(
310
- ['clang', '-c', '-O3', '-o', runtime_obj, _RUNTIME_C],
373
+ cmd,
311
374
  capture_output=True, text=True
312
375
  )
313
376
  if r.returncode == 0:
@@ -315,7 +378,8 @@ def run_aot(module, output="program.o", opt_level=3, src_files=None, no_userspac
315
378
 
316
379
  out_name = output.rsplit('.', 1)[0] if '.' in output else output
317
380
  r = subprocess.run(
318
- ['clang', '-O3', '-o', out_name] + objs + ['-lm'],
381
+ ['clang', '-O3', '-o', out_name] + objs + ['-lgc', '-lm',
382
+ '-L/opt/homebrew/opt/bdw-gc/lib', '-I/opt/homebrew/opt/bdw-gc/include'],
319
383
  capture_output=True, text=True
320
384
  )
321
385
  if r.returncode != 0:
@@ -209,8 +209,13 @@ class HookRegistry:
209
209
  """Get the current compiler context."""
210
210
  return self._context.copy()
211
211
 
212
+ def _is_duplicate(self, existing_list, package_name, hook_path) -> bool:
213
+ return any(r.package_name == package_name and r.hook_path == hook_path for r in existing_list)
214
+
212
215
  def register_lexer_hook(self, hook: LexerHook, priority: int = 0) -> None:
213
216
  """Register a lexer hook."""
217
+ if self._is_duplicate(self._lexer_hooks, hook.package_name, hook.hook_path):
218
+ return
214
219
  registration = HookRegistration(
215
220
  hook=hook,
216
221
  priority=priority,
@@ -219,9 +224,11 @@ class HookRegistry:
219
224
  )
220
225
  self._lexer_hooks.append(registration)
221
226
  self._lexer_hooks.sort(key=lambda r: r.priority, reverse=True)
222
-
227
+
223
228
  def register_parser_hook(self, hook: ParserHook, priority: int = 0) -> None:
224
229
  """Register a parser hook."""
230
+ if self._is_duplicate(self._parser_hooks, hook.package_name, hook.hook_path):
231
+ return
225
232
  registration = HookRegistration(
226
233
  hook=hook,
227
234
  priority=priority,
@@ -230,9 +237,11 @@ class HookRegistry:
230
237
  )
231
238
  self._parser_hooks.append(registration)
232
239
  self._parser_hooks.sort(key=lambda r: r.priority, reverse=True)
233
-
240
+
234
241
  def register_semantic_hook(self, hook: SemanticHook, priority: int = 0) -> None:
235
242
  """Register a semantic analysis hook."""
243
+ if self._is_duplicate(self._semantic_hooks, hook.package_name, hook.hook_path):
244
+ return
236
245
  registration = HookRegistration(
237
246
  hook=hook,
238
247
  priority=priority,
@@ -244,6 +253,8 @@ class HookRegistry:
244
253
 
245
254
  def register_codegen_hook(self, hook: CodegenHook, priority: int = 0) -> None:
246
255
  """Register a code generation hook."""
256
+ if self._is_duplicate(self._codegen_hooks, hook.package_name, hook.hook_path):
257
+ return
247
258
  registration = HookRegistration(
248
259
  hook=hook,
249
260
  priority=priority,
@@ -252,9 +263,11 @@ class HookRegistry:
252
263
  )
253
264
  self._codegen_hooks.append(registration)
254
265
  self._codegen_hooks.sort(key=lambda r: r.priority, reverse=True)
255
-
266
+
256
267
  def register_runtime_hook(self, hook: RuntimeHook, priority: int = 0) -> None:
257
268
  """Register a runtime hook."""
269
+ if self._is_duplicate(self._runtime_hooks, hook.package_name, hook.hook_path):
270
+ return
258
271
  registration = HookRegistration(
259
272
  hook=hook,
260
273
  priority=priority,
@@ -72,6 +72,7 @@ _BASE_KEYWORDS = {
72
72
  'let',
73
73
  'try', 'except', 'raise',
74
74
  'asm',
75
+ 'enum', 'type',
75
76
  }
76
77
 
77
78
  # Additional keywords registered by packages
@@ -35,13 +35,15 @@ class Linker:
35
35
  def cc(self):
36
36
  return self._cc
37
37
 
38
- def compile_c(self, src, output=None, opt_level=3, debug=False):
38
+ def compile_c(self, src, output=None, opt_level=3, debug=False, pic=False):
39
39
  if output is None:
40
40
  base = src.rsplit('.', 1)[0] if '.' in src else src
41
41
  output = base + '.o'
42
42
  cmd = [self._cc, '-c']
43
43
  if debug:
44
44
  cmd.append('-g')
45
+ if pic:
46
+ cmd.append('-fPIC')
45
47
  if opt_level is not None:
46
48
  cmd.append(f'-O{opt_level}')
47
49
  cmd.extend(['-o', output, src])
@@ -52,10 +54,12 @@ class Linker:
52
54
  return output
53
55
 
54
56
  def link(self, objects, output, libraries=None, library_paths=None,
55
- shared=False, debug=False, opt_level=3, frameworks=None):
57
+ shared=False, debug=False, opt_level=3, frameworks=None, pic=False):
56
58
  cmd = [self._cc]
57
59
  if shared:
58
60
  cmd.append('-shared')
61
+ if pic:
62
+ cmd.append('-fPIC')
59
63
  if debug:
60
64
  cmd.append('-g')
61
65
  if opt_level is not None:
@@ -77,13 +81,13 @@ class Linker:
77
81
 
78
82
 
79
83
  def build(objects, output=None, libraries=None, library_paths=None,
80
- shared=False, debug=False, opt_level=3, cc=None, frameworks=None):
84
+ shared=False, debug=False, opt_level=3, cc=None, frameworks=None, pic=False):
81
85
  linker = Linker(cc)
82
86
 
83
87
  final_objects = []
84
88
  for src in objects:
85
89
  if src.endswith('.c'):
86
- obj = linker.compile_c(src, opt_level=opt_level, debug=debug)
90
+ obj = linker.compile_c(src, opt_level=opt_level, debug=debug, pic=pic)
87
91
  final_objects.append(obj)
88
92
  else:
89
93
  final_objects.append(src)
@@ -100,5 +104,5 @@ def build(objects, output=None, libraries=None, library_paths=None,
100
104
  libraries=libraries, library_paths=library_paths,
101
105
  shared=shared,
102
106
  debug=debug, opt_level=opt_level,
103
- frameworks=frameworks,
107
+ frameworks=frameworks, pic=pic,
104
108
  )
@@ -283,9 +283,9 @@ def _collect_frameworks(nodes):
283
283
  return list(set(frameworks))
284
284
 
285
285
 
286
- def cmd_build(args, tab_size=4, strict=False, no_userspace=False):
286
+ def cmd_build(args, tab_size=4, strict=False, no_userspace=False, pic=False):
287
287
  if not args:
288
- print('Usage: cpy build [--output O] [--debug] [--opt N] [--no-userspace] <source.cpy>', file=sys.stderr)
288
+ print('Usage: cpy build [--output O] [--debug] [--opt N] [--no-userspace] [--pic] <source.cpy>', file=sys.stderr)
289
289
  sys.exit(1)
290
290
 
291
291
  output = None
@@ -308,6 +308,9 @@ def cmd_build(args, tab_size=4, strict=False, no_userspace=False):
308
308
  elif a == '--no-userspace':
309
309
  no_userspace = True
310
310
  i += 1
311
+ elif a == '--pic':
312
+ pic = True
313
+ i += 1
311
314
  elif a == '--opt' and i + 1 < len(args):
312
315
  opt = int(args[i + 1])
313
316
  i += 2
@@ -319,7 +322,7 @@ def cmd_build(args, tab_size=4, strict=False, no_userspace=False):
319
322
  sys.exit(1)
320
323
 
321
324
  if not src_file:
322
- print('Usage: cpy build [--output O] [--debug] [--opt N] <source.cpy>', file=sys.stderr)
325
+ print('Usage: cpy build [--output O] [--debug] [--opt N] [--pic] <source.cpy>', file=sys.stderr)
323
326
  sys.exit(1)
324
327
 
325
328
  with open(src_file) as f:
@@ -345,7 +348,10 @@ def cmd_build(args, tab_size=4, strict=False, no_userspace=False):
345
348
  mod.verify()
346
349
 
347
350
  target = binding.Target.from_default_triple()
348
- target_machine = target.create_target_machine()
351
+ if pic:
352
+ target_machine = target.create_target_machine(reloc='pic')
353
+ else:
354
+ target_machine = target.create_target_machine()
349
355
  obj = target_machine.emit_object(mod)
350
356
  with open(obj_file, 'wb') as f:
351
357
  f.write(obj)
@@ -354,16 +360,17 @@ def cmd_build(args, tab_size=4, strict=False, no_userspace=False):
354
360
  objs = [obj_file]
355
361
  for src in (src_files or []):
356
362
  src_obj = src.rsplit('.', 1)[0] + '.o'
357
- l.compile_c(src, output=src_obj, opt_level=opt, debug=debug)
363
+ l.compile_c(src, output=src_obj, opt_level=opt, debug=debug, pic=pic)
358
364
  objs.append(src_obj)
359
365
 
360
366
  if not no_userspace:
361
367
  runtime_obj = out_base + '.runtime.o'
362
- l.compile_c(_RUNTIME_C, output=runtime_obj, opt_level=opt, debug=debug)
368
+ l.compile_c(_RUNTIME_C, output=runtime_obj, opt_level=opt, debug=debug, pic=pic)
363
369
  objs.append(runtime_obj)
364
370
 
365
371
  executable = output or out_base
366
- l.link(objs, executable, opt_level=opt, debug=debug, frameworks=frameworks)
372
+ l.link(objs, executable, libraries=['gc'], library_paths=['/opt/homebrew/opt/bdw-gc/lib'],
373
+ opt_level=opt, debug=debug, frameworks=frameworks, pic=pic)
367
374
  print(f'Wrote {executable}')
368
375
 
369
376
 
@@ -374,6 +381,7 @@ def main():
374
381
 
375
382
  strict = False
376
383
  no_userspace = False
384
+ pic = False
377
385
  while args and args[0].startswith('--'):
378
386
  flag = args.pop(0)
379
387
  if flag == '--tab-size':
@@ -382,6 +390,8 @@ def main():
382
390
  strict = True
383
391
  elif flag == '--no-userspace':
384
392
  no_userspace = True
393
+ elif flag == '--pic':
394
+ pic = True
385
395
  elif flag == '--ast':
386
396
  mode = 'ast'
387
397
  elif flag == '--emit-llvm':
@@ -395,12 +405,12 @@ def main():
395
405
  sys.exit(1)
396
406
 
397
407
  if not args:
398
- print('Usage: cpy [--tab-size N] [--strict] [--no-userspace] [--ast|--emit-llvm|--jit|--aot] <source file>', file=sys.stderr)
399
- print(' cpy build [--output O] [--debug] [--opt N] [--no-userspace] <source.cpy>', file=sys.stderr)
408
+ print('Usage: cpy [--tab-size N] [--strict] [--no-userspace] [--pic] [--ast|--emit-llvm|--jit|--aot] <source file>', file=sys.stderr)
409
+ print(' cpy build [--output O] [--debug] [--opt N] [--no-userspace] [--pic] <source.cpy>', file=sys.stderr)
400
410
  sys.exit(1)
401
411
 
402
412
  if args[0] == 'build':
403
- cmd_build(args[1:], tab_size=tab_size, strict=strict, no_userspace=no_userspace)
413
+ cmd_build(args[1:], tab_size=tab_size, strict=strict, no_userspace=no_userspace, pic=pic)
404
414
  return
405
415
 
406
416
  with open(args[0]) as f:
@@ -419,10 +429,10 @@ def main():
419
429
  elif mode == 'aot':
420
430
  out_base = args[0].rsplit('.', 1)[0] if '.' in args[0] else 'program'
421
431
  obj_file = 'program.o'
422
- run_aot(prog, output=obj_file, src_files=src_files, no_userspace=no_userspace)
432
+ run_aot(prog, output=obj_file, src_files=src_files, no_userspace=no_userspace, pic=pic)
423
433
  print(f'Wrote {out_base}')
424
434
  else:
425
- run_jit(prog, src_files=src_files, no_userspace=no_userspace)
435
+ run_jit(prog, src_files=src_files, no_userspace=no_userspace, pic=pic)
426
436
 
427
437
 
428
438
  if __name__ == '__main__':
@@ -27,6 +27,7 @@ from .astparse import (
27
27
  VarDecl, Break, Continue, Switch, Import, While,
28
28
  NewExpr, Deref, AddrOf, SizeOf, StructDef, Field, Input,
29
29
  InputStr, Signed67, Try, Raise, ExceptHandler, InlineAsm,
30
+ EnumDef, TypeAlias,
30
31
  parse_file, ParseError,
31
32
  )
32
33
  from .clib import resolve_library, parse_header_file, parse_c_source
@@ -360,6 +361,12 @@ class SemanticAnalyzer:
360
361
  return None
361
362
  if sym.const_value is not None:
362
363
  node.const_value = sym.const_value
364
+ if sym.kind in ('enum', 'struct'):
365
+ node.inferred_type = node.name
366
+ return node.name
367
+ if sym.kind == 'type_alias':
368
+ node.inferred_type = sym.type
369
+ return sym.type
363
370
  node.inferred_type = sym.type
364
371
  return sym.type
365
372
 
@@ -581,6 +588,17 @@ class SemanticAnalyzer:
581
588
  obj_t = self._infer_type(node.obj)
582
589
  if obj_t:
583
590
  lookup_t = obj_t[:-1] if obj_t.endswith('*') else obj_t
591
+ sym = self.current_scope.lookup(lookup_t)
592
+ if sym and sym.kind == 'enum':
593
+ member_sym = self.current_scope.lookup(f'{lookup_t}.{node.name}')
594
+ if member_sym and member_sym.kind == 'enum_member':
595
+ node._enum_member_value = member_sym.const_value
596
+ return 'int'
597
+ self.error(
598
+ f'enum `{lookup_t}` has no member `{node.name}`',
599
+ node
600
+ )
601
+ return None
584
602
  struct_sym = self.current_scope.lookup(lookup_t)
585
603
  if struct_sym and struct_sym.kind == 'struct' and struct_sym.node:
586
604
  for field in struct_sym.node.fields:
@@ -707,6 +725,10 @@ class SemanticAnalyzer:
707
725
  self._visit_import(node)
708
726
  elif isinstance(node, StructDef):
709
727
  self._visit_struct(node, scope)
728
+ elif isinstance(node, EnumDef):
729
+ self._visit_enum(node, scope)
730
+ elif isinstance(node, TypeAlias):
731
+ self._visit_type_alias(node, scope)
710
732
  elif isinstance(node, Try):
711
733
  self._visit_try(node, scope)
712
734
  elif isinstance(node, Raise):
@@ -1160,7 +1182,8 @@ class SemanticAnalyzer:
1160
1182
  self._infer_type(node.target)
1161
1183
 
1162
1184
  def _visit_vardecl(self, node: VarDecl, scope: Scope | None = None):
1163
- val_type = node.var_type
1185
+ val_type = self._resolve_type_alias(node.var_type)
1186
+ node.var_type = val_type
1164
1187
  s = scope or self.current_scope
1165
1188
  existing = s.lookup_local(node.name)
1166
1189
  if existing:
@@ -1283,6 +1306,54 @@ class SemanticAnalyzer:
1283
1306
  for field in node.fields:
1284
1307
  struct_scope.define(field.name, Symbol('field', field.type_expr, node))
1285
1308
 
1309
+ def _visit_enum(self, node: EnumDef, scope: Scope | None = None):
1310
+ s = scope or self.globals
1311
+ existing = s.lookup_local(node.name)
1312
+ if existing:
1313
+ self.error(f'redefinition of enum `{node.name}`', node)
1314
+ return
1315
+ s.define(node.name, Symbol('enum', None, node))
1316
+ for i, member in enumerate(node.members):
1317
+ if member['value'] is not None:
1318
+ val = self._eval_const_expr(member['value'])
1319
+ member['_const_value'] = val
1320
+ else:
1321
+ member['_const_value'] = i
1322
+ member['_auto_index'] = i
1323
+ sym = Symbol('enum_member', node.name, node)
1324
+ sym.const_value = member['_const_value']
1325
+ s.define(f'{node.name}.{member["name"]}', sym)
1326
+
1327
+ def _visit_type_alias(self, node: TypeAlias, scope: Scope | None = None):
1328
+ s = scope or self.globals
1329
+ existing = s.lookup_local(node.name)
1330
+ if existing:
1331
+ self.error(f'redefinition of type alias `{node.name}`', node)
1332
+ return
1333
+ s.define(node.name, Symbol('type_alias', node.target_type, node))
1334
+
1335
+ def _eval_const_expr(self, node):
1336
+ if isinstance(node, Number):
1337
+ return node.value
1338
+ if isinstance(node, UnaryOp):
1339
+ val = self._eval_const_expr(node.operand)
1340
+ if node.op.name == 'MINUS':
1341
+ return -val
1342
+ if node.op.name == 'NOT':
1343
+ return not val
1344
+ if isinstance(node, BinOp):
1345
+ left = self._eval_const_expr(node.left)
1346
+ right = self._eval_const_expr(node.right)
1347
+ match node.op.name:
1348
+ case 'PLUS': return left + right
1349
+ case 'MINUS': return left - right
1350
+ case 'STAR': return left * right
1351
+ case 'SLASH': return left // right
1352
+ case 'PERCENT': return left % right
1353
+ if isinstance(node, Variable):
1354
+ return node.const_value
1355
+ return None
1356
+
1286
1357
  def _visit_while(self, node, scope: Scope | None = None):
1287
1358
  if isinstance(node, While):
1288
1359
  self._infer_type(node.cond)
@@ -1314,6 +1385,14 @@ class SemanticAnalyzer:
1314
1385
  self._loop_depth -= 1
1315
1386
  self.locals = old_locals
1316
1387
 
1388
+ def _resolve_type_alias(self, type_name):
1389
+ if type_name is None:
1390
+ return None
1391
+ sym = self.current_scope.lookup(type_name)
1392
+ if sym and sym.kind == 'type_alias':
1393
+ return self._resolve_type_alias(sym.type)
1394
+ return type_name
1395
+
1317
1396
 
1318
1397
  def analyze(source: str, nodes: list, strict: bool = False, workspace_root: str | None = None, enable_extensions: bool = True) -> str | None:
1319
1398
  analyzer = SemanticAnalyzer(source, strict=strict, workspace_root=workspace_root, enable_extensions=enable_extensions)
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: cpyte
3
- Version: 2.0.2
3
+ Version: 2.2.0
4
4
  Summary: The Cpyte programming language compiler
5
5
  Author: Hoang Duy Tung
6
6
  License: MIT
@@ -81,3 +81,9 @@ python source/mainpie.py --jit examples/<filename>.cpy
81
81
  Cpyte is experimental software. The compiler is continuously tested with fuzzing, and we're still discovering and fixing correctness bugs. While many programs compile and run correctly, I can't make any guarantees about correctness or stability.
82
82
 
83
83
  If you decide to use Cpyte, always use the latest version — older versions have bugs that are pretty easy to run into.
84
+
85
+ ## Memory Management
86
+
87
+ Cpyte uses **Boehm GC** (bdw-gc) for automatic garbage collection. Heap-allocated objects (via `new`) are managed automatically — no manual `free` needed.
88
+
89
+ **Note:** Boehm GC adds a small runtime overhead (~5-10%) compared to manual `malloc/free`. For performance-critical ecosystem projects or embedded use cases, this trade-off may be significant. The GC is required for safety but will slow down programs that do heavy heap allocation.
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes