cpyte 2.0.2__tar.gz → 2.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {cpyte-2.0.2 → cpyte-2.2.0}/PKG-INFO +7 -1
- {cpyte-2.0.2 → cpyte-2.2.0}/pyproject.toml +1 -1
- {cpyte-2.0.2 → cpyte-2.2.0}/readme.md +7 -1
- {cpyte-2.0.2 → cpyte-2.2.0}/source/cpyte/astparse.py +82 -4
- {cpyte-2.0.2 → cpyte-2.2.0}/source/cpyte/bytecoding.py +62 -19
- {cpyte-2.0.2 → cpyte-2.2.0}/source/cpyte/compiling.py +76 -12
- {cpyte-2.0.2 → cpyte-2.2.0}/source/cpyte/extension_hooks.py +16 -3
- {cpyte-2.0.2 → cpyte-2.2.0}/source/cpyte/lexar.py +1 -0
- {cpyte-2.0.2 → cpyte-2.2.0}/source/cpyte/linker.py +9 -5
- {cpyte-2.0.2 → cpyte-2.2.0}/source/cpyte/mainpie.py +22 -12
- {cpyte-2.0.2 → cpyte-2.2.0}/source/cpyte/semantic_analasis.py +80 -1
- {cpyte-2.0.2 → cpyte-2.2.0}/source/cpyte.egg-info/PKG-INFO +7 -1
- {cpyte-2.0.2 → cpyte-2.2.0}/setup.cfg +0 -0
- {cpyte-2.0.2 → cpyte-2.2.0}/source/cpyte/__init__.py +0 -0
- {cpyte-2.0.2 → cpyte-2.2.0}/source/cpyte/__main__.py +0 -0
- {cpyte-2.0.2 → cpyte-2.2.0}/source/cpyte/_bignum_bc.py +0 -0
- {cpyte-2.0.2 → cpyte-2.2.0}/source/cpyte/_runtime_bc.py +0 -0
- {cpyte-2.0.2 → cpyte-2.2.0}/source/cpyte/clib.py +0 -0
- {cpyte-2.0.2 → cpyte-2.2.0}/source/cpyte/generate_bc.py +0 -0
- {cpyte-2.0.2 → cpyte-2.2.0}/source/cpyte/lsp_server.py +0 -0
- {cpyte-2.0.2 → cpyte-2.2.0}/source/cpyte/package_manifest.py +0 -0
- {cpyte-2.0.2 → cpyte-2.2.0}/source/cpyte.egg-info/SOURCES.txt +0 -0
- {cpyte-2.0.2 → cpyte-2.2.0}/source/cpyte.egg-info/dependency_links.txt +0 -0
- {cpyte-2.0.2 → cpyte-2.2.0}/source/cpyte.egg-info/entry_points.txt +0 -0
- {cpyte-2.0.2 → cpyte-2.2.0}/source/cpyte.egg-info/requires.txt +0 -0
- {cpyte-2.0.2 → cpyte-2.2.0}/source/cpyte.egg-info/top_level.txt +0 -0
- {cpyte-2.0.2 → cpyte-2.2.0}/test/test_bignum_jit.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: cpyte
|
|
3
|
-
Version: 2.0
|
|
3
|
+
Version: 2.2.0
|
|
4
4
|
Summary: The Cpyte programming language compiler
|
|
5
5
|
Author: Hoang Duy Tung
|
|
6
6
|
License: MIT
|
|
@@ -81,3 +81,9 @@ python source/mainpie.py --jit examples/<filename>.cpy
|
|
|
81
81
|
Cpyte is experimental software. The compiler is continuously tested with fuzzing, and we're still discovering and fixing correctness bugs. While many programs compile and run correctly, I can't make any guarantees about correctness or stability.
|
|
82
82
|
|
|
83
83
|
If you decide to use Cpyte, always use the latest version — older versions have bugs that are pretty easy to run into.
|
|
84
|
+
|
|
85
|
+
## Memory Management
|
|
86
|
+
|
|
87
|
+
Cpyte uses **Boehm GC** (bdw-gc) for automatic garbage collection. Heap-allocated objects (via `new`) are managed automatically — no manual `free` needed.
|
|
88
|
+
|
|
89
|
+
**Note:** Boehm GC adds a small runtime overhead (~5-10%) compared to manual `malloc/free`. For performance-critical ecosystem projects or embedded use cases, this trade-off may be significant. The GC is required for safety but will slow down programs that do heavy heap allocation.
|
|
@@ -69,4 +69,10 @@ python source/mainpie.py --jit examples/<filename>.cpy
|
|
|
69
69
|
## Note
|
|
70
70
|
Cpyte is experimental software. The compiler is continuously tested with fuzzing, and we're still discovering and fixing correctness bugs. While many programs compile and run correctly, I can't make any guarantees about correctness or stability.
|
|
71
71
|
|
|
72
|
-
If you decide to use Cpyte, always use the latest version — older versions have bugs that are pretty easy to run into.
|
|
72
|
+
If you decide to use Cpyte, always use the latest version — older versions have bugs that are pretty easy to run into.
|
|
73
|
+
|
|
74
|
+
## Memory Management
|
|
75
|
+
|
|
76
|
+
Cpyte uses **Boehm GC** (bdw-gc) for automatic garbage collection. Heap-allocated objects (via `new`) are managed automatically — no manual `free` needed.
|
|
77
|
+
|
|
78
|
+
**Note:** Boehm GC adds a small runtime overhead (~5-10%) compared to manual `malloc/free`. For performance-critical ecosystem projects or embedded use cases, this trade-off may be significant. The GC is required for safety but will slow down programs that do heavy heap allocation.
|
|
@@ -116,11 +116,12 @@ class Index:
|
|
|
116
116
|
return f'Index({self.obj}, {self.index})'
|
|
117
117
|
|
|
118
118
|
class Attr:
|
|
119
|
-
__slots__ = ('obj', 'name', '_token')
|
|
119
|
+
__slots__ = ('obj', 'name', '_token', '_enum_member_value')
|
|
120
120
|
def __init__(self, obj, name: str, token=None):
|
|
121
121
|
self.obj = obj
|
|
122
122
|
self.name = name
|
|
123
123
|
self._token = token
|
|
124
|
+
self._enum_member_value = None
|
|
124
125
|
def __repr__(self):
|
|
125
126
|
return f'Attr({self.obj}, {self.name})'
|
|
126
127
|
|
|
@@ -426,6 +427,10 @@ def _parse_standard_statement(tokens: list[Token], pos: int):
|
|
|
426
427
|
node, pos = parse_try(tokens, pos)
|
|
427
428
|
elif tok.type == TokenType.KEYWORD and tok.value == 'raise':
|
|
428
429
|
node, pos = parse_raise(tokens, pos)
|
|
430
|
+
elif tok.type == TokenType.KEYWORD and tok.value == 'enum':
|
|
431
|
+
node, pos = parse_enum(tokens, pos)
|
|
432
|
+
elif tok.type == TokenType.KEYWORD and tok.value == 'type':
|
|
433
|
+
node, pos = parse_type_alias(tokens, pos)
|
|
429
434
|
elif tok.type == TokenType.IDENTIFIER and pos + 1 < len(tokens):
|
|
430
435
|
if tok.value in _TYPE_NAMES or _looks_like_type(tokens, pos):
|
|
431
436
|
try:
|
|
@@ -1158,8 +1163,8 @@ def parse_statement(tokens: list[Token], pos: int):
|
|
|
1158
1163
|
if tok.type == TokenType.KEYWORD and tok.value == 'switch':
|
|
1159
1164
|
return parse_switch(tokens, pos)
|
|
1160
1165
|
|
|
1161
|
-
if tok.type == TokenType.KEYWORD and tok.value in ('while', 'for', 'class', 'struct', 'import'):
|
|
1162
|
-
handler = {'while': parse_while, 'for': parse_for, 'class': parse_class, 'struct': parse_struct_def, 'import': parse_import}[tok.value]
|
|
1166
|
+
if tok.type == TokenType.KEYWORD and tok.value in ('while', 'for', 'class', 'struct', 'import', 'enum', 'type'):
|
|
1167
|
+
handler = {'while': parse_while, 'for': parse_for, 'class': parse_class, 'struct': parse_struct_def, 'import': parse_import, 'enum': parse_enum, 'type': parse_type_alias}[tok.value]
|
|
1163
1168
|
return handler(tokens, pos)
|
|
1164
1169
|
|
|
1165
1170
|
if tok.type == TokenType.KEYWORD and tok.value == 'try':
|
|
@@ -1268,8 +1273,26 @@ class Switch:
|
|
|
1268
1273
|
self._token = token
|
|
1269
1274
|
def __repr__(self):
|
|
1270
1275
|
return f'Switch({self.value}, {self.cases})'
|
|
1276
|
+
|
|
1277
|
+
|
|
1278
|
+
class EnumDef:
|
|
1279
|
+
__slots__ = ('name', 'members', '_token')
|
|
1280
|
+
def __init__(self, name: str, members: list, token=None):
|
|
1281
|
+
self.name = name
|
|
1282
|
+
self.members = members
|
|
1283
|
+
self._token = token
|
|
1271
1284
|
def __repr__(self):
|
|
1272
|
-
return f'
|
|
1285
|
+
return f'EnumDef({self.name}, {self.members})'
|
|
1286
|
+
|
|
1287
|
+
|
|
1288
|
+
class TypeAlias:
|
|
1289
|
+
__slots__ = ('name', 'target_type', '_token')
|
|
1290
|
+
def __init__(self, name: str, target_type, token=None):
|
|
1291
|
+
self.name = name
|
|
1292
|
+
self.target_type = target_type
|
|
1293
|
+
self._token = token
|
|
1294
|
+
def __repr__(self):
|
|
1295
|
+
return f'TypeAlias({self.name}, {self.target_type})'
|
|
1273
1296
|
|
|
1274
1297
|
|
|
1275
1298
|
def parse_switch(tokens: list[Token], pos: int):
|
|
@@ -1348,6 +1371,61 @@ def parse_generic_params(tokens: list[Token], pos: int):
|
|
|
1348
1371
|
return params, pos + 1
|
|
1349
1372
|
|
|
1350
1373
|
|
|
1374
|
+
def parse_enum(tokens: list[Token], pos: int):
|
|
1375
|
+
tok = tokens[pos]
|
|
1376
|
+
pos += 1
|
|
1377
|
+
if pos >= len(tokens) or tokens[pos].type != TokenType.IDENTIFIER:
|
|
1378
|
+
raise ParseError('Expected enum name', tokens[pos] if pos < len(tokens) else None)
|
|
1379
|
+
name = tokens[pos].value
|
|
1380
|
+
pos += 1
|
|
1381
|
+
if pos >= len(tokens) or tokens[pos].type != TokenType.COLON:
|
|
1382
|
+
raise ParseError('Expected ":" after enum name', tokens[pos] if pos < len(tokens) else None)
|
|
1383
|
+
pos += 1
|
|
1384
|
+
if pos >= len(tokens) or tokens[pos].type != TokenType.NEWLINE:
|
|
1385
|
+
raise ParseError('Expected newline after ":"', tokens[pos] if pos < len(tokens) else None)
|
|
1386
|
+
pos += 1
|
|
1387
|
+
if pos >= len(tokens) or tokens[pos].type != TokenType.INDENT:
|
|
1388
|
+
raise ParseError('Expected indented block', tokens[pos] if pos < len(tokens) else None)
|
|
1389
|
+
pos += 1
|
|
1390
|
+
|
|
1391
|
+
members = []
|
|
1392
|
+
while pos < len(tokens) and tokens[pos].type not in (TokenType.DEDENT, TokenType.EOF):
|
|
1393
|
+
if tokens[pos].type != TokenType.IDENTIFIER:
|
|
1394
|
+
raise ParseError('Expected enum member name', tokens[pos])
|
|
1395
|
+
member_name = tokens[pos].value
|
|
1396
|
+
pos += 1
|
|
1397
|
+
member_value = None
|
|
1398
|
+
if pos < len(tokens) and tokens[pos].type == TokenType.KEYWORD and tokens[pos].value == '=':
|
|
1399
|
+
pos += 1
|
|
1400
|
+
member_value, pos = parse_expression(tokens, pos)
|
|
1401
|
+
members.append({'name': member_name, 'value': member_value, '_token': tokens[pos - 1]})
|
|
1402
|
+
if pos < len(tokens) and tokens[pos].type == TokenType.NEWLINE:
|
|
1403
|
+
pos += 1
|
|
1404
|
+
|
|
1405
|
+
if pos < len(tokens) and tokens[pos].type == TokenType.DEDENT:
|
|
1406
|
+
pos += 1
|
|
1407
|
+
|
|
1408
|
+
return EnumDef(name, members, token=tok), pos
|
|
1409
|
+
|
|
1410
|
+
|
|
1411
|
+
def parse_type_alias(tokens: list[Token], pos: int):
|
|
1412
|
+
tok = tokens[pos]
|
|
1413
|
+
pos += 1
|
|
1414
|
+
if pos >= len(tokens) or tokens[pos].type != TokenType.IDENTIFIER:
|
|
1415
|
+
raise ParseError('Expected type alias name', tokens[pos] if pos < len(tokens) else None)
|
|
1416
|
+
name = tokens[pos].value
|
|
1417
|
+
pos += 1
|
|
1418
|
+
if pos >= len(tokens) or tokens[pos].type != TokenType.EQUAL:
|
|
1419
|
+
raise ParseError('Expected "="', tokens[pos] if pos < len(tokens) else None)
|
|
1420
|
+
pos += 1
|
|
1421
|
+
target_type, pos = parse_type(tokens, pos)
|
|
1422
|
+
target_type_str = _type_to_str(target_type) if isinstance(target_type, tuple) else target_type
|
|
1423
|
+
_expect_newline(tokens, pos, tok)
|
|
1424
|
+
if pos < len(tokens) and tokens[pos].type == TokenType.NEWLINE:
|
|
1425
|
+
pos += 1
|
|
1426
|
+
return TypeAlias(name, target_type_str, token=tok), pos
|
|
1427
|
+
|
|
1428
|
+
|
|
1351
1429
|
def parse_def(tokens: list[Token], pos: int):
|
|
1352
1430
|
tok = tokens[pos]
|
|
1353
1431
|
pos += 1
|
|
@@ -107,6 +107,15 @@ class LLVM:
|
|
|
107
107
|
|
|
108
108
|
def __init__(self, no_userspace=False, enable_extensions=True):
|
|
109
109
|
self.module = ir.Module("main")
|
|
110
|
+
try:
|
|
111
|
+
import llvmlite.binding as _binding
|
|
112
|
+
_binding.initialize_native_target()
|
|
113
|
+
_target = _binding.Target.from_default_triple()
|
|
114
|
+
_tm = _target.create_target_machine()
|
|
115
|
+
self.module.triple = _tm.triple
|
|
116
|
+
self.module.data_layout = _tm.target_data.get_data_layout()
|
|
117
|
+
except Exception:
|
|
118
|
+
pass
|
|
110
119
|
self.builder = None
|
|
111
120
|
self.functions = {}
|
|
112
121
|
self.global_vars = {}
|
|
@@ -122,6 +131,9 @@ class LLVM:
|
|
|
122
131
|
self.loop_stack = []
|
|
123
132
|
self._malloc_fn = None
|
|
124
133
|
self._free_fn = None
|
|
134
|
+
self._gc_alloc_fn = None
|
|
135
|
+
self._gc_mark_fn = None
|
|
136
|
+
self._gc_root_register_fn = None
|
|
125
137
|
self._strlen_fn = None
|
|
126
138
|
self._memcpy_fn = None
|
|
127
139
|
self.no_userspace = no_userspace
|
|
@@ -171,6 +183,24 @@ class LLVM:
|
|
|
171
183
|
fn = ir.Function(self.module, ir.FunctionType(ret, args), name=name)
|
|
172
184
|
self.functions[name] = fn
|
|
173
185
|
|
|
186
|
+
# Boehm GC runtime functions
|
|
187
|
+
gc_fns = [
|
|
188
|
+
('GC_init', _void, []),
|
|
189
|
+
('GC_malloc', _i8ptr, [_i64]),
|
|
190
|
+
('GC_realloc', _i8ptr, [_i8ptr, _i64]),
|
|
191
|
+
('GC_free', _void, [_i8ptr]),
|
|
192
|
+
('GC_gcollect', _void, []),
|
|
193
|
+
('GC_get_heap_size', _i64, []),
|
|
194
|
+
('GC_get_free_bytes', _i64, []),
|
|
195
|
+
('GC_disable', _void, []),
|
|
196
|
+
('GC_enable', _void, []),
|
|
197
|
+
('GC_add_roots', _void, [_i8ptr, _i8ptr]),
|
|
198
|
+
('GC_remove_roots', _void, [_i8ptr, _i8ptr]),
|
|
199
|
+
]
|
|
200
|
+
for name, ret, args in gc_fns:
|
|
201
|
+
fn = ir.Function(self.module, ir.FunctionType(ret, args), name=name)
|
|
202
|
+
self.functions[name] = fn
|
|
203
|
+
|
|
174
204
|
# Exception handling globals
|
|
175
205
|
self._exc_buf_ptr = ir.GlobalVariable(self.module, _i8ptr, "_exc_buf_ptr")
|
|
176
206
|
self._exc_buf_ptr.initializer = ir.Constant(_i8ptr, None)
|
|
@@ -413,6 +443,10 @@ class LLVM:
|
|
|
413
443
|
|
|
414
444
|
if wrapper_builder is not None:
|
|
415
445
|
self.builder = wrapper_builder
|
|
446
|
+
# Initialize Boehm GC at program start
|
|
447
|
+
gc_init_fn = self.functions.get('GC_init')
|
|
448
|
+
if gc_init_fn:
|
|
449
|
+
self.builder.call(gc_init_fn, [])
|
|
416
450
|
for node in toplevel:
|
|
417
451
|
self.emit(node)
|
|
418
452
|
self.builder.ret(ir.Constant(ir.IntType(32), 0))
|
|
@@ -547,19 +581,21 @@ class LLVM:
|
|
|
547
581
|
return ptr
|
|
548
582
|
|
|
549
583
|
def _get_malloc_fn(self):
|
|
550
|
-
|
|
584
|
+
# Use Boehm GC allocator instead of malloc
|
|
585
|
+
fn = self._gc_alloc_fn
|
|
551
586
|
if fn is not None:
|
|
552
587
|
return fn
|
|
553
588
|
for f in self.module.functions:
|
|
554
|
-
if f.name == '
|
|
555
|
-
self.
|
|
589
|
+
if f.name == 'GC_malloc':
|
|
590
|
+
self._gc_alloc_fn = f
|
|
556
591
|
return f
|
|
557
592
|
fnty = ir.FunctionType(_i8ptr, [_i64])
|
|
558
|
-
fn = ir.Function(self.module, fnty, '
|
|
559
|
-
self.
|
|
593
|
+
fn = ir.Function(self.module, fnty, 'GC_malloc')
|
|
594
|
+
self._gc_alloc_fn = fn
|
|
560
595
|
return fn
|
|
561
596
|
|
|
562
597
|
def _get_free_fn(self):
|
|
598
|
+
# Boehm GC manages memory - free is a no-op but keep for compatibility
|
|
563
599
|
fn = self._free_fn
|
|
564
600
|
if fn is not None:
|
|
565
601
|
return fn
|
|
@@ -674,6 +710,9 @@ class LLVM:
|
|
|
674
710
|
return self.builder.gep(obj, [idx], inbounds=True)
|
|
675
711
|
|
|
676
712
|
def emit_attr(self, node: Attr):
|
|
713
|
+
enum_val = getattr(node, '_enum_member_value', None)
|
|
714
|
+
if enum_val is not None:
|
|
715
|
+
return ir.Constant(ir.IntType(32), enum_val)
|
|
677
716
|
ptr = self._emit_lvalue_attr(node)
|
|
678
717
|
return self.builder.load(ptr)
|
|
679
718
|
|
|
@@ -801,6 +840,12 @@ class LLVM:
|
|
|
801
840
|
self.builder = ir.IRBuilder(entry)
|
|
802
841
|
self.builder.position_at_end(entry)
|
|
803
842
|
|
|
843
|
+
# Initialize Boehm GC in main function
|
|
844
|
+
if node.name == 'main':
|
|
845
|
+
gc_init_fn = self.functions.get('GC_init')
|
|
846
|
+
if gc_init_fn:
|
|
847
|
+
self.builder.call(gc_init_fn, [])
|
|
848
|
+
|
|
804
849
|
old_locals = self.locals
|
|
805
850
|
old_local_types = self.local_types
|
|
806
851
|
self.locals = {}
|
|
@@ -1067,6 +1112,18 @@ class LLVM:
|
|
|
1067
1112
|
cmp = self.builder.call(self.functions['bigint_cmp'], [left, right])
|
|
1068
1113
|
return self.builder.icmp_signed('>=', cmp, ir.Constant(_i64, 0))
|
|
1069
1114
|
|
|
1115
|
+
# String equality via strcmp (content, not pointer comparison)
|
|
1116
|
+
if node.op in (TokenType.EQ_EQ, TokenType.NOT_EQ):
|
|
1117
|
+
left_str = getattr(node.left, 'inferred_type', None) == 'str'
|
|
1118
|
+
right_str = getattr(node.right, 'inferred_type', None) == 'str'
|
|
1119
|
+
if left_str and right_str:
|
|
1120
|
+
cmp = self.builder.call(self._strcmp_fn, [left, right])
|
|
1121
|
+
zero = ir.Constant(_i32, 0)
|
|
1122
|
+
if node.op == TokenType.EQ_EQ:
|
|
1123
|
+
return self.builder.icmp_signed('==', cmp, zero)
|
|
1124
|
+
else:
|
|
1125
|
+
return self.builder.icmp_signed('!=', cmp, zero)
|
|
1126
|
+
|
|
1070
1127
|
left, right = self._promote(left, right)
|
|
1071
1128
|
is_float = isinstance(left.type, ir.DoubleType) or isinstance(right.type, ir.DoubleType)
|
|
1072
1129
|
if is_float:
|
|
@@ -1096,20 +1153,6 @@ class LLVM:
|
|
|
1096
1153
|
case TokenType.NOT_EQ:
|
|
1097
1154
|
return self.builder.fcmp_ordered('!=', left, right)
|
|
1098
1155
|
|
|
1099
|
-
# String equality via strcmp (content, not pointer comparison)
|
|
1100
|
-
if node.op in (TokenType.EQ_EQ, TokenType.NOT_EQ):
|
|
1101
|
-
left_str = getattr(node.left, 'inferred_type', None) == 'str'
|
|
1102
|
-
right_str = getattr(node.right, 'inferred_type', None) == 'str'
|
|
1103
|
-
if left_str and right_str:
|
|
1104
|
-
fnty = ir.FunctionType(_i32, [_i8ptr, _i8ptr])
|
|
1105
|
-
fn = ir.Function(self.module, fnty, '__cpy_strcmp')
|
|
1106
|
-
cmp = self.builder.call(fn, [left, right])
|
|
1107
|
-
zero = ir.Constant(_i32, 0)
|
|
1108
|
-
if node.op == TokenType.EQ_EQ:
|
|
1109
|
-
return self.builder.icmp_signed('==', cmp, zero)
|
|
1110
|
-
else:
|
|
1111
|
-
return self.builder.icmp_signed('!=', cmp, zero)
|
|
1112
|
-
|
|
1113
1156
|
match node.op:
|
|
1114
1157
|
case TokenType.PLUS:
|
|
1115
1158
|
return self.builder.add(left, right)
|
|
@@ -82,13 +82,15 @@ def optimize(mod, opt_level=3):
|
|
|
82
82
|
npm.run(mod, pb)
|
|
83
83
|
|
|
84
84
|
|
|
85
|
-
def _compile_sources_object(src_files, target_triple=None):
|
|
85
|
+
def _compile_sources_object(src_files, target_triple=None, pic=False):
|
|
86
86
|
objs = []
|
|
87
87
|
for src in src_files:
|
|
88
88
|
obj = src.rsplit('.', 1)[0] + '.o'
|
|
89
89
|
cmd = ['clang', '-c', '-O3', '-o', obj, src]
|
|
90
90
|
if target_triple:
|
|
91
91
|
cmd.extend(['-target', target_triple])
|
|
92
|
+
if pic:
|
|
93
|
+
cmd.append('-fPIC')
|
|
92
94
|
r = subprocess.run(cmd, capture_output=True, text=True)
|
|
93
95
|
if r.returncode != 0:
|
|
94
96
|
print(f'error compiling {src}: {r.stderr}', file=__import__('sys').stderr)
|
|
@@ -110,7 +112,7 @@ def _maybe_compile(module):
|
|
|
110
112
|
return prog, src_files
|
|
111
113
|
return module, None
|
|
112
114
|
|
|
113
|
-
def run_jit(module, opt_level=3, src_files=None, no_userspace=False):
|
|
115
|
+
def run_jit(module, opt_level=3, src_files=None, no_userspace=False, pic=False):
|
|
114
116
|
module, src_files_auto = _maybe_compile(module)
|
|
115
117
|
if src_files_auto is not None:
|
|
116
118
|
src_files = src_files_auto
|
|
@@ -128,16 +130,15 @@ def run_jit(module, opt_level=3, src_files=None, no_userspace=False):
|
|
|
128
130
|
with open(src) as f:
|
|
129
131
|
src_ir = f.read()
|
|
130
132
|
else:
|
|
131
|
-
ll = src.rsplit('.', 1)[0] + '.ll'
|
|
132
133
|
r = subprocess.run(
|
|
133
134
|
['clang', '-S', '-emit-llvm', '-O0', '-target', target.triple,
|
|
134
|
-
'-fno-stack-protector', '-o',
|
|
135
|
-
capture_output=True, text=True
|
|
136
|
-
|
|
135
|
+
'-fno-stack-protector', '-o', '-', src],
|
|
136
|
+
capture_output=True, text=True)
|
|
137
|
+
|
|
137
138
|
if r.returncode != 0:
|
|
138
139
|
print(f'error compiling {src}: {r.stderr}', file=__import__('sys').stderr)
|
|
139
140
|
raise SystemExit(1)
|
|
140
|
-
src_ir =
|
|
141
|
+
src_ir = r.stdout
|
|
141
142
|
src_ir = _remove_probe_stack_ir(src_ir)
|
|
142
143
|
src_mod = binding.parse_assembly(src_ir)
|
|
143
144
|
binding.link_modules(mod, src_mod)
|
|
@@ -228,10 +229,20 @@ def run_jit(module, opt_level=3, src_files=None, no_userspace=False):
|
|
|
228
229
|
pass
|
|
229
230
|
|
|
230
231
|
_map_libc_fn(engine, mod, 'malloc', ctypes.c_size_t, ctypes.c_void_p)
|
|
232
|
+
_map_libc_fn(engine, mod, 'free', None, None, argtypes=[ctypes.c_void_p])
|
|
231
233
|
_map_libc_fn(engine, mod, 'strlen', ctypes.c_char_p, ctypes.c_int)
|
|
232
234
|
_map_libc_fn(engine, mod, 'memcpy', None, ctypes.c_void_p,
|
|
233
235
|
argtypes=[ctypes.c_void_p, ctypes.c_void_p, ctypes.c_int])
|
|
234
236
|
|
|
237
|
+
try:
|
|
238
|
+
fn = mod.get_function('strcmp')
|
|
239
|
+
cfunctype = ctypes.CFUNCTYPE(ctypes.c_int, ctypes.c_char_p, ctypes.c_char_p)
|
|
240
|
+
cfn = cfunctype(_libc.strcmp)
|
|
241
|
+
_callbacks.append(cfn)
|
|
242
|
+
engine.add_global_mapping(fn, ctypes.cast(cfn, ctypes.c_void_p).value)
|
|
243
|
+
except NameError:
|
|
244
|
+
pass
|
|
245
|
+
|
|
235
246
|
try:
|
|
236
247
|
fn = mod.get_function('__cpy_strcmp')
|
|
237
248
|
cfunctype = ctypes.CFUNCTYPE(ctypes.c_int, ctypes.c_char_p, ctypes.c_char_p)
|
|
@@ -241,6 +252,49 @@ def run_jit(module, opt_level=3, src_files=None, no_userspace=False):
|
|
|
241
252
|
except NameError:
|
|
242
253
|
pass
|
|
243
254
|
|
|
255
|
+
# Map Boehm GC functions
|
|
256
|
+
# NOTE: Boehm GC cannot safely scan JIT'd code stacks on ARM64 macOS
|
|
257
|
+
# because JIT'd frames lack DWARF unwind info. For JIT, GC_malloc
|
|
258
|
+
# falls back to malloc (no collection needed - process exits immediately).
|
|
259
|
+
# For AOT, Boehm GC works normally via libgc linkage (-lgc).
|
|
260
|
+
try:
|
|
261
|
+
_libgc = ctypes.CDLL('/opt/homebrew/lib/libgc.dylib')
|
|
262
|
+
_gc_fns = {
|
|
263
|
+
'GC_init': (None, []),
|
|
264
|
+
'GC_realloc': (ctypes.c_void_p, [ctypes.c_void_p, ctypes.c_size_t]),
|
|
265
|
+
'GC_free': (None, [ctypes.c_void_p]),
|
|
266
|
+
'GC_gcollect': (None, []),
|
|
267
|
+
'GC_get_heap_size': (ctypes.c_size_t, []),
|
|
268
|
+
'GC_get_free_bytes': (ctypes.c_size_t, []),
|
|
269
|
+
'GC_disable': (None, []),
|
|
270
|
+
'GC_enable': (None, []),
|
|
271
|
+
'GC_add_roots': (None, [ctypes.c_void_p, ctypes.c_void_p]),
|
|
272
|
+
'GC_remove_roots': (None, [ctypes.c_void_p, ctypes.c_void_p]),
|
|
273
|
+
}
|
|
274
|
+
for gc_name, (gc_restype, gc_argtypes) in _gc_fns.items():
|
|
275
|
+
try:
|
|
276
|
+
fn = mod.get_function(gc_name)
|
|
277
|
+
if gc_argtypes:
|
|
278
|
+
cfunctype = ctypes.CFUNCTYPE(gc_restype, *gc_argtypes)
|
|
279
|
+
else:
|
|
280
|
+
cfunctype = ctypes.CFUNCTYPE(gc_restype)
|
|
281
|
+
cfn = cfunctype(getattr(_libgc, gc_name))
|
|
282
|
+
_callbacks.append(cfn)
|
|
283
|
+
engine.add_global_mapping(fn, ctypes.cast(cfn, ctypes.c_void_p).value)
|
|
284
|
+
except (NameError, AttributeError):
|
|
285
|
+
pass
|
|
286
|
+
# For JIT: map GC_malloc to malloc (avoids Boehm stack scanning crashes)
|
|
287
|
+
try:
|
|
288
|
+
fn = mod.get_function('GC_malloc')
|
|
289
|
+
cfunctype = ctypes.CFUNCTYPE(ctypes.c_void_p, ctypes.c_size_t)
|
|
290
|
+
cfn = cfunctype(_libc.malloc)
|
|
291
|
+
_callbacks.append(cfn)
|
|
292
|
+
engine.add_global_mapping(fn, ctypes.cast(cfn, ctypes.c_void_p).value)
|
|
293
|
+
except (NameError, AttributeError):
|
|
294
|
+
pass
|
|
295
|
+
except OSError:
|
|
296
|
+
pass
|
|
297
|
+
|
|
244
298
|
engine.finalize_object()
|
|
245
299
|
engine.run_static_constructors()
|
|
246
300
|
|
|
@@ -271,7 +325,7 @@ def _map_libc_fn(engine, mod, name, argtype, restype, argtypes=None):
|
|
|
271
325
|
pass
|
|
272
326
|
|
|
273
327
|
|
|
274
|
-
def run_aot(module, output="program.o", opt_level=3, src_files=None, no_userspace=False):
|
|
328
|
+
def run_aot(module, output="program.o", opt_level=3, src_files=None, no_userspace=False, pic=False):
|
|
275
329
|
llvm_ir = str(module)
|
|
276
330
|
import llvmlite.binding as binding
|
|
277
331
|
binding.initialize_native_target()
|
|
@@ -285,7 +339,10 @@ def run_aot(module, output="program.o", opt_level=3, src_files=None, no_userspac
|
|
|
285
339
|
mod.verify()
|
|
286
340
|
|
|
287
341
|
target = binding.Target.from_default_triple()
|
|
288
|
-
|
|
342
|
+
if pic:
|
|
343
|
+
target_machine = target.create_target_machine(reloc='pic')
|
|
344
|
+
else:
|
|
345
|
+
target_machine = target.create_target_machine()
|
|
289
346
|
|
|
290
347
|
obj = target_machine.emit_object(mod)
|
|
291
348
|
|
|
@@ -295,8 +352,11 @@ def run_aot(module, output="program.o", opt_level=3, src_files=None, no_userspac
|
|
|
295
352
|
objs = [output]
|
|
296
353
|
for src in (src_files or []):
|
|
297
354
|
src_obj = src.rsplit('.', 1)[0] + '.o'
|
|
355
|
+
cmd = ['clang', '-c', '-O3', '-o', src_obj, src]
|
|
356
|
+
if pic:
|
|
357
|
+
cmd.append('-fPIC')
|
|
298
358
|
r = subprocess.run(
|
|
299
|
-
|
|
359
|
+
cmd,
|
|
300
360
|
capture_output=True, text=True
|
|
301
361
|
)
|
|
302
362
|
if r.returncode != 0:
|
|
@@ -306,8 +366,11 @@ def run_aot(module, output="program.o", opt_level=3, src_files=None, no_userspac
|
|
|
306
366
|
|
|
307
367
|
if not no_userspace:
|
|
308
368
|
runtime_obj = output + '.runtime.o'
|
|
369
|
+
cmd = ['clang', '-c', '-O3', '-o', runtime_obj, _RUNTIME_C]
|
|
370
|
+
if pic:
|
|
371
|
+
cmd.append('-fPIC')
|
|
309
372
|
r = subprocess.run(
|
|
310
|
-
|
|
373
|
+
cmd,
|
|
311
374
|
capture_output=True, text=True
|
|
312
375
|
)
|
|
313
376
|
if r.returncode == 0:
|
|
@@ -315,7 +378,8 @@ def run_aot(module, output="program.o", opt_level=3, src_files=None, no_userspac
|
|
|
315
378
|
|
|
316
379
|
out_name = output.rsplit('.', 1)[0] if '.' in output else output
|
|
317
380
|
r = subprocess.run(
|
|
318
|
-
['clang', '-O3', '-o', out_name] + objs + ['-lm'
|
|
381
|
+
['clang', '-O3', '-o', out_name] + objs + ['-lgc', '-lm',
|
|
382
|
+
'-L/opt/homebrew/opt/bdw-gc/lib', '-I/opt/homebrew/opt/bdw-gc/include'],
|
|
319
383
|
capture_output=True, text=True
|
|
320
384
|
)
|
|
321
385
|
if r.returncode != 0:
|
|
@@ -209,8 +209,13 @@ class HookRegistry:
|
|
|
209
209
|
"""Get the current compiler context."""
|
|
210
210
|
return self._context.copy()
|
|
211
211
|
|
|
212
|
+
def _is_duplicate(self, existing_list, package_name, hook_path) -> bool:
|
|
213
|
+
return any(r.package_name == package_name and r.hook_path == hook_path for r in existing_list)
|
|
214
|
+
|
|
212
215
|
def register_lexer_hook(self, hook: LexerHook, priority: int = 0) -> None:
|
|
213
216
|
"""Register a lexer hook."""
|
|
217
|
+
if self._is_duplicate(self._lexer_hooks, hook.package_name, hook.hook_path):
|
|
218
|
+
return
|
|
214
219
|
registration = HookRegistration(
|
|
215
220
|
hook=hook,
|
|
216
221
|
priority=priority,
|
|
@@ -219,9 +224,11 @@ class HookRegistry:
|
|
|
219
224
|
)
|
|
220
225
|
self._lexer_hooks.append(registration)
|
|
221
226
|
self._lexer_hooks.sort(key=lambda r: r.priority, reverse=True)
|
|
222
|
-
|
|
227
|
+
|
|
223
228
|
def register_parser_hook(self, hook: ParserHook, priority: int = 0) -> None:
|
|
224
229
|
"""Register a parser hook."""
|
|
230
|
+
if self._is_duplicate(self._parser_hooks, hook.package_name, hook.hook_path):
|
|
231
|
+
return
|
|
225
232
|
registration = HookRegistration(
|
|
226
233
|
hook=hook,
|
|
227
234
|
priority=priority,
|
|
@@ -230,9 +237,11 @@ class HookRegistry:
|
|
|
230
237
|
)
|
|
231
238
|
self._parser_hooks.append(registration)
|
|
232
239
|
self._parser_hooks.sort(key=lambda r: r.priority, reverse=True)
|
|
233
|
-
|
|
240
|
+
|
|
234
241
|
def register_semantic_hook(self, hook: SemanticHook, priority: int = 0) -> None:
|
|
235
242
|
"""Register a semantic analysis hook."""
|
|
243
|
+
if self._is_duplicate(self._semantic_hooks, hook.package_name, hook.hook_path):
|
|
244
|
+
return
|
|
236
245
|
registration = HookRegistration(
|
|
237
246
|
hook=hook,
|
|
238
247
|
priority=priority,
|
|
@@ -244,6 +253,8 @@ class HookRegistry:
|
|
|
244
253
|
|
|
245
254
|
def register_codegen_hook(self, hook: CodegenHook, priority: int = 0) -> None:
|
|
246
255
|
"""Register a code generation hook."""
|
|
256
|
+
if self._is_duplicate(self._codegen_hooks, hook.package_name, hook.hook_path):
|
|
257
|
+
return
|
|
247
258
|
registration = HookRegistration(
|
|
248
259
|
hook=hook,
|
|
249
260
|
priority=priority,
|
|
@@ -252,9 +263,11 @@ class HookRegistry:
|
|
|
252
263
|
)
|
|
253
264
|
self._codegen_hooks.append(registration)
|
|
254
265
|
self._codegen_hooks.sort(key=lambda r: r.priority, reverse=True)
|
|
255
|
-
|
|
266
|
+
|
|
256
267
|
def register_runtime_hook(self, hook: RuntimeHook, priority: int = 0) -> None:
|
|
257
268
|
"""Register a runtime hook."""
|
|
269
|
+
if self._is_duplicate(self._runtime_hooks, hook.package_name, hook.hook_path):
|
|
270
|
+
return
|
|
258
271
|
registration = HookRegistration(
|
|
259
272
|
hook=hook,
|
|
260
273
|
priority=priority,
|
|
@@ -35,13 +35,15 @@ class Linker:
|
|
|
35
35
|
def cc(self):
|
|
36
36
|
return self._cc
|
|
37
37
|
|
|
38
|
-
def compile_c(self, src, output=None, opt_level=3, debug=False):
|
|
38
|
+
def compile_c(self, src, output=None, opt_level=3, debug=False, pic=False):
|
|
39
39
|
if output is None:
|
|
40
40
|
base = src.rsplit('.', 1)[0] if '.' in src else src
|
|
41
41
|
output = base + '.o'
|
|
42
42
|
cmd = [self._cc, '-c']
|
|
43
43
|
if debug:
|
|
44
44
|
cmd.append('-g')
|
|
45
|
+
if pic:
|
|
46
|
+
cmd.append('-fPIC')
|
|
45
47
|
if opt_level is not None:
|
|
46
48
|
cmd.append(f'-O{opt_level}')
|
|
47
49
|
cmd.extend(['-o', output, src])
|
|
@@ -52,10 +54,12 @@ class Linker:
|
|
|
52
54
|
return output
|
|
53
55
|
|
|
54
56
|
def link(self, objects, output, libraries=None, library_paths=None,
|
|
55
|
-
shared=False, debug=False, opt_level=3, frameworks=None):
|
|
57
|
+
shared=False, debug=False, opt_level=3, frameworks=None, pic=False):
|
|
56
58
|
cmd = [self._cc]
|
|
57
59
|
if shared:
|
|
58
60
|
cmd.append('-shared')
|
|
61
|
+
if pic:
|
|
62
|
+
cmd.append('-fPIC')
|
|
59
63
|
if debug:
|
|
60
64
|
cmd.append('-g')
|
|
61
65
|
if opt_level is not None:
|
|
@@ -77,13 +81,13 @@ class Linker:
|
|
|
77
81
|
|
|
78
82
|
|
|
79
83
|
def build(objects, output=None, libraries=None, library_paths=None,
|
|
80
|
-
shared=False, debug=False, opt_level=3, cc=None, frameworks=None):
|
|
84
|
+
shared=False, debug=False, opt_level=3, cc=None, frameworks=None, pic=False):
|
|
81
85
|
linker = Linker(cc)
|
|
82
86
|
|
|
83
87
|
final_objects = []
|
|
84
88
|
for src in objects:
|
|
85
89
|
if src.endswith('.c'):
|
|
86
|
-
obj = linker.compile_c(src, opt_level=opt_level, debug=debug)
|
|
90
|
+
obj = linker.compile_c(src, opt_level=opt_level, debug=debug, pic=pic)
|
|
87
91
|
final_objects.append(obj)
|
|
88
92
|
else:
|
|
89
93
|
final_objects.append(src)
|
|
@@ -100,5 +104,5 @@ def build(objects, output=None, libraries=None, library_paths=None,
|
|
|
100
104
|
libraries=libraries, library_paths=library_paths,
|
|
101
105
|
shared=shared,
|
|
102
106
|
debug=debug, opt_level=opt_level,
|
|
103
|
-
frameworks=frameworks,
|
|
107
|
+
frameworks=frameworks, pic=pic,
|
|
104
108
|
)
|
|
@@ -283,9 +283,9 @@ def _collect_frameworks(nodes):
|
|
|
283
283
|
return list(set(frameworks))
|
|
284
284
|
|
|
285
285
|
|
|
286
|
-
def cmd_build(args, tab_size=4, strict=False, no_userspace=False):
|
|
286
|
+
def cmd_build(args, tab_size=4, strict=False, no_userspace=False, pic=False):
|
|
287
287
|
if not args:
|
|
288
|
-
print('Usage: cpy build [--output O] [--debug] [--opt N] [--no-userspace] <source.cpy>', file=sys.stderr)
|
|
288
|
+
print('Usage: cpy build [--output O] [--debug] [--opt N] [--no-userspace] [--pic] <source.cpy>', file=sys.stderr)
|
|
289
289
|
sys.exit(1)
|
|
290
290
|
|
|
291
291
|
output = None
|
|
@@ -308,6 +308,9 @@ def cmd_build(args, tab_size=4, strict=False, no_userspace=False):
|
|
|
308
308
|
elif a == '--no-userspace':
|
|
309
309
|
no_userspace = True
|
|
310
310
|
i += 1
|
|
311
|
+
elif a == '--pic':
|
|
312
|
+
pic = True
|
|
313
|
+
i += 1
|
|
311
314
|
elif a == '--opt' and i + 1 < len(args):
|
|
312
315
|
opt = int(args[i + 1])
|
|
313
316
|
i += 2
|
|
@@ -319,7 +322,7 @@ def cmd_build(args, tab_size=4, strict=False, no_userspace=False):
|
|
|
319
322
|
sys.exit(1)
|
|
320
323
|
|
|
321
324
|
if not src_file:
|
|
322
|
-
print('Usage: cpy build [--output O] [--debug] [--opt N] <source.cpy>', file=sys.stderr)
|
|
325
|
+
print('Usage: cpy build [--output O] [--debug] [--opt N] [--pic] <source.cpy>', file=sys.stderr)
|
|
323
326
|
sys.exit(1)
|
|
324
327
|
|
|
325
328
|
with open(src_file) as f:
|
|
@@ -345,7 +348,10 @@ def cmd_build(args, tab_size=4, strict=False, no_userspace=False):
|
|
|
345
348
|
mod.verify()
|
|
346
349
|
|
|
347
350
|
target = binding.Target.from_default_triple()
|
|
348
|
-
|
|
351
|
+
if pic:
|
|
352
|
+
target_machine = target.create_target_machine(reloc='pic')
|
|
353
|
+
else:
|
|
354
|
+
target_machine = target.create_target_machine()
|
|
349
355
|
obj = target_machine.emit_object(mod)
|
|
350
356
|
with open(obj_file, 'wb') as f:
|
|
351
357
|
f.write(obj)
|
|
@@ -354,16 +360,17 @@ def cmd_build(args, tab_size=4, strict=False, no_userspace=False):
|
|
|
354
360
|
objs = [obj_file]
|
|
355
361
|
for src in (src_files or []):
|
|
356
362
|
src_obj = src.rsplit('.', 1)[0] + '.o'
|
|
357
|
-
l.compile_c(src, output=src_obj, opt_level=opt, debug=debug)
|
|
363
|
+
l.compile_c(src, output=src_obj, opt_level=opt, debug=debug, pic=pic)
|
|
358
364
|
objs.append(src_obj)
|
|
359
365
|
|
|
360
366
|
if not no_userspace:
|
|
361
367
|
runtime_obj = out_base + '.runtime.o'
|
|
362
|
-
l.compile_c(_RUNTIME_C, output=runtime_obj, opt_level=opt, debug=debug)
|
|
368
|
+
l.compile_c(_RUNTIME_C, output=runtime_obj, opt_level=opt, debug=debug, pic=pic)
|
|
363
369
|
objs.append(runtime_obj)
|
|
364
370
|
|
|
365
371
|
executable = output or out_base
|
|
366
|
-
l.link(objs, executable,
|
|
372
|
+
l.link(objs, executable, libraries=['gc'], library_paths=['/opt/homebrew/opt/bdw-gc/lib'],
|
|
373
|
+
opt_level=opt, debug=debug, frameworks=frameworks, pic=pic)
|
|
367
374
|
print(f'Wrote {executable}')
|
|
368
375
|
|
|
369
376
|
|
|
@@ -374,6 +381,7 @@ def main():
|
|
|
374
381
|
|
|
375
382
|
strict = False
|
|
376
383
|
no_userspace = False
|
|
384
|
+
pic = False
|
|
377
385
|
while args and args[0].startswith('--'):
|
|
378
386
|
flag = args.pop(0)
|
|
379
387
|
if flag == '--tab-size':
|
|
@@ -382,6 +390,8 @@ def main():
|
|
|
382
390
|
strict = True
|
|
383
391
|
elif flag == '--no-userspace':
|
|
384
392
|
no_userspace = True
|
|
393
|
+
elif flag == '--pic':
|
|
394
|
+
pic = True
|
|
385
395
|
elif flag == '--ast':
|
|
386
396
|
mode = 'ast'
|
|
387
397
|
elif flag == '--emit-llvm':
|
|
@@ -395,12 +405,12 @@ def main():
|
|
|
395
405
|
sys.exit(1)
|
|
396
406
|
|
|
397
407
|
if not args:
|
|
398
|
-
print('Usage: cpy [--tab-size N] [--strict] [--no-userspace] [--ast|--emit-llvm|--jit|--aot] <source file>', file=sys.stderr)
|
|
399
|
-
print(' cpy build [--output O] [--debug] [--opt N] [--no-userspace] <source.cpy>', file=sys.stderr)
|
|
408
|
+
print('Usage: cpy [--tab-size N] [--strict] [--no-userspace] [--pic] [--ast|--emit-llvm|--jit|--aot] <source file>', file=sys.stderr)
|
|
409
|
+
print(' cpy build [--output O] [--debug] [--opt N] [--no-userspace] [--pic] <source.cpy>', file=sys.stderr)
|
|
400
410
|
sys.exit(1)
|
|
401
411
|
|
|
402
412
|
if args[0] == 'build':
|
|
403
|
-
cmd_build(args[1:], tab_size=tab_size, strict=strict, no_userspace=no_userspace)
|
|
413
|
+
cmd_build(args[1:], tab_size=tab_size, strict=strict, no_userspace=no_userspace, pic=pic)
|
|
404
414
|
return
|
|
405
415
|
|
|
406
416
|
with open(args[0]) as f:
|
|
@@ -419,10 +429,10 @@ def main():
|
|
|
419
429
|
elif mode == 'aot':
|
|
420
430
|
out_base = args[0].rsplit('.', 1)[0] if '.' in args[0] else 'program'
|
|
421
431
|
obj_file = 'program.o'
|
|
422
|
-
run_aot(prog, output=obj_file, src_files=src_files, no_userspace=no_userspace)
|
|
432
|
+
run_aot(prog, output=obj_file, src_files=src_files, no_userspace=no_userspace, pic=pic)
|
|
423
433
|
print(f'Wrote {out_base}')
|
|
424
434
|
else:
|
|
425
|
-
run_jit(prog, src_files=src_files, no_userspace=no_userspace)
|
|
435
|
+
run_jit(prog, src_files=src_files, no_userspace=no_userspace, pic=pic)
|
|
426
436
|
|
|
427
437
|
|
|
428
438
|
if __name__ == '__main__':
|
|
@@ -27,6 +27,7 @@ from .astparse import (
|
|
|
27
27
|
VarDecl, Break, Continue, Switch, Import, While,
|
|
28
28
|
NewExpr, Deref, AddrOf, SizeOf, StructDef, Field, Input,
|
|
29
29
|
InputStr, Signed67, Try, Raise, ExceptHandler, InlineAsm,
|
|
30
|
+
EnumDef, TypeAlias,
|
|
30
31
|
parse_file, ParseError,
|
|
31
32
|
)
|
|
32
33
|
from .clib import resolve_library, parse_header_file, parse_c_source
|
|
@@ -360,6 +361,12 @@ class SemanticAnalyzer:
|
|
|
360
361
|
return None
|
|
361
362
|
if sym.const_value is not None:
|
|
362
363
|
node.const_value = sym.const_value
|
|
364
|
+
if sym.kind in ('enum', 'struct'):
|
|
365
|
+
node.inferred_type = node.name
|
|
366
|
+
return node.name
|
|
367
|
+
if sym.kind == 'type_alias':
|
|
368
|
+
node.inferred_type = sym.type
|
|
369
|
+
return sym.type
|
|
363
370
|
node.inferred_type = sym.type
|
|
364
371
|
return sym.type
|
|
365
372
|
|
|
@@ -581,6 +588,17 @@ class SemanticAnalyzer:
|
|
|
581
588
|
obj_t = self._infer_type(node.obj)
|
|
582
589
|
if obj_t:
|
|
583
590
|
lookup_t = obj_t[:-1] if obj_t.endswith('*') else obj_t
|
|
591
|
+
sym = self.current_scope.lookup(lookup_t)
|
|
592
|
+
if sym and sym.kind == 'enum':
|
|
593
|
+
member_sym = self.current_scope.lookup(f'{lookup_t}.{node.name}')
|
|
594
|
+
if member_sym and member_sym.kind == 'enum_member':
|
|
595
|
+
node._enum_member_value = member_sym.const_value
|
|
596
|
+
return 'int'
|
|
597
|
+
self.error(
|
|
598
|
+
f'enum `{lookup_t}` has no member `{node.name}`',
|
|
599
|
+
node
|
|
600
|
+
)
|
|
601
|
+
return None
|
|
584
602
|
struct_sym = self.current_scope.lookup(lookup_t)
|
|
585
603
|
if struct_sym and struct_sym.kind == 'struct' and struct_sym.node:
|
|
586
604
|
for field in struct_sym.node.fields:
|
|
@@ -707,6 +725,10 @@ class SemanticAnalyzer:
|
|
|
707
725
|
self._visit_import(node)
|
|
708
726
|
elif isinstance(node, StructDef):
|
|
709
727
|
self._visit_struct(node, scope)
|
|
728
|
+
elif isinstance(node, EnumDef):
|
|
729
|
+
self._visit_enum(node, scope)
|
|
730
|
+
elif isinstance(node, TypeAlias):
|
|
731
|
+
self._visit_type_alias(node, scope)
|
|
710
732
|
elif isinstance(node, Try):
|
|
711
733
|
self._visit_try(node, scope)
|
|
712
734
|
elif isinstance(node, Raise):
|
|
@@ -1160,7 +1182,8 @@ class SemanticAnalyzer:
|
|
|
1160
1182
|
self._infer_type(node.target)
|
|
1161
1183
|
|
|
1162
1184
|
def _visit_vardecl(self, node: VarDecl, scope: Scope | None = None):
|
|
1163
|
-
val_type = node.var_type
|
|
1185
|
+
val_type = self._resolve_type_alias(node.var_type)
|
|
1186
|
+
node.var_type = val_type
|
|
1164
1187
|
s = scope or self.current_scope
|
|
1165
1188
|
existing = s.lookup_local(node.name)
|
|
1166
1189
|
if existing:
|
|
@@ -1283,6 +1306,54 @@ class SemanticAnalyzer:
|
|
|
1283
1306
|
for field in node.fields:
|
|
1284
1307
|
struct_scope.define(field.name, Symbol('field', field.type_expr, node))
|
|
1285
1308
|
|
|
1309
|
+
def _visit_enum(self, node: EnumDef, scope: Scope | None = None):
|
|
1310
|
+
s = scope or self.globals
|
|
1311
|
+
existing = s.lookup_local(node.name)
|
|
1312
|
+
if existing:
|
|
1313
|
+
self.error(f'redefinition of enum `{node.name}`', node)
|
|
1314
|
+
return
|
|
1315
|
+
s.define(node.name, Symbol('enum', None, node))
|
|
1316
|
+
for i, member in enumerate(node.members):
|
|
1317
|
+
if member['value'] is not None:
|
|
1318
|
+
val = self._eval_const_expr(member['value'])
|
|
1319
|
+
member['_const_value'] = val
|
|
1320
|
+
else:
|
|
1321
|
+
member['_const_value'] = i
|
|
1322
|
+
member['_auto_index'] = i
|
|
1323
|
+
sym = Symbol('enum_member', node.name, node)
|
|
1324
|
+
sym.const_value = member['_const_value']
|
|
1325
|
+
s.define(f'{node.name}.{member["name"]}', sym)
|
|
1326
|
+
|
|
1327
|
+
def _visit_type_alias(self, node: TypeAlias, scope: Scope | None = None):
|
|
1328
|
+
s = scope or self.globals
|
|
1329
|
+
existing = s.lookup_local(node.name)
|
|
1330
|
+
if existing:
|
|
1331
|
+
self.error(f'redefinition of type alias `{node.name}`', node)
|
|
1332
|
+
return
|
|
1333
|
+
s.define(node.name, Symbol('type_alias', node.target_type, node))
|
|
1334
|
+
|
|
1335
|
+
def _eval_const_expr(self, node):
|
|
1336
|
+
if isinstance(node, Number):
|
|
1337
|
+
return node.value
|
|
1338
|
+
if isinstance(node, UnaryOp):
|
|
1339
|
+
val = self._eval_const_expr(node.operand)
|
|
1340
|
+
if node.op.name == 'MINUS':
|
|
1341
|
+
return -val
|
|
1342
|
+
if node.op.name == 'NOT':
|
|
1343
|
+
return not val
|
|
1344
|
+
if isinstance(node, BinOp):
|
|
1345
|
+
left = self._eval_const_expr(node.left)
|
|
1346
|
+
right = self._eval_const_expr(node.right)
|
|
1347
|
+
match node.op.name:
|
|
1348
|
+
case 'PLUS': return left + right
|
|
1349
|
+
case 'MINUS': return left - right
|
|
1350
|
+
case 'STAR': return left * right
|
|
1351
|
+
case 'SLASH': return left // right
|
|
1352
|
+
case 'PERCENT': return left % right
|
|
1353
|
+
if isinstance(node, Variable):
|
|
1354
|
+
return node.const_value
|
|
1355
|
+
return None
|
|
1356
|
+
|
|
1286
1357
|
def _visit_while(self, node, scope: Scope | None = None):
|
|
1287
1358
|
if isinstance(node, While):
|
|
1288
1359
|
self._infer_type(node.cond)
|
|
@@ -1314,6 +1385,14 @@ class SemanticAnalyzer:
|
|
|
1314
1385
|
self._loop_depth -= 1
|
|
1315
1386
|
self.locals = old_locals
|
|
1316
1387
|
|
|
1388
|
+
def _resolve_type_alias(self, type_name):
|
|
1389
|
+
if type_name is None:
|
|
1390
|
+
return None
|
|
1391
|
+
sym = self.current_scope.lookup(type_name)
|
|
1392
|
+
if sym and sym.kind == 'type_alias':
|
|
1393
|
+
return self._resolve_type_alias(sym.type)
|
|
1394
|
+
return type_name
|
|
1395
|
+
|
|
1317
1396
|
|
|
1318
1397
|
def analyze(source: str, nodes: list, strict: bool = False, workspace_root: str | None = None, enable_extensions: bool = True) -> str | None:
|
|
1319
1398
|
analyzer = SemanticAnalyzer(source, strict=strict, workspace_root=workspace_root, enable_extensions=enable_extensions)
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: cpyte
|
|
3
|
-
Version: 2.0
|
|
3
|
+
Version: 2.2.0
|
|
4
4
|
Summary: The Cpyte programming language compiler
|
|
5
5
|
Author: Hoang Duy Tung
|
|
6
6
|
License: MIT
|
|
@@ -81,3 +81,9 @@ python source/mainpie.py --jit examples/<filename>.cpy
|
|
|
81
81
|
Cpyte is experimental software. The compiler is continuously tested with fuzzing, and we're still discovering and fixing correctness bugs. While many programs compile and run correctly, I can't make any guarantees about correctness or stability.
|
|
82
82
|
|
|
83
83
|
If you decide to use Cpyte, always use the latest version — older versions have bugs that are pretty easy to run into.
|
|
84
|
+
|
|
85
|
+
## Memory Management
|
|
86
|
+
|
|
87
|
+
Cpyte uses **Boehm GC** (bdw-gc) for automatic garbage collection. Heap-allocated objects (via `new`) are managed automatically — no manual `free` needed.
|
|
88
|
+
|
|
89
|
+
**Note:** Boehm GC adds a small runtime overhead (~5-10%) compared to manual `malloc/free`. For performance-critical ecosystem projects or embedded use cases, this trade-off may be significant. The GC is required for safety but will slow down programs that do heavy heap allocation.
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|