codecin 5.4.2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- codecin/__init__.py +27 -0
- codecin/aot.py +229 -0
- codecin/assembler.py +548 -0
- codecin/cache.py +71 -0
- codecin/cin.py +2558 -0
- codecin/cli.py +282 -0
- codecin/config.py +58 -0
- codecin/console.py +183 -0
- codecin/cpu.py +1654 -0
- codecin/crom.py +222 -0
- codecin/debugger.py +659 -0
- codecin/disasm.py +86 -0
- codecin/errors.py +58 -0
- codecin/isa.py +457 -0
- codecin/jit.py +270 -0
- codecin/libcodecin_native.so +0 -0
- codecin/logger.py +145 -0
- codecin/memory.py +348 -0
- codecin/native/aot/aot.go +70 -0
- codecin/native/aot/build.go +219 -0
- codecin/native/aot/stub_main.go.txt +22 -0
- codecin/native/build.ps1 +33 -0
- codecin/native/build.sh +48 -0
- codecin/native/cmd/codecin/main.go +287 -0
- codecin/native/compiler/codegen.go +1938 -0
- codecin/native/compiler/compile_test.go +164 -0
- codecin/native/compiler/parser.go +1071 -0
- codecin/native/compiler/syscalls.go +86 -0
- codecin/native/compiler/tokenizer.go +336 -0
- codecin/native/compiler/types.go +213 -0
- codecin/native/engine/audio.go +152 -0
- codecin/native/engine/audio_other.go +46 -0
- codecin/native/engine/audio_windows.go +43 -0
- codecin/native/engine/canvas.go +271 -0
- codecin/native/engine/crom.go +86 -0
- codecin/native/engine/encode.go +93 -0
- codecin/native/engine/engine_test.go +198 -0
- codecin/native/engine/isa_gen.go +445 -0
- codecin/native/engine/system.go +190 -0
- codecin/native/engine/termux.go +113 -0
- codecin/native/engine/version_gen.go +7 -0
- codecin/native/engine/vm.go +1157 -0
- codecin/native/go.mod +3 -0
- codecin/native/ir/ir.go +48 -0
- codecin/native/main.go +225 -0
- codecin/native.py +336 -0
- codecin/registers.py +130 -0
- codecin/stats.py +235 -0
- codecin-5.4.2.dist-info/METADATA +985 -0
- codecin-5.4.2.dist-info/RECORD +54 -0
- codecin-5.4.2.dist-info/WHEEL +5 -0
- codecin-5.4.2.dist-info/entry_points.txt +2 -0
- codecin-5.4.2.dist-info/licenses/LICENSE +7 -0
- codecin-5.4.2.dist-info/top_level.txt +1 -0
codecin/assembler.py
ADDED
|
@@ -0,0 +1,548 @@
|
|
|
1
|
+
import os
|
|
2
|
+
import re
|
|
3
|
+
from typing import Any, Dict, List, Optional, Set, Tuple
|
|
4
|
+
|
|
5
|
+
from .console import Console
|
|
6
|
+
from .errors import AssemblerError
|
|
7
|
+
from .isa import Constants
|
|
8
|
+
from .memory import FastMemory
|
|
9
|
+
|
|
10
|
+
Operand = Tuple[Any, ...]
|
|
11
|
+
Instruction = Tuple[str, List[Operand]]
|
|
12
|
+
|
|
13
|
+
_RE_XREG = re.compile(r'^[xXrRwW]([0-9]|[12][0-9]|3[01])$')
|
|
14
|
+
_RE_VREG = re.compile(r'^[vV]([0-9]|[12][0-9]|3[01])(?:\.([0-3]))?$')
|
|
15
|
+
_RE_LABEL = re.compile(r'^[a-zA-Z_.$][a-zA-Z0-9_.$]*$')
|
|
16
|
+
_RE_EXPR_NAME = re.compile(r'\b[a-zA-Z_.$][a-zA-Z0-9_.$]*\b')
|
|
17
|
+
_RE_EQU_DIRECTIVE = {'EQU', 'SET'}
|
|
18
|
+
# 表达式只允许: 数字 / 已定义符号 / + - * / % ( ) 空白
|
|
19
|
+
_RE_EXPR_OK = re.compile(r'^[0-9+\-*/%()\s]+$')
|
|
20
|
+
# 表达式内的数值字面量 (十进制 / 0x / 0b / 0o, 允许下划线)
|
|
21
|
+
_RE_EXPR_NUM = re.compile(
|
|
22
|
+
r'0[xX][0-9a-fA-F_]+|0[bB][01_]+|0[oO][0-7_]+|\d[\d_]*')
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def _eval_expr(text: str, symbols: Dict[str, int]) -> Optional[int]:
|
|
26
|
+
"""求值简单算术表达式 (数值/已定义符号/四则/取模/括号/一元符号)。
|
|
27
|
+
|
|
28
|
+
失败返回 None (调用方回退其它解析路径)。
|
|
29
|
+
"""
|
|
30
|
+
text = text.strip().lstrip('#')
|
|
31
|
+
if not text:
|
|
32
|
+
return None
|
|
33
|
+
|
|
34
|
+
# 先把数值字面量替换为整数字符串, 避免 0xFF 中的字母与符号名混淆
|
|
35
|
+
def repl_num(match):
|
|
36
|
+
raw = match.group(0).replace('_', '')
|
|
37
|
+
if len(raw) > 1 and raw[:2].lower() in ('0x', '0b', '0o'):
|
|
38
|
+
raw = '0' + raw[1].lower() + raw[2:]
|
|
39
|
+
try:
|
|
40
|
+
return str(int(raw, 0))
|
|
41
|
+
except ValueError:
|
|
42
|
+
return '0'
|
|
43
|
+
|
|
44
|
+
text = _RE_EXPR_NUM.sub(repl_num, text)
|
|
45
|
+
|
|
46
|
+
def repl_name(match):
|
|
47
|
+
name = match.group(0)
|
|
48
|
+
if name not in symbols:
|
|
49
|
+
raise KeyError(name)
|
|
50
|
+
return str(symbols[name])
|
|
51
|
+
|
|
52
|
+
try:
|
|
53
|
+
expr = _RE_EXPR_NAME.sub(repl_name, text)
|
|
54
|
+
except KeyError:
|
|
55
|
+
return None
|
|
56
|
+
if not _RE_EXPR_OK.match(expr):
|
|
57
|
+
return None
|
|
58
|
+
try:
|
|
59
|
+
return int(eval(expr, {'__builtins__': {}}, {})) # noqa: S307 - 本地汇编器输入
|
|
60
|
+
except Exception:
|
|
61
|
+
return None
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
class Assembler:
|
|
65
|
+
def __init__(self, memory: FastMemory, console: Optional[Console] = None,
|
|
66
|
+
strict: bool = False, logger=None):
|
|
67
|
+
self.memory = memory
|
|
68
|
+
self.console = console or Console()
|
|
69
|
+
self.strict = strict
|
|
70
|
+
self.logger = logger
|
|
71
|
+
self.instructions: List[Instruction] = []
|
|
72
|
+
self.labels: Dict[str, int] = {}
|
|
73
|
+
self.data_labels: Dict[str, int] = {}
|
|
74
|
+
self.equ: Dict[str, int] = {} # .equ/.set 常量
|
|
75
|
+
|
|
76
|
+
def _dbg(self, msg: str) -> None:
|
|
77
|
+
if self.logger is not None:
|
|
78
|
+
self.logger.debug(msg)
|
|
79
|
+
|
|
80
|
+
def assemble_file(self, filename: str) -> Tuple[List[Instruction], Dict[str, int], Dict[str, int]]:
|
|
81
|
+
lines = self._preprocess(filename, set())
|
|
82
|
+
self._dbg(f"ASM preprocess: {len(lines)} lines from {filename}")
|
|
83
|
+
result = self._assemble_lines(lines, filename)
|
|
84
|
+
self._dbg(f"ASM assembled: {len(result[0])} instructions, "
|
|
85
|
+
f"{len(result[1])} labels, {len(result[2])} data labels")
|
|
86
|
+
return result
|
|
87
|
+
|
|
88
|
+
def assemble_source(self, source: str, filename: str = '<source>'
|
|
89
|
+
) -> Tuple[List[Instruction], Dict[str, int], Dict[str, int]]:
|
|
90
|
+
lines = []
|
|
91
|
+
for line_num, raw in enumerate(source.splitlines(), 1):
|
|
92
|
+
cleaned = self._strip_comment(raw)
|
|
93
|
+
if cleaned.strip():
|
|
94
|
+
lines.append((cleaned.strip(), line_num, filename))
|
|
95
|
+
return self._assemble_lines(lines, filename)
|
|
96
|
+
|
|
97
|
+
def _strip_comment(self, raw: str) -> str:
|
|
98
|
+
for marker in (';', '//'):
|
|
99
|
+
idx = raw.find(marker)
|
|
100
|
+
if idx >= 0:
|
|
101
|
+
raw = raw[:idx]
|
|
102
|
+
result = []
|
|
103
|
+
i = 0
|
|
104
|
+
while i < len(raw):
|
|
105
|
+
ch = raw[i]
|
|
106
|
+
if ch == '"':
|
|
107
|
+
result.append(ch)
|
|
108
|
+
i += 1
|
|
109
|
+
while i < len(raw):
|
|
110
|
+
result.append(raw[i])
|
|
111
|
+
if raw[i] == '"' and raw[i - 1] != '\\':
|
|
112
|
+
i += 1
|
|
113
|
+
break
|
|
114
|
+
i += 1
|
|
115
|
+
continue
|
|
116
|
+
if ch == '#':
|
|
117
|
+
at_start = len(''.join(result).strip()) == 0
|
|
118
|
+
rest = raw[i + 1:]
|
|
119
|
+
if at_start:
|
|
120
|
+
if rest.startswith('include'):
|
|
121
|
+
result.append(raw[i:])
|
|
122
|
+
return ''.join(result)
|
|
123
|
+
break
|
|
124
|
+
prev = raw[i - 1] if i > 0 else ' '
|
|
125
|
+
if prev in ' \t' and (rest == '' or rest[0] in ' \t'):
|
|
126
|
+
break
|
|
127
|
+
result.append(ch)
|
|
128
|
+
i += 1
|
|
129
|
+
return ''.join(result)
|
|
130
|
+
|
|
131
|
+
def _preprocess(self, filename: str, loaded: Set[str]) -> List[Tuple[str, int, str]]:
|
|
132
|
+
abs_path = os.path.abspath(filename)
|
|
133
|
+
if abs_path in loaded:
|
|
134
|
+
return []
|
|
135
|
+
loaded.add(abs_path)
|
|
136
|
+
dir_path = os.path.dirname(abs_path)
|
|
137
|
+
|
|
138
|
+
if not os.path.exists(abs_path):
|
|
139
|
+
raise AssemblerError('', f"File '{filename}' not found", filename=filename)
|
|
140
|
+
|
|
141
|
+
out: List[Tuple[str, int, str]] = []
|
|
142
|
+
with open(abs_path, encoding='utf-8') as f:
|
|
143
|
+
for line_num, raw in enumerate(f, 1):
|
|
144
|
+
stripped = raw.strip()
|
|
145
|
+
if stripped.startswith('#include'):
|
|
146
|
+
parts = stripped.split(None, 1)
|
|
147
|
+
if len(parts) < 2:
|
|
148
|
+
raise AssemblerError(stripped, "#include format error",
|
|
149
|
+
line_num, filename)
|
|
150
|
+
inc_file = parts[1].strip().strip('"<>')
|
|
151
|
+
inc_path = os.path.join(dir_path, inc_file)
|
|
152
|
+
if not os.path.exists(inc_path):
|
|
153
|
+
raise AssemblerError(stripped,
|
|
154
|
+
f"Include file '{inc_file}' not found",
|
|
155
|
+
line_num, filename)
|
|
156
|
+
out.extend(self._preprocess(inc_path, loaded))
|
|
157
|
+
continue
|
|
158
|
+
cleaned = self._strip_comment(raw)
|
|
159
|
+
if cleaned.strip():
|
|
160
|
+
out.append((cleaned.strip(), line_num, filename))
|
|
161
|
+
return out
|
|
162
|
+
|
|
163
|
+
def _parse_equ(self, line: str, line_num: int, fname: str) -> None:
|
|
164
|
+
""".equ/.set NAME <expr> (名称/值/四则/括号/已定义符号)。"""
|
|
165
|
+
head = line.split(None, 1)
|
|
166
|
+
rest = (head[1] if len(head) > 1 else '').lstrip(',').strip()
|
|
167
|
+
m = re.match(r'^([A-Za-z_.$][A-Za-z0-9_.$]*)\s*(?:=\s*)?(.*)$', rest)
|
|
168
|
+
if not m or not m.group(2).strip():
|
|
169
|
+
raise AssemblerError(
|
|
170
|
+
line, ".equ format: .equ NAME <expr> (例: .equ N, 8*4)",
|
|
171
|
+
line_num, fname)
|
|
172
|
+
name, expr = m.group(1), m.group(2).strip().lstrip(',').strip()
|
|
173
|
+
sym = dict(self.equ)
|
|
174
|
+
sym.update(self.labels)
|
|
175
|
+
sym.update(self.data_labels)
|
|
176
|
+
val = _eval_expr(expr, sym)
|
|
177
|
+
if val is None:
|
|
178
|
+
raise AssemblerError(line, f"Bad .equ expression: {expr!r}",
|
|
179
|
+
line_num, fname)
|
|
180
|
+
self.equ[name] = val
|
|
181
|
+
|
|
182
|
+
def _assemble_lines(self, lines: List[Tuple[str, int, str]], filename: str
|
|
183
|
+
) -> Tuple[List[Instruction], Dict[str, int], Dict[str, int]]:
|
|
184
|
+
self.instructions = []
|
|
185
|
+
self.labels = {}
|
|
186
|
+
self.data_labels = {}
|
|
187
|
+
self.equ = {}
|
|
188
|
+
instr_index = 0
|
|
189
|
+
data_addr = 0
|
|
190
|
+
section = 'TEXT'
|
|
191
|
+
text_lines: List[Tuple[str, int, int, str]] = []
|
|
192
|
+
|
|
193
|
+
i = 0
|
|
194
|
+
while i < len(lines):
|
|
195
|
+
line, line_num, fname = lines[i]
|
|
196
|
+
|
|
197
|
+
upper = line.upper()
|
|
198
|
+
if upper in ('.TEXT', '.CODE', 'TEXT', 'CODE'):
|
|
199
|
+
section = 'TEXT'
|
|
200
|
+
i += 1
|
|
201
|
+
continue
|
|
202
|
+
if upper in ('.DATA', 'DATA'):
|
|
203
|
+
section = 'DATA'
|
|
204
|
+
i += 1
|
|
205
|
+
continue
|
|
206
|
+
|
|
207
|
+
# .equ/.set 常量定义 (需带点前缀, 避免与 PL 关键字 'set' 冲突)
|
|
208
|
+
head_tok = line.split(None, 1)
|
|
209
|
+
if head_tok and head_tok[0].lower() in ('.equ', '.set'):
|
|
210
|
+
self._parse_equ(line, line_num, fname)
|
|
211
|
+
i += 1
|
|
212
|
+
continue
|
|
213
|
+
|
|
214
|
+
rest_line = line
|
|
215
|
+
if ':' in line and not line.startswith('['):
|
|
216
|
+
before, after = line.split(':', 1)
|
|
217
|
+
label = before.strip()
|
|
218
|
+
if _RE_LABEL.match(label):
|
|
219
|
+
if section == 'TEXT':
|
|
220
|
+
self.labels[label] = instr_index
|
|
221
|
+
else:
|
|
222
|
+
self.data_labels[label] = data_addr
|
|
223
|
+
rest_line = after.strip()
|
|
224
|
+
if not rest_line:
|
|
225
|
+
i += 1
|
|
226
|
+
continue
|
|
227
|
+
else:
|
|
228
|
+
raise AssemblerError(line, f"Invalid label: {label}", line_num, fname)
|
|
229
|
+
|
|
230
|
+
if section == 'DATA':
|
|
231
|
+
sym = dict(self.equ)
|
|
232
|
+
sym.update(self.labels)
|
|
233
|
+
sym.update(self.data_labels)
|
|
234
|
+
data_addr = self._handle_data(rest_line, data_addr, line_num,
|
|
235
|
+
fname, sym)
|
|
236
|
+
i += 1
|
|
237
|
+
continue
|
|
238
|
+
|
|
239
|
+
text_lines.append((rest_line, instr_index, line_num, fname))
|
|
240
|
+
instr_index += 1
|
|
241
|
+
i += 1
|
|
242
|
+
|
|
243
|
+
# 指令二次解析: 符号表 = 标签 + 数据标签 + .equ (标签优先)
|
|
244
|
+
sym_final = dict(self.equ)
|
|
245
|
+
sym_final.update(self.labels)
|
|
246
|
+
sym_final.update(self.data_labels)
|
|
247
|
+
for line, _idx, line_num, fname in text_lines:
|
|
248
|
+
instr = self._parse_instruction(line, sym_final, line_num, fname)
|
|
249
|
+
self.instructions.append(instr)
|
|
250
|
+
|
|
251
|
+
return self.instructions, self.labels, self.data_labels
|
|
252
|
+
|
|
253
|
+
def _handle_data(self, line: str, data_addr: int, line_num: int, fname: str,
|
|
254
|
+
symbols: Optional[Dict[str, int]] = None) -> int:
|
|
255
|
+
# 先按空白拆出指令名 (如 ASCIZ "..."), 其余按逗号拆分
|
|
256
|
+
head_tokens = line.split(None, 1)
|
|
257
|
+
if not head_tokens:
|
|
258
|
+
return data_addr
|
|
259
|
+
directive = head_tokens[0].upper().lstrip('.')
|
|
260
|
+
rest = head_tokens[1] if len(head_tokens) > 1 else ''
|
|
261
|
+
parts = [directive] + self._split_operands(rest)
|
|
262
|
+
if not parts:
|
|
263
|
+
return data_addr
|
|
264
|
+
if directive in ('BYTE',):
|
|
265
|
+
directive = 'DB'
|
|
266
|
+
elif directive in ('WORD',):
|
|
267
|
+
directive = 'DW'
|
|
268
|
+
elif directive in ('DWORD',):
|
|
269
|
+
directive = 'DD'
|
|
270
|
+
elif directive in ('QWORD',):
|
|
271
|
+
directive = 'DQ'
|
|
272
|
+
if directive not in Constants.DATA_DIRECTIVES:
|
|
273
|
+
raise AssemblerError(line, f"Unknown data directive: {parts[0]}",
|
|
274
|
+
line_num, fname)
|
|
275
|
+
|
|
276
|
+
width = {'DB': 1, 'DW': 2, 'DD': 4, 'DQ': 8}.get(directive)
|
|
277
|
+
|
|
278
|
+
for val_str in parts[1:]:
|
|
279
|
+
val_str = val_str.strip()
|
|
280
|
+
if not val_str:
|
|
281
|
+
continue
|
|
282
|
+
if directive in ('ASCII', 'ASCIZ', 'STRING'):
|
|
283
|
+
text = self._unquote(val_str)
|
|
284
|
+
raw = text.encode('utf-8')
|
|
285
|
+
self.memory.write_block(data_addr, raw)
|
|
286
|
+
data_addr += len(raw)
|
|
287
|
+
if directive in ('ASCIZ', 'STRING'):
|
|
288
|
+
self.memory.write_byte(data_addr, 0)
|
|
289
|
+
data_addr += 1
|
|
290
|
+
continue
|
|
291
|
+
for piece in val_str.split(','):
|
|
292
|
+
piece = piece.strip()
|
|
293
|
+
if not piece:
|
|
294
|
+
continue
|
|
295
|
+
values = self._parse_data_values(piece, line, line_num, fname,
|
|
296
|
+
symbols)
|
|
297
|
+
for val in values:
|
|
298
|
+
if width == 1:
|
|
299
|
+
self.memory.write_byte(data_addr, val & 0xFF)
|
|
300
|
+
elif width == 2:
|
|
301
|
+
self.memory.write_word(data_addr, val & 0xFFFF)
|
|
302
|
+
elif width == 4:
|
|
303
|
+
self.memory.write_dword(data_addr, val & 0xFFFFFFFF)
|
|
304
|
+
else:
|
|
305
|
+
self.memory.write_qword(data_addr, val & 0xFFFFFFFFFFFFFFFF)
|
|
306
|
+
data_addr += width
|
|
307
|
+
return data_addr
|
|
308
|
+
|
|
309
|
+
def _parse_data_values(self, token: str, line: str, line_num: int,
|
|
310
|
+
fname: str,
|
|
311
|
+
symbols: Optional[Dict[str, int]] = None) -> List[int]:
|
|
312
|
+
token = token.strip().rstrip(',')
|
|
313
|
+
if token.startswith("'") and token.endswith("'") and len(token) >= 3:
|
|
314
|
+
body = token[1:-1]
|
|
315
|
+
if body.startswith('\\'):
|
|
316
|
+
return [{'n': 10, 't': 9, 'r': 13, '0': 0, '\\': 92,
|
|
317
|
+
"'": 39, '"': 34}.get(body[1], ord(body[1]))]
|
|
318
|
+
return [ord(body[0])]
|
|
319
|
+
if token.startswith('"') and token.endswith('"'):
|
|
320
|
+
vals = list(token[1:-1].encode('utf-8'))
|
|
321
|
+
vals.append(0)
|
|
322
|
+
return vals
|
|
323
|
+
try:
|
|
324
|
+
return [self.parse_immediate(token)]
|
|
325
|
+
except ValueError:
|
|
326
|
+
val = _eval_expr(token, symbols or {})
|
|
327
|
+
if val is not None:
|
|
328
|
+
return [val]
|
|
329
|
+
raise AssemblerError(line, f"Invalid data value: {token}",
|
|
330
|
+
line_num, fname) from None
|
|
331
|
+
|
|
332
|
+
@staticmethod
|
|
333
|
+
def _unquote(token: str) -> str:
|
|
334
|
+
token = token.strip()
|
|
335
|
+
if len(token) >= 2 and token[0] == '"' and token[-1] == '"':
|
|
336
|
+
return token[1:-1].encode('utf-8').decode('unicode_escape')
|
|
337
|
+
return token
|
|
338
|
+
|
|
339
|
+
@staticmethod
|
|
340
|
+
def parse_immediate(val: str) -> int:
|
|
341
|
+
val = val.strip().lstrip('#').replace('_', '')
|
|
342
|
+
if not val:
|
|
343
|
+
raise ValueError("empty immediate")
|
|
344
|
+
neg = False
|
|
345
|
+
if val[0] in '+-':
|
|
346
|
+
neg = val[0] == '-'
|
|
347
|
+
val = val[1:]
|
|
348
|
+
if not val:
|
|
349
|
+
raise ValueError("empty immediate")
|
|
350
|
+
lower = val.lower()
|
|
351
|
+
if lower.startswith('0x') or lower.startswith('0b') or lower.startswith('0o'):
|
|
352
|
+
# 进制前缀字面量: u/U/l/L 不是任何受支持进制的数字, 可安全剥离;
|
|
353
|
+
# 但 f/F 是合法的十六进制数字, 绝不能剥离
|
|
354
|
+
# (否则 #0x1F 会被截成 0x1, #0xABCDEF 会被截成 0xABCD)。
|
|
355
|
+
end = len(val)
|
|
356
|
+
while end > 0 and val[end - 1] in 'uUlL':
|
|
357
|
+
end -= 1
|
|
358
|
+
body = val[:end]
|
|
359
|
+
if lower.startswith('0x'):
|
|
360
|
+
n = int(body, 16)
|
|
361
|
+
elif lower.startswith('0b'):
|
|
362
|
+
n = int(body[2:], 2)
|
|
363
|
+
else:
|
|
364
|
+
n = int(body[2:], 8)
|
|
365
|
+
else:
|
|
366
|
+
# 十进制: f/F 只能是类型后缀, 在 64 位槽模型下无宽度差异, 直接忽略
|
|
367
|
+
while val and val[-1] in 'uUlLfF':
|
|
368
|
+
val = val[:-1]
|
|
369
|
+
if not val:
|
|
370
|
+
raise ValueError("empty immediate")
|
|
371
|
+
n = int(val)
|
|
372
|
+
return -n if neg else n
|
|
373
|
+
|
|
374
|
+
def _split_operands(self, text: str) -> List[str]:
|
|
375
|
+
parts = []
|
|
376
|
+
depth = 0
|
|
377
|
+
current = []
|
|
378
|
+
in_str = False
|
|
379
|
+
for ch in text:
|
|
380
|
+
if ch == '"':
|
|
381
|
+
in_str = not in_str
|
|
382
|
+
current.append(ch)
|
|
383
|
+
elif not in_str and ch == '[':
|
|
384
|
+
depth += 1
|
|
385
|
+
current.append(ch)
|
|
386
|
+
elif not in_str and ch == ']':
|
|
387
|
+
depth -= 1
|
|
388
|
+
current.append(ch)
|
|
389
|
+
elif ch == ',' and depth == 0 and not in_str:
|
|
390
|
+
parts.append(''.join(current).strip())
|
|
391
|
+
current = []
|
|
392
|
+
else:
|
|
393
|
+
current.append(ch)
|
|
394
|
+
tail = ''.join(current).strip()
|
|
395
|
+
if tail:
|
|
396
|
+
parts.append(tail)
|
|
397
|
+
return parts
|
|
398
|
+
|
|
399
|
+
def _parse_instruction(self, line: str, symbols: Dict[str, int],
|
|
400
|
+
line_num: int, fname: str) -> Instruction:
|
|
401
|
+
tokens = line.split(None, 1)
|
|
402
|
+
mnemonic_raw = tokens[0]
|
|
403
|
+
rest = tokens[1] if len(tokens) > 1 else ''
|
|
404
|
+
|
|
405
|
+
cond_prefix = None
|
|
406
|
+
mnemonic = mnemonic_raw
|
|
407
|
+
if '.' in mnemonic_raw:
|
|
408
|
+
base, suffix = mnemonic_raw.split('.', 1)
|
|
409
|
+
suffix_up = suffix.upper()
|
|
410
|
+
if suffix_up in Constants.CONDITIONS:
|
|
411
|
+
mnemonic = base
|
|
412
|
+
cond_prefix = suffix_up
|
|
413
|
+
|
|
414
|
+
keyword = mnemonic.lower()
|
|
415
|
+
if keyword in Constants.PL_KEYWORDS:
|
|
416
|
+
opcode = Constants.PL_KEYWORDS[keyword]
|
|
417
|
+
else:
|
|
418
|
+
opcode = mnemonic.upper()
|
|
419
|
+
if opcode not in Constants.OPCODE_NAME_TO_ENUM:
|
|
420
|
+
raise AssemblerError(line, f"Unknown instruction: {mnemonic_raw}",
|
|
421
|
+
line_num, fname)
|
|
422
|
+
|
|
423
|
+
operands: List[Operand] = []
|
|
424
|
+
if cond_prefix:
|
|
425
|
+
operands.append(('cond', cond_prefix))
|
|
426
|
+
|
|
427
|
+
for tok in self._split_operands(rest):
|
|
428
|
+
operands.append(self._parse_operand(tok, symbols, line, line_num, fname))
|
|
429
|
+
|
|
430
|
+
if self.strict:
|
|
431
|
+
expected = Constants.ARG_COUNTS.get(Constants.OPCODE_NAME_TO_ENUM[opcode], -1)
|
|
432
|
+
if expected >= 0 and len(operands) != expected:
|
|
433
|
+
raise AssemblerError(
|
|
434
|
+
line,
|
|
435
|
+
f"Argument count mismatch for {opcode}: expected {expected}, "
|
|
436
|
+
f"got {len(operands)}", line_num, fname)
|
|
437
|
+
|
|
438
|
+
return (opcode, operands)
|
|
439
|
+
|
|
440
|
+
def _parse_operand(self, tok: str, symbols: Dict[str, int], line: str,
|
|
441
|
+
line_num: int, fname: str) -> Operand:
|
|
442
|
+
tok = tok.strip()
|
|
443
|
+
if not tok:
|
|
444
|
+
raise AssemblerError(line, "Empty operand", line_num, fname)
|
|
445
|
+
|
|
446
|
+
if tok.startswith('['):
|
|
447
|
+
if not tok.endswith(']'):
|
|
448
|
+
raise AssemblerError(line, f"Malformed memory operand: {tok}",
|
|
449
|
+
line_num, fname)
|
|
450
|
+
return self._parse_mem(tok[1:-1], symbols, line, line_num, fname)
|
|
451
|
+
|
|
452
|
+
m = _RE_VREG.match(tok)
|
|
453
|
+
if m:
|
|
454
|
+
reg = int(m.group(1))
|
|
455
|
+
if m.group(2) is not None:
|
|
456
|
+
return ('veclane', reg, int(m.group(2)))
|
|
457
|
+
return ('vec', reg)
|
|
458
|
+
|
|
459
|
+
if _RE_XREG.match(tok):
|
|
460
|
+
return ('reg', int(tok[1:]))
|
|
461
|
+
if tok.lower() == 'sp':
|
|
462
|
+
return ('reg', Constants.SP_REG)
|
|
463
|
+
if tok.lower() in ('fp',):
|
|
464
|
+
return ('reg', 29)
|
|
465
|
+
if tok.lower() in ('lr',):
|
|
466
|
+
return ('reg', 30)
|
|
467
|
+
if tok.lower() == 'xzr':
|
|
468
|
+
return ('reg', 31)
|
|
469
|
+
|
|
470
|
+
if tok.upper() in Constants.CONDITIONS:
|
|
471
|
+
return ('cond', tok.upper())
|
|
472
|
+
|
|
473
|
+
if re.match(r'^[+-]?\d+\.\d+([eE][+-]?\d+)?$', tok):
|
|
474
|
+
return ('float', float(tok))
|
|
475
|
+
|
|
476
|
+
if tok.startswith('='):
|
|
477
|
+
name = tok[1:].strip()
|
|
478
|
+
if name in symbols:
|
|
479
|
+
return ('imm', symbols[name])
|
|
480
|
+
val = _eval_expr(name, symbols)
|
|
481
|
+
if val is not None:
|
|
482
|
+
return ('imm', val)
|
|
483
|
+
raise AssemblerError(line, f"Undefined symbol: {name}", line_num, fname)
|
|
484
|
+
|
|
485
|
+
if tok.startswith('#'):
|
|
486
|
+
inner = tok[1:].strip()
|
|
487
|
+
try:
|
|
488
|
+
return ('imm', self.parse_immediate(tok))
|
|
489
|
+
except ValueError:
|
|
490
|
+
if inner in symbols:
|
|
491
|
+
return ('imm', symbols[inner])
|
|
492
|
+
val = _eval_expr(inner, symbols)
|
|
493
|
+
if val is not None:
|
|
494
|
+
return ('imm', val)
|
|
495
|
+
raise AssemblerError(line, f"Bad immediate: {tok}",
|
|
496
|
+
line_num, fname) from None
|
|
497
|
+
|
|
498
|
+
try:
|
|
499
|
+
return ('imm', self.parse_immediate(tok))
|
|
500
|
+
except ValueError:
|
|
501
|
+
pass
|
|
502
|
+
|
|
503
|
+
# 表达式立即数 / 符号算术: 8+4*2 / SIZE-1 / loop+4 / 等
|
|
504
|
+
val = _eval_expr(tok, symbols)
|
|
505
|
+
if val is not None:
|
|
506
|
+
return ('imm', val)
|
|
507
|
+
|
|
508
|
+
if _RE_LABEL.match(tok):
|
|
509
|
+
if tok in symbols:
|
|
510
|
+
return ('imm', symbols[tok])
|
|
511
|
+
raise AssemblerError(line, f"Undefined label: {tok}", line_num, fname)
|
|
512
|
+
|
|
513
|
+
raise AssemblerError(line, f"Cannot parse operand: {tok}", line_num, fname)
|
|
514
|
+
|
|
515
|
+
def _parse_mem(self, inner: str, symbols: Dict[str, int], line: str,
|
|
516
|
+
line_num: int, fname: str) -> Operand:
|
|
517
|
+
parts = [p.strip() for p in self._split_operands(inner) if p.strip()]
|
|
518
|
+
if not parts:
|
|
519
|
+
raise AssemblerError(line, "Empty memory operand", line_num, fname)
|
|
520
|
+
|
|
521
|
+
base = -1
|
|
522
|
+
offset = 0
|
|
523
|
+
|
|
524
|
+
first = parts[0]
|
|
525
|
+
if _RE_XREG.match(first) or first.lower() == 'sp':
|
|
526
|
+
base = Constants.SP_REG if first.lower() == 'sp' else int(first[1:])
|
|
527
|
+
else:
|
|
528
|
+
try:
|
|
529
|
+
offset = self.parse_immediate(first)
|
|
530
|
+
except ValueError:
|
|
531
|
+
val = _eval_expr(first.lstrip('#'), symbols)
|
|
532
|
+
if val is None:
|
|
533
|
+
raise AssemblerError(line, f"Bad memory base: {first}",
|
|
534
|
+
line_num, fname) from None
|
|
535
|
+
offset = val
|
|
536
|
+
|
|
537
|
+
if len(parts) > 1:
|
|
538
|
+
second = parts[1]
|
|
539
|
+
try:
|
|
540
|
+
offset += self.parse_immediate(second)
|
|
541
|
+
except ValueError:
|
|
542
|
+
val = _eval_expr(second.lstrip('#'), symbols)
|
|
543
|
+
if val is None:
|
|
544
|
+
raise AssemblerError(line, f"Bad memory offset: {second}",
|
|
545
|
+
line_num, fname) from None
|
|
546
|
+
offset += val
|
|
547
|
+
|
|
548
|
+
return ('mem', base, offset)
|
codecin/cache.py
ADDED
|
@@ -0,0 +1,71 @@
|
|
|
1
|
+
from collections import OrderedDict
|
|
2
|
+
from typing import Any, Dict, List
|
|
3
|
+
|
|
4
|
+
|
|
5
|
+
class CacheLine:
|
|
6
|
+
__slots__ = ('tag', 'valid', 'dirty')
|
|
7
|
+
|
|
8
|
+
def __init__(self, tag: int = 0, valid: bool = False, dirty: bool = False):
|
|
9
|
+
self.tag = tag
|
|
10
|
+
self.valid = valid
|
|
11
|
+
self.dirty = dirty
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class Cache:
|
|
15
|
+
def __init__(self, size: int = 64, assoc: int = 4, line_size: int = 16):
|
|
16
|
+
self.size = max(assoc, size)
|
|
17
|
+
self.assoc = assoc
|
|
18
|
+
self.line_size = line_size
|
|
19
|
+
self.num_sets = max(1, self.size // assoc)
|
|
20
|
+
self._sets: List[OrderedDict] = [OrderedDict() for _ in range(self.num_sets)]
|
|
21
|
+
self.hits = 0
|
|
22
|
+
self.misses = 0
|
|
23
|
+
self.writes = 0
|
|
24
|
+
|
|
25
|
+
def _set_index(self, addr: int) -> int:
|
|
26
|
+
return (addr // self.line_size) % self.num_sets
|
|
27
|
+
|
|
28
|
+
def _tag(self, addr: int) -> int:
|
|
29
|
+
return addr // (self.line_size * self.num_sets)
|
|
30
|
+
|
|
31
|
+
def access(self, addr: int, is_write: bool = False) -> bool:
|
|
32
|
+
s = self._sets[self._set_index(addr)]
|
|
33
|
+
tag = self._tag(addr)
|
|
34
|
+
if tag in s:
|
|
35
|
+
s.move_to_end(tag)
|
|
36
|
+
if is_write:
|
|
37
|
+
s[tag] = True
|
|
38
|
+
self.writes += 1
|
|
39
|
+
self.hits += 1
|
|
40
|
+
return True
|
|
41
|
+
s[tag] = is_write
|
|
42
|
+
s.move_to_end(tag)
|
|
43
|
+
while len(s) > self.assoc:
|
|
44
|
+
s.popitem(last=False)
|
|
45
|
+
self.misses += 1
|
|
46
|
+
return False
|
|
47
|
+
|
|
48
|
+
def read(self, addr: int) -> bool:
|
|
49
|
+
return self.access(addr, False)
|
|
50
|
+
|
|
51
|
+
def write(self, addr: int) -> bool:
|
|
52
|
+
return self.access(addr, True)
|
|
53
|
+
|
|
54
|
+
def flush(self) -> None:
|
|
55
|
+
for s in self._sets:
|
|
56
|
+
s.clear()
|
|
57
|
+
|
|
58
|
+
def warmup(self, instructions) -> None:
|
|
59
|
+
for pc in range(len(instructions)):
|
|
60
|
+
self.access(pc * self.line_size)
|
|
61
|
+
|
|
62
|
+
def get_stats(self) -> Dict[str, Any]:
|
|
63
|
+
total = self.hits + self.misses
|
|
64
|
+
return {
|
|
65
|
+
'hits': self.hits,
|
|
66
|
+
'misses': self.misses,
|
|
67
|
+
'hit_rate': self.hits / total if total > 0 else 0.0,
|
|
68
|
+
'miss_rate': self.misses / total if total > 0 else 0.0,
|
|
69
|
+
'total_accesses': total,
|
|
70
|
+
'dirty_writes': self.writes,
|
|
71
|
+
}
|