codecin 5.4.2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (54) hide show
  1. codecin/__init__.py +27 -0
  2. codecin/aot.py +229 -0
  3. codecin/assembler.py +548 -0
  4. codecin/cache.py +71 -0
  5. codecin/cin.py +2558 -0
  6. codecin/cli.py +282 -0
  7. codecin/config.py +58 -0
  8. codecin/console.py +183 -0
  9. codecin/cpu.py +1654 -0
  10. codecin/crom.py +222 -0
  11. codecin/debugger.py +659 -0
  12. codecin/disasm.py +86 -0
  13. codecin/errors.py +58 -0
  14. codecin/isa.py +457 -0
  15. codecin/jit.py +270 -0
  16. codecin/libcodecin_native.so +0 -0
  17. codecin/logger.py +145 -0
  18. codecin/memory.py +348 -0
  19. codecin/native/aot/aot.go +70 -0
  20. codecin/native/aot/build.go +219 -0
  21. codecin/native/aot/stub_main.go.txt +22 -0
  22. codecin/native/build.ps1 +33 -0
  23. codecin/native/build.sh +48 -0
  24. codecin/native/cmd/codecin/main.go +287 -0
  25. codecin/native/compiler/codegen.go +1938 -0
  26. codecin/native/compiler/compile_test.go +164 -0
  27. codecin/native/compiler/parser.go +1071 -0
  28. codecin/native/compiler/syscalls.go +86 -0
  29. codecin/native/compiler/tokenizer.go +336 -0
  30. codecin/native/compiler/types.go +213 -0
  31. codecin/native/engine/audio.go +152 -0
  32. codecin/native/engine/audio_other.go +46 -0
  33. codecin/native/engine/audio_windows.go +43 -0
  34. codecin/native/engine/canvas.go +271 -0
  35. codecin/native/engine/crom.go +86 -0
  36. codecin/native/engine/encode.go +93 -0
  37. codecin/native/engine/engine_test.go +198 -0
  38. codecin/native/engine/isa_gen.go +445 -0
  39. codecin/native/engine/system.go +190 -0
  40. codecin/native/engine/termux.go +113 -0
  41. codecin/native/engine/version_gen.go +7 -0
  42. codecin/native/engine/vm.go +1157 -0
  43. codecin/native/go.mod +3 -0
  44. codecin/native/ir/ir.go +48 -0
  45. codecin/native/main.go +225 -0
  46. codecin/native.py +336 -0
  47. codecin/registers.py +130 -0
  48. codecin/stats.py +235 -0
  49. codecin-5.4.2.dist-info/METADATA +985 -0
  50. codecin-5.4.2.dist-info/RECORD +54 -0
  51. codecin-5.4.2.dist-info/WHEEL +5 -0
  52. codecin-5.4.2.dist-info/entry_points.txt +2 -0
  53. codecin-5.4.2.dist-info/licenses/LICENSE +7 -0
  54. codecin-5.4.2.dist-info/top_level.txt +1 -0
codecin/assembler.py ADDED
@@ -0,0 +1,548 @@
1
+ import os
2
+ import re
3
+ from typing import Any, Dict, List, Optional, Set, Tuple
4
+
5
+ from .console import Console
6
+ from .errors import AssemblerError
7
+ from .isa import Constants
8
+ from .memory import FastMemory
9
+
10
+ Operand = Tuple[Any, ...]
11
+ Instruction = Tuple[str, List[Operand]]
12
+
13
+ _RE_XREG = re.compile(r'^[xXrRwW]([0-9]|[12][0-9]|3[01])$')
14
+ _RE_VREG = re.compile(r'^[vV]([0-9]|[12][0-9]|3[01])(?:\.([0-3]))?$')
15
+ _RE_LABEL = re.compile(r'^[a-zA-Z_.$][a-zA-Z0-9_.$]*$')
16
+ _RE_EXPR_NAME = re.compile(r'\b[a-zA-Z_.$][a-zA-Z0-9_.$]*\b')
17
+ _RE_EQU_DIRECTIVE = {'EQU', 'SET'}
18
+ # 表达式只允许: 数字 / 已定义符号 / + - * / % ( ) 空白
19
+ _RE_EXPR_OK = re.compile(r'^[0-9+\-*/%()\s]+$')
20
+ # 表达式内的数值字面量 (十进制 / 0x / 0b / 0o, 允许下划线)
21
+ _RE_EXPR_NUM = re.compile(
22
+ r'0[xX][0-9a-fA-F_]+|0[bB][01_]+|0[oO][0-7_]+|\d[\d_]*')
23
+
24
+
25
+ def _eval_expr(text: str, symbols: Dict[str, int]) -> Optional[int]:
26
+ """求值简单算术表达式 (数值/已定义符号/四则/取模/括号/一元符号)。
27
+
28
+ 失败返回 None (调用方回退其它解析路径)。
29
+ """
30
+ text = text.strip().lstrip('#')
31
+ if not text:
32
+ return None
33
+
34
+ # 先把数值字面量替换为整数字符串, 避免 0xFF 中的字母与符号名混淆
35
+ def repl_num(match):
36
+ raw = match.group(0).replace('_', '')
37
+ if len(raw) > 1 and raw[:2].lower() in ('0x', '0b', '0o'):
38
+ raw = '0' + raw[1].lower() + raw[2:]
39
+ try:
40
+ return str(int(raw, 0))
41
+ except ValueError:
42
+ return '0'
43
+
44
+ text = _RE_EXPR_NUM.sub(repl_num, text)
45
+
46
+ def repl_name(match):
47
+ name = match.group(0)
48
+ if name not in symbols:
49
+ raise KeyError(name)
50
+ return str(symbols[name])
51
+
52
+ try:
53
+ expr = _RE_EXPR_NAME.sub(repl_name, text)
54
+ except KeyError:
55
+ return None
56
+ if not _RE_EXPR_OK.match(expr):
57
+ return None
58
+ try:
59
+ return int(eval(expr, {'__builtins__': {}}, {})) # noqa: S307 - 本地汇编器输入
60
+ except Exception:
61
+ return None
62
+
63
+
64
+ class Assembler:
65
+ def __init__(self, memory: FastMemory, console: Optional[Console] = None,
66
+ strict: bool = False, logger=None):
67
+ self.memory = memory
68
+ self.console = console or Console()
69
+ self.strict = strict
70
+ self.logger = logger
71
+ self.instructions: List[Instruction] = []
72
+ self.labels: Dict[str, int] = {}
73
+ self.data_labels: Dict[str, int] = {}
74
+ self.equ: Dict[str, int] = {} # .equ/.set 常量
75
+
76
+ def _dbg(self, msg: str) -> None:
77
+ if self.logger is not None:
78
+ self.logger.debug(msg)
79
+
80
+ def assemble_file(self, filename: str) -> Tuple[List[Instruction], Dict[str, int], Dict[str, int]]:
81
+ lines = self._preprocess(filename, set())
82
+ self._dbg(f"ASM preprocess: {len(lines)} lines from {filename}")
83
+ result = self._assemble_lines(lines, filename)
84
+ self._dbg(f"ASM assembled: {len(result[0])} instructions, "
85
+ f"{len(result[1])} labels, {len(result[2])} data labels")
86
+ return result
87
+
88
+ def assemble_source(self, source: str, filename: str = '<source>'
89
+ ) -> Tuple[List[Instruction], Dict[str, int], Dict[str, int]]:
90
+ lines = []
91
+ for line_num, raw in enumerate(source.splitlines(), 1):
92
+ cleaned = self._strip_comment(raw)
93
+ if cleaned.strip():
94
+ lines.append((cleaned.strip(), line_num, filename))
95
+ return self._assemble_lines(lines, filename)
96
+
97
+ def _strip_comment(self, raw: str) -> str:
98
+ for marker in (';', '//'):
99
+ idx = raw.find(marker)
100
+ if idx >= 0:
101
+ raw = raw[:idx]
102
+ result = []
103
+ i = 0
104
+ while i < len(raw):
105
+ ch = raw[i]
106
+ if ch == '"':
107
+ result.append(ch)
108
+ i += 1
109
+ while i < len(raw):
110
+ result.append(raw[i])
111
+ if raw[i] == '"' and raw[i - 1] != '\\':
112
+ i += 1
113
+ break
114
+ i += 1
115
+ continue
116
+ if ch == '#':
117
+ at_start = len(''.join(result).strip()) == 0
118
+ rest = raw[i + 1:]
119
+ if at_start:
120
+ if rest.startswith('include'):
121
+ result.append(raw[i:])
122
+ return ''.join(result)
123
+ break
124
+ prev = raw[i - 1] if i > 0 else ' '
125
+ if prev in ' \t' and (rest == '' or rest[0] in ' \t'):
126
+ break
127
+ result.append(ch)
128
+ i += 1
129
+ return ''.join(result)
130
+
131
+ def _preprocess(self, filename: str, loaded: Set[str]) -> List[Tuple[str, int, str]]:
132
+ abs_path = os.path.abspath(filename)
133
+ if abs_path in loaded:
134
+ return []
135
+ loaded.add(abs_path)
136
+ dir_path = os.path.dirname(abs_path)
137
+
138
+ if not os.path.exists(abs_path):
139
+ raise AssemblerError('', f"File '{filename}' not found", filename=filename)
140
+
141
+ out: List[Tuple[str, int, str]] = []
142
+ with open(abs_path, encoding='utf-8') as f:
143
+ for line_num, raw in enumerate(f, 1):
144
+ stripped = raw.strip()
145
+ if stripped.startswith('#include'):
146
+ parts = stripped.split(None, 1)
147
+ if len(parts) < 2:
148
+ raise AssemblerError(stripped, "#include format error",
149
+ line_num, filename)
150
+ inc_file = parts[1].strip().strip('"<>')
151
+ inc_path = os.path.join(dir_path, inc_file)
152
+ if not os.path.exists(inc_path):
153
+ raise AssemblerError(stripped,
154
+ f"Include file '{inc_file}' not found",
155
+ line_num, filename)
156
+ out.extend(self._preprocess(inc_path, loaded))
157
+ continue
158
+ cleaned = self._strip_comment(raw)
159
+ if cleaned.strip():
160
+ out.append((cleaned.strip(), line_num, filename))
161
+ return out
162
+
163
+ def _parse_equ(self, line: str, line_num: int, fname: str) -> None:
164
+ """.equ/.set NAME <expr> (名称/值/四则/括号/已定义符号)。"""
165
+ head = line.split(None, 1)
166
+ rest = (head[1] if len(head) > 1 else '').lstrip(',').strip()
167
+ m = re.match(r'^([A-Za-z_.$][A-Za-z0-9_.$]*)\s*(?:=\s*)?(.*)$', rest)
168
+ if not m or not m.group(2).strip():
169
+ raise AssemblerError(
170
+ line, ".equ format: .equ NAME <expr> (例: .equ N, 8*4)",
171
+ line_num, fname)
172
+ name, expr = m.group(1), m.group(2).strip().lstrip(',').strip()
173
+ sym = dict(self.equ)
174
+ sym.update(self.labels)
175
+ sym.update(self.data_labels)
176
+ val = _eval_expr(expr, sym)
177
+ if val is None:
178
+ raise AssemblerError(line, f"Bad .equ expression: {expr!r}",
179
+ line_num, fname)
180
+ self.equ[name] = val
181
+
182
+ def _assemble_lines(self, lines: List[Tuple[str, int, str]], filename: str
183
+ ) -> Tuple[List[Instruction], Dict[str, int], Dict[str, int]]:
184
+ self.instructions = []
185
+ self.labels = {}
186
+ self.data_labels = {}
187
+ self.equ = {}
188
+ instr_index = 0
189
+ data_addr = 0
190
+ section = 'TEXT'
191
+ text_lines: List[Tuple[str, int, int, str]] = []
192
+
193
+ i = 0
194
+ while i < len(lines):
195
+ line, line_num, fname = lines[i]
196
+
197
+ upper = line.upper()
198
+ if upper in ('.TEXT', '.CODE', 'TEXT', 'CODE'):
199
+ section = 'TEXT'
200
+ i += 1
201
+ continue
202
+ if upper in ('.DATA', 'DATA'):
203
+ section = 'DATA'
204
+ i += 1
205
+ continue
206
+
207
+ # .equ/.set 常量定义 (需带点前缀, 避免与 PL 关键字 'set' 冲突)
208
+ head_tok = line.split(None, 1)
209
+ if head_tok and head_tok[0].lower() in ('.equ', '.set'):
210
+ self._parse_equ(line, line_num, fname)
211
+ i += 1
212
+ continue
213
+
214
+ rest_line = line
215
+ if ':' in line and not line.startswith('['):
216
+ before, after = line.split(':', 1)
217
+ label = before.strip()
218
+ if _RE_LABEL.match(label):
219
+ if section == 'TEXT':
220
+ self.labels[label] = instr_index
221
+ else:
222
+ self.data_labels[label] = data_addr
223
+ rest_line = after.strip()
224
+ if not rest_line:
225
+ i += 1
226
+ continue
227
+ else:
228
+ raise AssemblerError(line, f"Invalid label: {label}", line_num, fname)
229
+
230
+ if section == 'DATA':
231
+ sym = dict(self.equ)
232
+ sym.update(self.labels)
233
+ sym.update(self.data_labels)
234
+ data_addr = self._handle_data(rest_line, data_addr, line_num,
235
+ fname, sym)
236
+ i += 1
237
+ continue
238
+
239
+ text_lines.append((rest_line, instr_index, line_num, fname))
240
+ instr_index += 1
241
+ i += 1
242
+
243
+ # 指令二次解析: 符号表 = 标签 + 数据标签 + .equ (标签优先)
244
+ sym_final = dict(self.equ)
245
+ sym_final.update(self.labels)
246
+ sym_final.update(self.data_labels)
247
+ for line, _idx, line_num, fname in text_lines:
248
+ instr = self._parse_instruction(line, sym_final, line_num, fname)
249
+ self.instructions.append(instr)
250
+
251
+ return self.instructions, self.labels, self.data_labels
252
+
253
+ def _handle_data(self, line: str, data_addr: int, line_num: int, fname: str,
254
+ symbols: Optional[Dict[str, int]] = None) -> int:
255
+ # 先按空白拆出指令名 (如 ASCIZ "..."), 其余按逗号拆分
256
+ head_tokens = line.split(None, 1)
257
+ if not head_tokens:
258
+ return data_addr
259
+ directive = head_tokens[0].upper().lstrip('.')
260
+ rest = head_tokens[1] if len(head_tokens) > 1 else ''
261
+ parts = [directive] + self._split_operands(rest)
262
+ if not parts:
263
+ return data_addr
264
+ if directive in ('BYTE',):
265
+ directive = 'DB'
266
+ elif directive in ('WORD',):
267
+ directive = 'DW'
268
+ elif directive in ('DWORD',):
269
+ directive = 'DD'
270
+ elif directive in ('QWORD',):
271
+ directive = 'DQ'
272
+ if directive not in Constants.DATA_DIRECTIVES:
273
+ raise AssemblerError(line, f"Unknown data directive: {parts[0]}",
274
+ line_num, fname)
275
+
276
+ width = {'DB': 1, 'DW': 2, 'DD': 4, 'DQ': 8}.get(directive)
277
+
278
+ for val_str in parts[1:]:
279
+ val_str = val_str.strip()
280
+ if not val_str:
281
+ continue
282
+ if directive in ('ASCII', 'ASCIZ', 'STRING'):
283
+ text = self._unquote(val_str)
284
+ raw = text.encode('utf-8')
285
+ self.memory.write_block(data_addr, raw)
286
+ data_addr += len(raw)
287
+ if directive in ('ASCIZ', 'STRING'):
288
+ self.memory.write_byte(data_addr, 0)
289
+ data_addr += 1
290
+ continue
291
+ for piece in val_str.split(','):
292
+ piece = piece.strip()
293
+ if not piece:
294
+ continue
295
+ values = self._parse_data_values(piece, line, line_num, fname,
296
+ symbols)
297
+ for val in values:
298
+ if width == 1:
299
+ self.memory.write_byte(data_addr, val & 0xFF)
300
+ elif width == 2:
301
+ self.memory.write_word(data_addr, val & 0xFFFF)
302
+ elif width == 4:
303
+ self.memory.write_dword(data_addr, val & 0xFFFFFFFF)
304
+ else:
305
+ self.memory.write_qword(data_addr, val & 0xFFFFFFFFFFFFFFFF)
306
+ data_addr += width
307
+ return data_addr
308
+
309
+ def _parse_data_values(self, token: str, line: str, line_num: int,
310
+ fname: str,
311
+ symbols: Optional[Dict[str, int]] = None) -> List[int]:
312
+ token = token.strip().rstrip(',')
313
+ if token.startswith("'") and token.endswith("'") and len(token) >= 3:
314
+ body = token[1:-1]
315
+ if body.startswith('\\'):
316
+ return [{'n': 10, 't': 9, 'r': 13, '0': 0, '\\': 92,
317
+ "'": 39, '"': 34}.get(body[1], ord(body[1]))]
318
+ return [ord(body[0])]
319
+ if token.startswith('"') and token.endswith('"'):
320
+ vals = list(token[1:-1].encode('utf-8'))
321
+ vals.append(0)
322
+ return vals
323
+ try:
324
+ return [self.parse_immediate(token)]
325
+ except ValueError:
326
+ val = _eval_expr(token, symbols or {})
327
+ if val is not None:
328
+ return [val]
329
+ raise AssemblerError(line, f"Invalid data value: {token}",
330
+ line_num, fname) from None
331
+
332
+ @staticmethod
333
+ def _unquote(token: str) -> str:
334
+ token = token.strip()
335
+ if len(token) >= 2 and token[0] == '"' and token[-1] == '"':
336
+ return token[1:-1].encode('utf-8').decode('unicode_escape')
337
+ return token
338
+
339
+ @staticmethod
340
+ def parse_immediate(val: str) -> int:
341
+ val = val.strip().lstrip('#').replace('_', '')
342
+ if not val:
343
+ raise ValueError("empty immediate")
344
+ neg = False
345
+ if val[0] in '+-':
346
+ neg = val[0] == '-'
347
+ val = val[1:]
348
+ if not val:
349
+ raise ValueError("empty immediate")
350
+ lower = val.lower()
351
+ if lower.startswith('0x') or lower.startswith('0b') or lower.startswith('0o'):
352
+ # 进制前缀字面量: u/U/l/L 不是任何受支持进制的数字, 可安全剥离;
353
+ # 但 f/F 是合法的十六进制数字, 绝不能剥离
354
+ # (否则 #0x1F 会被截成 0x1, #0xABCDEF 会被截成 0xABCD)。
355
+ end = len(val)
356
+ while end > 0 and val[end - 1] in 'uUlL':
357
+ end -= 1
358
+ body = val[:end]
359
+ if lower.startswith('0x'):
360
+ n = int(body, 16)
361
+ elif lower.startswith('0b'):
362
+ n = int(body[2:], 2)
363
+ else:
364
+ n = int(body[2:], 8)
365
+ else:
366
+ # 十进制: f/F 只能是类型后缀, 在 64 位槽模型下无宽度差异, 直接忽略
367
+ while val and val[-1] in 'uUlLfF':
368
+ val = val[:-1]
369
+ if not val:
370
+ raise ValueError("empty immediate")
371
+ n = int(val)
372
+ return -n if neg else n
373
+
374
+ def _split_operands(self, text: str) -> List[str]:
375
+ parts = []
376
+ depth = 0
377
+ current = []
378
+ in_str = False
379
+ for ch in text:
380
+ if ch == '"':
381
+ in_str = not in_str
382
+ current.append(ch)
383
+ elif not in_str and ch == '[':
384
+ depth += 1
385
+ current.append(ch)
386
+ elif not in_str and ch == ']':
387
+ depth -= 1
388
+ current.append(ch)
389
+ elif ch == ',' and depth == 0 and not in_str:
390
+ parts.append(''.join(current).strip())
391
+ current = []
392
+ else:
393
+ current.append(ch)
394
+ tail = ''.join(current).strip()
395
+ if tail:
396
+ parts.append(tail)
397
+ return parts
398
+
399
+ def _parse_instruction(self, line: str, symbols: Dict[str, int],
400
+ line_num: int, fname: str) -> Instruction:
401
+ tokens = line.split(None, 1)
402
+ mnemonic_raw = tokens[0]
403
+ rest = tokens[1] if len(tokens) > 1 else ''
404
+
405
+ cond_prefix = None
406
+ mnemonic = mnemonic_raw
407
+ if '.' in mnemonic_raw:
408
+ base, suffix = mnemonic_raw.split('.', 1)
409
+ suffix_up = suffix.upper()
410
+ if suffix_up in Constants.CONDITIONS:
411
+ mnemonic = base
412
+ cond_prefix = suffix_up
413
+
414
+ keyword = mnemonic.lower()
415
+ if keyword in Constants.PL_KEYWORDS:
416
+ opcode = Constants.PL_KEYWORDS[keyword]
417
+ else:
418
+ opcode = mnemonic.upper()
419
+ if opcode not in Constants.OPCODE_NAME_TO_ENUM:
420
+ raise AssemblerError(line, f"Unknown instruction: {mnemonic_raw}",
421
+ line_num, fname)
422
+
423
+ operands: List[Operand] = []
424
+ if cond_prefix:
425
+ operands.append(('cond', cond_prefix))
426
+
427
+ for tok in self._split_operands(rest):
428
+ operands.append(self._parse_operand(tok, symbols, line, line_num, fname))
429
+
430
+ if self.strict:
431
+ expected = Constants.ARG_COUNTS.get(Constants.OPCODE_NAME_TO_ENUM[opcode], -1)
432
+ if expected >= 0 and len(operands) != expected:
433
+ raise AssemblerError(
434
+ line,
435
+ f"Argument count mismatch for {opcode}: expected {expected}, "
436
+ f"got {len(operands)}", line_num, fname)
437
+
438
+ return (opcode, operands)
439
+
440
+ def _parse_operand(self, tok: str, symbols: Dict[str, int], line: str,
441
+ line_num: int, fname: str) -> Operand:
442
+ tok = tok.strip()
443
+ if not tok:
444
+ raise AssemblerError(line, "Empty operand", line_num, fname)
445
+
446
+ if tok.startswith('['):
447
+ if not tok.endswith(']'):
448
+ raise AssemblerError(line, f"Malformed memory operand: {tok}",
449
+ line_num, fname)
450
+ return self._parse_mem(tok[1:-1], symbols, line, line_num, fname)
451
+
452
+ m = _RE_VREG.match(tok)
453
+ if m:
454
+ reg = int(m.group(1))
455
+ if m.group(2) is not None:
456
+ return ('veclane', reg, int(m.group(2)))
457
+ return ('vec', reg)
458
+
459
+ if _RE_XREG.match(tok):
460
+ return ('reg', int(tok[1:]))
461
+ if tok.lower() == 'sp':
462
+ return ('reg', Constants.SP_REG)
463
+ if tok.lower() in ('fp',):
464
+ return ('reg', 29)
465
+ if tok.lower() in ('lr',):
466
+ return ('reg', 30)
467
+ if tok.lower() == 'xzr':
468
+ return ('reg', 31)
469
+
470
+ if tok.upper() in Constants.CONDITIONS:
471
+ return ('cond', tok.upper())
472
+
473
+ if re.match(r'^[+-]?\d+\.\d+([eE][+-]?\d+)?$', tok):
474
+ return ('float', float(tok))
475
+
476
+ if tok.startswith('='):
477
+ name = tok[1:].strip()
478
+ if name in symbols:
479
+ return ('imm', symbols[name])
480
+ val = _eval_expr(name, symbols)
481
+ if val is not None:
482
+ return ('imm', val)
483
+ raise AssemblerError(line, f"Undefined symbol: {name}", line_num, fname)
484
+
485
+ if tok.startswith('#'):
486
+ inner = tok[1:].strip()
487
+ try:
488
+ return ('imm', self.parse_immediate(tok))
489
+ except ValueError:
490
+ if inner in symbols:
491
+ return ('imm', symbols[inner])
492
+ val = _eval_expr(inner, symbols)
493
+ if val is not None:
494
+ return ('imm', val)
495
+ raise AssemblerError(line, f"Bad immediate: {tok}",
496
+ line_num, fname) from None
497
+
498
+ try:
499
+ return ('imm', self.parse_immediate(tok))
500
+ except ValueError:
501
+ pass
502
+
503
+ # 表达式立即数 / 符号算术: 8+4*2 / SIZE-1 / loop+4 / 等
504
+ val = _eval_expr(tok, symbols)
505
+ if val is not None:
506
+ return ('imm', val)
507
+
508
+ if _RE_LABEL.match(tok):
509
+ if tok in symbols:
510
+ return ('imm', symbols[tok])
511
+ raise AssemblerError(line, f"Undefined label: {tok}", line_num, fname)
512
+
513
+ raise AssemblerError(line, f"Cannot parse operand: {tok}", line_num, fname)
514
+
515
+ def _parse_mem(self, inner: str, symbols: Dict[str, int], line: str,
516
+ line_num: int, fname: str) -> Operand:
517
+ parts = [p.strip() for p in self._split_operands(inner) if p.strip()]
518
+ if not parts:
519
+ raise AssemblerError(line, "Empty memory operand", line_num, fname)
520
+
521
+ base = -1
522
+ offset = 0
523
+
524
+ first = parts[0]
525
+ if _RE_XREG.match(first) or first.lower() == 'sp':
526
+ base = Constants.SP_REG if first.lower() == 'sp' else int(first[1:])
527
+ else:
528
+ try:
529
+ offset = self.parse_immediate(first)
530
+ except ValueError:
531
+ val = _eval_expr(first.lstrip('#'), symbols)
532
+ if val is None:
533
+ raise AssemblerError(line, f"Bad memory base: {first}",
534
+ line_num, fname) from None
535
+ offset = val
536
+
537
+ if len(parts) > 1:
538
+ second = parts[1]
539
+ try:
540
+ offset += self.parse_immediate(second)
541
+ except ValueError:
542
+ val = _eval_expr(second.lstrip('#'), symbols)
543
+ if val is None:
544
+ raise AssemblerError(line, f"Bad memory offset: {second}",
545
+ line_num, fname) from None
546
+ offset += val
547
+
548
+ return ('mem', base, offset)
codecin/cache.py ADDED
@@ -0,0 +1,71 @@
1
+ from collections import OrderedDict
2
+ from typing import Any, Dict, List
3
+
4
+
5
+ class CacheLine:
6
+ __slots__ = ('tag', 'valid', 'dirty')
7
+
8
+ def __init__(self, tag: int = 0, valid: bool = False, dirty: bool = False):
9
+ self.tag = tag
10
+ self.valid = valid
11
+ self.dirty = dirty
12
+
13
+
14
+ class Cache:
15
+ def __init__(self, size: int = 64, assoc: int = 4, line_size: int = 16):
16
+ self.size = max(assoc, size)
17
+ self.assoc = assoc
18
+ self.line_size = line_size
19
+ self.num_sets = max(1, self.size // assoc)
20
+ self._sets: List[OrderedDict] = [OrderedDict() for _ in range(self.num_sets)]
21
+ self.hits = 0
22
+ self.misses = 0
23
+ self.writes = 0
24
+
25
+ def _set_index(self, addr: int) -> int:
26
+ return (addr // self.line_size) % self.num_sets
27
+
28
+ def _tag(self, addr: int) -> int:
29
+ return addr // (self.line_size * self.num_sets)
30
+
31
+ def access(self, addr: int, is_write: bool = False) -> bool:
32
+ s = self._sets[self._set_index(addr)]
33
+ tag = self._tag(addr)
34
+ if tag in s:
35
+ s.move_to_end(tag)
36
+ if is_write:
37
+ s[tag] = True
38
+ self.writes += 1
39
+ self.hits += 1
40
+ return True
41
+ s[tag] = is_write
42
+ s.move_to_end(tag)
43
+ while len(s) > self.assoc:
44
+ s.popitem(last=False)
45
+ self.misses += 1
46
+ return False
47
+
48
+ def read(self, addr: int) -> bool:
49
+ return self.access(addr, False)
50
+
51
+ def write(self, addr: int) -> bool:
52
+ return self.access(addr, True)
53
+
54
+ def flush(self) -> None:
55
+ for s in self._sets:
56
+ s.clear()
57
+
58
+ def warmup(self, instructions) -> None:
59
+ for pc in range(len(instructions)):
60
+ self.access(pc * self.line_size)
61
+
62
+ def get_stats(self) -> Dict[str, Any]:
63
+ total = self.hits + self.misses
64
+ return {
65
+ 'hits': self.hits,
66
+ 'misses': self.misses,
67
+ 'hit_rate': self.hits / total if total > 0 else 0.0,
68
+ 'miss_rate': self.misses / total if total > 0 else 0.0,
69
+ 'total_accesses': total,
70
+ 'dirty_writes': self.writes,
71
+ }