cpyte 4.3.1__tar.gz → 4.3.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (44) hide show
  1. {cpyte-4.3.1/source/cpyte.egg-info → cpyte-4.3.2}/PKG-INFO +1 -1
  2. {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte/__init__.py +1 -1
  3. {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte/bytecoding.py +10 -19
  4. {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte/compiling.py +21 -16
  5. cpyte-4.3.2/source/cpyte/elf2sef.py +832 -0
  6. cpyte-4.3.2/source/cpyte/mksef.py +134 -0
  7. {cpyte-4.3.1 → cpyte-4.3.2/source/cpyte.egg-info}/PKG-INFO +1 -1
  8. {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte.egg-info/SOURCES.txt +2 -0
  9. {cpyte-4.3.1 → cpyte-4.3.2}/MANIFEST.in +0 -0
  10. {cpyte-4.3.1 → cpyte-4.3.2}/pyproject.toml +0 -0
  11. {cpyte-4.3.1 → cpyte-4.3.2}/readme.md +0 -0
  12. {cpyte-4.3.1 → cpyte-4.3.2}/setup.cfg +0 -0
  13. {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte/__main__.py +0 -0
  14. {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte/_bignum_bc.py +0 -0
  15. {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte/_runtime_bc.py +0 -0
  16. {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte/astparse.py +0 -0
  17. {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte/bignum.c +0 -0
  18. {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte/clib.py +0 -0
  19. {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte/extension_hooks.py +0 -0
  20. {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte/formatter.py +0 -0
  21. {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte/gc_runtime.c +0 -0
  22. {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte/generate_bc.py +0 -0
  23. {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte/lexar.py +0 -0
  24. {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte/linker.py +0 -0
  25. {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte/lsp_server.py +0 -0
  26. {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte/mainpie.py +0 -0
  27. {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte/optimizations.py +0 -0
  28. {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte/package_manifest.py +0 -0
  29. {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte/runtime.c +0 -0
  30. {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte/runtime_scorpion.c +0 -0
  31. {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte/sef.py +0 -0
  32. {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte/semantic_analasis.py +0 -0
  33. {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte/ugc.h +0 -0
  34. {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte/ui.py +0 -0
  35. {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte/update_check.py +0 -0
  36. {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte/winjit_patch.py +0 -0
  37. {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte/winjit_stubs.c +0 -0
  38. {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte.egg-info/dependency_links.txt +0 -0
  39. {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte.egg-info/entry_points.txt +0 -0
  40. {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte.egg-info/requires.txt +0 -0
  41. {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte.egg-info/top_level.txt +0 -0
  42. {cpyte-4.3.1 → cpyte-4.3.2}/test/test_bignum_jit.py +0 -0
  43. {cpyte-4.3.1 → cpyte-4.3.2}/test/test_fuzz_del_has.py +0 -0
  44. {cpyte-4.3.1 → cpyte-4.3.2}/test/test_sef_tools.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: cpyte
3
- Version: 4.3.1
3
+ Version: 4.3.2
4
4
  Summary: The Cpyte programming language compiler
5
5
  Author: Hoang Duy Tung
6
6
  License: MIT
@@ -1,4 +1,4 @@
1
- __version__ = "4.3.1"
1
+ __version__ = "4.3.2"
2
2
 
3
3
  # ---------------------------------------------------------------------------
4
4
  # Public library API.
@@ -2,7 +2,7 @@ import hashlib
2
2
  import hmac
3
3
  import os
4
4
  from traceback import print_exception
5
- from typing import Any, Optional, Protocol
5
+ from typing import Any, Protocol
6
6
 
7
7
  from llvmlite import binding, ir
8
8
  from llvmlite.ir import instructions
@@ -253,9 +253,7 @@ class LLVM:
253
253
  t = _bc_array_norm(t)
254
254
  if t == "int":
255
255
  result = ir.IntType(32)
256
- elif t == "int64":
257
- result = ir.IntType(64)
258
- elif t == "uint64":
256
+ elif t == "int64" or t == "uint64":
259
257
  result = ir.IntType(64)
260
258
  elif t == "size_t":
261
259
  # Pointer-sized unsigned integer (i64 on every 64-bit cpyte target,
@@ -273,11 +271,7 @@ class LLVM:
273
271
  result = ir.PointerType(ir.IntType(8))
274
272
  elif t == "char":
275
273
  result = ir.IntType(8)
276
- elif t == "void*":
277
- result = ir.PointerType(ir.IntType(8))
278
- elif t == "big":
279
- result = ir.PointerType(ir.IntType(8))
280
- elif t == "ubig":
274
+ elif t == "void*" or t == "big" or t == "ubig":
281
275
  result = ir.PointerType(ir.IntType(8))
282
276
  elif t == "dynamic":
283
277
  # A runtime-typed value: (kind, data) tag pair. Mirrors DynValue.
@@ -288,10 +282,7 @@ class LLVM:
288
282
  elif t.endswith("[]"):
289
283
  base = self.llvm_type(t[:-2])
290
284
  result = ir.PointerType(base)
291
- elif t.endswith("*"):
292
- base = self.llvm_type(t[:-1])
293
- result = ir.PointerType(base)
294
- elif t.endswith("&"):
285
+ elif t.endswith("*") or t.endswith("&"):
295
286
  base = self.llvm_type(t[:-1])
296
287
  result = ir.PointerType(base)
297
288
  elif t in self.structs:
@@ -1656,7 +1647,7 @@ class LLVM:
1656
1647
  for table in (self.locals, self.local_types, self.ssa_values):
1657
1648
  table.pop(name, None)
1658
1649
  self.const_vars.pop(name, None)
1659
- return None
1650
+ return
1660
1651
  if kind == "dynamic":
1661
1652
  name = target.name
1662
1653
  if name in self.locals and getattr(
@@ -1679,15 +1670,15 @@ class LLVM:
1679
1670
  ir.Constant(_i64, 0),
1680
1671
  ],
1681
1672
  )
1682
- return None
1673
+ return
1683
1674
  if kind == "heap":
1684
1675
  self._emit_builtin_free(target)
1685
- return None
1676
+ return
1686
1677
  if kind == "slot":
1687
1678
  slot = self._emit_lvalue(target)
1688
1679
  if isinstance(slot.type, ir.PointerType) and slot.type.pointee is not None:
1689
1680
  self.builder.store(ir.Constant(slot.type.pointee, None), slot)
1690
- return None
1681
+ return
1691
1682
  raise Exception(
1692
1683
  f"unhandled del_kind {kind!r} at L{node._token.line}:{node._token.column}"
1693
1684
  )
@@ -6409,7 +6400,7 @@ class LLVM:
6409
6400
  slot = self._emit_lvalue(target)
6410
6401
  cur = self.builder.load(slot, "append.cur")
6411
6402
 
6412
- elem_t = arr_t[:-2] if arr_t.endswith("[]") else arr_t
6403
+ elem_t = arr_t.removesuffix("[]")
6413
6404
  if arr_t == "dynamic":
6414
6405
  # Element slot holds a DynValue; its i64 data is the inner list ptr.
6415
6406
  elem_llvm = _DynValue
@@ -6778,7 +6769,7 @@ class LLVM:
6778
6769
  def _is_const_var(self, name: str) -> bool:
6779
6770
  return name in self._const_prop
6780
6771
 
6781
- def _const_var_value(self, name: str) -> Optional[int]:
6772
+ def _const_var_value(self, name: str) -> int | None:
6782
6773
  return self._const_prop.get(name)
6783
6774
 
6784
6775
  def _try_unroll_counted_loop(self, node: While, pending_ivs: dict) -> bool:
@@ -1806,6 +1806,25 @@ def _find_scorpion_tool(name, fallback):
1806
1806
  return fallback
1807
1807
 
1808
1808
 
1809
+ def _scorpion_tool(name):
1810
+ """Locate a bundled Scorpion converter script (elf2sef.py / mksef.py).
1811
+
1812
+ Prefers the copy shipped inside the cpyte package so the installed wheel
1813
+ works standalone; falls back to the WEW-scorpion sibling checkout for
1814
+ source-tree development.
1815
+ """
1816
+ bundled = os.path.join(os.path.dirname(__file__), name)
1817
+ if os.path.isfile(bundled):
1818
+ return bundled
1819
+ sibling = os.path.join(
1820
+ os.path.dirname(os.path.dirname(os.path.dirname(os.path.dirname(_RUNTIME_SCORPION_C)))),
1821
+ "WEW-scorpion",
1822
+ "tools" if name == "elf2sef.py" else "user",
1823
+ name,
1824
+ )
1825
+ return sibling
1826
+
1827
+
1809
1828
  def run_scorpion(
1810
1829
  module,
1811
1830
  output="program.sef",
@@ -1917,14 +1936,7 @@ def run_scorpion(
1917
1936
 
1918
1937
  if pic:
1919
1938
  # Dynamic SEF v2: relocation + import/export records
1920
- elf2sef = os.path.join(
1921
- os.path.dirname(os.path.dirname(_RUNTIME_SCORPION_C)),
1922
- "..",
1923
- "..",
1924
- "WEW-scorpion",
1925
- "tools",
1926
- "elf2sef.py",
1927
- )
1939
+ elf2sef = _scorpion_tool("elf2sef.py")
1928
1940
  cmd = [sys.executable, elf2sef, elf_file, output]
1929
1941
  if final_exports:
1930
1942
  for name in final_exports:
@@ -1935,14 +1947,7 @@ def run_scorpion(
1935
1947
  raise SystemExit(1)
1936
1948
  else:
1937
1949
  # Static SEF v1 via mksef.py
1938
- mksef = os.path.join(
1939
- os.path.dirname(os.path.dirname(_RUNTIME_SCORPION_C)),
1940
- "..",
1941
- "..",
1942
- "WEW-scorpion",
1943
- "user",
1944
- "mksef.py",
1945
- )
1950
+ mksef = _scorpion_tool("mksef.py")
1946
1951
  if not os.path.isfile(mksef):
1947
1952
  # Fallback: inline SEF generation using objdump/objcopy
1948
1953
  _elf_to_sef(elf_file, output, 0)
@@ -0,0 +1,832 @@
1
+ """
2
+ elf2sef.py — ELF (RV32, PIC, --emit-relocs) to Scorpion SEF v2.1 converter.
3
+
4
+ Produces a relocatable SEF image (docs/dynamic-linking.md):
5
+
6
+ * SEG_TEXT / SEG_DATA / SEG_BSS — flattened load image (linked at 0)
7
+ * SEG_RELOC — load-time relocations for absolute references that were
8
+ resolved at link time (R_RISCV_32 data words, lui-based
9
+ HI20/LO12 absolute pairs, .got/.got.plt slots)
10
+ * SEG_IMPORT — unresolved external symbols the loader must bind
11
+ * SEG_EXPORT — symbols other images may import (--export NAME)
12
+ * SEG_PLT — lazy-binding `ebreak` stubs (--lazy)
13
+ * SEG_VERDEF — version strings this image defines (--verdef VER)
14
+
15
+ An import record is emitted **per relocation site**, not per symbol name.
16
+ One symbol referenced from N places needs N records, because each site holds
17
+ its own copy of the address (a call site and a data word are separate words
18
+ in the image). Keying records by name binds only the first site and leaves
19
+ the rest pointing at whatever the linker left behind.
20
+
21
+ Link with: -fPIC ... -Wl,-q --unresolved-symbols=ignore-all --no-relax
22
+
23
+ Usage:
24
+ elf2sef.py [--lazy] [--weak NAME]... [--require NAME=VER]...
25
+ [--verdef VER]... [--no-versym]
26
+ [--export NAME]... [--export-file FILE] [--scope NAME]
27
+ <input.elf> <output.sef>
28
+ """
29
+
30
+ import struct
31
+ import sys
32
+
33
+ SEF_MAGIC = 0x00464553
34
+
35
+ SEG_TEXT = 0
36
+ SEG_DATA = 1
37
+ SEG_BSS = 2
38
+ SEG_RELOC = 3
39
+ SEG_IMPORT = 4
40
+ SEG_EXPORT = 5
41
+ SEG_PLT = 6
42
+ SEG_VERDEF = 7
43
+
44
+ SEF_MAX_SEGMENTS = 8
45
+
46
+ # There is deliberately no privilege flag: a SEF image never declares what it
47
+ # wants to become. The kernel's boot path grants manager privilege to the one
48
+ # image it finds bundled in flash; everything else is user code.
49
+ SEF_FLAG_DYNAMIC = 0x0002
50
+ SEF_FLAG_LAZY = 0x0004
51
+ SEF_FLAG_VERSYM = 0x0008
52
+
53
+ SEF_R_RELATIVE = 0
54
+ SEF_R_HI20 = 1
55
+ SEF_R_LO12I = 2
56
+ SEF_R_LO12S = 3
57
+ SEF_R_CALL = 4
58
+ SEF_R_LAZY_CALL = 5
59
+ SEF_R_PCREL_HI20 = 6
60
+ SEF_R_PCREL_LO12I = 7
61
+
62
+ SEF_IMPORT_WEAK = 0x00000001
63
+
64
+ SEF_PLT_STUB_SIZE = 8
65
+ SCOPE_SEP = "::"
66
+ VER_SEP = "@"
67
+
68
+ # ELF constants
69
+ SHT_NOBITS = 8
70
+ SHT_RELA = 4
71
+ SHT_SYMTAB = 2
72
+ SHF_ALLOC = 0x2
73
+ SHF_WRITE = 0x1
74
+ SHN_UNDEF = 0
75
+ SHN_ABS = 0xFFF1
76
+ SHN_COMMON = 0xFFF2
77
+ SHN_XINDEX = 0xFFFF
78
+
79
+ STT_SECTION = 3
80
+
81
+ # RISC-V relocation types (psABI)
82
+ R_RISCV_32 = 1
83
+ R_RISCV_HI20 = 26
84
+ R_RISCV_LO12_I = 27
85
+ R_RISCV_LO12_S = 28
86
+ R_RISCV_CALL = 18
87
+ R_RISCV_CALL_PLT = 19
88
+ R_RISCV_GOT_HI20 = 20
89
+ R_RISCV_GOT_LO12 = 23
90
+ R_RISCV_RELAX = 51
91
+
92
+ # relocations whose value is an absolute address; everything else is
93
+ # PC-relative / branch / relax / debug and is left alone at load time
94
+ ABSOLUTE_RELOCS = {
95
+ R_RISCV_32: SEF_R_RELATIVE,
96
+ R_RISCV_HI20: SEF_R_HI20,
97
+ R_RISCV_LO12_I: SEF_R_LO12I,
98
+ R_RISCV_LO12_S: SEF_R_LO12S,
99
+ R_RISCV_CALL: SEF_R_CALL,
100
+ R_RISCV_CALL_PLT: SEF_R_CALL,
101
+ }
102
+
103
+ # RISC-V encodings we need to recognise or synthesise.
104
+ OP_AUIPC = 0x17
105
+ OP_JALR = 0x67
106
+ OP_LW = 0x03
107
+ OP_LD = 0x03
108
+ INSN_EBREAK = 0x00100073
109
+
110
+ MAPPED_MAX = 0x00100000 # 1 MiB sanity bound for the flat image
111
+
112
+
113
+ def ver_hash(text):
114
+ """FNV-1a-32 of a version string. Must agree with sef_ver_hash() in
115
+ loader.c and scorpion_ver_hash() in abi/scorpion.h."""
116
+ h = 2166136261
117
+ for byte in text.encode("latin1"):
118
+ h = ((h ^ byte) * 16777619) & 0xFFFFFFFF
119
+ return h
120
+
121
+
122
+ def split_name(name):
123
+ """Split "[scope::]symbol[@version]" into (scope, symbol, version).
124
+
125
+ Only the parts we need hashes or sanity checks for are returned; the
126
+ record always stores `name` verbatim so both the inline spelling and the
127
+ out-of-band hash are available to the loader.
128
+ """
129
+ scope, sym, ver = "", name, ""
130
+ at = name.rfind(VER_SEP)
131
+ if at > 0:
132
+ sym, ver = name[:at], name[at + 1 :]
133
+ sep = sym.find(SCOPE_SEP)
134
+ if sep >= 0:
135
+ scope, sym = sym[:sep], sym[sep + 2 :]
136
+ return scope, sym, ver
137
+
138
+
139
+ def emit_call_pair(insn0, insn1, target, site):
140
+ """Rewrite an auipc+jalr pair at `site` to branch to `target`.
141
+
142
+ `disp` is the distance the linker would have encoded for a normal
143
+ PC-relative call, and the split is the same round-to-nearest-4K the
144
+ RISC-V psABI specifies, so this produces exactly the instruction pair
145
+ R_RISCV_CALL_PLT would have emitted.
146
+ """
147
+ rd = (insn0 >> 7) & 0x1F
148
+ disp = target - site
149
+ hi = (disp + 0x800) >> 12
150
+ lo = disp - (hi << 12)
151
+ new0 = ((hi & 0xFFFFF) << 12) | (rd << 7) | OP_AUIPC
152
+ new1 = ((lo & 0xFFF) << 20) | (rd << 7) | OP_JALR
153
+ return new0, new1
154
+
155
+
156
+ class Section:
157
+ def __init__(self, name, type_, flags, addr, offset, size, link, info, entsize):
158
+ self.name = name
159
+ self.type = type_
160
+ self.flags = flags
161
+ self.addr = addr
162
+ self.offset = offset
163
+ self.size = size
164
+ self.link = link
165
+ self.info = info
166
+ self.entsize = entsize
167
+
168
+
169
+ class Symbol:
170
+ def __init__(self, name, value, size, info, shndx):
171
+ self.name = name
172
+ self.value = value
173
+ self.size = size
174
+ self.info = info
175
+ self.shndx = shndx
176
+
177
+ @property
178
+ def is_defined(self):
179
+ return self.shndx != SHN_UNDEF and self.shndx != SHN_ABS
180
+
181
+ @property
182
+ def is_section(self):
183
+ return (self.info & 0xF) == STT_SECTION
184
+
185
+
186
+ def parse_elf(path):
187
+ with open(path, "rb") as f:
188
+ data = f.read()
189
+
190
+ if data[:4] != b"\x7fELF":
191
+ sys.exit(f"error: {path}: not an ELF file")
192
+ if data[4] != 1: # ELFCLASS32
193
+ sys.exit(f"error: {path}: not a 32-bit ELF")
194
+
195
+ ei_data = data[5]
196
+ endian = "<" if ei_data == 1 else ">"
197
+ if ei_data == 2:
198
+ sys.exit(f"error: {path}: big-endian ELF unsupported")
199
+
200
+ (
201
+ e_type,
202
+ e_machine,
203
+ e_version,
204
+ e_entry,
205
+ e_phoff,
206
+ e_shoff,
207
+ e_flags,
208
+ e_ehsize,
209
+ e_phentsize,
210
+ e_phnum,
211
+ e_shentsize,
212
+ e_shnum,
213
+ e_shstrndx,
214
+ ) = struct.unpack_from(endian + "HHIIIIIHHHHHH", data, 16)
215
+
216
+ if e_machine != 0xF3:
217
+ sys.exit(f"error: {path}: not RISC-V (machine={e_machine:#x})")
218
+
219
+ if e_shnum == 0 or e_shentsize != 40:
220
+ sys.exit(f"error: {path}: bad section headers")
221
+
222
+ shdr = []
223
+ for i in range(e_shnum):
224
+ shdr.append(struct.unpack_from(endian + "IIIIIIIIII", data, e_shoff + i * 40))
225
+
226
+ # section name string table
227
+ shstr = shdr[e_shstrndx]
228
+ shstr_data = data[shstr[4] : shstr[4] + shstr[5]]
229
+
230
+ def cstr(buf, off):
231
+ if off >= len(buf):
232
+ return ""
233
+ end = buf.find(b"\x00", off)
234
+ if end < 0:
235
+ return ""
236
+ return buf[off:end].decode("latin1")
237
+
238
+ sections = {}
239
+ by_index = []
240
+ for i, s in enumerate(shdr):
241
+ name = cstr(shstr_data, s[0])
242
+ sec = Section(name, s[1], s[2], s[3], s[4], s[5], s[6], s[7], s[9])
243
+ sections[name] = sec
244
+ by_index.append(sec)
245
+
246
+ def sym_by_index(idx):
247
+ return symbols[idx] if idx < len(symbols) else None
248
+
249
+ def section_by_index(idx):
250
+ return by_index[idx] if idx < len(by_index) else None
251
+
252
+ # symbol table (symbols[i] is ELF symbol index i; [0] is the null symbol)
253
+ symbols = []
254
+ if ".symtab" in sections:
255
+ st = sections[".symtab"]
256
+ strtab = section_by_index(st.link)
257
+ str_data = b""
258
+ if strtab is not None:
259
+ str_data = data[strtab.offset : strtab.offset + strtab.size]
260
+ for i in range(st.size // st.entsize):
261
+ st_name, st_value, st_size, st_info, st_other, st_shndx = (
262
+ struct.unpack_from(endian + "IIIBBH", data, st.offset + i * st.entsize)
263
+ )
264
+ name = cstr(str_data, st_name)
265
+ symbols.append(Symbol(name, st_value, st_size, st_info, st_shndx))
266
+
267
+ # alloc sections in address order
268
+ alloc = [s for s in sections.values() if (s.flags & SHF_ALLOC) and s.size > 0]
269
+ alloc.sort(key=lambda s: (s.addr, s.size))
270
+
271
+ if not alloc:
272
+ sys.exit(f"error: {path}: no allocated sections")
273
+
274
+ base = alloc[0].addr
275
+ mapped_end = max(s.addr + s.size for s in alloc)
276
+
277
+ if mapped_end - base > MAPPED_MAX:
278
+ sys.exit(f"error: {path}: image too large ({mapped_end - base:#x} bytes)")
279
+
280
+ flat = bytearray(mapped_end - base)
281
+ for s in alloc:
282
+ if s.type != SHT_NOBITS:
283
+ flat[s.addr - base : s.addr - base + s.size] = data[
284
+ s.offset : s.offset + s.size
285
+ ]
286
+
287
+ # relocations: site -> (type, sym_index, addend)
288
+ relocs = []
289
+ for sec in sections.values():
290
+ if sec.type != SHT_RELA or sec.link == 0:
291
+ continue
292
+ target = by_index[sec.info] if sec.info < len(by_index) else None
293
+ if target is None or (target.flags & SHF_ALLOC) == 0:
294
+ continue # .rela.debug* etc.: no memory image
295
+ for i in range(sec.size // sec.entsize):
296
+ r_offset, r_info, r_addend = struct.unpack_from(
297
+ endian + "IIi", data, sec.offset + i * sec.entsize
298
+ )
299
+ r_sym = r_info >> 8
300
+ r_type = r_info & 0xFF
301
+ relocs.append((r_offset, r_type, r_sym, r_addend))
302
+
303
+ return {
304
+ "entry": e_entry,
305
+ "base": base,
306
+ "mapped_end": mapped_end,
307
+ "flat": flat,
308
+ "sections": sections,
309
+ "by_index": by_index,
310
+ "symbols": symbols,
311
+ "sym_by_index": sym_by_index,
312
+ "section_by_index": section_by_index,
313
+ "relocs": relocs,
314
+ }
315
+
316
+
317
+ def is_absolute_pair(elf, sym_idx):
318
+ """True if the LO12 reloc's symbol points at a `lui` (absolute pair)
319
+ rather than an `auipc` (PC-relative pair)."""
320
+ sym = elf["sym_by_index"](sym_idx)
321
+ if sym is None or not sym.is_defined:
322
+ return False
323
+ off = sym.value - elf["base"]
324
+ flat = elf["flat"]
325
+ if off + 4 > len(flat):
326
+ return False
327
+ insn = struct.unpack_from("<I", flat, off)[0]
328
+ return (insn & 0x7F) == 0x37 # lui
329
+
330
+
331
+ def sign_extend(value, bits):
332
+ if value & (1 << (bits - 1)):
333
+ return value - (1 << bits)
334
+ return value
335
+
336
+
337
+ def got_slot_of(elf, site):
338
+ """Decode the GOT entry address a linker-emitted auipc+lo12 pair points at.
339
+
340
+ `-fPIC` reaches an *external* data symbol through the GOT, so the pair is
341
+ `auipc rd, hi; lw rd, lo(rd)` and the load-base-dependent part is the GOT
342
+ address, not the symbol itself. The linker has already resolved the pair,
343
+ so rather than re-deriving the psABI addend convention we read the
344
+ displacement back out of the two encodings and check that it lands in a
345
+ real GOT section. A pair that does not is not ours to rewrite.
346
+ """
347
+ flat = elf["flat"]
348
+ if site + 8 > len(flat):
349
+ return None
350
+ insn0 = struct.unpack_from("<I", flat, site)[0]
351
+ insn1 = struct.unpack_from("<I", flat, site + 4)[0]
352
+ if (insn0 & 0x7F) != OP_AUIPC:
353
+ return None
354
+ if (insn1 & 0x7F) not in (OP_LW, OP_LD):
355
+ return None
356
+
357
+ hi = sign_extend((insn0 >> 12) & 0xFFFFF, 20)
358
+ lo = sign_extend((insn1 >> 20) & 0xFFF, 12)
359
+ disp = site + (hi << 12) + lo
360
+ for name in (".got", ".got.plt"):
361
+ sec = elf["sections"].get(name)
362
+ if sec is None or (sec.flags & SHF_ALLOC) == 0:
363
+ continue
364
+ start = sec.addr - elf["base"]
365
+ if start <= disp < start + sec.size:
366
+ return disp
367
+ return None
368
+
369
+
370
+ class Import:
371
+ """One unresolved reference at one image offset."""
372
+
373
+ __slots__ = ("rtype", "site", "name", "ver_hash", "flags", "plt")
374
+
375
+ def __init__(self, rtype, site, name):
376
+ self.rtype = rtype
377
+ self.site = site
378
+ self.name = name
379
+ self.ver_hash = 0
380
+ self.flags = 0
381
+ self.plt = 0
382
+
383
+
384
+ def main(argv):
385
+ exports = []
386
+ export_file = None
387
+ weak = []
388
+ require = {} # import name -> version
389
+ verdefs = []
390
+ scope = ""
391
+ lazy = False
392
+ versym = True
393
+ positionals = []
394
+ args = list(argv)
395
+
396
+ i = 0
397
+ while i < len(args):
398
+ opt = args[i]
399
+ if opt == "--export":
400
+ exports.append(args[i + 1])
401
+ i += 2
402
+ elif opt == "--export-file":
403
+ export_file = args[i + 1]
404
+ i += 2
405
+ elif opt == "--weak":
406
+ weak.append(args[i + 1])
407
+ i += 2
408
+ elif opt == "--verdef":
409
+ verdefs.append(args[i + 1])
410
+ i += 2
411
+ elif opt == "--scope":
412
+ scope = args[i + 1]
413
+ i += 2
414
+ elif opt == "--require":
415
+ if "=" not in args[i + 1]:
416
+ sys.exit(f"error: --require wants NAME=VERSION, got {args[i+1]!r}")
417
+ name, ver = args[i + 1].split("=", 1)
418
+ require[name] = ver
419
+ i += 2
420
+ elif opt == "--lazy":
421
+ lazy = True
422
+ i += 1
423
+ elif opt == "--no-lazy":
424
+ lazy = False
425
+ i += 1
426
+ elif opt == "--no-versym":
427
+ versym = False
428
+ i += 1
429
+ elif opt.startswith("--"):
430
+ sys.exit(f"error: unknown option {opt}")
431
+ else:
432
+ positionals.append(opt)
433
+ i += 1
434
+
435
+ if len(positionals) != 2:
436
+ sys.exit(__doc__)
437
+
438
+ elf_path, sef_path = positionals
439
+ elf = parse_elf(elf_path)
440
+ flat = elf["flat"]
441
+ base = elf["base"]
442
+ sections = elf["sections"]
443
+
444
+ if export_file:
445
+ with open(export_file) as f:
446
+ exports.extend(line.strip() for line in f if line.strip())
447
+
448
+ # --- map / segment layout (linked at 0: vaddr == image offset) ---
449
+ text = sections.get(".text")
450
+ bss = sections.get(".bss")
451
+
452
+ if text is None or (text.flags & SHF_ALLOC) == 0:
453
+ sys.exit(f"error: {elf_path}: no allocated .text section")
454
+
455
+ text_vaddr = text.addr - base
456
+ text_size = text.size
457
+
458
+ # A zero-sized .bss still has an address, and using it as the end of the
459
+ # data segment would describe a hole the flat image never covers: the
460
+ # segment would claim bytes that `flat` does not have, silently truncating
461
+ # everything after it. Only a real .bss ends the data segment.
462
+ has_bss = (
463
+ bss is not None
464
+ and (bss.flags & SHF_ALLOC) != 0
465
+ and bss.size > 0
466
+ and bss.addr >= text.addr + text.size
467
+ )
468
+
469
+ if has_bss:
470
+ data_vaddr = text_vaddr + text_size
471
+ data_size = (bss.addr - base) - data_vaddr
472
+ bss_vaddr = bss.addr - base
473
+ bss_size = bss.size
474
+ else:
475
+ data_vaddr = text_vaddr + text_size
476
+ data_size = (elf["mapped_end"] - base) - data_vaddr
477
+ bss_vaddr = 0
478
+ bss_size = 0
479
+
480
+ if data_size < 0 or bss_size < 0:
481
+ sys.exit(f"error: {elf_path}: unexpected section order")
482
+
483
+ # Every mapped segment has to be backed by real bytes, or the header
484
+ # describes an image the file cannot supply.
485
+ for label, vaddr, size in (
486
+ ("SEG_TEXT", text_vaddr, text_size),
487
+ ("SEG_DATA", data_vaddr, data_size),
488
+ ("SEG_BSS", bss_vaddr, bss_size),
489
+ ):
490
+ if size and vaddr + size > len(flat):
491
+ sys.exit(
492
+ f"error: {elf_path}: {label} [0x{vaddr:x},0x{vaddr+size:x}) "
493
+ f"is past the end of the linked image (0x{len(flat):x})"
494
+ )
495
+
496
+ image_end = bss_vaddr + bss_size if bss_size else data_vaddr + data_size
497
+
498
+ # --- process relocations ---
499
+ reloc_records = [] # (type, offset, value)
500
+ imports = [] # one Import per relocation site
501
+ sym_by_index = elf["sym_by_index"]
502
+ section_by_index = elf["section_by_index"]
503
+ mapped = len(flat)
504
+
505
+ for r_offset, r_type, r_sym, r_addend in elf["relocs"]:
506
+ if r_type == R_RISCV_RELAX:
507
+ continue
508
+ if r_type not in ABSOLUTE_RELOCS and r_type != R_RISCV_GOT_HI20:
509
+ continue
510
+ if r_offset < base or r_offset - base + 4 > mapped:
511
+ continue # site outside the load image (debug etc.)
512
+
513
+ site = r_offset - base
514
+
515
+ if r_type == R_RISCV_LO12_I or r_type == R_RISCV_LO12_S:
516
+ if not is_absolute_pair(elf, r_sym):
517
+ continue # PC-relative pair; no load-time fixup needed
518
+
519
+ sym = sym_by_index(r_sym)
520
+ if sym is None:
521
+ continue
522
+
523
+ if not sym.is_defined:
524
+ # Unresolved external. Every site gets its own record: the same
525
+ # symbol called from N places occupies N words in the image, and
526
+ # each word has to be written for the call to reach the target.
527
+ if not sym.name:
528
+ continue
529
+ imp = Import(ABSOLUTE_RELOCS.get(r_type, SEF_R_RELATIVE), site, sym.name)
530
+ _, _, inline_ver = split_name(sym.name)
531
+ if inline_ver:
532
+ imp.ver_hash = ver_hash(inline_ver)
533
+ elif sym.name in require:
534
+ imp.ver_hash = ver_hash(require[sym.name])
535
+ if sym.name in weak:
536
+ imp.flags |= SEF_IMPORT_WEAK
537
+
538
+ if r_type == R_RISCV_GOT_HI20:
539
+ # External data binds through the GOT: the word the loader
540
+ # must write is the GOT slot, and the auipc/lo12 pair needs
541
+ # fixing up to reach it. That makes this import a plain
542
+ # 32-bit word write at the slot, not a patch of `site`.
543
+ got = got_slot_of(elf, site)
544
+ if got is None:
545
+ sys.exit(
546
+ f"error: {elf_path}: cannot resolve the GOT entry for "
547
+ f"{sym.name!r} referenced at 0x{site:x}"
548
+ )
549
+ imp.rtype = SEF_R_RELATIVE
550
+ imp.site = got
551
+ imports.append(imp)
552
+ # Two load-time relocs, one per half, each PC-relative to
553
+ # the auipc so the pair computes got_entry at the real base.
554
+ reloc_records.append((SEF_R_PCREL_HI20, site, got))
555
+ reloc_records.append((SEF_R_PCREL_LO12I, site + 4, got))
556
+ else:
557
+ imports.append(imp)
558
+ continue
559
+
560
+ if r_type == R_RISCV_GOT_HI20:
561
+ # GOT entry for a symbol this image defines: make the pair
562
+ # load-base relative, and let the .got scan below turn the slot
563
+ # into an absolute relocation.
564
+ got = got_slot_of(elf, site)
565
+ if got is None:
566
+ continue
567
+ reloc_records.append((SEF_R_PCREL_HI20, site, got))
568
+ reloc_records.append((SEF_R_PCREL_LO12I, site + 4, got))
569
+ continue
570
+
571
+ sec = section_by_index(sym.shndx)
572
+ if sec is not None and (sec.flags & SHF_ALLOC) == 0:
573
+ continue # e.g. debug symbol
574
+
575
+ value = sym.value + r_addend
576
+ reloc_records.append((ABSOLUTE_RELOCS[r_type], site, value))
577
+
578
+ # Two relocations can land on the same offset only if the linker emitted
579
+ # duplicates; collapse them so the loader never sees a double write.
580
+ imports.sort(key=lambda imp: (imp.site, imp.rtype))
581
+ deduped = []
582
+ for imp in imports:
583
+ if deduped and deduped[-1].site == imp.site:
584
+ continue
585
+ deduped.append(imp)
586
+ imports = deduped
587
+
588
+ # --- lazy PLT ---
589
+ #
590
+ # Each eligible call site gets a `ebreak` stub. The call site is
591
+ # rewritten to reach the stub; the first execution traps, the handler
592
+ # resolves the symbol and patches the site to branch straight at the
593
+ # target, so the stub is entered at most once.
594
+ plt_vaddr = 0
595
+ plt_size = 0
596
+ plt_bytes = b""
597
+
598
+ if lazy:
599
+ stubs = []
600
+ for imp in imports:
601
+ if imp.rtype != SEF_R_CALL:
602
+ continue
603
+ if imp.site + 8 > mapped:
604
+ continue
605
+ insn0 = struct.unpack_from("<I", flat, imp.site)[0]
606
+ insn1 = struct.unpack_from("<I", flat, imp.site + 4)[0]
607
+ if (insn0 & 0x7F) != OP_AUIPC or (insn1 & 0x7F) != OP_JALR:
608
+ continue # not the auipc+jalr shape a stub can intercept
609
+ stubs.append((imp, insn0, insn1))
610
+
611
+ if stubs:
612
+ plt_vaddr = (image_end + SEF_PLT_STUB_SIZE - 1) & ~(SEF_PLT_STUB_SIZE - 1)
613
+ plt_size = len(stubs) * SEF_PLT_STUB_SIZE
614
+ if plt_vaddr + plt_size > MAPPED_MAX:
615
+ sys.exit(f"error: {elf_path}: lazy PLT does not fit the image")
616
+
617
+ body = bytearray()
618
+ for index, (imp, insn0, insn1) in enumerate(stubs):
619
+ imp.rtype = SEF_R_LAZY_CALL
620
+ imp.plt = index * SEF_PLT_STUB_SIZE
621
+ new0, new1 = emit_call_pair(insn0, insn1, plt_vaddr + imp.plt, imp.site)
622
+ struct.pack_into("<I", flat, imp.site, new0)
623
+ struct.pack_into("<I", flat, imp.site + 4, new1)
624
+ # Both words are `ebreak`: the first traps, and if the handler
625
+ # ever failed to resolve, the second traps too rather than
626
+ # sliding into whatever follows as if it were code.
627
+ body += struct.pack("<II", INSN_EBREAK, INSN_EBREAK)
628
+ plt_bytes = bytes(body)
629
+
630
+ # The call relocations for these sites were queued above, before we
631
+ # knew they were lazy. The loader applies SEG_RELOC records at load
632
+ # time, so leaving them in would re-apply the *link-time* target
633
+ # and silently undo the stub rewrite above -- the call site would
634
+ # jump straight at the original PLT entry, never trap, and the
635
+ # lazy import would be marked bound while its stub is never
636
+ # reached. The import record alone now drives this site.
637
+ stub_sites = {imp.site for imp, _, _ in stubs}
638
+ reloc_records = [
639
+ rec for rec in reloc_records
640
+ if not (rec[0] == SEF_R_CALL and rec[1] in stub_sites)
641
+ ]
642
+
643
+ # --- scan .got/.got.plt for slots the linker resolved without a reloc ---
644
+ defined_values = set()
645
+ for sym in elf["symbols"]:
646
+ if sym is not None and sym.is_defined:
647
+ sec = section_by_index(sym.shndx)
648
+ if sec is not None and (sec.flags & SHF_ALLOC):
649
+ defined_values.add(sym.value)
650
+ if sym.is_section:
651
+ for delta in range(0, min(sec.size, 0x100), 4):
652
+ defined_values.add(sym.value + delta)
653
+
654
+ plt_sec_names = (".plt", ".plt.sec")
655
+ got_records = {}
656
+ for got_name in (".got", ".got.plt"):
657
+ got = sections.get(got_name)
658
+ if got is None or (got.flags & SHF_ALLOC) == 0:
659
+ continue
660
+ for off in range(0, got.size, 4):
661
+ slot_addr = got.addr + off
662
+ word = struct.unpack_from("<I", flat, slot_addr - base)[0]
663
+ if word == 0 or word == 0xFFFFFFFF:
664
+ continue
665
+ skip = False
666
+ for name in plt_sec_names:
667
+ sec = sections.get(name)
668
+ if sec is not None and sec.addr <= word < sec.addr + sec.size:
669
+ skip = True # PLT trampoline pointer (external call)
670
+ break
671
+ if skip:
672
+ continue
673
+ if word in defined_values:
674
+ got_records.setdefault(slot_addr - base, word)
675
+
676
+ for slot, word in got_records.items():
677
+ if not any(r[0] == SEF_R_RELATIVE and r[1] == slot for r in reloc_records):
678
+ reloc_records.append((SEF_R_RELATIVE, slot, word))
679
+
680
+ # --- exports ---
681
+ # Names are stored verbatim, so a caller may spell them "sym", "sym@ver",
682
+ # "scope::sym" or "scope::sym@ver" and the loader sees the same thing it
683
+ # would have seen in an import. ver_hash repeats the version out of band
684
+ # for producers that keep the two apart.
685
+ #
686
+ # --export takes either NAME (export the ELF symbol of that name) or
687
+ # SYMBOL=WIRE_NAME (export SYMBOL under a different published name, which
688
+ # is how a versioned alias like addone@2.0 is published from a plain
689
+ # `addone_v2` definition).
690
+ export_records = []
691
+ for raw in exports:
692
+ if "=" in raw:
693
+ elf_name, wire = raw.split("=", 1)
694
+ else:
695
+ elf_name, wire = raw, raw
696
+ # --scope applies to every name that does not already carry one,
697
+ # including versioned ones, so --export foo@1.0 under --scope util
698
+ # publishes "util::foo@1.0" rather than leaving it unscoped.
699
+ if scope and SCOPE_SEP not in wire.split(VER_SEP, 1)[0]:
700
+ wire = scope + SCOPE_SEP + wire
701
+
702
+ for sym in elf["symbols"]:
703
+ if sym is None or not sym.is_defined or sym.name != elf_name:
704
+ continue
705
+ _, _, inline_ver = split_name(wire)
706
+ export_records.append(
707
+ (sym.value, wire, ver_hash(inline_ver) if inline_ver else 0)
708
+ )
709
+ break
710
+ else:
711
+ sys.exit(f"error: {elf_path}: export {raw!r} not found")
712
+
713
+ # --- version definitions ---
714
+ # Anything named in an export's @ver tail is a version this image defines.
715
+ defined_vers = list(verdefs)
716
+ for _, name, _ in export_records:
717
+ _, _, inline_ver = split_name(name)
718
+ if inline_ver and inline_ver not in defined_vers:
719
+ defined_vers.append(inline_ver)
720
+
721
+ # --- build SEF ---
722
+ flags = 0
723
+ dynamic = bool(reloc_records or imports or export_records)
724
+ if dynamic:
725
+ flags |= SEF_FLAG_DYNAMIC
726
+ if plt_bytes:
727
+ flags |= SEF_FLAG_LAZY
728
+ if versym and (imports or export_records or defined_vers):
729
+ flags |= SEF_FLAG_VERSYM
730
+
731
+ segments = [] # (type, vaddr, size, data)
732
+ segments.append((SEG_TEXT, text_vaddr, text_size, None))
733
+ segments.append((SEG_DATA, data_vaddr, data_size, None))
734
+ if bss_size:
735
+ segments.append((SEG_BSS, bss_vaddr, bss_size, None))
736
+
737
+ reloc_data = b"".join(struct.pack("<III", t, o, v) for (t, o, v) in reloc_records)
738
+ if reloc_data:
739
+ segments.append((SEG_RELOC, 0, len(reloc_data), reloc_data))
740
+
741
+ import_data = b""
742
+ for imp in imports:
743
+ nb = imp.name.encode("latin1")
744
+ if versym:
745
+ import_data += struct.pack(
746
+ "<IIIIII",
747
+ imp.rtype,
748
+ imp.site,
749
+ imp.plt,
750
+ imp.ver_hash,
751
+ imp.flags,
752
+ len(nb),
753
+ )
754
+ else:
755
+ import_data += struct.pack("<III", imp.rtype, imp.site, len(nb))
756
+ import_data += nb
757
+ import_data += b"\x00" * ((4 - len(nb) % 4) % 4)
758
+ if import_data:
759
+ segments.append((SEG_IMPORT, 0, len(import_data), import_data))
760
+
761
+ export_data = b""
762
+ for value, name, vhash in export_records:
763
+ nb = name.encode("latin1")
764
+ if versym:
765
+ export_data += struct.pack("<III", value, vhash, len(nb))
766
+ else:
767
+ export_data += struct.pack("<II", value, len(nb))
768
+ export_data += nb
769
+ export_data += b"\x00" * ((4 - len(nb) % 4) % 4)
770
+ if export_data:
771
+ segments.append((SEG_EXPORT, 0, len(export_data), export_data))
772
+
773
+ if plt_bytes:
774
+ segments.append((SEG_PLT, plt_vaddr, plt_size, plt_bytes))
775
+
776
+ verdef_data = b""
777
+ for ver in defined_vers:
778
+ nb = ver.encode("latin1")
779
+ verdef_data += struct.pack("<I", len(nb)) + nb
780
+ verdef_data += b"\x00" * ((4 - len(nb) % 4) % 4)
781
+ if verdef_data:
782
+ segments.append((SEG_VERDEF, 0, len(verdef_data), verdef_data))
783
+
784
+ if len(segments) > SEF_MAX_SEGMENTS:
785
+ sys.exit(
786
+ f"error: {elf_path}: {len(segments)} segments exceeds "
787
+ f"SEF_MAX_SEGMENTS={SEF_MAX_SEGMENTS}"
788
+ )
789
+
790
+ out = bytearray()
791
+ out += struct.pack("<IIHH", SEF_MAGIC, elf["entry"], len(segments), flags)
792
+
793
+ dc = 12 + len(segments) * 16
794
+ body = bytearray()
795
+ for st, vaddr, size, payload in segments:
796
+ if payload is None:
797
+ start = dc
798
+ dc += size
799
+ out += struct.pack("<IIII", st, vaddr, size, start)
800
+ body += flat[vaddr : vaddr + size]
801
+ else:
802
+ out += struct.pack("<IIII", st, vaddr, size, dc)
803
+ dc += len(payload)
804
+ body += payload
805
+
806
+ out += body
807
+
808
+ with open(sef_path, "wb") as f:
809
+ f.write(out)
810
+
811
+ print(
812
+ f"Created {sef_path}: {len(out)} bytes, {len(segments)} segments, "
813
+ f"entry=0x{elf['entry']:x}, flags=0x{flags:x}"
814
+ )
815
+ print(
816
+ f" relocs={len(reloc_records)} imports={len(imports)} "
817
+ f"exports={len(export_records)}"
818
+ )
819
+ if plt_bytes:
820
+ print(f" lazy PLT: {len(plt_bytes) // SEF_PLT_STUB_SIZE} stubs at 0x{plt_vaddr:x}")
821
+ if defined_vers:
822
+ print(f" version defs: {', '.join(defined_vers)}")
823
+
824
+ # Import sites must be distinct: the whole point of the v2.1 record set is
825
+ # that every site binds, so a duplicate here is a tool bug worth failing on.
826
+ sites = [imp.site for imp in imports]
827
+ if len(sites) != len(set(sites)):
828
+ sys.exit(f"error: {elf_path}: internal error, duplicate import sites")
829
+
830
+
831
+ if __name__ == "__main__":
832
+ main(sys.argv[1:])
@@ -0,0 +1,134 @@
1
+ """
2
+ Scorpion SEF format builder — ELF to SEF conversion.
3
+
4
+ NOTE: This is a DEVELOPMENT CONVENIENCE TOOL and NOT the recommended route
5
+ for production use. It relies on objdump/readelf/objcopy to extract section
6
+ data and does not preserve ELF metadata beyond segment contents. For
7
+ production, author SEF binaries directly using the format spec in
8
+ docs/exec-format.md.
9
+ """
10
+
11
+ import os
12
+ import struct
13
+ import subprocess
14
+ import sys
15
+
16
+
17
+ def build(elf_path, sef_output, flags=0):
18
+ if not os.path.isfile(elf_path):
19
+ print(f"error: {elf_path}: not found", file=sys.stderr)
20
+ sys.exit(1)
21
+
22
+ sections = {}
23
+ result = subprocess.run(
24
+ ["riscv64-elf-objdump", "-h", elf_path], capture_output=True, text=True
25
+ )
26
+ if result.returncode != 0:
27
+ print(f"error: objdump failed on {elf_path}", file=sys.stderr)
28
+ sys.exit(1)
29
+
30
+ for line in result.stdout.split("\n"):
31
+ parts = line.split()
32
+ if len(parts) >= 4 and parts[0].isdigit():
33
+ name = parts[1]
34
+ if name in (".text", ".rodata", ".data", ".bss"):
35
+ sections[name] = int(parts[2], 16)
36
+
37
+ result = subprocess.run(
38
+ ["riscv64-elf-readelf", "-h", elf_path], capture_output=True, text=True
39
+ )
40
+ if result.returncode != 0:
41
+ print(f"error: readelf failed on {elf_path}", file=sys.stderr)
42
+ sys.exit(1)
43
+
44
+ entry = 0
45
+ for line in result.stdout.split("\n"):
46
+ if "Entry point address" in line:
47
+ entry = int(line.split(":")[1].strip(), 16)
48
+
49
+ bin_path = elf_path + ".bin"
50
+ result = subprocess.run(
51
+ ["riscv64-elf-objcopy", "-O", "binary", elf_path, bin_path],
52
+ capture_output=True,
53
+ text=True,
54
+ )
55
+ if result.returncode != 0:
56
+ print(f"error: objcopy failed on {elf_path}", file=sys.stderr)
57
+ sys.exit(1)
58
+
59
+ with open(bin_path, "rb") as f:
60
+ flat = f.read()
61
+ os.unlink(bin_path)
62
+
63
+ segments = []
64
+ off = 0
65
+ if ".text" in sections:
66
+ segments.append((0, off, sections[".text"]))
67
+ off += sections[".text"]
68
+ if ".rodata" in sections:
69
+ segments.append((1, off, sections[".rodata"]))
70
+ off += sections[".rodata"]
71
+ if ".data" in sections:
72
+ segments.append((1, off, sections[".data"]))
73
+ off += sections[".data"]
74
+ if ".bss" in sections:
75
+ segments.append((2, off, sections[".bss"]))
76
+
77
+ if not segments:
78
+ print(f"error: {elf_path}: no known sections found", file=sys.stderr)
79
+ sys.exit(1)
80
+
81
+ num = len(segments)
82
+ hdr = 12 + num * 16
83
+
84
+ out = bytearray()
85
+ out += struct.pack("<IIHH", 0x00464553, entry, num, flags)
86
+ dc = hdr
87
+ for st, sv, ss in segments:
88
+ out += struct.pack("<IIII", st, sv, ss, dc)
89
+ dc += ss
90
+ out += flat
91
+
92
+ with open(sef_output, "wb") as f:
93
+ f.write(out)
94
+
95
+ print(
96
+ f"Created {sef_output}: {len(out)} bytes, {num} segments, "
97
+ f"entry=0x{entry:x}, flags=0x{flags:x}"
98
+ )
99
+
100
+
101
+ def build_h(sef_path, h_output):
102
+ if not os.path.isfile(sef_path):
103
+ print(f"error: {sef_path}: not found", file=sys.stderr)
104
+ sys.exit(1)
105
+
106
+ with open(sef_path, "rb") as f:
107
+ data = f.read()
108
+ base = os.path.splitext(os.path.basename(sef_path))[0]
109
+ guard = f"SCORPION_{base.upper()}_SEF_H"
110
+ define = f"{base.upper()}_SEF_SIZE"
111
+ var = f"{base}_sef"
112
+ with open(h_output, "w") as f:
113
+ f.write(f"#ifndef {guard}\n#define {guard}\n\n")
114
+ f.write(f"#define {define} {len(data)}\n\n")
115
+ f.write(f"static const unsigned char {var}[] = {{\n")
116
+ for i in range(0, len(data), 12):
117
+ chunk = data[i : i + 12]
118
+ f.write(" " + ", ".join(f"0x{b:02x}" for b in chunk) + ",\n")
119
+ f.write("};\n\n#endif\n")
120
+ print(f"Generated {h_output} from {sef_path}")
121
+
122
+
123
+ if __name__ == "__main__":
124
+ args = sys.argv[1:]
125
+ if len(args) < 2:
126
+ print(
127
+ f"usage: {sys.argv[0]} <input.elf> <output.sef>",
128
+ file=sys.stderr,
129
+ )
130
+ sys.exit(1)
131
+ elf_path, sef_output = args[0], args[1]
132
+ # No --flags: privilege is not encoded in the image. Whether an image runs
133
+ # as the manager is decided by the kernel boot path that spawns it.
134
+ build(elf_path, sef_output, 0)
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: cpyte
3
- Version: 4.3.1
3
+ Version: 4.3.2
4
4
  Summary: The Cpyte programming language compiler
5
5
  Author: Hoang Duy Tung
6
6
  License: MIT
@@ -10,6 +10,7 @@ source/cpyte/bignum.c
10
10
  source/cpyte/bytecoding.py
11
11
  source/cpyte/clib.py
12
12
  source/cpyte/compiling.py
13
+ source/cpyte/elf2sef.py
13
14
  source/cpyte/extension_hooks.py
14
15
  source/cpyte/formatter.py
15
16
  source/cpyte/gc_runtime.c
@@ -18,6 +19,7 @@ source/cpyte/lexar.py
18
19
  source/cpyte/linker.py
19
20
  source/cpyte/lsp_server.py
20
21
  source/cpyte/mainpie.py
22
+ source/cpyte/mksef.py
21
23
  source/cpyte/optimizations.py
22
24
  source/cpyte/package_manifest.py
23
25
  source/cpyte/runtime.c
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes