cpyte 4.3.1__tar.gz → 4.3.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {cpyte-4.3.1/source/cpyte.egg-info → cpyte-4.3.2}/PKG-INFO +1 -1
- {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte/__init__.py +1 -1
- {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte/bytecoding.py +10 -19
- {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte/compiling.py +21 -16
- cpyte-4.3.2/source/cpyte/elf2sef.py +832 -0
- cpyte-4.3.2/source/cpyte/mksef.py +134 -0
- {cpyte-4.3.1 → cpyte-4.3.2/source/cpyte.egg-info}/PKG-INFO +1 -1
- {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte.egg-info/SOURCES.txt +2 -0
- {cpyte-4.3.1 → cpyte-4.3.2}/MANIFEST.in +0 -0
- {cpyte-4.3.1 → cpyte-4.3.2}/pyproject.toml +0 -0
- {cpyte-4.3.1 → cpyte-4.3.2}/readme.md +0 -0
- {cpyte-4.3.1 → cpyte-4.3.2}/setup.cfg +0 -0
- {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte/__main__.py +0 -0
- {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte/_bignum_bc.py +0 -0
- {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte/_runtime_bc.py +0 -0
- {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte/astparse.py +0 -0
- {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte/bignum.c +0 -0
- {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte/clib.py +0 -0
- {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte/extension_hooks.py +0 -0
- {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte/formatter.py +0 -0
- {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte/gc_runtime.c +0 -0
- {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte/generate_bc.py +0 -0
- {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte/lexar.py +0 -0
- {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte/linker.py +0 -0
- {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte/lsp_server.py +0 -0
- {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte/mainpie.py +0 -0
- {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte/optimizations.py +0 -0
- {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte/package_manifest.py +0 -0
- {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte/runtime.c +0 -0
- {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte/runtime_scorpion.c +0 -0
- {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte/sef.py +0 -0
- {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte/semantic_analasis.py +0 -0
- {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte/ugc.h +0 -0
- {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte/ui.py +0 -0
- {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte/update_check.py +0 -0
- {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte/winjit_patch.py +0 -0
- {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte/winjit_stubs.c +0 -0
- {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte.egg-info/dependency_links.txt +0 -0
- {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte.egg-info/entry_points.txt +0 -0
- {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte.egg-info/requires.txt +0 -0
- {cpyte-4.3.1 → cpyte-4.3.2}/source/cpyte.egg-info/top_level.txt +0 -0
- {cpyte-4.3.1 → cpyte-4.3.2}/test/test_bignum_jit.py +0 -0
- {cpyte-4.3.1 → cpyte-4.3.2}/test/test_fuzz_del_has.py +0 -0
- {cpyte-4.3.1 → cpyte-4.3.2}/test/test_sef_tools.py +0 -0
|
@@ -2,7 +2,7 @@ import hashlib
|
|
|
2
2
|
import hmac
|
|
3
3
|
import os
|
|
4
4
|
from traceback import print_exception
|
|
5
|
-
from typing import Any,
|
|
5
|
+
from typing import Any, Protocol
|
|
6
6
|
|
|
7
7
|
from llvmlite import binding, ir
|
|
8
8
|
from llvmlite.ir import instructions
|
|
@@ -253,9 +253,7 @@ class LLVM:
|
|
|
253
253
|
t = _bc_array_norm(t)
|
|
254
254
|
if t == "int":
|
|
255
255
|
result = ir.IntType(32)
|
|
256
|
-
elif t == "int64":
|
|
257
|
-
result = ir.IntType(64)
|
|
258
|
-
elif t == "uint64":
|
|
256
|
+
elif t == "int64" or t == "uint64":
|
|
259
257
|
result = ir.IntType(64)
|
|
260
258
|
elif t == "size_t":
|
|
261
259
|
# Pointer-sized unsigned integer (i64 on every 64-bit cpyte target,
|
|
@@ -273,11 +271,7 @@ class LLVM:
|
|
|
273
271
|
result = ir.PointerType(ir.IntType(8))
|
|
274
272
|
elif t == "char":
|
|
275
273
|
result = ir.IntType(8)
|
|
276
|
-
elif t == "void*":
|
|
277
|
-
result = ir.PointerType(ir.IntType(8))
|
|
278
|
-
elif t == "big":
|
|
279
|
-
result = ir.PointerType(ir.IntType(8))
|
|
280
|
-
elif t == "ubig":
|
|
274
|
+
elif t == "void*" or t == "big" or t == "ubig":
|
|
281
275
|
result = ir.PointerType(ir.IntType(8))
|
|
282
276
|
elif t == "dynamic":
|
|
283
277
|
# A runtime-typed value: (kind, data) tag pair. Mirrors DynValue.
|
|
@@ -288,10 +282,7 @@ class LLVM:
|
|
|
288
282
|
elif t.endswith("[]"):
|
|
289
283
|
base = self.llvm_type(t[:-2])
|
|
290
284
|
result = ir.PointerType(base)
|
|
291
|
-
elif t.endswith("*"):
|
|
292
|
-
base = self.llvm_type(t[:-1])
|
|
293
|
-
result = ir.PointerType(base)
|
|
294
|
-
elif t.endswith("&"):
|
|
285
|
+
elif t.endswith("*") or t.endswith("&"):
|
|
295
286
|
base = self.llvm_type(t[:-1])
|
|
296
287
|
result = ir.PointerType(base)
|
|
297
288
|
elif t in self.structs:
|
|
@@ -1656,7 +1647,7 @@ class LLVM:
|
|
|
1656
1647
|
for table in (self.locals, self.local_types, self.ssa_values):
|
|
1657
1648
|
table.pop(name, None)
|
|
1658
1649
|
self.const_vars.pop(name, None)
|
|
1659
|
-
return
|
|
1650
|
+
return
|
|
1660
1651
|
if kind == "dynamic":
|
|
1661
1652
|
name = target.name
|
|
1662
1653
|
if name in self.locals and getattr(
|
|
@@ -1679,15 +1670,15 @@ class LLVM:
|
|
|
1679
1670
|
ir.Constant(_i64, 0),
|
|
1680
1671
|
],
|
|
1681
1672
|
)
|
|
1682
|
-
return
|
|
1673
|
+
return
|
|
1683
1674
|
if kind == "heap":
|
|
1684
1675
|
self._emit_builtin_free(target)
|
|
1685
|
-
return
|
|
1676
|
+
return
|
|
1686
1677
|
if kind == "slot":
|
|
1687
1678
|
slot = self._emit_lvalue(target)
|
|
1688
1679
|
if isinstance(slot.type, ir.PointerType) and slot.type.pointee is not None:
|
|
1689
1680
|
self.builder.store(ir.Constant(slot.type.pointee, None), slot)
|
|
1690
|
-
return
|
|
1681
|
+
return
|
|
1691
1682
|
raise Exception(
|
|
1692
1683
|
f"unhandled del_kind {kind!r} at L{node._token.line}:{node._token.column}"
|
|
1693
1684
|
)
|
|
@@ -6409,7 +6400,7 @@ class LLVM:
|
|
|
6409
6400
|
slot = self._emit_lvalue(target)
|
|
6410
6401
|
cur = self.builder.load(slot, "append.cur")
|
|
6411
6402
|
|
|
6412
|
-
elem_t = arr_t
|
|
6403
|
+
elem_t = arr_t.removesuffix("[]")
|
|
6413
6404
|
if arr_t == "dynamic":
|
|
6414
6405
|
# Element slot holds a DynValue; its i64 data is the inner list ptr.
|
|
6415
6406
|
elem_llvm = _DynValue
|
|
@@ -6778,7 +6769,7 @@ class LLVM:
|
|
|
6778
6769
|
def _is_const_var(self, name: str) -> bool:
|
|
6779
6770
|
return name in self._const_prop
|
|
6780
6771
|
|
|
6781
|
-
def _const_var_value(self, name: str) ->
|
|
6772
|
+
def _const_var_value(self, name: str) -> int | None:
|
|
6782
6773
|
return self._const_prop.get(name)
|
|
6783
6774
|
|
|
6784
6775
|
def _try_unroll_counted_loop(self, node: While, pending_ivs: dict) -> bool:
|
|
@@ -1806,6 +1806,25 @@ def _find_scorpion_tool(name, fallback):
|
|
|
1806
1806
|
return fallback
|
|
1807
1807
|
|
|
1808
1808
|
|
|
1809
|
+
def _scorpion_tool(name):
|
|
1810
|
+
"""Locate a bundled Scorpion converter script (elf2sef.py / mksef.py).
|
|
1811
|
+
|
|
1812
|
+
Prefers the copy shipped inside the cpyte package so the installed wheel
|
|
1813
|
+
works standalone; falls back to the WEW-scorpion sibling checkout for
|
|
1814
|
+
source-tree development.
|
|
1815
|
+
"""
|
|
1816
|
+
bundled = os.path.join(os.path.dirname(__file__), name)
|
|
1817
|
+
if os.path.isfile(bundled):
|
|
1818
|
+
return bundled
|
|
1819
|
+
sibling = os.path.join(
|
|
1820
|
+
os.path.dirname(os.path.dirname(os.path.dirname(os.path.dirname(_RUNTIME_SCORPION_C)))),
|
|
1821
|
+
"WEW-scorpion",
|
|
1822
|
+
"tools" if name == "elf2sef.py" else "user",
|
|
1823
|
+
name,
|
|
1824
|
+
)
|
|
1825
|
+
return sibling
|
|
1826
|
+
|
|
1827
|
+
|
|
1809
1828
|
def run_scorpion(
|
|
1810
1829
|
module,
|
|
1811
1830
|
output="program.sef",
|
|
@@ -1917,14 +1936,7 @@ def run_scorpion(
|
|
|
1917
1936
|
|
|
1918
1937
|
if pic:
|
|
1919
1938
|
# Dynamic SEF v2: relocation + import/export records
|
|
1920
|
-
elf2sef =
|
|
1921
|
-
os.path.dirname(os.path.dirname(_RUNTIME_SCORPION_C)),
|
|
1922
|
-
"..",
|
|
1923
|
-
"..",
|
|
1924
|
-
"WEW-scorpion",
|
|
1925
|
-
"tools",
|
|
1926
|
-
"elf2sef.py",
|
|
1927
|
-
)
|
|
1939
|
+
elf2sef = _scorpion_tool("elf2sef.py")
|
|
1928
1940
|
cmd = [sys.executable, elf2sef, elf_file, output]
|
|
1929
1941
|
if final_exports:
|
|
1930
1942
|
for name in final_exports:
|
|
@@ -1935,14 +1947,7 @@ def run_scorpion(
|
|
|
1935
1947
|
raise SystemExit(1)
|
|
1936
1948
|
else:
|
|
1937
1949
|
# Static SEF v1 via mksef.py
|
|
1938
|
-
mksef =
|
|
1939
|
-
os.path.dirname(os.path.dirname(_RUNTIME_SCORPION_C)),
|
|
1940
|
-
"..",
|
|
1941
|
-
"..",
|
|
1942
|
-
"WEW-scorpion",
|
|
1943
|
-
"user",
|
|
1944
|
-
"mksef.py",
|
|
1945
|
-
)
|
|
1950
|
+
mksef = _scorpion_tool("mksef.py")
|
|
1946
1951
|
if not os.path.isfile(mksef):
|
|
1947
1952
|
# Fallback: inline SEF generation using objdump/objcopy
|
|
1948
1953
|
_elf_to_sef(elf_file, output, 0)
|
|
@@ -0,0 +1,832 @@
|
|
|
1
|
+
"""
|
|
2
|
+
elf2sef.py — ELF (RV32, PIC, --emit-relocs) to Scorpion SEF v2.1 converter.
|
|
3
|
+
|
|
4
|
+
Produces a relocatable SEF image (docs/dynamic-linking.md):
|
|
5
|
+
|
|
6
|
+
* SEG_TEXT / SEG_DATA / SEG_BSS — flattened load image (linked at 0)
|
|
7
|
+
* SEG_RELOC — load-time relocations for absolute references that were
|
|
8
|
+
resolved at link time (R_RISCV_32 data words, lui-based
|
|
9
|
+
HI20/LO12 absolute pairs, .got/.got.plt slots)
|
|
10
|
+
* SEG_IMPORT — unresolved external symbols the loader must bind
|
|
11
|
+
* SEG_EXPORT — symbols other images may import (--export NAME)
|
|
12
|
+
* SEG_PLT — lazy-binding `ebreak` stubs (--lazy)
|
|
13
|
+
* SEG_VERDEF — version strings this image defines (--verdef VER)
|
|
14
|
+
|
|
15
|
+
An import record is emitted **per relocation site**, not per symbol name.
|
|
16
|
+
One symbol referenced from N places needs N records, because each site holds
|
|
17
|
+
its own copy of the address (a call site and a data word are separate words
|
|
18
|
+
in the image). Keying records by name binds only the first site and leaves
|
|
19
|
+
the rest pointing at whatever the linker left behind.
|
|
20
|
+
|
|
21
|
+
Link with: -fPIC ... -Wl,-q --unresolved-symbols=ignore-all --no-relax
|
|
22
|
+
|
|
23
|
+
Usage:
|
|
24
|
+
elf2sef.py [--lazy] [--weak NAME]... [--require NAME=VER]...
|
|
25
|
+
[--verdef VER]... [--no-versym]
|
|
26
|
+
[--export NAME]... [--export-file FILE] [--scope NAME]
|
|
27
|
+
<input.elf> <output.sef>
|
|
28
|
+
"""
|
|
29
|
+
|
|
30
|
+
import struct
|
|
31
|
+
import sys
|
|
32
|
+
|
|
33
|
+
SEF_MAGIC = 0x00464553
|
|
34
|
+
|
|
35
|
+
SEG_TEXT = 0
|
|
36
|
+
SEG_DATA = 1
|
|
37
|
+
SEG_BSS = 2
|
|
38
|
+
SEG_RELOC = 3
|
|
39
|
+
SEG_IMPORT = 4
|
|
40
|
+
SEG_EXPORT = 5
|
|
41
|
+
SEG_PLT = 6
|
|
42
|
+
SEG_VERDEF = 7
|
|
43
|
+
|
|
44
|
+
SEF_MAX_SEGMENTS = 8
|
|
45
|
+
|
|
46
|
+
# There is deliberately no privilege flag: a SEF image never declares what it
|
|
47
|
+
# wants to become. The kernel's boot path grants manager privilege to the one
|
|
48
|
+
# image it finds bundled in flash; everything else is user code.
|
|
49
|
+
SEF_FLAG_DYNAMIC = 0x0002
|
|
50
|
+
SEF_FLAG_LAZY = 0x0004
|
|
51
|
+
SEF_FLAG_VERSYM = 0x0008
|
|
52
|
+
|
|
53
|
+
SEF_R_RELATIVE = 0
|
|
54
|
+
SEF_R_HI20 = 1
|
|
55
|
+
SEF_R_LO12I = 2
|
|
56
|
+
SEF_R_LO12S = 3
|
|
57
|
+
SEF_R_CALL = 4
|
|
58
|
+
SEF_R_LAZY_CALL = 5
|
|
59
|
+
SEF_R_PCREL_HI20 = 6
|
|
60
|
+
SEF_R_PCREL_LO12I = 7
|
|
61
|
+
|
|
62
|
+
SEF_IMPORT_WEAK = 0x00000001
|
|
63
|
+
|
|
64
|
+
SEF_PLT_STUB_SIZE = 8
|
|
65
|
+
SCOPE_SEP = "::"
|
|
66
|
+
VER_SEP = "@"
|
|
67
|
+
|
|
68
|
+
# ELF constants
|
|
69
|
+
SHT_NOBITS = 8
|
|
70
|
+
SHT_RELA = 4
|
|
71
|
+
SHT_SYMTAB = 2
|
|
72
|
+
SHF_ALLOC = 0x2
|
|
73
|
+
SHF_WRITE = 0x1
|
|
74
|
+
SHN_UNDEF = 0
|
|
75
|
+
SHN_ABS = 0xFFF1
|
|
76
|
+
SHN_COMMON = 0xFFF2
|
|
77
|
+
SHN_XINDEX = 0xFFFF
|
|
78
|
+
|
|
79
|
+
STT_SECTION = 3
|
|
80
|
+
|
|
81
|
+
# RISC-V relocation types (psABI)
|
|
82
|
+
R_RISCV_32 = 1
|
|
83
|
+
R_RISCV_HI20 = 26
|
|
84
|
+
R_RISCV_LO12_I = 27
|
|
85
|
+
R_RISCV_LO12_S = 28
|
|
86
|
+
R_RISCV_CALL = 18
|
|
87
|
+
R_RISCV_CALL_PLT = 19
|
|
88
|
+
R_RISCV_GOT_HI20 = 20
|
|
89
|
+
R_RISCV_GOT_LO12 = 23
|
|
90
|
+
R_RISCV_RELAX = 51
|
|
91
|
+
|
|
92
|
+
# relocations whose value is an absolute address; everything else is
|
|
93
|
+
# PC-relative / branch / relax / debug and is left alone at load time
|
|
94
|
+
ABSOLUTE_RELOCS = {
|
|
95
|
+
R_RISCV_32: SEF_R_RELATIVE,
|
|
96
|
+
R_RISCV_HI20: SEF_R_HI20,
|
|
97
|
+
R_RISCV_LO12_I: SEF_R_LO12I,
|
|
98
|
+
R_RISCV_LO12_S: SEF_R_LO12S,
|
|
99
|
+
R_RISCV_CALL: SEF_R_CALL,
|
|
100
|
+
R_RISCV_CALL_PLT: SEF_R_CALL,
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
# RISC-V encodings we need to recognise or synthesise.
|
|
104
|
+
OP_AUIPC = 0x17
|
|
105
|
+
OP_JALR = 0x67
|
|
106
|
+
OP_LW = 0x03
|
|
107
|
+
OP_LD = 0x03
|
|
108
|
+
INSN_EBREAK = 0x00100073
|
|
109
|
+
|
|
110
|
+
MAPPED_MAX = 0x00100000 # 1 MiB sanity bound for the flat image
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def ver_hash(text):
|
|
114
|
+
"""FNV-1a-32 of a version string. Must agree with sef_ver_hash() in
|
|
115
|
+
loader.c and scorpion_ver_hash() in abi/scorpion.h."""
|
|
116
|
+
h = 2166136261
|
|
117
|
+
for byte in text.encode("latin1"):
|
|
118
|
+
h = ((h ^ byte) * 16777619) & 0xFFFFFFFF
|
|
119
|
+
return h
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def split_name(name):
|
|
123
|
+
"""Split "[scope::]symbol[@version]" into (scope, symbol, version).
|
|
124
|
+
|
|
125
|
+
Only the parts we need hashes or sanity checks for are returned; the
|
|
126
|
+
record always stores `name` verbatim so both the inline spelling and the
|
|
127
|
+
out-of-band hash are available to the loader.
|
|
128
|
+
"""
|
|
129
|
+
scope, sym, ver = "", name, ""
|
|
130
|
+
at = name.rfind(VER_SEP)
|
|
131
|
+
if at > 0:
|
|
132
|
+
sym, ver = name[:at], name[at + 1 :]
|
|
133
|
+
sep = sym.find(SCOPE_SEP)
|
|
134
|
+
if sep >= 0:
|
|
135
|
+
scope, sym = sym[:sep], sym[sep + 2 :]
|
|
136
|
+
return scope, sym, ver
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def emit_call_pair(insn0, insn1, target, site):
|
|
140
|
+
"""Rewrite an auipc+jalr pair at `site` to branch to `target`.
|
|
141
|
+
|
|
142
|
+
`disp` is the distance the linker would have encoded for a normal
|
|
143
|
+
PC-relative call, and the split is the same round-to-nearest-4K the
|
|
144
|
+
RISC-V psABI specifies, so this produces exactly the instruction pair
|
|
145
|
+
R_RISCV_CALL_PLT would have emitted.
|
|
146
|
+
"""
|
|
147
|
+
rd = (insn0 >> 7) & 0x1F
|
|
148
|
+
disp = target - site
|
|
149
|
+
hi = (disp + 0x800) >> 12
|
|
150
|
+
lo = disp - (hi << 12)
|
|
151
|
+
new0 = ((hi & 0xFFFFF) << 12) | (rd << 7) | OP_AUIPC
|
|
152
|
+
new1 = ((lo & 0xFFF) << 20) | (rd << 7) | OP_JALR
|
|
153
|
+
return new0, new1
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
class Section:
|
|
157
|
+
def __init__(self, name, type_, flags, addr, offset, size, link, info, entsize):
|
|
158
|
+
self.name = name
|
|
159
|
+
self.type = type_
|
|
160
|
+
self.flags = flags
|
|
161
|
+
self.addr = addr
|
|
162
|
+
self.offset = offset
|
|
163
|
+
self.size = size
|
|
164
|
+
self.link = link
|
|
165
|
+
self.info = info
|
|
166
|
+
self.entsize = entsize
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
class Symbol:
|
|
170
|
+
def __init__(self, name, value, size, info, shndx):
|
|
171
|
+
self.name = name
|
|
172
|
+
self.value = value
|
|
173
|
+
self.size = size
|
|
174
|
+
self.info = info
|
|
175
|
+
self.shndx = shndx
|
|
176
|
+
|
|
177
|
+
@property
|
|
178
|
+
def is_defined(self):
|
|
179
|
+
return self.shndx != SHN_UNDEF and self.shndx != SHN_ABS
|
|
180
|
+
|
|
181
|
+
@property
|
|
182
|
+
def is_section(self):
|
|
183
|
+
return (self.info & 0xF) == STT_SECTION
|
|
184
|
+
|
|
185
|
+
|
|
186
|
+
def parse_elf(path):
|
|
187
|
+
with open(path, "rb") as f:
|
|
188
|
+
data = f.read()
|
|
189
|
+
|
|
190
|
+
if data[:4] != b"\x7fELF":
|
|
191
|
+
sys.exit(f"error: {path}: not an ELF file")
|
|
192
|
+
if data[4] != 1: # ELFCLASS32
|
|
193
|
+
sys.exit(f"error: {path}: not a 32-bit ELF")
|
|
194
|
+
|
|
195
|
+
ei_data = data[5]
|
|
196
|
+
endian = "<" if ei_data == 1 else ">"
|
|
197
|
+
if ei_data == 2:
|
|
198
|
+
sys.exit(f"error: {path}: big-endian ELF unsupported")
|
|
199
|
+
|
|
200
|
+
(
|
|
201
|
+
e_type,
|
|
202
|
+
e_machine,
|
|
203
|
+
e_version,
|
|
204
|
+
e_entry,
|
|
205
|
+
e_phoff,
|
|
206
|
+
e_shoff,
|
|
207
|
+
e_flags,
|
|
208
|
+
e_ehsize,
|
|
209
|
+
e_phentsize,
|
|
210
|
+
e_phnum,
|
|
211
|
+
e_shentsize,
|
|
212
|
+
e_shnum,
|
|
213
|
+
e_shstrndx,
|
|
214
|
+
) = struct.unpack_from(endian + "HHIIIIIHHHHHH", data, 16)
|
|
215
|
+
|
|
216
|
+
if e_machine != 0xF3:
|
|
217
|
+
sys.exit(f"error: {path}: not RISC-V (machine={e_machine:#x})")
|
|
218
|
+
|
|
219
|
+
if e_shnum == 0 or e_shentsize != 40:
|
|
220
|
+
sys.exit(f"error: {path}: bad section headers")
|
|
221
|
+
|
|
222
|
+
shdr = []
|
|
223
|
+
for i in range(e_shnum):
|
|
224
|
+
shdr.append(struct.unpack_from(endian + "IIIIIIIIII", data, e_shoff + i * 40))
|
|
225
|
+
|
|
226
|
+
# section name string table
|
|
227
|
+
shstr = shdr[e_shstrndx]
|
|
228
|
+
shstr_data = data[shstr[4] : shstr[4] + shstr[5]]
|
|
229
|
+
|
|
230
|
+
def cstr(buf, off):
|
|
231
|
+
if off >= len(buf):
|
|
232
|
+
return ""
|
|
233
|
+
end = buf.find(b"\x00", off)
|
|
234
|
+
if end < 0:
|
|
235
|
+
return ""
|
|
236
|
+
return buf[off:end].decode("latin1")
|
|
237
|
+
|
|
238
|
+
sections = {}
|
|
239
|
+
by_index = []
|
|
240
|
+
for i, s in enumerate(shdr):
|
|
241
|
+
name = cstr(shstr_data, s[0])
|
|
242
|
+
sec = Section(name, s[1], s[2], s[3], s[4], s[5], s[6], s[7], s[9])
|
|
243
|
+
sections[name] = sec
|
|
244
|
+
by_index.append(sec)
|
|
245
|
+
|
|
246
|
+
def sym_by_index(idx):
|
|
247
|
+
return symbols[idx] if idx < len(symbols) else None
|
|
248
|
+
|
|
249
|
+
def section_by_index(idx):
|
|
250
|
+
return by_index[idx] if idx < len(by_index) else None
|
|
251
|
+
|
|
252
|
+
# symbol table (symbols[i] is ELF symbol index i; [0] is the null symbol)
|
|
253
|
+
symbols = []
|
|
254
|
+
if ".symtab" in sections:
|
|
255
|
+
st = sections[".symtab"]
|
|
256
|
+
strtab = section_by_index(st.link)
|
|
257
|
+
str_data = b""
|
|
258
|
+
if strtab is not None:
|
|
259
|
+
str_data = data[strtab.offset : strtab.offset + strtab.size]
|
|
260
|
+
for i in range(st.size // st.entsize):
|
|
261
|
+
st_name, st_value, st_size, st_info, st_other, st_shndx = (
|
|
262
|
+
struct.unpack_from(endian + "IIIBBH", data, st.offset + i * st.entsize)
|
|
263
|
+
)
|
|
264
|
+
name = cstr(str_data, st_name)
|
|
265
|
+
symbols.append(Symbol(name, st_value, st_size, st_info, st_shndx))
|
|
266
|
+
|
|
267
|
+
# alloc sections in address order
|
|
268
|
+
alloc = [s for s in sections.values() if (s.flags & SHF_ALLOC) and s.size > 0]
|
|
269
|
+
alloc.sort(key=lambda s: (s.addr, s.size))
|
|
270
|
+
|
|
271
|
+
if not alloc:
|
|
272
|
+
sys.exit(f"error: {path}: no allocated sections")
|
|
273
|
+
|
|
274
|
+
base = alloc[0].addr
|
|
275
|
+
mapped_end = max(s.addr + s.size for s in alloc)
|
|
276
|
+
|
|
277
|
+
if mapped_end - base > MAPPED_MAX:
|
|
278
|
+
sys.exit(f"error: {path}: image too large ({mapped_end - base:#x} bytes)")
|
|
279
|
+
|
|
280
|
+
flat = bytearray(mapped_end - base)
|
|
281
|
+
for s in alloc:
|
|
282
|
+
if s.type != SHT_NOBITS:
|
|
283
|
+
flat[s.addr - base : s.addr - base + s.size] = data[
|
|
284
|
+
s.offset : s.offset + s.size
|
|
285
|
+
]
|
|
286
|
+
|
|
287
|
+
# relocations: site -> (type, sym_index, addend)
|
|
288
|
+
relocs = []
|
|
289
|
+
for sec in sections.values():
|
|
290
|
+
if sec.type != SHT_RELA or sec.link == 0:
|
|
291
|
+
continue
|
|
292
|
+
target = by_index[sec.info] if sec.info < len(by_index) else None
|
|
293
|
+
if target is None or (target.flags & SHF_ALLOC) == 0:
|
|
294
|
+
continue # .rela.debug* etc.: no memory image
|
|
295
|
+
for i in range(sec.size // sec.entsize):
|
|
296
|
+
r_offset, r_info, r_addend = struct.unpack_from(
|
|
297
|
+
endian + "IIi", data, sec.offset + i * sec.entsize
|
|
298
|
+
)
|
|
299
|
+
r_sym = r_info >> 8
|
|
300
|
+
r_type = r_info & 0xFF
|
|
301
|
+
relocs.append((r_offset, r_type, r_sym, r_addend))
|
|
302
|
+
|
|
303
|
+
return {
|
|
304
|
+
"entry": e_entry,
|
|
305
|
+
"base": base,
|
|
306
|
+
"mapped_end": mapped_end,
|
|
307
|
+
"flat": flat,
|
|
308
|
+
"sections": sections,
|
|
309
|
+
"by_index": by_index,
|
|
310
|
+
"symbols": symbols,
|
|
311
|
+
"sym_by_index": sym_by_index,
|
|
312
|
+
"section_by_index": section_by_index,
|
|
313
|
+
"relocs": relocs,
|
|
314
|
+
}
|
|
315
|
+
|
|
316
|
+
|
|
317
|
+
def is_absolute_pair(elf, sym_idx):
|
|
318
|
+
"""True if the LO12 reloc's symbol points at a `lui` (absolute pair)
|
|
319
|
+
rather than an `auipc` (PC-relative pair)."""
|
|
320
|
+
sym = elf["sym_by_index"](sym_idx)
|
|
321
|
+
if sym is None or not sym.is_defined:
|
|
322
|
+
return False
|
|
323
|
+
off = sym.value - elf["base"]
|
|
324
|
+
flat = elf["flat"]
|
|
325
|
+
if off + 4 > len(flat):
|
|
326
|
+
return False
|
|
327
|
+
insn = struct.unpack_from("<I", flat, off)[0]
|
|
328
|
+
return (insn & 0x7F) == 0x37 # lui
|
|
329
|
+
|
|
330
|
+
|
|
331
|
+
def sign_extend(value, bits):
|
|
332
|
+
if value & (1 << (bits - 1)):
|
|
333
|
+
return value - (1 << bits)
|
|
334
|
+
return value
|
|
335
|
+
|
|
336
|
+
|
|
337
|
+
def got_slot_of(elf, site):
|
|
338
|
+
"""Decode the GOT entry address a linker-emitted auipc+lo12 pair points at.
|
|
339
|
+
|
|
340
|
+
`-fPIC` reaches an *external* data symbol through the GOT, so the pair is
|
|
341
|
+
`auipc rd, hi; lw rd, lo(rd)` and the load-base-dependent part is the GOT
|
|
342
|
+
address, not the symbol itself. The linker has already resolved the pair,
|
|
343
|
+
so rather than re-deriving the psABI addend convention we read the
|
|
344
|
+
displacement back out of the two encodings and check that it lands in a
|
|
345
|
+
real GOT section. A pair that does not is not ours to rewrite.
|
|
346
|
+
"""
|
|
347
|
+
flat = elf["flat"]
|
|
348
|
+
if site + 8 > len(flat):
|
|
349
|
+
return None
|
|
350
|
+
insn0 = struct.unpack_from("<I", flat, site)[0]
|
|
351
|
+
insn1 = struct.unpack_from("<I", flat, site + 4)[0]
|
|
352
|
+
if (insn0 & 0x7F) != OP_AUIPC:
|
|
353
|
+
return None
|
|
354
|
+
if (insn1 & 0x7F) not in (OP_LW, OP_LD):
|
|
355
|
+
return None
|
|
356
|
+
|
|
357
|
+
hi = sign_extend((insn0 >> 12) & 0xFFFFF, 20)
|
|
358
|
+
lo = sign_extend((insn1 >> 20) & 0xFFF, 12)
|
|
359
|
+
disp = site + (hi << 12) + lo
|
|
360
|
+
for name in (".got", ".got.plt"):
|
|
361
|
+
sec = elf["sections"].get(name)
|
|
362
|
+
if sec is None or (sec.flags & SHF_ALLOC) == 0:
|
|
363
|
+
continue
|
|
364
|
+
start = sec.addr - elf["base"]
|
|
365
|
+
if start <= disp < start + sec.size:
|
|
366
|
+
return disp
|
|
367
|
+
return None
|
|
368
|
+
|
|
369
|
+
|
|
370
|
+
class Import:
|
|
371
|
+
"""One unresolved reference at one image offset."""
|
|
372
|
+
|
|
373
|
+
__slots__ = ("rtype", "site", "name", "ver_hash", "flags", "plt")
|
|
374
|
+
|
|
375
|
+
def __init__(self, rtype, site, name):
|
|
376
|
+
self.rtype = rtype
|
|
377
|
+
self.site = site
|
|
378
|
+
self.name = name
|
|
379
|
+
self.ver_hash = 0
|
|
380
|
+
self.flags = 0
|
|
381
|
+
self.plt = 0
|
|
382
|
+
|
|
383
|
+
|
|
384
|
+
def main(argv):
|
|
385
|
+
exports = []
|
|
386
|
+
export_file = None
|
|
387
|
+
weak = []
|
|
388
|
+
require = {} # import name -> version
|
|
389
|
+
verdefs = []
|
|
390
|
+
scope = ""
|
|
391
|
+
lazy = False
|
|
392
|
+
versym = True
|
|
393
|
+
positionals = []
|
|
394
|
+
args = list(argv)
|
|
395
|
+
|
|
396
|
+
i = 0
|
|
397
|
+
while i < len(args):
|
|
398
|
+
opt = args[i]
|
|
399
|
+
if opt == "--export":
|
|
400
|
+
exports.append(args[i + 1])
|
|
401
|
+
i += 2
|
|
402
|
+
elif opt == "--export-file":
|
|
403
|
+
export_file = args[i + 1]
|
|
404
|
+
i += 2
|
|
405
|
+
elif opt == "--weak":
|
|
406
|
+
weak.append(args[i + 1])
|
|
407
|
+
i += 2
|
|
408
|
+
elif opt == "--verdef":
|
|
409
|
+
verdefs.append(args[i + 1])
|
|
410
|
+
i += 2
|
|
411
|
+
elif opt == "--scope":
|
|
412
|
+
scope = args[i + 1]
|
|
413
|
+
i += 2
|
|
414
|
+
elif opt == "--require":
|
|
415
|
+
if "=" not in args[i + 1]:
|
|
416
|
+
sys.exit(f"error: --require wants NAME=VERSION, got {args[i+1]!r}")
|
|
417
|
+
name, ver = args[i + 1].split("=", 1)
|
|
418
|
+
require[name] = ver
|
|
419
|
+
i += 2
|
|
420
|
+
elif opt == "--lazy":
|
|
421
|
+
lazy = True
|
|
422
|
+
i += 1
|
|
423
|
+
elif opt == "--no-lazy":
|
|
424
|
+
lazy = False
|
|
425
|
+
i += 1
|
|
426
|
+
elif opt == "--no-versym":
|
|
427
|
+
versym = False
|
|
428
|
+
i += 1
|
|
429
|
+
elif opt.startswith("--"):
|
|
430
|
+
sys.exit(f"error: unknown option {opt}")
|
|
431
|
+
else:
|
|
432
|
+
positionals.append(opt)
|
|
433
|
+
i += 1
|
|
434
|
+
|
|
435
|
+
if len(positionals) != 2:
|
|
436
|
+
sys.exit(__doc__)
|
|
437
|
+
|
|
438
|
+
elf_path, sef_path = positionals
|
|
439
|
+
elf = parse_elf(elf_path)
|
|
440
|
+
flat = elf["flat"]
|
|
441
|
+
base = elf["base"]
|
|
442
|
+
sections = elf["sections"]
|
|
443
|
+
|
|
444
|
+
if export_file:
|
|
445
|
+
with open(export_file) as f:
|
|
446
|
+
exports.extend(line.strip() for line in f if line.strip())
|
|
447
|
+
|
|
448
|
+
# --- map / segment layout (linked at 0: vaddr == image offset) ---
|
|
449
|
+
text = sections.get(".text")
|
|
450
|
+
bss = sections.get(".bss")
|
|
451
|
+
|
|
452
|
+
if text is None or (text.flags & SHF_ALLOC) == 0:
|
|
453
|
+
sys.exit(f"error: {elf_path}: no allocated .text section")
|
|
454
|
+
|
|
455
|
+
text_vaddr = text.addr - base
|
|
456
|
+
text_size = text.size
|
|
457
|
+
|
|
458
|
+
# A zero-sized .bss still has an address, and using it as the end of the
|
|
459
|
+
# data segment would describe a hole the flat image never covers: the
|
|
460
|
+
# segment would claim bytes that `flat` does not have, silently truncating
|
|
461
|
+
# everything after it. Only a real .bss ends the data segment.
|
|
462
|
+
has_bss = (
|
|
463
|
+
bss is not None
|
|
464
|
+
and (bss.flags & SHF_ALLOC) != 0
|
|
465
|
+
and bss.size > 0
|
|
466
|
+
and bss.addr >= text.addr + text.size
|
|
467
|
+
)
|
|
468
|
+
|
|
469
|
+
if has_bss:
|
|
470
|
+
data_vaddr = text_vaddr + text_size
|
|
471
|
+
data_size = (bss.addr - base) - data_vaddr
|
|
472
|
+
bss_vaddr = bss.addr - base
|
|
473
|
+
bss_size = bss.size
|
|
474
|
+
else:
|
|
475
|
+
data_vaddr = text_vaddr + text_size
|
|
476
|
+
data_size = (elf["mapped_end"] - base) - data_vaddr
|
|
477
|
+
bss_vaddr = 0
|
|
478
|
+
bss_size = 0
|
|
479
|
+
|
|
480
|
+
if data_size < 0 or bss_size < 0:
|
|
481
|
+
sys.exit(f"error: {elf_path}: unexpected section order")
|
|
482
|
+
|
|
483
|
+
# Every mapped segment has to be backed by real bytes, or the header
|
|
484
|
+
# describes an image the file cannot supply.
|
|
485
|
+
for label, vaddr, size in (
|
|
486
|
+
("SEG_TEXT", text_vaddr, text_size),
|
|
487
|
+
("SEG_DATA", data_vaddr, data_size),
|
|
488
|
+
("SEG_BSS", bss_vaddr, bss_size),
|
|
489
|
+
):
|
|
490
|
+
if size and vaddr + size > len(flat):
|
|
491
|
+
sys.exit(
|
|
492
|
+
f"error: {elf_path}: {label} [0x{vaddr:x},0x{vaddr+size:x}) "
|
|
493
|
+
f"is past the end of the linked image (0x{len(flat):x})"
|
|
494
|
+
)
|
|
495
|
+
|
|
496
|
+
image_end = bss_vaddr + bss_size if bss_size else data_vaddr + data_size
|
|
497
|
+
|
|
498
|
+
# --- process relocations ---
|
|
499
|
+
reloc_records = [] # (type, offset, value)
|
|
500
|
+
imports = [] # one Import per relocation site
|
|
501
|
+
sym_by_index = elf["sym_by_index"]
|
|
502
|
+
section_by_index = elf["section_by_index"]
|
|
503
|
+
mapped = len(flat)
|
|
504
|
+
|
|
505
|
+
for r_offset, r_type, r_sym, r_addend in elf["relocs"]:
|
|
506
|
+
if r_type == R_RISCV_RELAX:
|
|
507
|
+
continue
|
|
508
|
+
if r_type not in ABSOLUTE_RELOCS and r_type != R_RISCV_GOT_HI20:
|
|
509
|
+
continue
|
|
510
|
+
if r_offset < base or r_offset - base + 4 > mapped:
|
|
511
|
+
continue # site outside the load image (debug etc.)
|
|
512
|
+
|
|
513
|
+
site = r_offset - base
|
|
514
|
+
|
|
515
|
+
if r_type == R_RISCV_LO12_I or r_type == R_RISCV_LO12_S:
|
|
516
|
+
if not is_absolute_pair(elf, r_sym):
|
|
517
|
+
continue # PC-relative pair; no load-time fixup needed
|
|
518
|
+
|
|
519
|
+
sym = sym_by_index(r_sym)
|
|
520
|
+
if sym is None:
|
|
521
|
+
continue
|
|
522
|
+
|
|
523
|
+
if not sym.is_defined:
|
|
524
|
+
# Unresolved external. Every site gets its own record: the same
|
|
525
|
+
# symbol called from N places occupies N words in the image, and
|
|
526
|
+
# each word has to be written for the call to reach the target.
|
|
527
|
+
if not sym.name:
|
|
528
|
+
continue
|
|
529
|
+
imp = Import(ABSOLUTE_RELOCS.get(r_type, SEF_R_RELATIVE), site, sym.name)
|
|
530
|
+
_, _, inline_ver = split_name(sym.name)
|
|
531
|
+
if inline_ver:
|
|
532
|
+
imp.ver_hash = ver_hash(inline_ver)
|
|
533
|
+
elif sym.name in require:
|
|
534
|
+
imp.ver_hash = ver_hash(require[sym.name])
|
|
535
|
+
if sym.name in weak:
|
|
536
|
+
imp.flags |= SEF_IMPORT_WEAK
|
|
537
|
+
|
|
538
|
+
if r_type == R_RISCV_GOT_HI20:
|
|
539
|
+
# External data binds through the GOT: the word the loader
|
|
540
|
+
# must write is the GOT slot, and the auipc/lo12 pair needs
|
|
541
|
+
# fixing up to reach it. That makes this import a plain
|
|
542
|
+
# 32-bit word write at the slot, not a patch of `site`.
|
|
543
|
+
got = got_slot_of(elf, site)
|
|
544
|
+
if got is None:
|
|
545
|
+
sys.exit(
|
|
546
|
+
f"error: {elf_path}: cannot resolve the GOT entry for "
|
|
547
|
+
f"{sym.name!r} referenced at 0x{site:x}"
|
|
548
|
+
)
|
|
549
|
+
imp.rtype = SEF_R_RELATIVE
|
|
550
|
+
imp.site = got
|
|
551
|
+
imports.append(imp)
|
|
552
|
+
# Two load-time relocs, one per half, each PC-relative to
|
|
553
|
+
# the auipc so the pair computes got_entry at the real base.
|
|
554
|
+
reloc_records.append((SEF_R_PCREL_HI20, site, got))
|
|
555
|
+
reloc_records.append((SEF_R_PCREL_LO12I, site + 4, got))
|
|
556
|
+
else:
|
|
557
|
+
imports.append(imp)
|
|
558
|
+
continue
|
|
559
|
+
|
|
560
|
+
if r_type == R_RISCV_GOT_HI20:
|
|
561
|
+
# GOT entry for a symbol this image defines: make the pair
|
|
562
|
+
# load-base relative, and let the .got scan below turn the slot
|
|
563
|
+
# into an absolute relocation.
|
|
564
|
+
got = got_slot_of(elf, site)
|
|
565
|
+
if got is None:
|
|
566
|
+
continue
|
|
567
|
+
reloc_records.append((SEF_R_PCREL_HI20, site, got))
|
|
568
|
+
reloc_records.append((SEF_R_PCREL_LO12I, site + 4, got))
|
|
569
|
+
continue
|
|
570
|
+
|
|
571
|
+
sec = section_by_index(sym.shndx)
|
|
572
|
+
if sec is not None and (sec.flags & SHF_ALLOC) == 0:
|
|
573
|
+
continue # e.g. debug symbol
|
|
574
|
+
|
|
575
|
+
value = sym.value + r_addend
|
|
576
|
+
reloc_records.append((ABSOLUTE_RELOCS[r_type], site, value))
|
|
577
|
+
|
|
578
|
+
# Two relocations can land on the same offset only if the linker emitted
|
|
579
|
+
# duplicates; collapse them so the loader never sees a double write.
|
|
580
|
+
imports.sort(key=lambda imp: (imp.site, imp.rtype))
|
|
581
|
+
deduped = []
|
|
582
|
+
for imp in imports:
|
|
583
|
+
if deduped and deduped[-1].site == imp.site:
|
|
584
|
+
continue
|
|
585
|
+
deduped.append(imp)
|
|
586
|
+
imports = deduped
|
|
587
|
+
|
|
588
|
+
# --- lazy PLT ---
|
|
589
|
+
#
|
|
590
|
+
# Each eligible call site gets a `ebreak` stub. The call site is
|
|
591
|
+
# rewritten to reach the stub; the first execution traps, the handler
|
|
592
|
+
# resolves the symbol and patches the site to branch straight at the
|
|
593
|
+
# target, so the stub is entered at most once.
|
|
594
|
+
plt_vaddr = 0
|
|
595
|
+
plt_size = 0
|
|
596
|
+
plt_bytes = b""
|
|
597
|
+
|
|
598
|
+
if lazy:
|
|
599
|
+
stubs = []
|
|
600
|
+
for imp in imports:
|
|
601
|
+
if imp.rtype != SEF_R_CALL:
|
|
602
|
+
continue
|
|
603
|
+
if imp.site + 8 > mapped:
|
|
604
|
+
continue
|
|
605
|
+
insn0 = struct.unpack_from("<I", flat, imp.site)[0]
|
|
606
|
+
insn1 = struct.unpack_from("<I", flat, imp.site + 4)[0]
|
|
607
|
+
if (insn0 & 0x7F) != OP_AUIPC or (insn1 & 0x7F) != OP_JALR:
|
|
608
|
+
continue # not the auipc+jalr shape a stub can intercept
|
|
609
|
+
stubs.append((imp, insn0, insn1))
|
|
610
|
+
|
|
611
|
+
if stubs:
|
|
612
|
+
plt_vaddr = (image_end + SEF_PLT_STUB_SIZE - 1) & ~(SEF_PLT_STUB_SIZE - 1)
|
|
613
|
+
plt_size = len(stubs) * SEF_PLT_STUB_SIZE
|
|
614
|
+
if plt_vaddr + plt_size > MAPPED_MAX:
|
|
615
|
+
sys.exit(f"error: {elf_path}: lazy PLT does not fit the image")
|
|
616
|
+
|
|
617
|
+
body = bytearray()
|
|
618
|
+
for index, (imp, insn0, insn1) in enumerate(stubs):
|
|
619
|
+
imp.rtype = SEF_R_LAZY_CALL
|
|
620
|
+
imp.plt = index * SEF_PLT_STUB_SIZE
|
|
621
|
+
new0, new1 = emit_call_pair(insn0, insn1, plt_vaddr + imp.plt, imp.site)
|
|
622
|
+
struct.pack_into("<I", flat, imp.site, new0)
|
|
623
|
+
struct.pack_into("<I", flat, imp.site + 4, new1)
|
|
624
|
+
# Both words are `ebreak`: the first traps, and if the handler
|
|
625
|
+
# ever failed to resolve, the second traps too rather than
|
|
626
|
+
# sliding into whatever follows as if it were code.
|
|
627
|
+
body += struct.pack("<II", INSN_EBREAK, INSN_EBREAK)
|
|
628
|
+
plt_bytes = bytes(body)
|
|
629
|
+
|
|
630
|
+
# The call relocations for these sites were queued above, before we
|
|
631
|
+
# knew they were lazy. The loader applies SEG_RELOC records at load
|
|
632
|
+
# time, so leaving them in would re-apply the *link-time* target
|
|
633
|
+
# and silently undo the stub rewrite above -- the call site would
|
|
634
|
+
# jump straight at the original PLT entry, never trap, and the
|
|
635
|
+
# lazy import would be marked bound while its stub is never
|
|
636
|
+
# reached. The import record alone now drives this site.
|
|
637
|
+
stub_sites = {imp.site for imp, _, _ in stubs}
|
|
638
|
+
reloc_records = [
|
|
639
|
+
rec for rec in reloc_records
|
|
640
|
+
if not (rec[0] == SEF_R_CALL and rec[1] in stub_sites)
|
|
641
|
+
]
|
|
642
|
+
|
|
643
|
+
# --- scan .got/.got.plt for slots the linker resolved without a reloc ---
|
|
644
|
+
defined_values = set()
|
|
645
|
+
for sym in elf["symbols"]:
|
|
646
|
+
if sym is not None and sym.is_defined:
|
|
647
|
+
sec = section_by_index(sym.shndx)
|
|
648
|
+
if sec is not None and (sec.flags & SHF_ALLOC):
|
|
649
|
+
defined_values.add(sym.value)
|
|
650
|
+
if sym.is_section:
|
|
651
|
+
for delta in range(0, min(sec.size, 0x100), 4):
|
|
652
|
+
defined_values.add(sym.value + delta)
|
|
653
|
+
|
|
654
|
+
plt_sec_names = (".plt", ".plt.sec")
|
|
655
|
+
got_records = {}
|
|
656
|
+
for got_name in (".got", ".got.plt"):
|
|
657
|
+
got = sections.get(got_name)
|
|
658
|
+
if got is None or (got.flags & SHF_ALLOC) == 0:
|
|
659
|
+
continue
|
|
660
|
+
for off in range(0, got.size, 4):
|
|
661
|
+
slot_addr = got.addr + off
|
|
662
|
+
word = struct.unpack_from("<I", flat, slot_addr - base)[0]
|
|
663
|
+
if word == 0 or word == 0xFFFFFFFF:
|
|
664
|
+
continue
|
|
665
|
+
skip = False
|
|
666
|
+
for name in plt_sec_names:
|
|
667
|
+
sec = sections.get(name)
|
|
668
|
+
if sec is not None and sec.addr <= word < sec.addr + sec.size:
|
|
669
|
+
skip = True # PLT trampoline pointer (external call)
|
|
670
|
+
break
|
|
671
|
+
if skip:
|
|
672
|
+
continue
|
|
673
|
+
if word in defined_values:
|
|
674
|
+
got_records.setdefault(slot_addr - base, word)
|
|
675
|
+
|
|
676
|
+
for slot, word in got_records.items():
|
|
677
|
+
if not any(r[0] == SEF_R_RELATIVE and r[1] == slot for r in reloc_records):
|
|
678
|
+
reloc_records.append((SEF_R_RELATIVE, slot, word))
|
|
679
|
+
|
|
680
|
+
# --- exports ---
|
|
681
|
+
# Names are stored verbatim, so a caller may spell them "sym", "sym@ver",
|
|
682
|
+
# "scope::sym" or "scope::sym@ver" and the loader sees the same thing it
|
|
683
|
+
# would have seen in an import. ver_hash repeats the version out of band
|
|
684
|
+
# for producers that keep the two apart.
|
|
685
|
+
#
|
|
686
|
+
# --export takes either NAME (export the ELF symbol of that name) or
|
|
687
|
+
# SYMBOL=WIRE_NAME (export SYMBOL under a different published name, which
|
|
688
|
+
# is how a versioned alias like addone@2.0 is published from a plain
|
|
689
|
+
# `addone_v2` definition).
|
|
690
|
+
export_records = []
|
|
691
|
+
for raw in exports:
|
|
692
|
+
if "=" in raw:
|
|
693
|
+
elf_name, wire = raw.split("=", 1)
|
|
694
|
+
else:
|
|
695
|
+
elf_name, wire = raw, raw
|
|
696
|
+
# --scope applies to every name that does not already carry one,
|
|
697
|
+
# including versioned ones, so --export foo@1.0 under --scope util
|
|
698
|
+
# publishes "util::foo@1.0" rather than leaving it unscoped.
|
|
699
|
+
if scope and SCOPE_SEP not in wire.split(VER_SEP, 1)[0]:
|
|
700
|
+
wire = scope + SCOPE_SEP + wire
|
|
701
|
+
|
|
702
|
+
for sym in elf["symbols"]:
|
|
703
|
+
if sym is None or not sym.is_defined or sym.name != elf_name:
|
|
704
|
+
continue
|
|
705
|
+
_, _, inline_ver = split_name(wire)
|
|
706
|
+
export_records.append(
|
|
707
|
+
(sym.value, wire, ver_hash(inline_ver) if inline_ver else 0)
|
|
708
|
+
)
|
|
709
|
+
break
|
|
710
|
+
else:
|
|
711
|
+
sys.exit(f"error: {elf_path}: export {raw!r} not found")
|
|
712
|
+
|
|
713
|
+
# --- version definitions ---
|
|
714
|
+
# Anything named in an export's @ver tail is a version this image defines.
|
|
715
|
+
defined_vers = list(verdefs)
|
|
716
|
+
for _, name, _ in export_records:
|
|
717
|
+
_, _, inline_ver = split_name(name)
|
|
718
|
+
if inline_ver and inline_ver not in defined_vers:
|
|
719
|
+
defined_vers.append(inline_ver)
|
|
720
|
+
|
|
721
|
+
# --- build SEF ---
|
|
722
|
+
flags = 0
|
|
723
|
+
dynamic = bool(reloc_records or imports or export_records)
|
|
724
|
+
if dynamic:
|
|
725
|
+
flags |= SEF_FLAG_DYNAMIC
|
|
726
|
+
if plt_bytes:
|
|
727
|
+
flags |= SEF_FLAG_LAZY
|
|
728
|
+
if versym and (imports or export_records or defined_vers):
|
|
729
|
+
flags |= SEF_FLAG_VERSYM
|
|
730
|
+
|
|
731
|
+
segments = [] # (type, vaddr, size, data)
|
|
732
|
+
segments.append((SEG_TEXT, text_vaddr, text_size, None))
|
|
733
|
+
segments.append((SEG_DATA, data_vaddr, data_size, None))
|
|
734
|
+
if bss_size:
|
|
735
|
+
segments.append((SEG_BSS, bss_vaddr, bss_size, None))
|
|
736
|
+
|
|
737
|
+
reloc_data = b"".join(struct.pack("<III", t, o, v) for (t, o, v) in reloc_records)
|
|
738
|
+
if reloc_data:
|
|
739
|
+
segments.append((SEG_RELOC, 0, len(reloc_data), reloc_data))
|
|
740
|
+
|
|
741
|
+
import_data = b""
|
|
742
|
+
for imp in imports:
|
|
743
|
+
nb = imp.name.encode("latin1")
|
|
744
|
+
if versym:
|
|
745
|
+
import_data += struct.pack(
|
|
746
|
+
"<IIIIII",
|
|
747
|
+
imp.rtype,
|
|
748
|
+
imp.site,
|
|
749
|
+
imp.plt,
|
|
750
|
+
imp.ver_hash,
|
|
751
|
+
imp.flags,
|
|
752
|
+
len(nb),
|
|
753
|
+
)
|
|
754
|
+
else:
|
|
755
|
+
import_data += struct.pack("<III", imp.rtype, imp.site, len(nb))
|
|
756
|
+
import_data += nb
|
|
757
|
+
import_data += b"\x00" * ((4 - len(nb) % 4) % 4)
|
|
758
|
+
if import_data:
|
|
759
|
+
segments.append((SEG_IMPORT, 0, len(import_data), import_data))
|
|
760
|
+
|
|
761
|
+
export_data = b""
|
|
762
|
+
for value, name, vhash in export_records:
|
|
763
|
+
nb = name.encode("latin1")
|
|
764
|
+
if versym:
|
|
765
|
+
export_data += struct.pack("<III", value, vhash, len(nb))
|
|
766
|
+
else:
|
|
767
|
+
export_data += struct.pack("<II", value, len(nb))
|
|
768
|
+
export_data += nb
|
|
769
|
+
export_data += b"\x00" * ((4 - len(nb) % 4) % 4)
|
|
770
|
+
if export_data:
|
|
771
|
+
segments.append((SEG_EXPORT, 0, len(export_data), export_data))
|
|
772
|
+
|
|
773
|
+
if plt_bytes:
|
|
774
|
+
segments.append((SEG_PLT, plt_vaddr, plt_size, plt_bytes))
|
|
775
|
+
|
|
776
|
+
verdef_data = b""
|
|
777
|
+
for ver in defined_vers:
|
|
778
|
+
nb = ver.encode("latin1")
|
|
779
|
+
verdef_data += struct.pack("<I", len(nb)) + nb
|
|
780
|
+
verdef_data += b"\x00" * ((4 - len(nb) % 4) % 4)
|
|
781
|
+
if verdef_data:
|
|
782
|
+
segments.append((SEG_VERDEF, 0, len(verdef_data), verdef_data))
|
|
783
|
+
|
|
784
|
+
if len(segments) > SEF_MAX_SEGMENTS:
|
|
785
|
+
sys.exit(
|
|
786
|
+
f"error: {elf_path}: {len(segments)} segments exceeds "
|
|
787
|
+
f"SEF_MAX_SEGMENTS={SEF_MAX_SEGMENTS}"
|
|
788
|
+
)
|
|
789
|
+
|
|
790
|
+
out = bytearray()
|
|
791
|
+
out += struct.pack("<IIHH", SEF_MAGIC, elf["entry"], len(segments), flags)
|
|
792
|
+
|
|
793
|
+
dc = 12 + len(segments) * 16
|
|
794
|
+
body = bytearray()
|
|
795
|
+
for st, vaddr, size, payload in segments:
|
|
796
|
+
if payload is None:
|
|
797
|
+
start = dc
|
|
798
|
+
dc += size
|
|
799
|
+
out += struct.pack("<IIII", st, vaddr, size, start)
|
|
800
|
+
body += flat[vaddr : vaddr + size]
|
|
801
|
+
else:
|
|
802
|
+
out += struct.pack("<IIII", st, vaddr, size, dc)
|
|
803
|
+
dc += len(payload)
|
|
804
|
+
body += payload
|
|
805
|
+
|
|
806
|
+
out += body
|
|
807
|
+
|
|
808
|
+
with open(sef_path, "wb") as f:
|
|
809
|
+
f.write(out)
|
|
810
|
+
|
|
811
|
+
print(
|
|
812
|
+
f"Created {sef_path}: {len(out)} bytes, {len(segments)} segments, "
|
|
813
|
+
f"entry=0x{elf['entry']:x}, flags=0x{flags:x}"
|
|
814
|
+
)
|
|
815
|
+
print(
|
|
816
|
+
f" relocs={len(reloc_records)} imports={len(imports)} "
|
|
817
|
+
f"exports={len(export_records)}"
|
|
818
|
+
)
|
|
819
|
+
if plt_bytes:
|
|
820
|
+
print(f" lazy PLT: {len(plt_bytes) // SEF_PLT_STUB_SIZE} stubs at 0x{plt_vaddr:x}")
|
|
821
|
+
if defined_vers:
|
|
822
|
+
print(f" version defs: {', '.join(defined_vers)}")
|
|
823
|
+
|
|
824
|
+
# Import sites must be distinct: the whole point of the v2.1 record set is
|
|
825
|
+
# that every site binds, so a duplicate here is a tool bug worth failing on.
|
|
826
|
+
sites = [imp.site for imp in imports]
|
|
827
|
+
if len(sites) != len(set(sites)):
|
|
828
|
+
sys.exit(f"error: {elf_path}: internal error, duplicate import sites")
|
|
829
|
+
|
|
830
|
+
|
|
831
|
+
if __name__ == "__main__":
|
|
832
|
+
main(sys.argv[1:])
|
|
@@ -0,0 +1,134 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Scorpion SEF format builder — ELF to SEF conversion.
|
|
3
|
+
|
|
4
|
+
NOTE: This is a DEVELOPMENT CONVENIENCE TOOL and NOT the recommended route
|
|
5
|
+
for production use. It relies on objdump/readelf/objcopy to extract section
|
|
6
|
+
data and does not preserve ELF metadata beyond segment contents. For
|
|
7
|
+
production, author SEF binaries directly using the format spec in
|
|
8
|
+
docs/exec-format.md.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
import os
|
|
12
|
+
import struct
|
|
13
|
+
import subprocess
|
|
14
|
+
import sys
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def build(elf_path, sef_output, flags=0):
|
|
18
|
+
if not os.path.isfile(elf_path):
|
|
19
|
+
print(f"error: {elf_path}: not found", file=sys.stderr)
|
|
20
|
+
sys.exit(1)
|
|
21
|
+
|
|
22
|
+
sections = {}
|
|
23
|
+
result = subprocess.run(
|
|
24
|
+
["riscv64-elf-objdump", "-h", elf_path], capture_output=True, text=True
|
|
25
|
+
)
|
|
26
|
+
if result.returncode != 0:
|
|
27
|
+
print(f"error: objdump failed on {elf_path}", file=sys.stderr)
|
|
28
|
+
sys.exit(1)
|
|
29
|
+
|
|
30
|
+
for line in result.stdout.split("\n"):
|
|
31
|
+
parts = line.split()
|
|
32
|
+
if len(parts) >= 4 and parts[0].isdigit():
|
|
33
|
+
name = parts[1]
|
|
34
|
+
if name in (".text", ".rodata", ".data", ".bss"):
|
|
35
|
+
sections[name] = int(parts[2], 16)
|
|
36
|
+
|
|
37
|
+
result = subprocess.run(
|
|
38
|
+
["riscv64-elf-readelf", "-h", elf_path], capture_output=True, text=True
|
|
39
|
+
)
|
|
40
|
+
if result.returncode != 0:
|
|
41
|
+
print(f"error: readelf failed on {elf_path}", file=sys.stderr)
|
|
42
|
+
sys.exit(1)
|
|
43
|
+
|
|
44
|
+
entry = 0
|
|
45
|
+
for line in result.stdout.split("\n"):
|
|
46
|
+
if "Entry point address" in line:
|
|
47
|
+
entry = int(line.split(":")[1].strip(), 16)
|
|
48
|
+
|
|
49
|
+
bin_path = elf_path + ".bin"
|
|
50
|
+
result = subprocess.run(
|
|
51
|
+
["riscv64-elf-objcopy", "-O", "binary", elf_path, bin_path],
|
|
52
|
+
capture_output=True,
|
|
53
|
+
text=True,
|
|
54
|
+
)
|
|
55
|
+
if result.returncode != 0:
|
|
56
|
+
print(f"error: objcopy failed on {elf_path}", file=sys.stderr)
|
|
57
|
+
sys.exit(1)
|
|
58
|
+
|
|
59
|
+
with open(bin_path, "rb") as f:
|
|
60
|
+
flat = f.read()
|
|
61
|
+
os.unlink(bin_path)
|
|
62
|
+
|
|
63
|
+
segments = []
|
|
64
|
+
off = 0
|
|
65
|
+
if ".text" in sections:
|
|
66
|
+
segments.append((0, off, sections[".text"]))
|
|
67
|
+
off += sections[".text"]
|
|
68
|
+
if ".rodata" in sections:
|
|
69
|
+
segments.append((1, off, sections[".rodata"]))
|
|
70
|
+
off += sections[".rodata"]
|
|
71
|
+
if ".data" in sections:
|
|
72
|
+
segments.append((1, off, sections[".data"]))
|
|
73
|
+
off += sections[".data"]
|
|
74
|
+
if ".bss" in sections:
|
|
75
|
+
segments.append((2, off, sections[".bss"]))
|
|
76
|
+
|
|
77
|
+
if not segments:
|
|
78
|
+
print(f"error: {elf_path}: no known sections found", file=sys.stderr)
|
|
79
|
+
sys.exit(1)
|
|
80
|
+
|
|
81
|
+
num = len(segments)
|
|
82
|
+
hdr = 12 + num * 16
|
|
83
|
+
|
|
84
|
+
out = bytearray()
|
|
85
|
+
out += struct.pack("<IIHH", 0x00464553, entry, num, flags)
|
|
86
|
+
dc = hdr
|
|
87
|
+
for st, sv, ss in segments:
|
|
88
|
+
out += struct.pack("<IIII", st, sv, ss, dc)
|
|
89
|
+
dc += ss
|
|
90
|
+
out += flat
|
|
91
|
+
|
|
92
|
+
with open(sef_output, "wb") as f:
|
|
93
|
+
f.write(out)
|
|
94
|
+
|
|
95
|
+
print(
|
|
96
|
+
f"Created {sef_output}: {len(out)} bytes, {num} segments, "
|
|
97
|
+
f"entry=0x{entry:x}, flags=0x{flags:x}"
|
|
98
|
+
)
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def build_h(sef_path, h_output):
|
|
102
|
+
if not os.path.isfile(sef_path):
|
|
103
|
+
print(f"error: {sef_path}: not found", file=sys.stderr)
|
|
104
|
+
sys.exit(1)
|
|
105
|
+
|
|
106
|
+
with open(sef_path, "rb") as f:
|
|
107
|
+
data = f.read()
|
|
108
|
+
base = os.path.splitext(os.path.basename(sef_path))[0]
|
|
109
|
+
guard = f"SCORPION_{base.upper()}_SEF_H"
|
|
110
|
+
define = f"{base.upper()}_SEF_SIZE"
|
|
111
|
+
var = f"{base}_sef"
|
|
112
|
+
with open(h_output, "w") as f:
|
|
113
|
+
f.write(f"#ifndef {guard}\n#define {guard}\n\n")
|
|
114
|
+
f.write(f"#define {define} {len(data)}\n\n")
|
|
115
|
+
f.write(f"static const unsigned char {var}[] = {{\n")
|
|
116
|
+
for i in range(0, len(data), 12):
|
|
117
|
+
chunk = data[i : i + 12]
|
|
118
|
+
f.write(" " + ", ".join(f"0x{b:02x}" for b in chunk) + ",\n")
|
|
119
|
+
f.write("};\n\n#endif\n")
|
|
120
|
+
print(f"Generated {h_output} from {sef_path}")
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
if __name__ == "__main__":
|
|
124
|
+
args = sys.argv[1:]
|
|
125
|
+
if len(args) < 2:
|
|
126
|
+
print(
|
|
127
|
+
f"usage: {sys.argv[0]} <input.elf> <output.sef>",
|
|
128
|
+
file=sys.stderr,
|
|
129
|
+
)
|
|
130
|
+
sys.exit(1)
|
|
131
|
+
elf_path, sef_output = args[0], args[1]
|
|
132
|
+
# No --flags: privilege is not encoded in the image. Whether an image runs
|
|
133
|
+
# as the manager is decided by the kernel boot path that spawns it.
|
|
134
|
+
build(elf_path, sef_output, 0)
|
|
@@ -10,6 +10,7 @@ source/cpyte/bignum.c
|
|
|
10
10
|
source/cpyte/bytecoding.py
|
|
11
11
|
source/cpyte/clib.py
|
|
12
12
|
source/cpyte/compiling.py
|
|
13
|
+
source/cpyte/elf2sef.py
|
|
13
14
|
source/cpyte/extension_hooks.py
|
|
14
15
|
source/cpyte/formatter.py
|
|
15
16
|
source/cpyte/gc_runtime.c
|
|
@@ -18,6 +19,7 @@ source/cpyte/lexar.py
|
|
|
18
19
|
source/cpyte/linker.py
|
|
19
20
|
source/cpyte/lsp_server.py
|
|
20
21
|
source/cpyte/mainpie.py
|
|
22
|
+
source/cpyte/mksef.py
|
|
21
23
|
source/cpyte/optimizations.py
|
|
22
24
|
source/cpyte/package_manifest.py
|
|
23
25
|
source/cpyte/runtime.c
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|