scientific-method-engine 0.4.0__tar.gz → 0.6.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- scientific_method_engine-0.4.0/README.md → scientific_method_engine-0.6.0/PKG-INFO +27 -2
- scientific_method_engine-0.4.0/PKG-INFO → scientific_method_engine-0.6.0/README.md +12 -14
- {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/pyproject.toml +8 -3
- {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/cli.py +1 -0
- {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/x86/dispatch.py +1 -1
- {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/x86/effect_order.py +9 -5
- scientific_method_engine-0.6.0/src/scientific_method_engine/x86/handwritten.py +391 -0
- scientific_method_engine-0.6.0/src/scientific_method_engine/x86/machine.py +404 -0
- scientific_method_engine-0.6.0/src/scientific_method_engine/x86/pcode.py +376 -0
- scientific_method_engine-0.6.0/src/scientific_method_engine/x86/pcode_backend.py +835 -0
- {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/x86/reports.py +5 -3
- scientific_method_engine-0.6.0/src/scientific_method_engine/x86/semantics.py +77 -0
- {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/x86/trace.py +114 -14
- scientific_method_engine-0.6.0/tests/differential.py +120 -0
- scientific_method_engine-0.6.0/tests/oracle.py +75 -0
- scientific_method_engine-0.6.0/tests/test_differential.py +71 -0
- {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/tests/test_dispatch.py +1 -1
- {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/tests/test_effect_order.py +3 -1
- scientific_method_engine-0.6.0/tests/test_nested_frame_request.py +61 -0
- scientific_method_engine-0.6.0/tests/test_oracle.py +155 -0
- {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/tests/test_pe.py +16 -1
- scientific_method_engine-0.6.0/tests/test_table_continuations.py +161 -0
- {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/tests/test_x86.py +55 -1
- scientific_method_engine-0.4.0/src/scientific_method_engine/x86/machine.py +0 -693
- {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/.gitignore +0 -0
- {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/LICENSE +0 -0
- {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/__init__.py +0 -0
- {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/__main__.py +0 -0
- {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/ghidra/ClearNoReturnFunctions.java +0 -0
- {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/ghidra/CreateFunctions.java +0 -0
- {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/ghidra/ExportBoundedFlow.java +0 -0
- {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/ghidra/ExportFunctionFingerprints.java +0 -0
- {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/ghidra/ExportFunctionInventory.java +0 -0
- {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/ghidra/MergeFallThroughFragment.java +0 -0
- {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/ghidra/RecoverCitedFunctions.java +0 -0
- {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/ghidra/RepairReturningCallers.java +0 -0
- {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/ghidra/ReportCallArguments.java +0 -0
- {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/ghidra/ReportCallPaths.java +0 -0
- {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/ghidra/ReportCallSitesWithScalars.java +0 -0
- {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/ghidra/ReportCallsToRange.java +0 -0
- {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/ghidra/ReportConstantFirstArgumentCalls.java +0 -0
- {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/ghidra/ReportDataBytes.java +0 -0
- {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/ghidra/ReportDecompileMatches.java +0 -0
- {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/ghidra/ReportDecompileWindow.java +0 -0
- {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/ghidra/ReportFilePatternInMemory.java +0 -0
- {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/ghidra/ReportFirstArgumentCallSummary.java +0 -0
- {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/ghidra/ReportFunctionScalarConstants.java +0 -0
- {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/ghidra/ReportFunctionSummary.java +0 -0
- {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/ghidra/ReportInstructionContext.java +0 -0
- {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/ghidra/ReportInstructionWindow.java +0 -0
- {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/ghidra/ReportMemoryBlockForFileOffset.java +0 -0
- {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/ghidra/ReportMemoryBlocks.java +0 -0
- {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/ghidra/ReportRandomnessCandidates.java +0 -0
- {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/ghidra/ReportReferences.java +0 -0
- {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/ghidra/ReportScalarConstants.java +0 -0
- {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/ghidra/ReportStringReferences.java +0 -0
- {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/ghidra/ReportSymbolReferences.java +0 -0
- {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/x86/__init__.py +0 -0
- {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/x86/image.py +0 -0
- {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/x86/pe.py +0 -0
- {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/x86/result_flow.py +0 -0
- {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/x86/values.py +0 -0
|
@@ -1,8 +1,24 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: scientific-method-engine
|
|
3
|
+
Version: 0.6.0
|
|
4
|
+
Summary: Bounded instruction-derived x86 evidence reports for segmented MZ/FBOV and PE32/i386 code.
|
|
5
|
+
Project-URL: Source, https://github.com/kibertoad/refurbished-dinosaurs-toolkit/tree/main/packages/scientific-method-engine
|
|
6
|
+
Author: kibertoad
|
|
7
|
+
License-Expression: MIT
|
|
8
|
+
License-File: LICENSE
|
|
9
|
+
Requires-Python: >=3.12
|
|
10
|
+
Requires-Dist: capstone==5.0.7
|
|
11
|
+
Requires-Dist: pypcode==4.0.0
|
|
12
|
+
Provides-Extra: test
|
|
13
|
+
Requires-Dist: unicorn==2.1.4; extra == 'test'
|
|
14
|
+
Description-Content-Type: text/markdown
|
|
15
|
+
|
|
1
16
|
# scientific-method-engine
|
|
2
17
|
|
|
3
18
|
Bounded instruction-derived x86 evidence reports for segmented 16-bit MZ/FBOV code and
|
|
4
|
-
PE32/i386 code. The engine decodes instructions with Capstone,
|
|
5
|
-
|
|
19
|
+
PE32/i386 code. The engine decodes instructions with Capstone, takes their values, flags and
|
|
20
|
+
branch conditions from Ghidra's SLEIGH specification through pypcode, follows bounded paths and
|
|
21
|
+
emits `bounded-x86-v1` JSON. It never runs the original program. It needs Python 3.12 or later.
|
|
6
22
|
|
|
7
23
|
```sh
|
|
8
24
|
uv add --group research scientific-method-engine # or: pip install scientific-method-engine
|
|
@@ -101,3 +117,12 @@ The `effects` command additionally emits path-local `effectOrdering` timelines,
|
|
|
101
117
|
pre-call write prefixes and bounded local restoration witnesses. Unknown modeled
|
|
102
118
|
or nested service effects stay separate; no return code establishes rollback or
|
|
103
119
|
transactionality. See [the full contract](https://github.com/kibertoad/refurbished-dinosaurs-toolkit/blob/main/docs/bounded-evidence-reporters.md#ordered-effect-path-summaries).
|
|
120
|
+
|
|
121
|
+
Evidenced segmented16 `indirectJumps` declarations also expose separate
|
|
122
|
+
`declaredContinuationPaths`. Ordinary paths retain their unresolved transfer;
|
|
123
|
+
conditional routes preserve prefix/child effects with target-choice, live-table
|
|
124
|
+
and selector assumptions. Effects summaries never mark those routes complete.
|
|
125
|
+
Concrete values/field addresses reject inconsistent rows; overlap and shared
|
|
126
|
+
path/step/visit/total/boundary limits remain explicit. Partial-table splits are
|
|
127
|
+
partial evidence, never complete dispatch or native-reachability claims. See the
|
|
128
|
+
[table contract](https://github.com/kibertoad/refurbished-dinosaurs-toolkit/blob/main/docs/bounded-evidence-reporters.md#evidenced-indirect-jump-tables).
|
|
@@ -1,20 +1,9 @@
|
|
|
1
|
-
Metadata-Version: 2.5
|
|
2
|
-
Name: scientific-method-engine
|
|
3
|
-
Version: 0.4.0
|
|
4
|
-
Summary: Bounded instruction-derived x86 evidence reports for segmented MZ/FBOV and PE32/i386 code.
|
|
5
|
-
Project-URL: Source, https://github.com/kibertoad/refurbished-dinosaurs-toolkit/tree/main/packages/scientific-method-engine
|
|
6
|
-
Author: kibertoad
|
|
7
|
-
License-Expression: MIT
|
|
8
|
-
License-File: LICENSE
|
|
9
|
-
Requires-Python: >=3.10
|
|
10
|
-
Requires-Dist: capstone==5.0.7
|
|
11
|
-
Description-Content-Type: text/markdown
|
|
12
|
-
|
|
13
1
|
# scientific-method-engine
|
|
14
2
|
|
|
15
3
|
Bounded instruction-derived x86 evidence reports for segmented 16-bit MZ/FBOV code and
|
|
16
|
-
PE32/i386 code. The engine decodes instructions with Capstone,
|
|
17
|
-
|
|
4
|
+
PE32/i386 code. The engine decodes instructions with Capstone, takes their values, flags and
|
|
5
|
+
branch conditions from Ghidra's SLEIGH specification through pypcode, follows bounded paths and
|
|
6
|
+
emits `bounded-x86-v1` JSON. It never runs the original program. It needs Python 3.12 or later.
|
|
18
7
|
|
|
19
8
|
```sh
|
|
20
9
|
uv add --group research scientific-method-engine # or: pip install scientific-method-engine
|
|
@@ -113,3 +102,12 @@ The `effects` command additionally emits path-local `effectOrdering` timelines,
|
|
|
113
102
|
pre-call write prefixes and bounded local restoration witnesses. Unknown modeled
|
|
114
103
|
or nested service effects stay separate; no return code establishes rollback or
|
|
115
104
|
transactionality. See [the full contract](https://github.com/kibertoad/refurbished-dinosaurs-toolkit/blob/main/docs/bounded-evidence-reporters.md#ordered-effect-path-summaries).
|
|
105
|
+
|
|
106
|
+
Evidenced segmented16 `indirectJumps` declarations also expose separate
|
|
107
|
+
`declaredContinuationPaths`. Ordinary paths retain their unresolved transfer;
|
|
108
|
+
conditional routes preserve prefix/child effects with target-choice, live-table
|
|
109
|
+
and selector assumptions. Effects summaries never mark those routes complete.
|
|
110
|
+
Concrete values/field addresses reject inconsistent rows; overlap and shared
|
|
111
|
+
path/step/visit/total/boundary limits remain explicit. Partial-table splits are
|
|
112
|
+
partial evidence, never complete dispatch or native-reachability claims. See the
|
|
113
|
+
[table contract](https://github.com/kibertoad/refurbished-dinosaurs-toolkit/blob/main/docs/bounded-evidence-reporters.md#evidenced-indirect-jump-tables).
|
|
@@ -5,13 +5,18 @@ build-backend = "hatchling.build"
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "scientific-method-engine"
|
|
7
7
|
# The release workflow writes the published version from the package's release tag.
|
|
8
|
-
version = "0.
|
|
8
|
+
version = "0.6.0"
|
|
9
9
|
description = "Bounded instruction-derived x86 evidence reports for segmented MZ/FBOV and PE32/i386 code."
|
|
10
10
|
readme = "README.md"
|
|
11
|
-
requires-python = ">=3.
|
|
11
|
+
requires-python = ">=3.12"
|
|
12
12
|
license = "MIT"
|
|
13
13
|
authors = [{ name = "kibertoad" }]
|
|
14
|
-
dependencies = ["capstone==5.0.7"]
|
|
14
|
+
dependencies = ["capstone==5.0.7", "pypcode==4.0.0"]
|
|
15
|
+
|
|
16
|
+
[project.optional-dependencies]
|
|
17
|
+
# Unicorn is the tests' concrete oracle (ADR 0003, decision 4). Its core is GPLv2, so it never
|
|
18
|
+
# becomes a runtime dependency.
|
|
19
|
+
test = ["unicorn==2.1.4"]
|
|
15
20
|
|
|
16
21
|
[project.scripts]
|
|
17
22
|
scientific-method-engine = "scientific_method_engine.cli:run"
|
|
@@ -10,6 +10,7 @@ PREPARED_CONFIG_LIMIT = 16 * 1024 * 1024
|
|
|
10
10
|
USAGE = ("Usage: scientific-method-engine <operand|operand-candidates|target|bounds|owner|callees|trace|uses|arguments|"
|
|
11
11
|
"effects|returns|memory|incoming|call-order|guards|allocation|dispatch> <config.json|->\n"
|
|
12
12
|
"effects includes ordered path writes/calls and local restoration witnesses; transactionality remains unestablished.\n"
|
|
13
|
+
"Declared table continuations are separate conditional paths; ordinary computed transfers remain stopped.\n"
|
|
13
14
|
" scientific-method-engine ghidra-scripts")
|
|
14
15
|
|
|
15
16
|
|
|
@@ -3,7 +3,7 @@ from bisect import bisect_right
|
|
|
3
3
|
|
|
4
4
|
|
|
5
5
|
KINDS = {"read", "write", "call", "call-return", "return", "branch", "compare", "flag-assumption",
|
|
6
|
-
"arithmetic", "value-transfer", "conversion", "flag-write", "flags-save", "flags-restore", "local-iret", "string-operation"}
|
|
6
|
+
"arithmetic", "value-transfer", "conversion", "flag-write", "flags-save", "flags-restore", "local-iret", "string-operation", "declared-jump-continuation"}
|
|
7
7
|
|
|
8
8
|
|
|
9
9
|
def _storage(event):
|
|
@@ -25,7 +25,10 @@ def effect_ordering(report):
|
|
|
25
25
|
aliases, external effects, resources or transactionality.
|
|
26
26
|
"""
|
|
27
27
|
summaries = []
|
|
28
|
-
|
|
28
|
+
conditional_summaries = []
|
|
29
|
+
combined = [(False, i, p) for i, p in enumerate(report["paths"])]
|
|
30
|
+
combined += [(True, i, p) for i, p in enumerate(report.get("declaredContinuationPaths", []))]
|
|
31
|
+
for conditional, index, path in combined:
|
|
29
32
|
timeline, writes, calls, witnesses = [], [], [], []
|
|
30
33
|
snapshots = {}
|
|
31
34
|
pending = {}
|
|
@@ -78,13 +81,14 @@ def effect_ordering(report):
|
|
|
78
81
|
if path.get("stop"):
|
|
79
82
|
boundary = {"site": path.get("stopSite"), "reason": path["stop"],
|
|
80
83
|
"writesBeforeCount": len(writes), "meaning": "later effects are not read"}
|
|
81
|
-
|
|
84
|
+
destination = conditional_summaries if conditional else summaries
|
|
85
|
+
destination.append({"path": index, "declaredJumpAssumptions": path.get("declaredJumpAssumptions", []), "returned": path["returned"], "stop": boundary,
|
|
82
86
|
"guards": path["guards"], "timeline": timeline,
|
|
83
87
|
"writeOrders": [w["order"] for w in writes], "calls": calls,
|
|
84
88
|
"localRestorationWitnesses": witnesses,
|
|
85
|
-
"effectCompleteWithinModel": bool(path["returned"] and not unknown_orders),
|
|
89
|
+
"effectCompleteWithinModel": bool(path["returned"] and not unknown_orders and not conditional),
|
|
86
90
|
"transactionality": "not established; local writes and result codes cannot prove external rollback"})
|
|
87
|
-
report["effectOrdering"] = {"paths": summaries,
|
|
91
|
+
report["effectOrdering"] = {"paths": summaries, "declaredContinuationPaths": conditional_summaries,
|
|
88
92
|
"allPathsRead": report["completeWithinModel"],
|
|
89
93
|
"nativeReachability": "unconfirmed",
|
|
90
94
|
"meaning": "separate conditional paths; write prefixes index each path's writeOrders"}
|
|
@@ -0,0 +1,391 @@
|
|
|
1
|
+
"""The handwritten semantics backend (ADR 0003): value computation, flag predicates and string bodies.
|
|
2
|
+
|
|
3
|
+
ADR 0003 decision 6 freezes this module. No mnemonic, flag rule or value computation is added here;
|
|
4
|
+
the pypcode backend replaces it group by group, and phase 5 deletes it.
|
|
5
|
+
"""
|
|
6
|
+
from capstone.x86 import X86_OP_IMM, X86_OP_REG, X86_OP_MEM
|
|
7
|
+
from . import semantics
|
|
8
|
+
from .machine import ALIASES, StopPath
|
|
9
|
+
from .values import Value, const, unknown, op, extract, join, resize, sources
|
|
10
|
+
|
|
11
|
+
CARRY_BRANCHES = {"jb": True, "jc": True, "jnae": True, "jae": False, "jnb": False, "jnc": False}
|
|
12
|
+
# Branches taken when CF or OF is set; logic operations clear both whatever their operands.
|
|
13
|
+
CLEARED_BY_LOGIC = {**CARRY_BRANCHES, "jo": True, "jno": False}
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def predicate(state, mnemonic):
|
|
17
|
+
flags = state.flags
|
|
18
|
+
if flags is None and state.carry is not None and mnemonic in CARRY_BRANCHES:
|
|
19
|
+
info = {"predicate": mnemonic, "flag": "CF", "carry": state.carry.report()}
|
|
20
|
+
if state.carry.number is None:
|
|
21
|
+
return None, {**info, "reason": "carry unresolved"}
|
|
22
|
+
return bool(state.carry.number) == CARRY_BRANCHES[mnemonic], info
|
|
23
|
+
if flags is None:
|
|
24
|
+
return None, {"predicate": mnemonic, "reason": "flag producer unresolved",
|
|
25
|
+
"flagProducer": state.unknown_flag_site, "flagGeneration": state.flag_epoch}
|
|
26
|
+
a, b, operation, site = flags
|
|
27
|
+
info = {"predicate": mnemonic, "flagProducer": site, "operation": operation,
|
|
28
|
+
"left": a.report(), "right": b.report()}
|
|
29
|
+
if a.number is not None and b.number is not None:
|
|
30
|
+
x, y = a.number, b.number
|
|
31
|
+
elif operation in ("cmp", "sub", "xor") and a.term == b.term:
|
|
32
|
+
# Any value compared with, subtracted from or XORed with itself yields zero.
|
|
33
|
+
x = y = 0
|
|
34
|
+
elif operation in ("test", "and", "or", "xor") and mnemonic in CLEARED_BY_LOGIC:
|
|
35
|
+
return not CLEARED_BY_LOGIC[mnemonic], info
|
|
36
|
+
else:
|
|
37
|
+
return None, info
|
|
38
|
+
bits = a.bits
|
|
39
|
+
if operation in ("cmp", "sub"):
|
|
40
|
+
raw = x - y
|
|
41
|
+
result = raw % (1 << bits)
|
|
42
|
+
cf = x < y
|
|
43
|
+
of = bool(((x ^ y) & (x ^ result)) & (1 << (bits - 1)))
|
|
44
|
+
elif operation == "add":
|
|
45
|
+
raw = x + y
|
|
46
|
+
result = raw % (1 << bits)
|
|
47
|
+
cf = raw >= 1 << bits
|
|
48
|
+
of = bool((~(x ^ y) & (x ^ result)) & (1 << (bits - 1)))
|
|
49
|
+
elif operation in ("test", "and", "or", "xor"):
|
|
50
|
+
result = {"test": x & y, "and": x & y, "or": x | y, "xor": x ^ y}[operation]
|
|
51
|
+
cf = of = False
|
|
52
|
+
else:
|
|
53
|
+
return None, info
|
|
54
|
+
zf, sf = result == 0, bool(result & (1 << (bits - 1)))
|
|
55
|
+
conditions = {"je": zf, "jz": zf, "jne": not zf, "jnz": not zf,
|
|
56
|
+
"jb": cf, "jc": cf, "jnae": cf, "jae": not cf, "jnb": not cf, "jnc": not cf,
|
|
57
|
+
"jbe": cf or zf, "jna": cf or zf, "ja": not cf and not zf, "jnbe": not cf and not zf,
|
|
58
|
+
"jl": sf != of, "jnge": sf != of, "jge": sf == of, "jnl": sf == of,
|
|
59
|
+
"jle": zf or sf != of, "jng": zf or sf != of, "jg": not zf and sf == of,
|
|
60
|
+
"jnle": not zf and sf == of, "js": sf, "jns": not sf, "jo": of, "jno": not of}
|
|
61
|
+
return conditions.get(mnemonic), info
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def ordinary(state, ins, image):
|
|
65
|
+
m, operands = ins.mnemonic, ins.operands
|
|
66
|
+
if m in ("cld", "std", "cli", "sti"):
|
|
67
|
+
value = const(1 if m in ("std", "sti") else 0, 1, state.at)
|
|
68
|
+
flag = "DF" if m in ("cld", "std") else "IF"
|
|
69
|
+
if flag == "DF": state.direction_flag = value
|
|
70
|
+
else: state.interrupt_flag = value
|
|
71
|
+
state.event("flag-write", flag=flag, value=value.report(),
|
|
72
|
+
interpretation="local flag effect only; interrupts and timing are not simulated")
|
|
73
|
+
return
|
|
74
|
+
if m in ("pushf", "pushfd", "popf", "popfd"):
|
|
75
|
+
bits = 32 if (0x66 in ins.prefix) != state.flat else 16
|
|
76
|
+
if m.startswith("push"): state.save_flags(bits)
|
|
77
|
+
else: state.restore_flags(bits)
|
|
78
|
+
return
|
|
79
|
+
if m == "nop":
|
|
80
|
+
return
|
|
81
|
+
if m in ("mov", "movzx", "movsx"):
|
|
82
|
+
if state.flat and operands[0].type == X86_OP_REG and ins.reg_name(operands[0].reg) in state.segment_bases:
|
|
83
|
+
raise StopPath("Segment selector assignment requires a descriptor model")
|
|
84
|
+
value = state.get(ins, operands[1], image)
|
|
85
|
+
result = resize(value, operands[0].size * 8, signed=m == "movsx")
|
|
86
|
+
state.put(ins, operands[0], result)
|
|
87
|
+
if not state.value_transfers:
|
|
88
|
+
return
|
|
89
|
+
|
|
90
|
+
def location(operand):
|
|
91
|
+
return {"kind": "register", "register": ins.reg_name(operand.reg)} if operand.type == X86_OP_REG else {"kind": "memory"} if operand.type == X86_OP_MEM else {"kind": "immediate"}
|
|
92
|
+
destination_container = ALIASES[ins.reg_name(operands[0].reg)][0] if operands[0].type == X86_OP_REG else None
|
|
93
|
+
state.event("value-transfer", operation=m, source=location(operands[1]), destination=location(operands[0]),
|
|
94
|
+
destinationContainer=destination_container, destinationContainerValue=state.reg(destination_container).report() if destination_container else None,
|
|
95
|
+
sourceBits=value.bits, destinationBits=result.bits, sourceValue=value.report(), resultValue=result.report(),
|
|
96
|
+
conversion="truncate" if result.bits < value.bits else "signExtend" if m == "movsx" else "zeroExtend" if result.bits > value.bits else "sameWidth")
|
|
97
|
+
return
|
|
98
|
+
if m == "xchg":
|
|
99
|
+
values = [state.get(ins, operand, image) for operand in operands]
|
|
100
|
+
addresses = [state.address(ins, operand) if operand.type == X86_OP_MEM else None for operand in operands]
|
|
101
|
+
for index, operand in enumerate(operands):
|
|
102
|
+
value = values[1-index]
|
|
103
|
+
if addresses[index] is None:
|
|
104
|
+
state.put(ins, operand, value)
|
|
105
|
+
else:
|
|
106
|
+
segment, offset, register = addresses[index]
|
|
107
|
+
state.access(segment, offset, operand.size, resize(value, operand.size * 8), addressing_register=register)
|
|
108
|
+
return
|
|
109
|
+
if m == "imul" and len(operands) in (2, 3):
|
|
110
|
+
left, right = (state.get(ins, operand, image) for operand in (operands if len(operands) == 2 else operands[1:]))
|
|
111
|
+
left = resize(left, operands[0].size * 8)
|
|
112
|
+
right = resize(right, left.bits, signed=True)
|
|
113
|
+
result = op("mul", left, right, state.at)
|
|
114
|
+
state.put(ins, operands[0], result)
|
|
115
|
+
state.forget_flags() # CF/OF require the full signed product; other flags are undefined.
|
|
116
|
+
state.event("arithmetic", operation="imul", left=left.report(), right=right.report(),
|
|
117
|
+
result=result.report(), modulus=1 << left.bits, flags="unresolved signed-product overflow")
|
|
118
|
+
return
|
|
119
|
+
if m == "lea":
|
|
120
|
+
segment, offset, register = state.address(ins, operands[1])
|
|
121
|
+
state.put(ins, operands[0], offset)
|
|
122
|
+
state.event("address-formation", value=offset.report(), addressingSegment=segment.report(),
|
|
123
|
+
addressingSegmentRegister=register, destinationRegister=ins.reg_name(operands[0].reg),
|
|
124
|
+
note="LEA does not access memory; this addressing default does not bind a later dereference")
|
|
125
|
+
return
|
|
126
|
+
if m in ("lds", "les"):
|
|
127
|
+
if state.flat:
|
|
128
|
+
raise StopPath("Descriptor loads are outside the PE32 flat model")
|
|
129
|
+
segment, offset, register = state.address(ins, operands[1])
|
|
130
|
+
if operands[0].size != 2:
|
|
131
|
+
raise StopPath("Only 16:16 pointer loads are supported")
|
|
132
|
+
value = state.access(segment, offset, 4, role="far-pointer", addressing_register=register)
|
|
133
|
+
state.put(ins, operands[0], extract(value, 0, 16))
|
|
134
|
+
state.setreg("ds" if m == "lds" else "es", extract(value, 16, 16), state.at)
|
|
135
|
+
return
|
|
136
|
+
if m == "push":
|
|
137
|
+
if state.flat and operands[0].type == X86_OP_REG and ins.reg_name(operands[0].reg) in state.segment_bases:
|
|
138
|
+
raise StopPath("Segment stack operations require a descriptor model")
|
|
139
|
+
state.push(state.get(ins, operands[0], image))
|
|
140
|
+
return
|
|
141
|
+
if m == "pop":
|
|
142
|
+
if state.flat and operands[0].type == X86_OP_REG and ins.reg_name(operands[0].reg) in state.segment_bases:
|
|
143
|
+
raise StopPath("Segment selector assignment requires a descriptor model")
|
|
144
|
+
state.put(ins, operands[0], state.pop(operands[0].size))
|
|
145
|
+
return
|
|
146
|
+
if m == "leave":
|
|
147
|
+
if 0x66 in ins.prefix:
|
|
148
|
+
raise StopPath("Operand-size override on LEAVE is unsupported")
|
|
149
|
+
state.setreg(state.sp, state.reg(state.bp), state.at)
|
|
150
|
+
state.setreg(state.bp, state.pop(state.bits // 8), state.at)
|
|
151
|
+
return
|
|
152
|
+
if m in ("cmp", "test"):
|
|
153
|
+
a, b = (state.get(ins, o, image) for o in operands)
|
|
154
|
+
state.set_flags(a, resize(b, a.bits), m)
|
|
155
|
+
state.event("compare", operation=m, left=a.report(), right=b.report())
|
|
156
|
+
return
|
|
157
|
+
if m in ("add", "sub", "and", "or", "xor", "shl", "sal", "shr", "sar"):
|
|
158
|
+
a, b = (state.get(ins, o, image) for o in operands)
|
|
159
|
+
b = resize(b, a.bits)
|
|
160
|
+
result = op("shl" if m == "sal" else m, a, b, state.at)
|
|
161
|
+
state.put(ins, operands[0], result)
|
|
162
|
+
if m in ("add", "sub", "and", "or", "xor"):
|
|
163
|
+
state.set_flags(a, b, m)
|
|
164
|
+
else:
|
|
165
|
+
shift_carry(state, m, a, b)
|
|
166
|
+
state.event("arithmetic", operation=m, left=a.report(), right=b.report(), result=result.report(), modulus=1 << a.bits)
|
|
167
|
+
return
|
|
168
|
+
if m in ("inc", "dec"):
|
|
169
|
+
a = state.get(ins, operands[0], image)
|
|
170
|
+
result = op("add" if m == "inc" else "sub", a, const(1, a.bits), state.at)
|
|
171
|
+
state.put(ins, operands[0], result)
|
|
172
|
+
# Carry is preserved; the other flags are not modeled for INC/DEC.
|
|
173
|
+
state.forget_flags(keep_carry=True)
|
|
174
|
+
return
|
|
175
|
+
if m in ("cbw", "cwde"):
|
|
176
|
+
# Capstone 5 names these inconsistently in 16-bit mode. Use effective size.
|
|
177
|
+
# The prefix toggles the mode's default operand size (16-bit real mode, 32-bit flat).
|
|
178
|
+
wide = (0x66 in ins.prefix) != state.flat
|
|
179
|
+
source, destination = ("ax", "eax") if wide else ("al", "ax")
|
|
180
|
+
source_value = state.reg(source)
|
|
181
|
+
value = resize(source_value, 32 if wide else 16, True)
|
|
182
|
+
state.setreg(destination, value, state.at)
|
|
183
|
+
state.event("conversion", sourceRegister=source, destinationRegister=destination,
|
|
184
|
+
effectiveOperandBits=32 if wide else 16, decoderMnemonic=m,
|
|
185
|
+
mnemonicWidthMismatch=m != ("cwde" if wide else "cbw"), result=value.report(),
|
|
186
|
+
sourceValue=source_value.report(), sourceBits=source_value.bits, destinationBits=value.bits, conversion="signExtend")
|
|
187
|
+
return
|
|
188
|
+
if m in ("cwd", "cdq"):
|
|
189
|
+
wide = (0x66 in ins.prefix) != state.flat
|
|
190
|
+
source, destination = ("eax", "edx") if wide else ("ax", "dx")
|
|
191
|
+
bits = 32 if wide else 16
|
|
192
|
+
source_value = state.reg(source)
|
|
193
|
+
value = resize(extract(source_value, bits-1, 1), bits, signed=True)
|
|
194
|
+
state.setreg(destination, value, state.at)
|
|
195
|
+
state.event("conversion", sourceRegister=source, destinationRegister=destination,
|
|
196
|
+
effectiveOperandBits=bits, decoderMnemonic=m,
|
|
197
|
+
mnemonicWidthMismatch=m != ("cdq" if wide else "cwd"), result=value.report(),
|
|
198
|
+
sourceValue=source_value.report(), sourceBits=source_value.bits, destinationBits=value.bits, conversion="signFillHighHalf")
|
|
199
|
+
return
|
|
200
|
+
if m in ("clc", "stc", "cmc"):
|
|
201
|
+
if m == "cmc":
|
|
202
|
+
value = op("xor", state.carry_value(), const(1, 1), state.at)
|
|
203
|
+
else:
|
|
204
|
+
value = const(int(m == "stc"), 1, state.at)
|
|
205
|
+
state.forget_flags()
|
|
206
|
+
state.carry = value
|
|
207
|
+
state.event("flag-write", flag="CF", value=value.report(), interpretation="local carry effect")
|
|
208
|
+
return
|
|
209
|
+
if m in ("not", "neg"):
|
|
210
|
+
a = state.get(ins, operands[0], image)
|
|
211
|
+
if m == "not":
|
|
212
|
+
state.put(ins, operands[0], op("xor", a, const((1 << a.bits) - 1, a.bits), state.at))
|
|
213
|
+
return
|
|
214
|
+
result = op("sub", const(0, a.bits), a, state.at)
|
|
215
|
+
state.put(ins, operands[0], result)
|
|
216
|
+
state.set_flags(const(0, a.bits, state.at), a, "sub")
|
|
217
|
+
state.event("arithmetic", operation="neg", left=a.report(), result=result.report(), modulus=1 << a.bits)
|
|
218
|
+
return
|
|
219
|
+
if m in ("adc", "sbb"):
|
|
220
|
+
a, b = (state.get(ins, o, image) for o in operands)
|
|
221
|
+
b = resize(b, a.bits)
|
|
222
|
+
carry = state.carry_value()
|
|
223
|
+
name = "add" if m == "adc" else "sub"
|
|
224
|
+
result = op(name, op(name, a, b, state.at), resize(carry, a.bits), state.at)
|
|
225
|
+
state.put(ins, operands[0], result)
|
|
226
|
+
state.forget_flags()
|
|
227
|
+
if None not in (a.number, b.number, carry.number):
|
|
228
|
+
raw = a.number + b.number + carry.number if m == "adc" else a.number - b.number - carry.number
|
|
229
|
+
state.carry = const(int(raw < 0 or raw >= 1 << a.bits), 1, state.at)
|
|
230
|
+
else:
|
|
231
|
+
state.carry = unknown(f"carry:{state.at}:{state.flag_serial}", 1, state.at)
|
|
232
|
+
state.event("arithmetic", operation=m, left=a.report(), right=b.report(), carryIn=carry.report(),
|
|
233
|
+
result=result.report(), carryOut=state.carry.report(), modulus=1 << a.bits)
|
|
234
|
+
return
|
|
235
|
+
if m in ("rol", "ror", "rcl", "rcr"):
|
|
236
|
+
a = state.get(ins, operands[0], image)
|
|
237
|
+
count = state.get(ins, operands[1], image) if len(operands) > 1 else const(1, 8)
|
|
238
|
+
if len(operands) > 1 and operands[1].type == X86_OP_IMM:
|
|
239
|
+
# Capstone gives RCL's implicit count of 1 on a memory operand a size of 0.
|
|
240
|
+
count = const(operands[1].imm, 8, state.at)
|
|
241
|
+
if count.number is None:
|
|
242
|
+
raise StopPath("rotate count unresolved")
|
|
243
|
+
masked = count.number & 31
|
|
244
|
+
if masked == 0:
|
|
245
|
+
return # The value and flags are unchanged.
|
|
246
|
+
bits = a.bits
|
|
247
|
+
through = m in ("rcl", "rcr")
|
|
248
|
+
n = masked % (bits + 1 if through else bits)
|
|
249
|
+
if through and n == 0:
|
|
250
|
+
# A full rotation through CF restores the value and CF; OF is undefined.
|
|
251
|
+
state.forget_flags(keep_carry=True)
|
|
252
|
+
return
|
|
253
|
+
carry_in = resize(state.carry_value(), bits)
|
|
254
|
+
# Each form is an OR of shifted copies (positive shifts left), built once so the
|
|
255
|
+
# expression does not repeat the operand for every bit rotated.
|
|
256
|
+
parts = {"rol": [(a, n), (a, n - bits)],
|
|
257
|
+
"ror": [(a, -n), (a, bits - n)],
|
|
258
|
+
"rcl": [(a, n), (carry_in, n - 1), (a, n - bits - 1)],
|
|
259
|
+
"rcr": [(a, -n), (carry_in, bits - n), (a, bits + 1 - n)]}[m]
|
|
260
|
+
value = None
|
|
261
|
+
for part, shift in parts:
|
|
262
|
+
if abs(shift) >= bits:
|
|
263
|
+
continue # Every bit leaves the operand; op() would mask the count instead.
|
|
264
|
+
if shift:
|
|
265
|
+
part = op("shl" if shift > 0 else "shr", part, const(abs(shift), bits), state.at)
|
|
266
|
+
value = part if value is None else op("or", value, part, state.at)
|
|
267
|
+
# CF is the last bit rotated out: the result's low bit for ROL/RCL and its high bit for ROR/RCR.
|
|
268
|
+
out = {"rol": (bits - n) % bits, "ror": (n - 1) % bits, "rcl": bits - n, "rcr": n - 1}[m]
|
|
269
|
+
carry = extract(a, out, 1)
|
|
270
|
+
# A count read from CL is an input of the result and of CF.
|
|
271
|
+
value = Value(value.bits, value.term, sources(value, count, site=state.at))
|
|
272
|
+
state.put(ins, operands[0], value)
|
|
273
|
+
state.forget_flags()
|
|
274
|
+
state.carry = Value(1, carry.term, sources(carry, count, site=state.at))
|
|
275
|
+
state.event("arithmetic", operation=m, left=a.report(), count=n, result=value.report(),
|
|
276
|
+
carryOut=state.carry.report(), modulus=1 << bits)
|
|
277
|
+
return
|
|
278
|
+
if m in ("mul", "imul") and len(operands) == 1:
|
|
279
|
+
source = state.get(ins, operands[0], image)
|
|
280
|
+
bits = source.bits
|
|
281
|
+
low_reg, high_reg = {8: ("al", "ah"), 16: ("ax", "dx"), 32: ("eax", "edx")}[bits]
|
|
282
|
+
signed = m == "imul"
|
|
283
|
+
multiplicand = state.reg(low_reg)
|
|
284
|
+
product = op("mul", resize(multiplicand, 2 * bits, signed), resize(source, 2 * bits, signed), state.at)
|
|
285
|
+
low = extract(product, 0, bits)
|
|
286
|
+
if bits == 8:
|
|
287
|
+
state.setreg("ax", product, state.at)
|
|
288
|
+
else:
|
|
289
|
+
state.setreg(low_reg, low, state.at)
|
|
290
|
+
state.setreg(high_reg, extract(product, bits, bits), state.at)
|
|
291
|
+
state.forget_flags()
|
|
292
|
+
if product.number is None:
|
|
293
|
+
state.carry = unknown(f"carry:{state.at}:{state.flag_serial}", 1, state.at)
|
|
294
|
+
else:
|
|
295
|
+
# CF and OF say whether the high half carries information beyond the low half.
|
|
296
|
+
state.carry = const(int(resize(low, 2 * bits, signed).number != product.number), 1, state.at)
|
|
297
|
+
state.event("arithmetic", operation=m, left=multiplicand.report(), right=source.report(),
|
|
298
|
+
result=product.report(), resultBits=2 * bits, carryOut=state.carry.report())
|
|
299
|
+
return
|
|
300
|
+
if m in ("div", "idiv"):
|
|
301
|
+
divisor = state.get(ins, operands[0], image)
|
|
302
|
+
bits = divisor.bits
|
|
303
|
+
signed = m == "idiv"
|
|
304
|
+
if bits == 8:
|
|
305
|
+
dividend = state.reg("ax")
|
|
306
|
+
else:
|
|
307
|
+
dividend = join([state.reg({16: "ax", 32: "eax"}[bits]), state.reg({16: "dx", 32: "edx"}[bits])])
|
|
308
|
+
quotient_reg, remainder_reg = {8: ("al", "ah"), 16: ("ax", "dx"), 32: ("eax", "edx")}[bits]
|
|
309
|
+
fault = None
|
|
310
|
+
if divisor.number == 0:
|
|
311
|
+
raise StopPath("divide by zero raises interrupt 0; its handler is not modeled")
|
|
312
|
+
if None not in (dividend.number, divisor.number):
|
|
313
|
+
x, y = dividend.number, divisor.number
|
|
314
|
+
if signed:
|
|
315
|
+
x -= (x >> (2 * bits - 1)) << (2 * bits)
|
|
316
|
+
y -= (y >> (bits - 1)) << bits
|
|
317
|
+
q = abs(x) // abs(y) * (1 if (x < 0) == (y < 0) else -1)
|
|
318
|
+
r = x - q * y
|
|
319
|
+
if not ((-(1 << (bits - 1)) <= q < 1 << (bits - 1)) if signed else q < 1 << bits):
|
|
320
|
+
raise StopPath("divide overflow raises interrupt 0; its handler is not modeled")
|
|
321
|
+
# Both results are computed from the dividend and divisor, so they name their producers.
|
|
322
|
+
origin = sources(dividend, divisor, site=state.at)
|
|
323
|
+
quotient, remainder = Value(bits, const(q, bits).term, origin), Value(bits, const(r, bits).term, origin)
|
|
324
|
+
else:
|
|
325
|
+
wide = resize(divisor, 2 * bits, signed)
|
|
326
|
+
origin = sources(dividend, divisor, site=state.at)
|
|
327
|
+
quotient = extract(Value(2 * bits, ("sdiv" if signed else "udiv", dividend.term, wide.term), origin), 0, bits)
|
|
328
|
+
remainder = extract(Value(2 * bits, ("smod" if signed else "umod", dividend.term, wide.term), origin), 0, bits)
|
|
329
|
+
fault = "possible divide error (interrupt 0) unresolved; this path assumes none"
|
|
330
|
+
state.conditional.append({"site": state.at, "assumption": "no divide error"})
|
|
331
|
+
state.setreg(quotient_reg, quotient, state.at)
|
|
332
|
+
state.setreg(remainder_reg, remainder, state.at)
|
|
333
|
+
state.forget_flags()
|
|
334
|
+
state.event("arithmetic", operation=m, dividend=dividend.report(), divisor=divisor.report(),
|
|
335
|
+
quotient=quotient.report(), remainder=remainder.report(), fault=fault)
|
|
336
|
+
return
|
|
337
|
+
raise StopPath("Unsupported instruction semantics: " + m)
|
|
338
|
+
|
|
339
|
+
|
|
340
|
+
def shift_carry(state, m, a, count):
|
|
341
|
+
"""CF after SHL/SHR/SAR: the last bit shifted out, when the count is known."""
|
|
342
|
+
n = None if count.number is None else count.number & 31
|
|
343
|
+
if n == 0:
|
|
344
|
+
return # A zero count leaves every flag unchanged.
|
|
345
|
+
state.forget_flags()
|
|
346
|
+
if n is None or n > a.bits:
|
|
347
|
+
state.carry = unknown(f"carry:{state.at}:{state.flag_serial}", 1, state.at)
|
|
348
|
+
elif m in ("shl", "sal"):
|
|
349
|
+
state.carry = Value(1, extract(a, a.bits - n, 1).term, sources(a, site=state.at))
|
|
350
|
+
else:
|
|
351
|
+
state.carry = Value(1, extract(a, n - 1, 1).term, sources(a, site=state.at))
|
|
352
|
+
|
|
353
|
+
|
|
354
|
+
def string_iteration(state, ins, operation, width, source_name, delta):
|
|
355
|
+
if operation in ("cmps", "scas"):
|
|
356
|
+
raise StopPath("Unsupported instruction semantics: " + ins.mnemonic)
|
|
357
|
+
si, di = ("esi", "edi") if state.flat else ("si", "di")
|
|
358
|
+
# DF chooses the step's sign, so the instruction that set DF is one of the step's producers.
|
|
359
|
+
step = Value(state.bits, const(delta, state.bits).term, state.direction_flag.sources)
|
|
360
|
+
if operation in ("movs", "lods"):
|
|
361
|
+
value = state.access(state.segment(source_name), state.reg(si), width, role="string-source", addressing_register=source_name)
|
|
362
|
+
state.setreg(si, op("add", state.reg(si), step, state.at), state.at)
|
|
363
|
+
else:
|
|
364
|
+
value = state.reg({1:"al",2:"ax",4:"eax"}[width])
|
|
365
|
+
if operation in ("movs", "stos"):
|
|
366
|
+
state.access(state.segment("es"), state.reg(di), width, value, role="string-destination", addressing_register="es")
|
|
367
|
+
state.setreg(di, op("add", state.reg(di), step, state.at), state.at)
|
|
368
|
+
else:
|
|
369
|
+
state.setreg({1:"al",2:"ax",4:"eax"}[width], value, state.at)
|
|
370
|
+
|
|
371
|
+
|
|
372
|
+
class Handwritten:
|
|
373
|
+
"""The handwritten semantics backend: ``ordinary``, ``predicate`` and ``string_iteration``."""
|
|
374
|
+
|
|
375
|
+
name = "handwritten"
|
|
376
|
+
|
|
377
|
+
def ordinary(self, state, ins, image):
|
|
378
|
+
ordinary(state, ins, image)
|
|
379
|
+
|
|
380
|
+
def condition(self, state, mnemonic):
|
|
381
|
+
return predicate(state, mnemonic)
|
|
382
|
+
|
|
383
|
+
def string_iteration(self, state, ins, operation, width, source_segment, delta):
|
|
384
|
+
string_iteration(state, ins, operation, width, source_segment, delta)
|
|
385
|
+
|
|
386
|
+
def __deepcopy__(self, memo):
|
|
387
|
+
# Backends hold no path state, so every copied path shares one.
|
|
388
|
+
return self
|
|
389
|
+
|
|
390
|
+
|
|
391
|
+
semantics.register(Handwritten(), default=True)
|