scientific-method-engine 0.7.0__tar.gz → 0.9.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {scientific_method_engine-0.7.0 → scientific_method_engine-0.9.0}/PKG-INFO +2 -1
- {scientific_method_engine-0.7.0 → scientific_method_engine-0.9.0}/README.md +1 -0
- {scientific_method_engine-0.7.0 → scientific_method_engine-0.9.0}/pyproject.toml +1 -1
- {scientific_method_engine-0.7.0 → scientific_method_engine-0.9.0}/src/scientific_method_engine/cli.py +9 -1
- scientific_method_engine-0.9.0/src/scientific_method_engine/ghidra/ExportCallEdges.java +138 -0
- {scientific_method_engine-0.7.0 → scientific_method_engine-0.9.0}/src/scientific_method_engine/x86/machine.py +13 -14
- {scientific_method_engine-0.7.0 → scientific_method_engine-0.9.0}/src/scientific_method_engine/x86/pcode_backend.py +63 -55
- {scientific_method_engine-0.7.0 → scientific_method_engine-0.9.0}/src/scientific_method_engine/x86/reports.py +105 -3
- {scientific_method_engine-0.7.0 → scientific_method_engine-0.9.0}/tests/oracle.py +6 -6
- {scientific_method_engine-0.7.0 → scientific_method_engine-0.9.0}/tests/test_dispatch.py +1 -1
- {scientific_method_engine-0.7.0 → scientific_method_engine-0.9.0}/tests/test_effect_order.py +1 -3
- {scientific_method_engine-0.7.0 → scientific_method_engine-0.9.0}/tests/test_oracle.py +29 -16
- {scientific_method_engine-0.7.0 → scientific_method_engine-0.9.0}/tests/test_pe.py +2 -3
- {scientific_method_engine-0.7.0 → scientific_method_engine-0.9.0}/tests/test_x86.py +153 -28
- scientific_method_engine-0.7.0/src/scientific_method_engine/x86/handwritten.py +0 -391
- scientific_method_engine-0.7.0/src/scientific_method_engine/x86/semantics.py +0 -77
- scientific_method_engine-0.7.0/tests/differential.py +0 -120
- scientific_method_engine-0.7.0/tests/test_differential.py +0 -71
- {scientific_method_engine-0.7.0 → scientific_method_engine-0.9.0}/.gitignore +0 -0
- {scientific_method_engine-0.7.0 → scientific_method_engine-0.9.0}/LICENSE +0 -0
- {scientific_method_engine-0.7.0 → scientific_method_engine-0.9.0}/src/scientific_method_engine/__init__.py +0 -0
- {scientific_method_engine-0.7.0 → scientific_method_engine-0.9.0}/src/scientific_method_engine/__main__.py +0 -0
- {scientific_method_engine-0.7.0 → scientific_method_engine-0.9.0}/src/scientific_method_engine/ghidra/ClearNoReturnFunctions.java +0 -0
- {scientific_method_engine-0.7.0 → scientific_method_engine-0.9.0}/src/scientific_method_engine/ghidra/CreateFunctions.java +0 -0
- {scientific_method_engine-0.7.0 → scientific_method_engine-0.9.0}/src/scientific_method_engine/ghidra/ExportBoundedFlow.java +0 -0
- {scientific_method_engine-0.7.0 → scientific_method_engine-0.9.0}/src/scientific_method_engine/ghidra/ExportFunctionFingerprints.java +0 -0
- {scientific_method_engine-0.7.0 → scientific_method_engine-0.9.0}/src/scientific_method_engine/ghidra/ExportFunctionInventory.java +0 -0
- {scientific_method_engine-0.7.0 → scientific_method_engine-0.9.0}/src/scientific_method_engine/ghidra/MergeFallThroughFragment.java +0 -0
- {scientific_method_engine-0.7.0 → scientific_method_engine-0.9.0}/src/scientific_method_engine/ghidra/RecoverCitedFunctions.java +0 -0
- {scientific_method_engine-0.7.0 → scientific_method_engine-0.9.0}/src/scientific_method_engine/ghidra/RepairReturningCallers.java +0 -0
- {scientific_method_engine-0.7.0 → scientific_method_engine-0.9.0}/src/scientific_method_engine/ghidra/ReportCallArguments.java +0 -0
- {scientific_method_engine-0.7.0 → scientific_method_engine-0.9.0}/src/scientific_method_engine/ghidra/ReportCallPaths.java +0 -0
- {scientific_method_engine-0.7.0 → scientific_method_engine-0.9.0}/src/scientific_method_engine/ghidra/ReportCallSitesWithScalars.java +0 -0
- {scientific_method_engine-0.7.0 → scientific_method_engine-0.9.0}/src/scientific_method_engine/ghidra/ReportCallsToRange.java +0 -0
- {scientific_method_engine-0.7.0 → scientific_method_engine-0.9.0}/src/scientific_method_engine/ghidra/ReportConstantFirstArgumentCalls.java +0 -0
- {scientific_method_engine-0.7.0 → scientific_method_engine-0.9.0}/src/scientific_method_engine/ghidra/ReportDataBytes.java +0 -0
- {scientific_method_engine-0.7.0 → scientific_method_engine-0.9.0}/src/scientific_method_engine/ghidra/ReportDecompileMatches.java +0 -0
- {scientific_method_engine-0.7.0 → scientific_method_engine-0.9.0}/src/scientific_method_engine/ghidra/ReportDecompileWindow.java +0 -0
- {scientific_method_engine-0.7.0 → scientific_method_engine-0.9.0}/src/scientific_method_engine/ghidra/ReportFilePatternInMemory.java +0 -0
- {scientific_method_engine-0.7.0 → scientific_method_engine-0.9.0}/src/scientific_method_engine/ghidra/ReportFirstArgumentCallSummary.java +0 -0
- {scientific_method_engine-0.7.0 → scientific_method_engine-0.9.0}/src/scientific_method_engine/ghidra/ReportFunctionScalarConstants.java +0 -0
- {scientific_method_engine-0.7.0 → scientific_method_engine-0.9.0}/src/scientific_method_engine/ghidra/ReportFunctionSummary.java +0 -0
- {scientific_method_engine-0.7.0 → scientific_method_engine-0.9.0}/src/scientific_method_engine/ghidra/ReportInstructionContext.java +0 -0
- {scientific_method_engine-0.7.0 → scientific_method_engine-0.9.0}/src/scientific_method_engine/ghidra/ReportInstructionWindow.java +0 -0
- {scientific_method_engine-0.7.0 → scientific_method_engine-0.9.0}/src/scientific_method_engine/ghidra/ReportMemoryBlockForFileOffset.java +0 -0
- {scientific_method_engine-0.7.0 → scientific_method_engine-0.9.0}/src/scientific_method_engine/ghidra/ReportMemoryBlocks.java +0 -0
- {scientific_method_engine-0.7.0 → scientific_method_engine-0.9.0}/src/scientific_method_engine/ghidra/ReportRandomnessCandidates.java +0 -0
- {scientific_method_engine-0.7.0 → scientific_method_engine-0.9.0}/src/scientific_method_engine/ghidra/ReportReferences.java +0 -0
- {scientific_method_engine-0.7.0 → scientific_method_engine-0.9.0}/src/scientific_method_engine/ghidra/ReportScalarConstants.java +0 -0
- {scientific_method_engine-0.7.0 → scientific_method_engine-0.9.0}/src/scientific_method_engine/ghidra/ReportStringReferences.java +0 -0
- {scientific_method_engine-0.7.0 → scientific_method_engine-0.9.0}/src/scientific_method_engine/ghidra/ReportSymbolReferences.java +0 -0
- {scientific_method_engine-0.7.0 → scientific_method_engine-0.9.0}/src/scientific_method_engine/x86/__init__.py +0 -0
- {scientific_method_engine-0.7.0 → scientific_method_engine-0.9.0}/src/scientific_method_engine/x86/dispatch.py +0 -0
- {scientific_method_engine-0.7.0 → scientific_method_engine-0.9.0}/src/scientific_method_engine/x86/effect_order.py +0 -0
- {scientific_method_engine-0.7.0 → scientific_method_engine-0.9.0}/src/scientific_method_engine/x86/image.py +0 -0
- {scientific_method_engine-0.7.0 → scientific_method_engine-0.9.0}/src/scientific_method_engine/x86/pcode.py +0 -0
- {scientific_method_engine-0.7.0 → scientific_method_engine-0.9.0}/src/scientific_method_engine/x86/pe.py +0 -0
- {scientific_method_engine-0.7.0 → scientific_method_engine-0.9.0}/src/scientific_method_engine/x86/result_flow.py +0 -0
- {scientific_method_engine-0.7.0 → scientific_method_engine-0.9.0}/src/scientific_method_engine/x86/trace.py +0 -0
- {scientific_method_engine-0.7.0 → scientific_method_engine-0.9.0}/src/scientific_method_engine/x86/values.py +0 -0
- {scientific_method_engine-0.7.0 → scientific_method_engine-0.9.0}/tests/test_nested_frame_request.py +0 -0
- {scientific_method_engine-0.7.0 → scientific_method_engine-0.9.0}/tests/test_table_continuations.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: scientific-method-engine
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.9.0
|
|
4
4
|
Summary: Bounded instruction-derived x86 evidence reports for segmented MZ/FBOV and PE32/i386 code.
|
|
5
5
|
Project-URL: Source, https://github.com/kibertoad/refurbished-dinosaurs-toolkit/tree/main/packages/scientific-method-engine
|
|
6
6
|
Author: kibertoad
|
|
@@ -83,6 +83,7 @@ Exporting for comparison (each writes one file and refuses to overwrite where no
|
|
|
83
83
|
|---|---|---|
|
|
84
84
|
| `ExportBoundedFlow` | entry, instruction limit (1..10000), output path under `analysis/original/` | instruction metadata of one bounded flow as JSON |
|
|
85
85
|
| `ExportFunctionInventory` | output TSV path (must not exist) | every function's start and body size |
|
|
86
|
+
| `ExportCallEdges` | output JSON path (must not exist), function limit (1..128), one or more function entries | the call and tail-jump edges of the functions Ghidra reaches breadth-first from the entries, with file offsets, as the `ghidraCallEdges` input of `callees` |
|
|
86
87
|
| `ExportFunctionFingerprints` | output TSV path | per-function and per-instruction fingerprints with addresses normalized, for matching functions across versions |
|
|
87
88
|
|
|
88
89
|
Repairing the analysis (these change the Ghidra program, so run them before reports and keep the
|
|
@@ -68,6 +68,7 @@ Exporting for comparison (each writes one file and refuses to overwrite where no
|
|
|
68
68
|
|---|---|---|
|
|
69
69
|
| `ExportBoundedFlow` | entry, instruction limit (1..10000), output path under `analysis/original/` | instruction metadata of one bounded flow as JSON |
|
|
70
70
|
| `ExportFunctionInventory` | output TSV path (must not exist) | every function's start and body size |
|
|
71
|
+
| `ExportCallEdges` | output JSON path (must not exist), function limit (1..128), one or more function entries | the call and tail-jump edges of the functions Ghidra reaches breadth-first from the entries, with file offsets, as the `ghidraCallEdges` input of `callees` |
|
|
71
72
|
| `ExportFunctionFingerprints` | output TSV path | per-function and per-instruction fingerprints with addresses normalized, for matching functions across versions |
|
|
72
73
|
|
|
73
74
|
Repairing the analysis (these change the Ghidra program, so run them before reports and keep the
|
|
@@ -5,7 +5,7 @@ build-backend = "hatchling.build"
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "scientific-method-engine"
|
|
7
7
|
# The release workflow writes the published version from the package's release tag.
|
|
8
|
-
version = "0.
|
|
8
|
+
version = "0.9.0"
|
|
9
9
|
description = "Bounded instruction-derived x86 evidence reports for segmented MZ/FBOV and PE32/i386 code."
|
|
10
10
|
readme = "README.md"
|
|
11
11
|
requires-python = ">=3.12"
|
|
@@ -3,13 +3,20 @@ import json
|
|
|
3
3
|
import sys
|
|
4
4
|
from pathlib import Path
|
|
5
5
|
|
|
6
|
+
import capstone
|
|
7
|
+
import pypcode
|
|
8
|
+
|
|
6
9
|
from . import PREPARED_PROTOCOL
|
|
7
10
|
|
|
8
11
|
CONFIG_LIMIT = 1024 * 1024
|
|
9
12
|
PREPARED_CONFIG_LIMIT = 16 * 1024 * 1024
|
|
13
|
+
# The header names the decoder and instruction semantics that actually ran, not the pins in pyproject.toml.
|
|
14
|
+
DECODER = "capstone " + capstone.__version__
|
|
15
|
+
INSTRUCTION_SEMANTICS = f"pypcode {pypcode.__version__} (Ghidra SLEIGH x86)"
|
|
10
16
|
USAGE = ("Usage: scientific-method-engine <operand|operand-candidates|target|bounds|owner|callees|trace|uses|arguments|"
|
|
11
17
|
"effects|returns|memory|incoming|call-order|guards|allocation|dispatch> <config.json|->\n"
|
|
12
18
|
"effects includes ordered path writes/calls and local restoration witnesses; transactionality remains unestablished.\n"
|
|
19
|
+
"callees compares its edges with an ExportCallEdges.java export given as ghidraCallEdges.\n"
|
|
13
20
|
"Declared table continuations are separate conditional paths; ordinary computed transfers remain stopped.\n"
|
|
14
21
|
" scientific-method-engine ghidra-scripts")
|
|
15
22
|
|
|
@@ -60,7 +67,8 @@ def main(argv):
|
|
|
60
67
|
raise ValueError("preparedProtocol is set by the reader and cannot be supplied")
|
|
61
68
|
data, identity = read_source(config, base)
|
|
62
69
|
result = run_report(data, config, command)
|
|
63
|
-
print(json.dumps({"schema": "bounded-x86-v1", "decoder":
|
|
70
|
+
print(json.dumps({"schema": "bounded-x86-v1", "decoder": DECODER,
|
|
71
|
+
"instructionSemantics": INSTRUCTION_SEMANTICS, "sourceIdentity": identity,
|
|
64
72
|
"status": "Conditional static report; never promotes an evidence entry", **result}, indent=2))
|
|
65
73
|
|
|
66
74
|
|
|
@@ -0,0 +1,138 @@
|
|
|
1
|
+
// Exports the call edges Ghidra recovers below given entries, as JSON for the engine's callees cross-check.
|
|
2
|
+
// @category Restoration
|
|
3
|
+
|
|
4
|
+
import java.io.BufferedWriter;
|
|
5
|
+
import java.nio.charset.StandardCharsets;
|
|
6
|
+
import java.nio.file.Files;
|
|
7
|
+
import java.nio.file.Path;
|
|
8
|
+
import java.util.ArrayDeque;
|
|
9
|
+
import java.util.ArrayList;
|
|
10
|
+
import java.util.LinkedHashSet;
|
|
11
|
+
import java.util.List;
|
|
12
|
+
import java.util.Set;
|
|
13
|
+
|
|
14
|
+
import ghidra.app.script.GhidraScript;
|
|
15
|
+
import ghidra.program.database.mem.AddressSourceInfo;
|
|
16
|
+
import ghidra.program.model.address.Address;
|
|
17
|
+
import ghidra.program.model.listing.Function;
|
|
18
|
+
import ghidra.program.model.listing.FunctionManager;
|
|
19
|
+
import ghidra.program.model.listing.Instruction;
|
|
20
|
+
import ghidra.program.model.listing.InstructionIterator;
|
|
21
|
+
import ghidra.program.model.symbol.FlowType;
|
|
22
|
+
|
|
23
|
+
public class ExportCallEdges extends GhidraScript {
|
|
24
|
+
private static final int MAX_FUNCTIONS = 128;
|
|
25
|
+
|
|
26
|
+
@Override
|
|
27
|
+
protected void run() throws Exception {
|
|
28
|
+
String[] arguments = getScriptArgs();
|
|
29
|
+
if (arguments.length < 3) {
|
|
30
|
+
throw new IllegalArgumentException(
|
|
31
|
+
"Supply an output JSON path, a function limit and at least one entry address.");
|
|
32
|
+
}
|
|
33
|
+
Path output = Path.of(arguments[0]).toAbsolutePath().normalize();
|
|
34
|
+
if (Files.exists(output)) {
|
|
35
|
+
throw new IllegalArgumentException("Call edge output already exists: " + output);
|
|
36
|
+
}
|
|
37
|
+
int functionLimit = Integer.parseInt(arguments[1]);
|
|
38
|
+
if (functionLimit < 1 || functionLimit > MAX_FUNCTIONS) {
|
|
39
|
+
throw new IllegalArgumentException("Function limit must be between 1 and " + MAX_FUNCTIONS + ".");
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
FunctionManager manager = currentProgram.getFunctionManager();
|
|
43
|
+
List<String> missing = new ArrayList<>();
|
|
44
|
+
Set<Function> queued = new LinkedHashSet<>();
|
|
45
|
+
ArrayDeque<Function> queue = new ArrayDeque<>();
|
|
46
|
+
for (int i = 2; i < arguments.length; i++) {
|
|
47
|
+
Address address = toAddr(arguments[i]);
|
|
48
|
+
Function function = address == null ? null : manager.getFunctionAt(address);
|
|
49
|
+
if (function == null) missing.add(arguments[i]);
|
|
50
|
+
else if (queued.add(function)) queue.add(function);
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
// Breadth first from the entries, so the function limit keeps the shallowest functions.
|
|
54
|
+
StringBuilder functions = new StringBuilder();
|
|
55
|
+
List<String> unread = new ArrayList<>();
|
|
56
|
+
int exported = 0;
|
|
57
|
+
while (!queue.isEmpty()) {
|
|
58
|
+
if (monitor.isCancelled()) throw new InterruptedException("Call edge export cancelled.");
|
|
59
|
+
Function function = queue.poll();
|
|
60
|
+
if (exported >= functionLimit) {
|
|
61
|
+
unread.add(quote(function.getEntryPoint().toString()));
|
|
62
|
+
continue;
|
|
63
|
+
}
|
|
64
|
+
List<String> edges = new ArrayList<>();
|
|
65
|
+
InstructionIterator instructions = currentProgram.getListing()
|
|
66
|
+
.getInstructions(function.getBody(), true);
|
|
67
|
+
while (instructions.hasNext()) {
|
|
68
|
+
Instruction instruction = instructions.next();
|
|
69
|
+
FlowType flow = instruction.getFlowType();
|
|
70
|
+
Address[] destinations = instruction.getFlows();
|
|
71
|
+
if (flow.isCall() && destinations.length == 0) {
|
|
72
|
+
edges.add(edge(instruction, null, flow));
|
|
73
|
+
}
|
|
74
|
+
for (Address destination : destinations) {
|
|
75
|
+
Function callee = manager.getFunctionAt(destination);
|
|
76
|
+
// A jump counts when it enters another function at its entry: a tail transfer.
|
|
77
|
+
if (!flow.isCall() && (callee == null || callee.equals(function))) continue;
|
|
78
|
+
edges.add(edge(instruction, destination, flow));
|
|
79
|
+
// An external function has no body to read; its edge keeps the external address.
|
|
80
|
+
if (callee != null && !callee.isExternal() && queued.add(callee)) queue.add(callee);
|
|
81
|
+
}
|
|
82
|
+
}
|
|
83
|
+
if (functions.length() > 0) functions.append(",\n");
|
|
84
|
+
functions.append(" {\"entry\": ").append(offset(function.getEntryPoint()))
|
|
85
|
+
.append(", \"address\": ").append(quote(function.getEntryPoint().toString()))
|
|
86
|
+
.append(", \"edges\": [").append(String.join(", ", edges)).append("]}");
|
|
87
|
+
exported++;
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
Files.createDirectories(output.getParent());
|
|
91
|
+
Path temporary = Files.createTempFile(output.getParent(), "call-edges-", ".partial");
|
|
92
|
+
try {
|
|
93
|
+
try (BufferedWriter writer = Files.newBufferedWriter(temporary, StandardCharsets.UTF_8)) {
|
|
94
|
+
writer.write("{\n \"format\": \"scientific-method-ghidra-call-edges\",\n \"version\": 1,\n");
|
|
95
|
+
writer.write(" \"sha256\": " + quote(currentProgram.getExecutableSHA256()) + ",\n");
|
|
96
|
+
writer.write(" \"functionLimit\": " + functionLimit + ",\n");
|
|
97
|
+
writer.write(" \"missingEntries\": [" + quoteAll(missing) + "],\n");
|
|
98
|
+
writer.write(" \"unreadFunctions\": [" + String.join(", ", unread) + "],\n");
|
|
99
|
+
writer.write(" \"functions\": [\n" + functions + "\n ]\n}\n");
|
|
100
|
+
}
|
|
101
|
+
Files.move(temporary, output); // No replacement, and no final file until the walk completes.
|
|
102
|
+
}
|
|
103
|
+
finally { Files.deleteIfExists(temporary); }
|
|
104
|
+
println("Exported the call edges of " + exported + " functions to " + output);
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
private String edge(Instruction instruction, Address target, FlowType flow) {
|
|
108
|
+
return "{\"site\": " + offset(instruction.getAddress())
|
|
109
|
+
+ ", \"siteAddress\": " + quote(instruction.getAddress().toString())
|
|
110
|
+
+ ", \"target\": " + (target == null ? "null" : offset(target))
|
|
111
|
+
+ ", \"targetAddress\": " + (target == null ? "null" : quote(target.toString()))
|
|
112
|
+
+ ", \"flow\": " + quote(flow.toString()) + "}";
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
// The file offset the engine uses for an address, or null when the address has no file bytes.
|
|
116
|
+
private String offset(Address address) {
|
|
117
|
+
AddressSourceInfo info = currentProgram.getMemory().getAddressSourceInfo(address);
|
|
118
|
+
if (info == null || info.getFileOffset() < 0) return "null";
|
|
119
|
+
return Long.toString(info.getFileOffset());
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
private static String quoteAll(List<String> values) {
|
|
123
|
+
List<String> quoted = new ArrayList<>();
|
|
124
|
+
for (String value : values) quoted.add(quote(value));
|
|
125
|
+
return String.join(", ", quoted);
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
private static String quote(String value) {
|
|
129
|
+
if (value == null) return "null";
|
|
130
|
+
StringBuilder result = new StringBuilder("\"");
|
|
131
|
+
for (char c : value.toCharArray()) {
|
|
132
|
+
if (c == '"' || c == '\\') result.append('\\').append(c);
|
|
133
|
+
else if (c < 0x20) result.append(String.format("\\u%04x", (int) c));
|
|
134
|
+
else result.append(c);
|
|
135
|
+
}
|
|
136
|
+
return result.append('"').toString();
|
|
137
|
+
}
|
|
138
|
+
}
|
|
@@ -1,8 +1,7 @@
|
|
|
1
|
-
"""Path state for the evidence layer. Instruction semantics come from
|
|
1
|
+
"""Path state for the evidence layer. Instruction semantics come from pypcode (pcode_backend.py)."""
|
|
2
2
|
from copy import deepcopy
|
|
3
3
|
from capstone.x86 import X86_OP_REG, X86_OP_IMM, X86_OP_MEM
|
|
4
4
|
from .values import Value, const, unknown, op, extract, join, resize, sources, address_parts, producers
|
|
5
|
-
from . import semantics
|
|
6
5
|
|
|
7
6
|
REGISTERS = ("eax", "ebx", "ecx", "edx", "esi", "edi", "ebp", "esp", "cs", "ds", "es", "ss", "fs", "gs")
|
|
8
7
|
ALIASES = {}
|
|
@@ -39,7 +38,10 @@ def alias(name):
|
|
|
39
38
|
class State:
|
|
40
39
|
def __init__(self, entry, image, config):
|
|
41
40
|
self.bits, self.flat, self.mask = image.bits, image.flat, image.mask
|
|
42
|
-
|
|
41
|
+
# Imported here: the backend imports this module, so a module-level import would make the
|
|
42
|
+
# import order matter.
|
|
43
|
+
from .pcode_backend import BACKEND
|
|
44
|
+
self.semantics = BACKEND
|
|
43
45
|
self.sp, self.bp = ("esp", "ebp") if self.flat else ("sp", "bp")
|
|
44
46
|
self.at = entry
|
|
45
47
|
self.regs = {r: unknown("initial:" + r, ALIASES[r][2]) for r in REGISTERS}
|
|
@@ -64,7 +66,8 @@ class State:
|
|
|
64
66
|
self.guards = []
|
|
65
67
|
self.assumptions = {}
|
|
66
68
|
self.flags = None
|
|
67
|
-
# Arithmetic flags as values
|
|
69
|
+
# Arithmetic flags as values from the last flag-writing instruction's p-code; None when no
|
|
70
|
+
# instruction computed them yet or the last one forgot them.
|
|
68
71
|
self.flag_values = None
|
|
69
72
|
# CF when an instruction sets it without leaving a comparable flag producer; None defers to flags.
|
|
70
73
|
self.carry = None
|
|
@@ -123,9 +126,10 @@ class State:
|
|
|
123
126
|
def carry_value(self):
|
|
124
127
|
"""CF as a one-bit value: from the last comparable flag producer, an explicit carry, or unknown."""
|
|
125
128
|
if self.flags is not None:
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
+
if self.flag_values is not None and "CF" in self.flag_values:
|
|
130
|
+
answer, _ = self.semantics.condition(self, "jb")
|
|
131
|
+
if answer is not None:
|
|
132
|
+
return const(int(answer), 1, self.flags[3])
|
|
129
133
|
# Name the carry by its producer's operands, so every reading of one comparison shares an assumption.
|
|
130
134
|
a, b, operation, site = self.flags
|
|
131
135
|
return unknown(f"carry:{site}:{(operation, a.term, b.term)!r}", 1, site)
|
|
@@ -367,11 +371,10 @@ def string_effect(state, ins, count, remaining, charge=None):
|
|
|
367
371
|
# any segment override.
|
|
368
372
|
source_name = (segment_register(ins, ins.operands[1].mem) if operation in ("movs", "lods") else
|
|
369
373
|
segment_register(ins, ins.operands[0].mem) if operation == "cmps" else None)
|
|
370
|
-
delta = -width if state.direction_flag.number else width
|
|
371
374
|
counter = "ecx" if state.flat else "cx"
|
|
372
375
|
if not compare or not repeated(ins):
|
|
373
376
|
for _ in range(count.number):
|
|
374
|
-
state.semantics.string_iteration(state, ins, operation, width, source_name
|
|
377
|
+
state.semantics.string_iteration(state, ins, operation, width, source_name)
|
|
375
378
|
if repeated(ins):
|
|
376
379
|
state.setreg(counter, const(0, state.bits, state.at), state.at)
|
|
377
380
|
return count.number
|
|
@@ -381,7 +384,7 @@ def string_effect(state, ins, count, remaining, charge=None):
|
|
|
381
384
|
raise StopPath("String iteration budget exhausted; remaining effects unresolved")
|
|
382
385
|
if charge is not None:
|
|
383
386
|
charge(1)
|
|
384
|
-
holds = state.semantics.string_iteration(state, ins, operation, width, source_name
|
|
387
|
+
holds = state.semantics.string_iteration(state, ins, operation, width, source_name)
|
|
385
388
|
iterations += 1
|
|
386
389
|
outcomes.append(holds)
|
|
387
390
|
if iterations == count.number:
|
|
@@ -398,7 +401,3 @@ def string_effect(state, ins, count, remaining, charge=None):
|
|
|
398
401
|
state.event("string-compare-exit", iterations=iterations, exit=reason, counter=state.reg(counter).report())
|
|
399
402
|
return iterations
|
|
400
403
|
|
|
401
|
-
|
|
402
|
-
# The handwritten backend registers itself as the default; it imports names defined above.
|
|
403
|
-
from . import handwritten # noqa: E402,F401
|
|
404
|
-
from . import pcode_backend # noqa: E402,F401
|
|
@@ -1,31 +1,30 @@
|
|
|
1
|
-
"""
|
|
2
|
-
|
|
3
|
-
Values come from the p-code pypcode lifts
|
|
4
|
-
|
|
5
|
-
|
|
6
|
-
|
|
1
|
+
"""Instruction semantics from pypcode (ADR 0003).
|
|
2
|
+
|
|
3
|
+
Values, flags and branch conditions come from the p-code pypcode lifts from Ghidra's SLEIGH
|
|
4
|
+
specification. The evidence layer keeps what it always decided: which segment register an access
|
|
5
|
+
uses (from Capstone, decision 3), access roles, the model's acceptance rules, flag-producer records
|
|
6
|
+
and the events each instruction reports. A ``State`` reaches this module through ``BACKEND``.
|
|
7
|
+
|
|
8
|
+
What this module and ``pcode.Run`` use of a ``State``:
|
|
9
|
+
|
|
10
|
+
- Methods: ``get``, ``put``, ``reg``, ``setreg``, ``segment``, ``access``, ``address``, ``event``,
|
|
11
|
+
``set_flags``, ``forget_flags``, ``carry_value``, ``save_flags`` and ``restore_flags``.
|
|
12
|
+
- Read only: ``at``, ``bits``, ``flat``, ``sp``, ``flags``, ``flag_serial``, ``flag_epoch``,
|
|
13
|
+
``unknown_flag_site``, ``segment_bases`` and ``value_transfers``.
|
|
14
|
+
- Read and assigned: ``flag_values`` (the arithmetic flags p-code computed, or None), ``carry``,
|
|
15
|
+
``direction_flag`` and ``interrupt_flag``. ``conditional`` is appended to (the divide-error
|
|
16
|
+
assumption).
|
|
17
|
+
|
|
18
|
+
Every member above belongs to one path, and ``trace`` copies a path with ``deepcopy`` when a branch
|
|
19
|
+
splits it. The backend keeps no path state of its own: ``Pypcode.__deepcopy__`` returns the same
|
|
20
|
+
object, so a value this module stores must live on the ``State`` to be copied with its path.
|
|
7
21
|
"""
|
|
8
22
|
from capstone.x86 import X86_OP_REG, X86_OP_IMM, X86_OP_MEM
|
|
9
23
|
|
|
10
|
-
from . import handwritten, semantics
|
|
11
24
|
from .machine import ALIASES, StopPath
|
|
12
25
|
from .pcode import LIFTER, Address, Run, FLAGS, SEGMENT_BASES, segment_base
|
|
13
26
|
from .values import Value, const, unknown, op, extract, join, resize, sources
|
|
14
27
|
|
|
15
|
-
GROUPS = {
|
|
16
|
-
"data movement": ("mov", "movzx", "movsx", "xchg", "nop"),
|
|
17
|
-
"address forms": ("lea", "lds", "les"),
|
|
18
|
-
"stack": ("push", "pop", "leave", "pushf", "pushfd", "popf", "popfd"),
|
|
19
|
-
"compare": ("cmp", "test"),
|
|
20
|
-
"arithmetic and logic": ("add", "sub", "and", "or", "xor", "inc", "dec", "not", "neg"),
|
|
21
|
-
"carry chain": ("adc", "sbb", "clc", "stc", "cmc"),
|
|
22
|
-
"shifts and rotates": ("shl", "sal", "shr", "sar", "rol", "ror", "rcl", "rcr"),
|
|
23
|
-
"multiply and divide": ("mul", "imul", "div", "idiv"),
|
|
24
|
-
"conversions": ("cbw", "cwde", "cwd", "cdq"),
|
|
25
|
-
"flags and direction": ("cld", "std", "cli", "sti"),
|
|
26
|
-
"string operations": ("movs", "stos", "lods", "cmps", "scas"),
|
|
27
|
-
}
|
|
28
|
-
|
|
29
28
|
# Conditional branches by the condition code SLEIGH decodes from 0x70 + code.
|
|
30
29
|
CONDITION_CODES = {}
|
|
31
30
|
for code, names in enumerate((("jo",), ("jno",), ("jb", "jc", "jnae"), ("jae", "jnb", "jnc"), ("je", "jz"),
|
|
@@ -281,7 +280,7 @@ class Frame:
|
|
|
281
280
|
self.addresses = {}
|
|
282
281
|
if ins.mnemonic != "pop":
|
|
283
282
|
# Operands address with the registers the instruction started with. POP's destination
|
|
284
|
-
# is addressed after the stack pointer moves, as the
|
|
283
|
+
# is addressed after the stack pointer moves, as the CPU addresses it.
|
|
285
284
|
for index, operand in enumerate(ins.operands):
|
|
286
285
|
if operand.type == X86_OP_MEM:
|
|
287
286
|
self.address(index)
|
|
@@ -422,47 +421,59 @@ def carry_out(state, run, site):
|
|
|
422
421
|
|
|
423
422
|
|
|
424
423
|
class Pypcode:
|
|
425
|
-
"""
|
|
426
|
-
|
|
427
|
-
name = "pypcode"
|
|
428
|
-
|
|
429
|
-
def __init__(self, groups):
|
|
430
|
-
self.groups = tuple(groups)
|
|
431
|
-
self.mnemonics = {m for group in self.groups for m in GROUPS[group]}
|
|
424
|
+
"""The engine's instruction semantics: ordinary instructions, branch conditions and string bodies."""
|
|
432
425
|
|
|
433
426
|
def __deepcopy__(self, memo):
|
|
427
|
+
# The backend holds no path state, so every copied path shares it.
|
|
434
428
|
return self
|
|
435
429
|
|
|
436
430
|
def ordinary(self, state, ins, image):
|
|
437
|
-
|
|
438
|
-
|
|
439
|
-
|
|
440
|
-
|
|
431
|
+
"""Apply one instruction that is neither a control transfer nor a string operation."""
|
|
432
|
+
handler = HANDLERS.get(ins.mnemonic)
|
|
433
|
+
if handler is None:
|
|
434
|
+
raise StopPath("Unsupported instruction semantics: " + ins.mnemonic)
|
|
435
|
+
handler(state, ins, image)
|
|
441
436
|
|
|
442
437
|
def condition(self, state, mnemonic):
|
|
443
|
-
|
|
444
|
-
|
|
445
|
-
|
|
446
|
-
|
|
447
|
-
|
|
448
|
-
|
|
438
|
+
"""Evaluate a conditional branch on the current flags.
|
|
439
|
+
|
|
440
|
+
Returns ``(answer, info)``: True, False or None when unresolved, and the ``branch`` event
|
|
441
|
+
fields. They describe the evidence layer's record of the flag producer, then either
|
|
442
|
+
``decidedBy: "p-code flags"`` for a decided branch or a ``reason`` for an undecided one.
|
|
443
|
+
"""
|
|
449
444
|
ops, _ = LIFTER.ops(state.flat, bytes((0x70 + CONDITION_CODES[mnemonic], 0)), 0x100)
|
|
450
445
|
needed = {LIFTER.register(state.flat, v[1], v[2]) for o in ops for v in o.inputs if v[0] == "register"}
|
|
451
|
-
|
|
452
|
-
|
|
453
|
-
|
|
454
|
-
|
|
446
|
+
carry_only = needed == {"CF"} and state.flags is None and state.carry is not None
|
|
447
|
+
if carry_only:
|
|
448
|
+
info = {"predicate": mnemonic, "flag": "CF", "carry": state.carry.report()}
|
|
449
|
+
elif state.flags is None:
|
|
450
|
+
info = {"predicate": mnemonic, "reason": "flag producer unresolved",
|
|
451
|
+
"flagProducer": state.unknown_flag_site, "flagGeneration": state.flag_epoch}
|
|
452
|
+
else:
|
|
453
|
+
a, b, operation, site = state.flags
|
|
454
|
+
info = {"predicate": mnemonic, "flagProducer": site, "operation": operation,
|
|
455
|
+
"left": a.report(), "right": b.report()}
|
|
456
|
+
# CF is always readable: the evidence layer names it when no instruction resolved it.
|
|
457
|
+
if not needed <= set(state.flag_values or ()) | {"CF"}:
|
|
458
|
+
info.setdefault("reason", "flags unresolved")
|
|
459
|
+
return None, info
|
|
460
|
+
condition = Run(state, ops, None, flags=state.flag_values or {}).execute(stop_at_branch=True)
|
|
455
461
|
if condition.number is None:
|
|
462
|
+
# Every undecided branch says why: the carry, the producer or its flags are unknown.
|
|
463
|
+
info.setdefault("reason", "carry unresolved" if carry_only else "flags unresolved")
|
|
456
464
|
return None, info
|
|
457
|
-
|
|
458
|
-
|
|
459
|
-
|
|
460
|
-
info["decidedBy"] = "p-code flags"
|
|
465
|
+
# p-code decides every branch the engine resolves; the record above only describes the producer.
|
|
466
|
+
info.pop("reason", None)
|
|
467
|
+
info["decidedBy"] = "p-code flags"
|
|
461
468
|
return bool(condition.number), info
|
|
462
469
|
|
|
463
|
-
def string_iteration(self, state, ins, operation, width, source_segment
|
|
464
|
-
|
|
465
|
-
|
|
470
|
+
def string_iteration(self, state, ins, operation, width, source_segment):
|
|
471
|
+
"""Apply one iteration of an accepted string form; see ``machine.string_effect``.
|
|
472
|
+
|
|
473
|
+
``source_segment`` names the source operand's segment register (MOVS, LODS and CMPS only).
|
|
474
|
+
p-code steps SI and DI by the direction flag. For a repeated CMPS or SCAS, returns whether
|
|
475
|
+
the repeat condition holds afterwards (1, 0 or unknown); otherwise None.
|
|
476
|
+
"""
|
|
466
477
|
si, di = ("esi", "edi") if state.flat else ("si", "di")
|
|
467
478
|
source, destination = state.reg(si), state.reg(di)
|
|
468
479
|
loaded = {}
|
|
@@ -910,8 +921,5 @@ for names, handler in ((("mov", "movzx", "movsx", "xchg"), move), (("nop",), nop
|
|
|
910
921
|
for name in names:
|
|
911
922
|
HANDLERS[name] = handler
|
|
912
923
|
|
|
913
|
-
|
|
914
|
-
|
|
915
|
-
"string operations")
|
|
916
|
-
|
|
917
|
-
semantics.register(Pypcode(MOVED), default=bool(MOVED))
|
|
924
|
+
# The backend every State uses.
|
|
925
|
+
BACKEND = Pypcode()
|
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
"""Focused reports derived from instruction paths and explicit source bounds."""
|
|
2
|
+
import hashlib
|
|
2
3
|
from bisect import bisect_right
|
|
3
4
|
from collections import deque
|
|
4
5
|
from capstone import CS_AC_READ, CS_AC_WRITE
|
|
@@ -859,8 +860,13 @@ def callees(image, config):
|
|
|
859
860
|
depth_limit = integer(config.get("depthLimit", 16), 1, 128, "callee depth limit")
|
|
860
861
|
instruction_limit = integer(config.get("instructionLimit", 10000), 1, 100000, "instruction limit")
|
|
861
862
|
controls = config.get("controls", {})
|
|
862
|
-
if not isinstance(controls, dict) or set(controls) - {"sharedSites", "recursiveSites", "writeSites"}:
|
|
863
|
+
if not isinstance(controls, dict) or set(controls) - {"sharedSites", "recursiveSites", "writeSites", "ghidraAgreementSites"}:
|
|
863
864
|
raise ValueError("Invalid callee controls")
|
|
865
|
+
export = config.get("ghidraCallEdges")
|
|
866
|
+
if export is not None:
|
|
867
|
+
export = _ghidra_call_edges(image, export)
|
|
868
|
+
elif "ghidraAgreementSites" in controls:
|
|
869
|
+
raise ValueError("ghidraAgreementSites needs ghidraCallEdges")
|
|
864
870
|
nodes, edges, omitted = {}, [], []
|
|
865
871
|
|
|
866
872
|
def read(entry):
|
|
@@ -984,15 +990,17 @@ def callees(image, config):
|
|
|
984
990
|
and not s["omittedRoutes"]
|
|
985
991
|
and not any(d.get("reason") in capped for at in s["entries"] for d in nodes[at]["dependencies"])
|
|
986
992
|
and not any(d.get("reason") in capped for i in s["dependencyEdges"] for d in edges[i]["dependencies"]))
|
|
993
|
+
cross_check = _ghidra_cross_check(export, nodes, outgoing, omitted) if export is not None else None
|
|
987
994
|
known = {"sharedSites": {e["site"] for e in edges if shared_control(e)},
|
|
988
995
|
"recursiveSites": {e["site"] for e in edges if e["classification"] == "recursivePath"},
|
|
989
|
-
"writeSites": {o["site"] for n in nodes.values() for o in n["memoryObservations"] if o["boundaryUsable"] and "write" in o["access"]}
|
|
996
|
+
"writeSites": {o["site"] for n in nodes.values() for o in n["memoryObservations"] if o["boundaryUsable"] and "write" in o["access"]},
|
|
997
|
+
"ghidraAgreementSites": cross_check and cross_check["agreementSites"]}
|
|
990
998
|
for kind, sites in controls.items():
|
|
991
999
|
if not isinstance(sites, list) or len(sites) > 256 or any(type(at) is not int or at not in known[kind] for at in sites):
|
|
992
1000
|
raise ValueError("Callee positive control missed: " + kind)
|
|
993
1001
|
return {"root": root, "nodes": [{k: v for k, v in n.items() if k != "body"} | {"body": _body_report(n["body"])} for n in nodes.values()],
|
|
994
1002
|
"edges": edges, "calleeSummaries": list(summaries.values()), "omittedRoutes": omitted, "uncheckedEntries": unchecked,
|
|
995
|
-
"controls": controls,
|
|
1003
|
+
"controls": controls, "ghidraCrossCheck": cross_check and {k: v for k, v in cross_check.items() if k != "agreementSites"},
|
|
996
1004
|
"completeWithinDeclaredGraph": not omitted and all(n["boundaryUsable"] for n in nodes.values()) and not any(e["dependencies"] for e in edges),
|
|
997
1005
|
"exclusions": ["implicit memory effects", "computed/unestablished targets", "argument-sensitive effects", "runtime reachability"],
|
|
998
1006
|
"interpretation": "Nodes are read breadth-first; path is the shortest read route to the caller. A recursivePath is a "
|
|
@@ -1000,6 +1008,100 @@ def callees(image, config):
|
|
|
1000
1008
|
"that does not. Neither proves runtime recursion."}
|
|
1001
1009
|
|
|
1002
1010
|
|
|
1011
|
+
GHIDRA_CALL_EDGES = "scientific-method-ghidra-call-edges"
|
|
1012
|
+
|
|
1013
|
+
|
|
1014
|
+
def _ghidra_call_edges(image, export):
|
|
1015
|
+
"""Validate an ExportCallEdges.java export against the image; returns its edges keyed by caller file offset."""
|
|
1016
|
+
if not isinstance(export, dict) or export.get("format") != GHIDRA_CALL_EDGES or export.get("version") != 1:
|
|
1017
|
+
raise ValueError("ghidraCallEdges must be an ExportCallEdges.java export, format version 1")
|
|
1018
|
+
if not isinstance(export.get("sha256"), str) or export["sha256"].lower() != hashlib.sha256(image.data).hexdigest():
|
|
1019
|
+
raise ValueError("ghidraCallEdges was exported from a different file")
|
|
1020
|
+
functions = export.get("functions")
|
|
1021
|
+
if not isinstance(functions, list) or len(functions) > 128:
|
|
1022
|
+
raise ValueError("ghidraCallEdges functions must be a list of at most 128")
|
|
1023
|
+
for key in ("missingEntries", "unreadFunctions"):
|
|
1024
|
+
if not isinstance(export.get(key), list) or not all(isinstance(v, str) for v in export[key]):
|
|
1025
|
+
raise ValueError(f"ghidraCallEdges {key} must be a list of addresses")
|
|
1026
|
+
|
|
1027
|
+
def offset(value, label):
|
|
1028
|
+
if value is None:
|
|
1029
|
+
return None
|
|
1030
|
+
return integer(value, 0, len(image.data) - 1, "Ghidra " + label + " file offset")
|
|
1031
|
+
callers, unmapped, total = {}, [], 0
|
|
1032
|
+
for function in functions:
|
|
1033
|
+
if not isinstance(function, dict) or not isinstance(function.get("address"), str) or not isinstance(function.get("edges"), list):
|
|
1034
|
+
raise ValueError("Invalid ghidraCallEdges function")
|
|
1035
|
+
total += len(function["edges"])
|
|
1036
|
+
if total > 8192:
|
|
1037
|
+
raise ValueError("ghidraCallEdges holds more than 8192 edges")
|
|
1038
|
+
rows = []
|
|
1039
|
+
for edge in function["edges"]:
|
|
1040
|
+
if (not isinstance(edge, dict) or not isinstance(edge.get("siteAddress"), str) or not isinstance(edge.get("flow"), str)
|
|
1041
|
+
or not (edge.get("targetAddress") is None or isinstance(edge["targetAddress"], str))):
|
|
1042
|
+
raise ValueError("Invalid ghidraCallEdges edge")
|
|
1043
|
+
rows.append({"site": offset(edge.get("site"), "site"), "siteAddress": edge["siteAddress"],
|
|
1044
|
+
"target": offset(edge.get("target"), "target"), "targetAddress": edge["targetAddress"], "flow": edge["flow"]})
|
|
1045
|
+
entry = offset(function.get("entry"), "entry")
|
|
1046
|
+
if entry is None:
|
|
1047
|
+
unmapped.append(function["address"])
|
|
1048
|
+
elif entry in callers:
|
|
1049
|
+
raise ValueError("ghidraCallEdges exports one function twice")
|
|
1050
|
+
else:
|
|
1051
|
+
callers[entry] = rows
|
|
1052
|
+
return {"callers": callers, "unmappedFunctions": unmapped, "missingEntries": export["missingEntries"],
|
|
1053
|
+
"unreadFunctions": export["unreadFunctions"]}
|
|
1054
|
+
|
|
1055
|
+
|
|
1056
|
+
def _ghidra_key(g):
|
|
1057
|
+
"""The (site, target) an exported edge matches on.
|
|
1058
|
+
|
|
1059
|
+
A null target matches the engine's unresolved call only when Ghidra resolved no address either;
|
|
1060
|
+
a target address without file bytes (an import, uninitialized memory) matches nothing.
|
|
1061
|
+
"""
|
|
1062
|
+
if g["target"] is None and g["targetAddress"] is not None:
|
|
1063
|
+
return g["site"], ("withoutFileOffset", g["targetAddress"])
|
|
1064
|
+
return g["site"], g["target"]
|
|
1065
|
+
|
|
1066
|
+
|
|
1067
|
+
def _ghidra_cross_check(export, nodes, outgoing, omitted):
|
|
1068
|
+
"""Compare the engine's edges with Ghidra's for each caller both read; a Ghidra-only edge stays unchecked."""
|
|
1069
|
+
callers = export["callers"]
|
|
1070
|
+
compared = sorted(nodes.keys() & callers.keys())
|
|
1071
|
+
rows = []
|
|
1072
|
+
for caller in compared:
|
|
1073
|
+
ours = outgoing.get(caller, [])
|
|
1074
|
+
theirs = callers[caller]
|
|
1075
|
+
# A call neither analysis resolved matches on its site with no target.
|
|
1076
|
+
flows = {_ghidra_key(g): g["flow"] for g in theirs if g["site"] is not None}
|
|
1077
|
+
read = {(e["site"], e["target"]) for e in ours}
|
|
1078
|
+
for e in ours:
|
|
1079
|
+
flow = flows.get((e["site"], e["target"]))
|
|
1080
|
+
rows.append({"caller": caller, "site": e["site"], "target": e["target"], "engineEdge": e["id"],
|
|
1081
|
+
"result": "engineOnly" if flow is None else "agreement", "ghidraFlow": flow})
|
|
1082
|
+
for g in theirs:
|
|
1083
|
+
if g["site"] is not None and _ghidra_key(g) in read:
|
|
1084
|
+
continue
|
|
1085
|
+
# Ghidra's edge is evidence the engine did not check; it never becomes an engine edge.
|
|
1086
|
+
rows.append({"caller": caller, "site": g["site"], "target": g["target"], "siteAddress": g["siteAddress"],
|
|
1087
|
+
"targetAddress": g["targetAddress"], "ghidraFlow": g["flow"], "result": "ghidraOnly", "checked": False,
|
|
1088
|
+
"engineEdge": next((e["id"] for e in ours if g["site"] is not None and e["site"] == g["site"]), None)})
|
|
1089
|
+
counts = {kind: sum(r["result"] == kind for r in rows) for kind in ("agreement", "engineOnly", "ghidraOnly")}
|
|
1090
|
+
not_compared = {"engineCallers": sorted(nodes.keys() - callers.keys()), "ghidraCallers": sorted(callers.keys() - nodes.keys()),
|
|
1091
|
+
"unmappedGhidraFunctions": export["unmappedFunctions"], "missingGhidraEntries": export["missingEntries"],
|
|
1092
|
+
"unreadGhidraFunctions": export["unreadFunctions"],
|
|
1093
|
+
"omittedEngineRoutes": [o["id"] for o in omitted if o["entry"] in callers]}
|
|
1094
|
+
# A site agrees only when every edge either analysis read there agrees.
|
|
1095
|
+
disputed = {r["site"] for r in rows if r["result"] != "agreement"}
|
|
1096
|
+
return {"comparedCallers": compared, "edges": rows, "counts": counts, "notCompared": not_compared,
|
|
1097
|
+
"agreed": not counts["engineOnly"] and not counts["ghidraOnly"] and not any(not_compared.values()),
|
|
1098
|
+
"agreementSites": {r["site"] for r in rows if r["result"] == "agreement"} - disputed,
|
|
1099
|
+
"interpretation": "Edges of each caller that both the engine and the Ghidra export read, matched by site and target "
|
|
1100
|
+
"file offset; an unresolved call matches an unresolved call at its site, and a Ghidra target without a file offset "
|
|
1101
|
+
"matches no engine edge. A ghidraOnly edge is Ghidra's claim: the engine did not check it and never adds it to "
|
|
1102
|
+
"its graph. Agreement means both analyses read the edge, not that it executes."}
|
|
1103
|
+
|
|
1104
|
+
|
|
1003
1105
|
# These branches test CX/ECX (LOOPE/LOOPNE also ZF), so an adjacent CMP/TEST never describes their predicate.
|
|
1004
1106
|
COUNT_BRANCHES = frozenset(("jcxz", "jecxz", "jrcxz", "loop", "loope", "loopne", "loopz", "loopnz"))
|
|
1005
1107
|
|
|
@@ -4,12 +4,14 @@ Unicorn is a test dependency only. ``check`` runs one synthetic routine on the e
|
|
|
4
4
|
Unicorn with the same segment layout and concrete registers, and compares every register the
|
|
5
5
|
engine resolved to a constant at the routine's return with Unicorn's value there.
|
|
6
6
|
"""
|
|
7
|
-
import
|
|
7
|
+
import sys
|
|
8
|
+
from pathlib import Path
|
|
8
9
|
|
|
9
10
|
from unicorn import UC_ARCH_X86, UC_MODE_16, Uc
|
|
10
11
|
from unicorn import x86_const as U
|
|
11
12
|
|
|
12
|
-
|
|
13
|
+
sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "src"))
|
|
14
|
+
from scientific_method_engine.x86.reports import run_report # noqa: E402
|
|
13
15
|
|
|
14
16
|
SEGMENT = 0x1000
|
|
15
17
|
STACK = {"ss": 0x2000, "sp": 0xFFF0}
|
|
@@ -47,18 +49,16 @@ def unicorn(data, registers, direction=None):
|
|
|
47
49
|
return stopped.get("at"), {name: uc.reg_read(getattr(U, "UC_X86_REG_" + name.upper())) for name in names}
|
|
48
50
|
|
|
49
51
|
|
|
50
|
-
def check(test, code, registers=None, direction=None, resolved=()
|
|
52
|
+
def check(test, code, registers=None, direction=None, resolved=()):
|
|
51
53
|
"""Compare the engine's resolved registers with Unicorn's for one synthetic routine.
|
|
52
54
|
|
|
53
55
|
Routines write any memory they read, because the engine starts with memory unknown and
|
|
54
56
|
Unicorn with zeros. ``resolved`` names registers the engine must resolve.
|
|
55
|
-
``extended`` states why the default backend may resolve more than the handwritten one here.
|
|
56
57
|
"""
|
|
57
58
|
data = bytes.fromhex(code)
|
|
58
59
|
registers = {**STACK, **(registers or {})}
|
|
59
60
|
flags = {} if direction is None else {"flags": {"direction": direction}}
|
|
60
|
-
|
|
61
|
-
result = run_report(data, configuration(data, registers, **flags), "trace")
|
|
61
|
+
result = run_report(data, configuration(data, registers, **flags), "trace")
|
|
62
62
|
test.assertEqual(len(result["paths"]), 1, "the oracle compares one resolved path")
|
|
63
63
|
path = result["paths"][0]
|
|
64
64
|
test.assertTrue(path["returned"], path["stop"])
|
|
@@ -5,7 +5,7 @@ import sys
|
|
|
5
5
|
import unittest
|
|
6
6
|
SRC = Path(__file__).resolve().parents[1] / "src"
|
|
7
7
|
sys.path.insert(0, str(SRC))
|
|
8
|
-
from
|
|
8
|
+
from scientific_method_engine.x86.reports import run_report
|
|
9
9
|
from scientific_method_engine.x86.image import Image
|
|
10
10
|
from scientific_method_engine.x86.trace import walk, OVERLAP_REASON
|
|
11
11
|
|