scientific-method-engine 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. scientific_method_engine/__init__.py +5 -0
  2. scientific_method_engine/__main__.py +3 -0
  3. scientific_method_engine/cli.py +75 -0
  4. scientific_method_engine/ghidra/ClearNoReturnFunctions.java +22 -0
  5. scientific_method_engine/ghidra/CreateFunctions.java +27 -0
  6. scientific_method_engine/ghidra/ExportBoundedFlow.java +59 -0
  7. scientific_method_engine/ghidra/ExportFunctionFingerprints.java +135 -0
  8. scientific_method_engine/ghidra/ExportFunctionInventory.java +50 -0
  9. scientific_method_engine/ghidra/MergeFallThroughFragment.java +63 -0
  10. scientific_method_engine/ghidra/RecoverCitedFunctions.java +93 -0
  11. scientific_method_engine/ghidra/RepairReturningCallers.java +76 -0
  12. scientific_method_engine/ghidra/ReportCallArguments.java +65 -0
  13. scientific_method_engine/ghidra/ReportCallPaths.java +106 -0
  14. scientific_method_engine/ghidra/ReportCallSitesWithScalars.java +95 -0
  15. scientific_method_engine/ghidra/ReportCallsToRange.java +67 -0
  16. scientific_method_engine/ghidra/ReportConstantFirstArgumentCalls.java +55 -0
  17. scientific_method_engine/ghidra/ReportDataBytes.java +36 -0
  18. scientific_method_engine/ghidra/ReportDecompileMatches.java +71 -0
  19. scientific_method_engine/ghidra/ReportDecompileWindow.java +61 -0
  20. scientific_method_engine/ghidra/ReportFilePatternInMemory.java +102 -0
  21. scientific_method_engine/ghidra/ReportFirstArgumentCallSummary.java +63 -0
  22. scientific_method_engine/ghidra/ReportFunctionScalarConstants.java +56 -0
  23. scientific_method_engine/ghidra/ReportFunctionSummary.java +64 -0
  24. scientific_method_engine/ghidra/ReportInstructionContext.java +64 -0
  25. scientific_method_engine/ghidra/ReportInstructionWindow.java +36 -0
  26. scientific_method_engine/ghidra/ReportMemoryBlockForFileOffset.java +105 -0
  27. scientific_method_engine/ghidra/ReportMemoryBlocks.java +52 -0
  28. scientific_method_engine/ghidra/ReportRandomnessCandidates.java +74 -0
  29. scientific_method_engine/ghidra/ReportReferences.java +42 -0
  30. scientific_method_engine/ghidra/ReportScalarConstants.java +55 -0
  31. scientific_method_engine/ghidra/ReportStringReferences.java +102 -0
  32. scientific_method_engine/ghidra/ReportSymbolReferences.java +72 -0
  33. scientific_method_engine/x86/__init__.py +0 -0
  34. scientific_method_engine/x86/dispatch.py +51 -0
  35. scientific_method_engine/x86/image.py +172 -0
  36. scientific_method_engine/x86/machine.py +659 -0
  37. scientific_method_engine/x86/pe.py +96 -0
  38. scientific_method_engine/x86/reports.py +1259 -0
  39. scientific_method_engine/x86/trace.py +547 -0
  40. scientific_method_engine/x86/values.py +123 -0
  41. scientific_method_engine-0.1.0.dist-info/METADATA +110 -0
  42. scientific_method_engine-0.1.0.dist-info/RECORD +45 -0
  43. scientific_method_engine-0.1.0.dist-info/WHEEL +4 -0
  44. scientific_method_engine-0.1.0.dist-info/entry_points.txt +2 -0
  45. scientific_method_engine-0.1.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,102 @@
1
+ // Finds bounded references to explicitly requested string fragments.
2
+ // @category Restoration
3
+
4
+ import ghidra.app.script.GhidraScript;
5
+ import ghidra.program.model.listing.Data;
6
+ import ghidra.program.model.listing.DataIterator;
7
+ import ghidra.program.model.listing.Function;
8
+ import ghidra.program.model.mem.Memory;
9
+ import ghidra.program.model.mem.MemoryBlock;
10
+ import ghidra.program.model.symbol.Reference;
11
+ import ghidra.program.model.symbol.ReferenceIterator;
12
+
13
+ import java.util.Locale;
14
+
15
+ public class ReportStringReferences extends GhidraScript {
16
+ private static final int MAX_MATCHES = 100;
17
+ private static final int MAX_REFERENCES_PER_MATCH = 100;
18
+ private static final int MAX_POINTERS_PER_MATCH = 100;
19
+
20
+ @Override
21
+ protected void run() throws Exception {
22
+ String[] arguments = getScriptArgs();
23
+ if (arguments.length == 0) {
24
+ printerr("Supply one or more literal string fragments.");
25
+ return;
26
+ }
27
+
28
+ var needles = new String[arguments.length];
29
+ for (var index = 0; index < arguments.length; index++)
30
+ needles[index] = arguments[index].toUpperCase(Locale.ROOT);
31
+
32
+ var matches = 0;
33
+ DataIterator data = currentProgram.getListing().getDefinedData(true);
34
+ while (data.hasNext() && matches < MAX_MATCHES && !monitor.isCancelled()) {
35
+ Data candidate = data.next();
36
+ Object value = candidate.getValue();
37
+ if (!(value instanceof String text)) continue;
38
+ String upper = text.toUpperCase(Locale.ROOT);
39
+ var requested = false;
40
+ for (String needle : needles) {
41
+ if (!upper.contains(needle)) continue;
42
+ requested = true;
43
+ break;
44
+ }
45
+ if (!requested) continue;
46
+
47
+ matches++;
48
+ println("===== " + candidate.getAddress() + " \"" + text + "\" =====");
49
+ ReferenceIterator references = currentProgram.getReferenceManager()
50
+ .getReferencesTo(candidate.getAddress());
51
+ var emitted = 0;
52
+ while (references.hasNext() && emitted < MAX_REFERENCES_PER_MATCH) {
53
+ Reference reference = references.next();
54
+ Function function = currentProgram.getFunctionManager()
55
+ .getFunctionContaining(reference.getFromAddress());
56
+ println(" " + reference.getFromAddress() + " " + reference.getReferenceType()
57
+ + (function == null ? "" : " in " + function.getEntryPoint() + " " + function.getName()));
58
+ emitted++;
59
+ }
60
+ if (references.hasNext())
61
+ println(" ... references capped at " + MAX_REFERENCES_PER_MATCH);
62
+ reportRawPointers(candidate);
63
+ }
64
+
65
+ if (matches == 0) println("No requested strings matched.");
66
+ else if (matches == MAX_MATCHES) println("Output capped at " + MAX_MATCHES + " matching strings.");
67
+ }
68
+
69
+ private void reportRawPointers(Data candidate) throws Exception {
70
+ long offset = candidate.getAddress().getOffset();
71
+ byte[] pointer = new byte[] {
72
+ (byte) offset,
73
+ (byte) (offset >>> 8),
74
+ (byte) (offset >>> 16),
75
+ (byte) (offset >>> 24)
76
+ };
77
+ Memory memory = currentProgram.getMemory();
78
+ var emitted = 0;
79
+ for (MemoryBlock block : memory.getBlocks()) {
80
+ var cursor = block.getStart();
81
+ while (cursor.compareTo(block.getEnd()) <= 0 && emitted < MAX_POINTERS_PER_MATCH) {
82
+ var found = memory.findBytes(cursor, pointer, null, true, monitor);
83
+ if (found == null || !block.contains(found)) break;
84
+ println(" raw pointer at " + found + " in " + block.getName());
85
+ ReferenceIterator references = currentProgram.getReferenceManager().getReferencesTo(found);
86
+ while (references.hasNext() && emitted < MAX_POINTERS_PER_MATCH) {
87
+ Reference reference = references.next();
88
+ Function function = currentProgram.getFunctionManager()
89
+ .getFunctionContaining(reference.getFromAddress());
90
+ println(" used at " + reference.getFromAddress()
91
+ + (function == null ? "" : " in " + function.getEntryPoint() + " " + function.getName()));
92
+ emitted++;
93
+ }
94
+ emitted++;
95
+ cursor = found.add(1);
96
+ }
97
+ if (emitted >= MAX_POINTERS_PER_MATCH) break;
98
+ }
99
+ if (emitted >= MAX_POINTERS_PER_MATCH)
100
+ println(" ... raw pointer output capped at " + MAX_POINTERS_PER_MATCH);
101
+ }
102
+ }
@@ -0,0 +1,72 @@
1
+ // Reports bounded references to explicitly requested symbol-name fragments.
2
+ // @category Restoration
3
+
4
+ import ghidra.app.script.GhidraScript;
5
+ import ghidra.program.model.listing.Function;
6
+ import ghidra.program.model.symbol.Reference;
7
+ import ghidra.program.model.symbol.ReferenceIterator;
8
+ import ghidra.program.model.symbol.Symbol;
9
+ import ghidra.program.model.symbol.SymbolIterator;
10
+
11
+ import java.util.Locale;
12
+
13
+ public class ReportSymbolReferences extends GhidraScript {
14
+ private static final int MAX_MATCHES = 100;
15
+ private static final int MAX_REFERENCES_PER_MATCH = 100;
16
+
17
+ @Override
18
+ protected void run() throws Exception {
19
+ String[] arguments = getScriptArgs();
20
+ if (arguments.length == 0) {
21
+ printerr("Supply one or more literal symbol-name fragments.");
22
+ return;
23
+ }
24
+
25
+ String[] needles = new String[arguments.length];
26
+ for (int index = 0; index < arguments.length; index++) {
27
+ if (arguments[index].isBlank()) {
28
+ printerr("Symbol-name fragments must not be blank.");
29
+ return;
30
+ }
31
+ needles[index] = arguments[index].toUpperCase(Locale.ROOT);
32
+ }
33
+
34
+ int matches = 0;
35
+ SymbolIterator symbols = currentProgram.getSymbolTable().getAllSymbols(true);
36
+ while (symbols.hasNext() && matches < MAX_MATCHES && !monitor.isCancelled()) {
37
+ Symbol symbol = symbols.next();
38
+ String name = symbol.getName(true);
39
+ if (!matchesAny(name.toUpperCase(Locale.ROOT), needles)) continue;
40
+
41
+ matches++;
42
+ println("===== " + symbol.getAddress() + " " + name + " =====");
43
+ ReferenceIterator references = currentProgram.getReferenceManager()
44
+ .getReferencesTo(symbol.getAddress());
45
+ int emitted = 0;
46
+ while (references.hasNext() && emitted < MAX_REFERENCES_PER_MATCH) {
47
+ Reference reference = references.next();
48
+ Function function = currentProgram.getFunctionManager()
49
+ .getFunctionContaining(reference.getFromAddress());
50
+ println(" " + reference.getFromAddress() + " " + reference.getReferenceType()
51
+ + (function == null ? "" : " in " + function.getEntryPoint()
52
+ + " " + function.getName()));
53
+ emitted++;
54
+ }
55
+ if (references.hasNext()) {
56
+ println(" ... references capped at " + MAX_REFERENCES_PER_MATCH);
57
+ }
58
+ }
59
+
60
+ if (matches == 0) println("No requested symbols matched.");
61
+ else if (matches == MAX_MATCHES) {
62
+ println("Output capped at " + MAX_MATCHES + " matching symbols.");
63
+ }
64
+ }
65
+
66
+ private static boolean matchesAny(String name, String[] needles) {
67
+ for (String needle : needles) {
68
+ if (name.contains(needle)) return true;
69
+ }
70
+ return false;
71
+ }
72
+ }
File without changes
@@ -0,0 +1,51 @@
1
+ """Explicit, source-derived indirect jump tables for CFG discovery only."""
2
+ from capstone.x86 import X86_OP_IMM
3
+ from .image import integer
4
+
5
+
6
+ def read_indirect_jumps(image):
7
+ declarations = image.config.get("indirectJumps", [])
8
+ if not isinstance(declarations, list) or len(declarations) > 256:
9
+ raise ValueError("indirectJumps must contain at most 256 declarations")
10
+ if declarations and image.flat:
11
+ raise ValueError("indirectJumps currently requires segmented16")
12
+ result = {}
13
+ for claim in declarations:
14
+ if not isinstance(claim, dict):
15
+ raise ValueError("Each indirect jump needs an object")
16
+ site = integer(claim.get("site"), 0, len(image.data) - 1, "indirect jump site")
17
+ if site in result:
18
+ raise ValueError("Duplicate indirect jump site")
19
+ if not isinstance(claim.get("evidence"), str) or not claim["evidence"].strip():
20
+ raise ValueError("An indirect jump needs consumer/mapping evidence")
21
+ if type(claim.get("exhaustive")) is not bool:
22
+ raise ValueError("An indirect jump needs explicit exhaustive true or false")
23
+ ins = image.decode(site)
24
+ if (ins is None or ins.mnemonic != "jmp" or len(ins.operands) != 1
25
+ or ins.operands[0].type == X86_OP_IMM or ins.operands[0].size != 2
26
+ or 0x66 in ins.prefix or 0x67 in ins.prefix):
27
+ raise ValueError("indirectJumps site must decode as an unprefixed computed near word jump")
28
+ table = claim.get("table")
29
+ if not isinstance(table, dict) or not isinstance(table.get("evidence"), str) or not table["evidence"].strip():
30
+ raise ValueError("Indirect table needs layout/count evidence")
31
+ if type(table.get("width", 2)) is not int or table.get("width", 2) != 2:
32
+ raise ValueError("Indirect table targets must be word width 2")
33
+ start = integer(table.get("start"), 0, len(image.data) - 1, "indirect table start")
34
+ count = integer(table.get("count"), 1, 256, "indirect table count")
35
+ stride = integer(table.get("stride"), 2, 65536, "indirect table stride")
36
+ field = integer(table.get("fieldOffset", 0), 0, stride - 2, "indirect target field offset")
37
+ if start + (count - 1) * stride + field + 2 > len(image.data):
38
+ raise ValueError("Indirect table leaves source bounds")
39
+ rows = []
40
+ for index in range(count):
41
+ operand = start + index * stride + field
42
+ raw = int.from_bytes(image.data[operand:operand + 2], "little")
43
+ target = image.near_target(site, raw)
44
+ if target is None:
45
+ raise ValueError("Indirect table target leaves declared code mappings")
46
+ rows.append({"index": index, "operandSite": operand, "rawOffset": raw, "target": target})
47
+ result[site] = {"site": site, "evidence": claim["evidence"], "table": table,
48
+ "exhaustive": claim["exhaustive"], "rows": rows,
49
+ "interpretation": "source words under supplied consumer, mapping and exhaustiveness evidence; "
50
+ "not executed dispatch or an overlap-boundary proof"}
51
+ return result
@@ -0,0 +1,172 @@
1
+ """Bounded explicit code mappings. No guessed linear disassembly domains."""
2
+ import hashlib
3
+ from pathlib import Path
4
+ from capstone import Cs, CS_ARCH_X86, CS_MODE_16, CS_MODE_32
5
+ from .pe import prepare_pe
6
+ import capstone
7
+
8
+ MAX_SOURCE = 256 * 1024 * 1024
9
+
10
+
11
+ def integer(value, low, high, label):
12
+ if type(value) is not int or not low <= value <= high:
13
+ raise ValueError(f"{label} must be an integer in {low}..{high}")
14
+ return value
15
+
16
+
17
+ def read_source(config, base):
18
+ """Read the source a report config names and check its hash.
19
+
20
+ ``config["source"]`` is a path relative to ``base``; the file must be at most 256 MiB and hash to
21
+ ``config["sha256"]``. Returns ``(data, identity)`` where ``identity`` is
22
+ ``{"size": ..., "sha256": ...}``. Raises ``ValueError`` otherwise.
23
+ """
24
+ path = config.get("source")
25
+ if not isinstance(path, str) or not path:
26
+ raise ValueError("source path required")
27
+ path = Path(base) / path
28
+ if not path.is_file() or path.stat().st_size > MAX_SOURCE:
29
+ raise ValueError("source must be a regular file of at most 256 MiB")
30
+ data = path.read_bytes()
31
+ digest = hashlib.sha256(data).hexdigest()
32
+ if digest != config.get("sha256"):
33
+ raise ValueError("Source SHA-256 differs from the supplied baseline")
34
+ return data, {"size": len(data), "sha256": digest}
35
+
36
+
37
+ class Image:
38
+ def __init__(self, data, config):
39
+ if capstone.__version__ != "5.0.7":
40
+ raise ValueError("This reporter requires capstone==5.0.7")
41
+ if config.get("sourceKind") == "pe32":
42
+ config = prepare_pe(data, config)
43
+ self.config = config
44
+ if config.get("addressModel", "segmented16") not in ("segmented16", "flat32"):
45
+ raise ValueError("Unknown address model")
46
+ self.bits = config.get("bits", 16)
47
+ self.flat = config.get("addressModel", "segmented16") == "flat32"
48
+ if (self.bits, self.flat) not in ((16, False), (32, True)):
49
+ raise ValueError("Only segmented16 and flat32 instruction models are supported")
50
+ if self.flat and config.get("sourceKind") not in ("pe32", "synthetic-raw"):
51
+ raise ValueError("flat32 requires pe32 or explicit synthetic-raw input")
52
+ self.mask = (1 << self.bits) - 1
53
+ self.data = data
54
+ self.regions = config.get("regions", [])
55
+ if not isinstance(self.regions, list) or not 1 <= len(self.regions) <= 256:
56
+ raise ValueError("Declare 1..256 code regions")
57
+ names = set()
58
+ for r in self.regions:
59
+ if not isinstance(r, dict):
60
+ raise ValueError("Each region must be an object")
61
+ name = r.get("name")
62
+ if not isinstance(name, str) or not name or name in names:
63
+ raise ValueError("Region names must be unique")
64
+ names.add(name)
65
+ integer(r.get("start"), 0, len(data), "region start")
66
+ integer(r.get("end"), r["start"] + 1, len(data), "region end")
67
+ integer(r.get("ip"), 0, self.mask, "region IP")
68
+ integer(r.get("segment"), 0, 65535, "region segment")
69
+ if r["ip"] + r["end"] - r["start"] > 1 << self.bits:
70
+ raise ValueError("Code region crosses the instruction address boundary")
71
+ if not isinstance(r.get("evidence"), str) or not r["evidence"].strip():
72
+ raise ValueError("Each region needs mapping/bounds evidence")
73
+ entries = r.get("entries")
74
+ if not isinstance(entries, list) or not entries or len(entries) > 4096:
75
+ raise ValueError("Each region needs 1..4096 established entry offsets")
76
+ for at in entries:
77
+ integer(at, r["start"], r["end"] - 1, "entry")
78
+ container = r.get("container")
79
+ if container is not None:
80
+ # The complete overlay or segment that holds this region, as the Node loader supplies it.
81
+ if not isinstance(container, dict) or not isinstance(container.get("view"), str) or not container["view"]:
82
+ raise ValueError("A region container needs view, start and end")
83
+ integer(container.get("start"), 0, r["start"], "container start")
84
+ integer(container.get("end"), r["end"], len(data), "container end")
85
+ self.segments = config.get("segments", [])
86
+ if not isinstance(self.segments, list) or len(self.segments) > 256:
87
+ raise ValueError("segments must be a list of at most 256 declared segment bounds")
88
+ for d in self.segments:
89
+ if (not isinstance(d, dict) or not isinstance(d.get("name"), str) or not d["name"]
90
+ or not isinstance(d.get("evidence"), str) or not d["evidence"].strip()):
91
+ raise ValueError("Each declared segment needs name, start, end and evidence")
92
+ integer(d.get("start"), 0, len(data), "segment start")
93
+ integer(d.get("end"), d["start"] + 1, len(data), "segment end")
94
+ for i, r in enumerate(self.regions):
95
+ for s in self.regions[i + 1:]:
96
+ if max(r["start"], s["start"]) < min(r["end"], s["end"]):
97
+ raise ValueError("Overlapping source regions")
98
+ if r["segment"] == s["segment"] and max(r["ip"], s["ip"]) < min(r["ip"] + r["end"] - r["start"], s["ip"] + s["end"] - s["start"]):
99
+ raise ValueError("Ambiguous loaded region mapping")
100
+ self.decoder = Cs(CS_ARCH_X86, CS_MODE_32 if self.flat else CS_MODE_16)
101
+ self.decoder.detail = True
102
+ self.cache = {}
103
+ self.relocations = config.get("relocations", [])
104
+ if not isinstance(self.relocations, list) or len(self.relocations) > 100000:
105
+ raise ValueError("Invalid relocation metadata")
106
+ self.fixups = {}
107
+ for f in self.relocations:
108
+ integer(f.get("site"), 0, len(data) - 2, "relocation site")
109
+ integer(f.get("segment"), 0, 65535, "resolved segment")
110
+ if f["site"] in self.fixups or not f.get("evidence"):
111
+ raise ValueError("Duplicate relocation or missing provenance")
112
+ self.fixups[f["site"]] = f
113
+ from .dispatch import read_indirect_jumps
114
+ self.indirect_jumps = read_indirect_jumps(self)
115
+
116
+ def region(self, site):
117
+ return next((r for r in self.regions if r["start"] <= site < r["end"]), None)
118
+
119
+ def offset(self, segment, ip):
120
+ exact = [r for r in self.regions if r["segment"] == segment and r["ip"] <= ip < r["ip"] + r["end"] - r["start"]]
121
+ if exact:
122
+ r = exact[0]
123
+ return r["start"] + ip - r["ip"]
124
+ if self.flat:
125
+ return None
126
+ linear = segment * 16 + ip
127
+ aliases = [r for r in self.regions if r.get("resident", False) and r["segment"] * 16 + r["ip"] <= linear < r["segment"] * 16 + r["ip"] + r["end"] - r["start"]]
128
+ if len(aliases) == 1:
129
+ r = aliases[0]
130
+ return r["start"] + linear - r["segment"] * 16 - r["ip"]
131
+ return None
132
+
133
+ def decode(self, site):
134
+ if site in self.cache:
135
+ return self.cache[site]
136
+ r = self.region(site)
137
+ if r is None:
138
+ return None
139
+ ip = r["ip"] + site - r["start"]
140
+ instruction = next(self.decoder.disasm(self.data[site:min(site + 15, r["end"])], ip, count=1), None)
141
+ self.cache[site] = instruction
142
+ return instruction
143
+
144
+ def near_target(self, site, ip):
145
+ r = self.region(site)
146
+ return self.offset(r["segment"], ip & self.mask)
147
+
148
+ def far_target(self, site, ins):
149
+ if self.flat:
150
+ return None, {"reason": "far transfer is outside the PE32 flat model"}
151
+ # ptr16:16 immediate only. Operand-size-prefixed far calls are unsupported.
152
+ if ins.size != 5 or self.data[site] not in (0x9a, 0xea):
153
+ return None, {"reason": "unsupported far transfer encoding"}
154
+ raw = int.from_bytes(self.data[site + 3:site + 5], "little")
155
+ ip = int.from_bytes(self.data[site + 1:site + 3], "little")
156
+ fixup = self.fixups.get(site + 3)
157
+ if not fixup:
158
+ return None, {"rawSegment": raw, "offset": ip, "reason": "no declared relocation/fixup"}
159
+ target = self.offset(fixup["segment"], ip)
160
+ if "target" in fixup:
161
+ target = integer(fixup["target"], 0, len(self.data) - 1, "canonical target")
162
+ return target, {"rawSegment": raw, "offset": ip, "resolvedSegment": fixup["segment"], "relocation": fixup}
163
+
164
+ def file_offset(self, va, width=1):
165
+ """Only loaded raw PE bytes; zero-fill and alignment padding are not source extents."""
166
+ metadata = self.config.get("peMetadata")
167
+ if not metadata:
168
+ return None
169
+ for section in metadata["sections"]:
170
+ if section["va"] <= va and va + width <= section["va"] + section["loadedRawSize"]:
171
+ return section["rawStart"] + va - section["va"]
172
+ return None