scientific-method-engine 0.1.0__tar.gz → 0.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/PKG-INFO +1 -1
- {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/pyproject.toml +1 -1
- {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/cli.py +1 -1
- {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/x86/machine.py +43 -9
- {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/x86/reports.py +205 -2
- scientific_method_engine-0.3.0/src/scientific_method_engine/x86/result_flow.py +131 -0
- {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/x86/trace.py +11 -33
- {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/x86/values.py +20 -2
- {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/tests/test_x86.py +288 -0
- {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/.gitignore +0 -0
- {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/LICENSE +0 -0
- {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/README.md +0 -0
- {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/__init__.py +0 -0
- {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/__main__.py +0 -0
- {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/ghidra/ClearNoReturnFunctions.java +0 -0
- {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/ghidra/CreateFunctions.java +0 -0
- {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/ghidra/ExportBoundedFlow.java +0 -0
- {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/ghidra/ExportFunctionFingerprints.java +0 -0
- {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/ghidra/ExportFunctionInventory.java +0 -0
- {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/ghidra/MergeFallThroughFragment.java +0 -0
- {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/ghidra/RecoverCitedFunctions.java +0 -0
- {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/ghidra/RepairReturningCallers.java +0 -0
- {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/ghidra/ReportCallArguments.java +0 -0
- {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/ghidra/ReportCallPaths.java +0 -0
- {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/ghidra/ReportCallSitesWithScalars.java +0 -0
- {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/ghidra/ReportCallsToRange.java +0 -0
- {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/ghidra/ReportConstantFirstArgumentCalls.java +0 -0
- {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/ghidra/ReportDataBytes.java +0 -0
- {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/ghidra/ReportDecompileMatches.java +0 -0
- {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/ghidra/ReportDecompileWindow.java +0 -0
- {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/ghidra/ReportFilePatternInMemory.java +0 -0
- {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/ghidra/ReportFirstArgumentCallSummary.java +0 -0
- {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/ghidra/ReportFunctionScalarConstants.java +0 -0
- {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/ghidra/ReportFunctionSummary.java +0 -0
- {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/ghidra/ReportInstructionContext.java +0 -0
- {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/ghidra/ReportInstructionWindow.java +0 -0
- {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/ghidra/ReportMemoryBlockForFileOffset.java +0 -0
- {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/ghidra/ReportMemoryBlocks.java +0 -0
- {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/ghidra/ReportRandomnessCandidates.java +0 -0
- {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/ghidra/ReportReferences.java +0 -0
- {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/ghidra/ReportScalarConstants.java +0 -0
- {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/ghidra/ReportStringReferences.java +0 -0
- {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/ghidra/ReportSymbolReferences.java +0 -0
- {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/x86/__init__.py +0 -0
- {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/x86/dispatch.py +0 -0
- {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/x86/image.py +0 -0
- {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/x86/pe.py +0 -0
- {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/tests/test_dispatch.py +0 -0
- {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/tests/test_pe.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: scientific-method-engine
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.3.0
|
|
4
4
|
Summary: Bounded instruction-derived x86 evidence reports for segmented MZ/FBOV and PE32/i386 code.
|
|
5
5
|
Project-URL: Source, https://github.com/kibertoad/refurbished-dinosaurs-toolkit/tree/main/packages/scientific-method-engine
|
|
6
6
|
Author: kibertoad
|
|
@@ -5,7 +5,7 @@ build-backend = "hatchling.build"
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "scientific-method-engine"
|
|
7
7
|
# The release workflow writes the published version from the package's release tag.
|
|
8
|
-
version = "0.
|
|
8
|
+
version = "0.3.0"
|
|
9
9
|
description = "Bounded instruction-derived x86 evidence reports for segmented MZ/FBOV and PE32/i386 code."
|
|
10
10
|
readme = "README.md"
|
|
11
11
|
requires-python = ">=3.10"
|
|
@@ -8,7 +8,7 @@ from . import PREPARED_PROTOCOL
|
|
|
8
8
|
CONFIG_LIMIT = 1024 * 1024
|
|
9
9
|
PREPARED_CONFIG_LIMIT = 16 * 1024 * 1024
|
|
10
10
|
USAGE = ("Usage: scientific-method-engine <operand|operand-candidates|target|bounds|owner|callees|trace|uses|arguments|"
|
|
11
|
-
"effects|returns|memory|incoming|guards|allocation|dispatch> <config.json|->\n"
|
|
11
|
+
"effects|returns|memory|incoming|call-order|guards|allocation|dispatch> <config.json|->\n"
|
|
12
12
|
" scientific-method-engine ghidra-scripts")
|
|
13
13
|
|
|
14
14
|
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
"""Path-specific instruction effects. Unsupported semantics stop the path."""
|
|
2
2
|
from copy import deepcopy
|
|
3
3
|
from capstone.x86 import X86_OP_REG, X86_OP_IMM, X86_OP_MEM
|
|
4
|
-
from .values import Value, const, unknown, op, extract, join, resize, sources, address_parts
|
|
4
|
+
from .values import Value, const, unknown, op, extract, join, resize, sources, address_parts, producers
|
|
5
5
|
|
|
6
6
|
REGISTERS = ("eax", "ebx", "ecx", "edx", "esi", "edi", "ebp", "esp", "cs", "ds", "es", "ss", "fs", "gs")
|
|
7
7
|
ALIASES = {}
|
|
@@ -41,6 +41,8 @@ class State:
|
|
|
41
41
|
self.sp, self.bp = ("esp", "ebp") if self.flat else ("sp", "bp")
|
|
42
42
|
self.at = entry
|
|
43
43
|
self.regs = {r: unknown("initial:" + r, ALIASES[r][2]) for r in REGISTERS}
|
|
44
|
+
# Producers per register byte, so a partial write replaces only the bytes it stores.
|
|
45
|
+
self.reg_sources = {r: [()] * (ALIASES[r][2] // 8) for r in REGISTERS}
|
|
44
46
|
self.regs["esp"] = resize(unknown("entry:sp", self.bits), 32)
|
|
45
47
|
self.setreg(self.sp, unknown("entry:sp", self.bits), None)
|
|
46
48
|
self.segment_bases = {r: const(0, 32) if r in ("cs", "ds", "es", "ss") else unknown("initial-base:" + r, 32)
|
|
@@ -55,6 +57,8 @@ class State:
|
|
|
55
57
|
self.memory_groups = {}
|
|
56
58
|
self.memory_epoch = 0
|
|
57
59
|
self.events = []
|
|
60
|
+
# Value transfers are recorded only for queries that trace declared return results.
|
|
61
|
+
self.value_transfers = bool(config.get("returnContracts"))
|
|
58
62
|
self.guards = []
|
|
59
63
|
self.assumptions = {}
|
|
60
64
|
self.flags = None
|
|
@@ -134,7 +138,8 @@ class State:
|
|
|
134
138
|
|
|
135
139
|
def reg(self, name):
|
|
136
140
|
root, low, bits = alias(name)
|
|
137
|
-
|
|
141
|
+
value = extract(self.regs[root], low, bits)
|
|
142
|
+
return Value(bits, value.term, tuple(sorted(set().union(*self.reg_sources[root][low // 8:(low + bits) // 8]))))
|
|
138
143
|
|
|
139
144
|
def setreg(self, name, value, site):
|
|
140
145
|
root, low, bits = alias(name)
|
|
@@ -147,6 +152,7 @@ class State:
|
|
|
147
152
|
chunks = [extract(old, n, 8) for n in range(0, old.bits, 8)]
|
|
148
153
|
chunks[low // 8:(low + bits) // 8] = [extract(value, n, 8) for n in range(0, bits, 8)]
|
|
149
154
|
self.regs[root] = join(chunks)
|
|
155
|
+
self.reg_sources[root][low // 8:(low + bits) // 8] = [value.sources] * (bits // 8)
|
|
150
156
|
|
|
151
157
|
def event(self, kind, **fields):
|
|
152
158
|
event = {"kind": kind, "site": self.at, "entry": self.frames[-1]["entry"],
|
|
@@ -225,7 +231,7 @@ class State:
|
|
|
225
231
|
event = self.event("write" if write is not None else "read", segment=segment.report(), offset=offset.report(),
|
|
226
232
|
width=width, effectiveSegmentRegister=addressing_register, segmentInterpretation="base" if self.flat else "selector-paragraph", interval={"segment": seg, "base": base, "start": delta, "end": delta + width},
|
|
227
233
|
value=value.report(), missingByteProducers=missing,
|
|
228
|
-
byteProducers=[{"index": i, "producers":
|
|
234
|
+
byteProducers=[{"index": i, "producers": producers(self.memory[key]) if key in self.memory else []} for i, key in enumerate(keys)],
|
|
229
235
|
guards=deepcopy(relevant), role=role,
|
|
230
236
|
uncertainAliasesInvalidated=len(uncertain))
|
|
231
237
|
f = self.frames[-1]
|
|
@@ -234,7 +240,7 @@ class State:
|
|
|
234
240
|
relative = (mem_delta - stack_delta) % (1 << self.bits)
|
|
235
241
|
if write is None and segment.term == self.segment("ss").term and mem_base == stack_base and f["returnBytes"] <= relative < 1 << (self.bits - 1):
|
|
236
242
|
event["argument"] = {"offsetFromEntrySP": relative, "width": width, "returnFrameBytes": f["returnBytes"],
|
|
237
|
-
"pushProducers":
|
|
243
|
+
"pushProducers": producers(value), "grouping": role or "consumed width only"}
|
|
238
244
|
return value
|
|
239
245
|
|
|
240
246
|
def address(self, ins, operand):
|
|
@@ -287,6 +293,19 @@ class State:
|
|
|
287
293
|
return value
|
|
288
294
|
|
|
289
295
|
|
|
296
|
+
# Synonymous and complementary branches on one flag producer share a single assumption.
|
|
297
|
+
BRANCH_CONDITIONS = {}
|
|
298
|
+
for names, condition in ((("je", "jz"), "z"), (("jb", "jc", "jnae"), "c"), (("jbe", "jna"), "be"),
|
|
299
|
+
(("jl", "jnge"), "l"), (("jle", "jng"), "le"), (("js",), "s"),
|
|
300
|
+
(("jo",), "o"), (("jp", "jpe"), "p")):
|
|
301
|
+
for name in names:
|
|
302
|
+
BRANCH_CONDITIONS[name] = (condition, False)
|
|
303
|
+
for names, condition in ((("jne", "jnz"), "z"), (("jae", "jnb", "jnc"), "c"), (("ja", "jnbe"), "be"),
|
|
304
|
+
(("jge", "jnl"), "l"), (("jg", "jnle"), "le"), (("jns",), "s"),
|
|
305
|
+
(("jno",), "o"), (("jnp", "jpo"), "p")):
|
|
306
|
+
for name in names:
|
|
307
|
+
BRANCH_CONDITIONS[name] = (condition, True)
|
|
308
|
+
|
|
290
309
|
CARRY_BRANCHES = {"jb": True, "jc": True, "jnae": True, "jae": False, "jnb": False, "jnc": False}
|
|
291
310
|
# Branches taken when CF or OF is set; logic operations clear both whatever their operands.
|
|
292
311
|
CLEARED_BY_LOGIC = {**CARRY_BRANCHES, "jo": True, "jno": False}
|
|
@@ -361,7 +380,18 @@ def ordinary(state, ins, image):
|
|
|
361
380
|
if state.flat and operands[0].type == X86_OP_REG and ins.reg_name(operands[0].reg) in state.segment_bases:
|
|
362
381
|
raise StopPath("Segment selector assignment requires a descriptor model")
|
|
363
382
|
value = state.get(ins, operands[1], image)
|
|
364
|
-
|
|
383
|
+
result = resize(value, operands[0].size * 8, signed=m == "movsx")
|
|
384
|
+
state.put(ins, operands[0], result)
|
|
385
|
+
if not state.value_transfers:
|
|
386
|
+
return
|
|
387
|
+
|
|
388
|
+
def location(operand):
|
|
389
|
+
return {"kind": "register", "register": ins.reg_name(operand.reg)} if operand.type == X86_OP_REG else {"kind": "memory"} if operand.type == X86_OP_MEM else {"kind": "immediate"}
|
|
390
|
+
destination_container = ALIASES[ins.reg_name(operands[0].reg)][0] if operands[0].type == X86_OP_REG else None
|
|
391
|
+
state.event("value-transfer", operation=m, source=location(operands[1]), destination=location(operands[0]),
|
|
392
|
+
destinationContainer=destination_container, destinationContainerValue=state.reg(destination_container).report() if destination_container else None,
|
|
393
|
+
sourceBits=value.bits, destinationBits=result.bits, sourceValue=value.report(), resultValue=result.report(),
|
|
394
|
+
conversion="truncate" if result.bits < value.bits else "signExtend" if m == "movsx" else "zeroExtend" if result.bits > value.bits else "sameWidth")
|
|
365
395
|
return
|
|
366
396
|
if m == "xchg":
|
|
367
397
|
values = [state.get(ins, operand, image) for operand in operands]
|
|
@@ -445,21 +475,25 @@ def ordinary(state, ins, image):
|
|
|
445
475
|
# The prefix toggles the mode's default operand size (16-bit real mode, 32-bit flat).
|
|
446
476
|
wide = (0x66 in ins.prefix) != state.flat
|
|
447
477
|
source, destination = ("ax", "eax") if wide else ("al", "ax")
|
|
448
|
-
|
|
478
|
+
source_value = state.reg(source)
|
|
479
|
+
value = resize(source_value, 32 if wide else 16, True)
|
|
449
480
|
state.setreg(destination, value, state.at)
|
|
450
481
|
state.event("conversion", sourceRegister=source, destinationRegister=destination,
|
|
451
482
|
effectiveOperandBits=32 if wide else 16, decoderMnemonic=m,
|
|
452
|
-
mnemonicWidthMismatch=m != ("cwde" if wide else "cbw"), result=value.report()
|
|
483
|
+
mnemonicWidthMismatch=m != ("cwde" if wide else "cbw"), result=value.report(),
|
|
484
|
+
sourceValue=source_value.report(), sourceBits=source_value.bits, destinationBits=value.bits, conversion="signExtend")
|
|
453
485
|
return
|
|
454
486
|
if m in ("cwd", "cdq"):
|
|
455
487
|
wide = (0x66 in ins.prefix) != state.flat
|
|
456
488
|
source, destination = ("eax", "edx") if wide else ("ax", "dx")
|
|
457
489
|
bits = 32 if wide else 16
|
|
458
|
-
|
|
490
|
+
source_value = state.reg(source)
|
|
491
|
+
value = resize(extract(source_value, bits-1, 1), bits, signed=True)
|
|
459
492
|
state.setreg(destination, value, state.at)
|
|
460
493
|
state.event("conversion", sourceRegister=source, destinationRegister=destination,
|
|
461
494
|
effectiveOperandBits=bits, decoderMnemonic=m,
|
|
462
|
-
mnemonicWidthMismatch=m != ("cdq" if wide else "cwd"), result=value.report()
|
|
495
|
+
mnemonicWidthMismatch=m != ("cdq" if wide else "cwd"), result=value.report(),
|
|
496
|
+
sourceValue=source_value.report(), sourceBits=source_value.bits, destinationBits=value.bits, conversion="signFillHighHalf")
|
|
463
497
|
return
|
|
464
498
|
if m in ("clc", "stc", "cmc"):
|
|
465
499
|
if m == "cmc":
|
|
@@ -5,6 +5,7 @@ from capstone import CS_AC_READ, CS_AC_WRITE
|
|
|
5
5
|
from capstone.x86 import X86_OP_IMM, X86_OP_MEM, X86_OP_REG
|
|
6
6
|
from .machine import State, StopPath, REGISTERS, ALIASES, segment_register
|
|
7
7
|
from .values import unknown
|
|
8
|
+
from .result_flow import return_flows
|
|
8
9
|
from .image import Image, integer
|
|
9
10
|
from .trace import (trace, walk, call_target, unsupported_transfer, uncovered, base_mnemonic, OVERLAP_REASON, CONTESTED_REASON,
|
|
10
11
|
RETURNS, INTERRUPTS, PORTS)
|
|
@@ -298,7 +299,9 @@ def uses(image, config):
|
|
|
298
299
|
if state is None:
|
|
299
300
|
# Registers are unknown here; name them for this operand site so no entry value is implied.
|
|
300
301
|
state = State(at, image, {})
|
|
301
|
-
|
|
302
|
+
for r in REGISTERS:
|
|
303
|
+
if r != "cs":
|
|
304
|
+
state.setreg(r, unknown(f"CFG-operand:{at}:{r}", ALIASES[r][2]), None)
|
|
302
305
|
try:
|
|
303
306
|
segment_value, offset_value, segment_name = state.address(ins, operand)
|
|
304
307
|
except StopPath as error:
|
|
@@ -994,6 +997,200 @@ def callees(image, config):
|
|
|
994
997
|
"that does not. Neither proves runtime recursion."}
|
|
995
998
|
|
|
996
999
|
|
|
1000
|
+
# These branches test CX/ECX (LOOPE/LOOPNE also ZF), so an adjacent CMP/TEST never describes their predicate.
|
|
1001
|
+
COUNT_BRANCHES = frozenset(("jcxz", "jecxz", "jrcxz", "loop", "loope", "loopne", "loopz", "loopnz"))
|
|
1002
|
+
|
|
1003
|
+
|
|
1004
|
+
def _operand_view(ins, o):
|
|
1005
|
+
if o.type == X86_OP_REG:
|
|
1006
|
+
return {"kind": "register", "name": ins.reg_name(o.reg), "width": o.size}
|
|
1007
|
+
if o.type == X86_OP_IMM:
|
|
1008
|
+
return {"kind": "immediate", "value": o.imm, "width": o.size}
|
|
1009
|
+
if o.type == X86_OP_MEM:
|
|
1010
|
+
return {"kind": "memory", "width": memory_width(ins, o), "segmentRegister": segment_register(ins, o.mem),
|
|
1011
|
+
"baseRegister": ins.reg_name(o.mem.base) or None, "indexRegister": ins.reg_name(o.mem.index) or None,
|
|
1012
|
+
"displacement": o.mem.disp}
|
|
1013
|
+
return {"kind": "unresolved"}
|
|
1014
|
+
|
|
1015
|
+
|
|
1016
|
+
def _stack_cleanup(ins):
|
|
1017
|
+
"""The bytes an immediate ADD SP/ESP releases, or None for any other instruction or a negative immediate."""
|
|
1018
|
+
if (ins is None or ins.mnemonic != "add" or len(ins.operands) != 2 or ins.operands[0].type != X86_OP_REG
|
|
1019
|
+
or ins.reg_name(ins.operands[0].reg) not in ("sp", "esp") or ins.operands[1].type != X86_OP_IMM):
|
|
1020
|
+
return None
|
|
1021
|
+
bits = 8 * ins.operands[0].size
|
|
1022
|
+
amount = ins.operands[1].imm & ((1 << bits) - 1)
|
|
1023
|
+
return amount if 0 < amount < 1 << (bits - 1) else None
|
|
1024
|
+
|
|
1025
|
+
|
|
1026
|
+
def call_order(image, config):
|
|
1027
|
+
"""Group the confirmed incoming calls of each containing entry by necessary guards and CFG order."""
|
|
1028
|
+
flat = incoming(image, config)
|
|
1029
|
+
entry_limit = integer(config.get("entryLimit", 64), 1, 256, "entry limit")
|
|
1030
|
+
analysis_limit = integer(config.get("analysisLimit", 1000000), 1, 10000000, "call order analysis limit")
|
|
1031
|
+
established = entries(image)
|
|
1032
|
+
bodies = {e: body(image, e, config.get("instructionLimit", 10000)) for e in established[:entry_limit]}
|
|
1033
|
+
conflicts = _cross_entry_overlaps(bodies)
|
|
1034
|
+
unchecked = established[entry_limit:]
|
|
1035
|
+
sites = {r["site"] for r in flat["confirmed"]}
|
|
1036
|
+
owners = {at: [e for e, b in bodies.items() if at in b["instructions"]] for at in sites}
|
|
1037
|
+
reports = []
|
|
1038
|
+
for entry, b in bodies.items():
|
|
1039
|
+
selected = sorted(at for at in sites if entry in owners[at])
|
|
1040
|
+
if not selected:
|
|
1041
|
+
continue
|
|
1042
|
+
usable = b["complete"] and not unchecked and not conflicts.get(entry) and not flat["truncated"] and all(owners[at] == [entry] for at in selected)
|
|
1043
|
+
instructions, successors, branches, exits = b["instructions"], {}, [], set()
|
|
1044
|
+
ending = {}
|
|
1045
|
+
for at, ins in instructions.items():
|
|
1046
|
+
ending.setdefault(at + ins.size, []).append(at)
|
|
1047
|
+
for at, ins in instructions.items():
|
|
1048
|
+
m, following = base_mnemonic(ins), at + ins.size
|
|
1049
|
+
targets = []
|
|
1050
|
+
if m in RETURNS or m == "hlt" or unsupported_transfer(image, ins):
|
|
1051
|
+
pass
|
|
1052
|
+
elif m in ("jmp", "ljmp"):
|
|
1053
|
+
d = image.indirect_jumps.get(at)
|
|
1054
|
+
targets = [r["target"] for r in d["rows"]] if d else [call_target(image, at, ins)[0]]
|
|
1055
|
+
elif m.startswith("j") or m.startswith("loop"):
|
|
1056
|
+
target = call_target(image, at, ins)[0]
|
|
1057
|
+
targets = [target, following]
|
|
1058
|
+
if target != following:
|
|
1059
|
+
producer = ending.get(at, []) if m not in COUNT_BRANCHES and at != entry else []
|
|
1060
|
+
comparison = instructions[producer[0]] if len(producer) == 1 else None
|
|
1061
|
+
context = ({"site": producer[0], "mnemonic": comparison.mnemonic,
|
|
1062
|
+
"operands": [_operand_view(comparison, o) for o in comparison.operands]}
|
|
1063
|
+
if comparison and comparison.mnemonic in ("cmp", "test") else None)
|
|
1064
|
+
for taken, to in ((True, target), (False, following)):
|
|
1065
|
+
if to is not None:
|
|
1066
|
+
branches.append({"site": at, "to": to, "taken": taken, "predicate": m,
|
|
1067
|
+
"comparison": context, "interpretation": "necessary caller CFG edge, not a runtime value or preserved guard"})
|
|
1068
|
+
else:
|
|
1069
|
+
targets = [following]
|
|
1070
|
+
successors[at] = sorted(set(t for t in targets if t in instructions))
|
|
1071
|
+
# Returns, halts, unsupported frames and transfers out of the body leave the caller CFG.
|
|
1072
|
+
if not targets or any(t not in instructions for t in targets):
|
|
1073
|
+
exits.add(at)
|
|
1074
|
+
predecessors = {}
|
|
1075
|
+
for source, targets in successors.items():
|
|
1076
|
+
for to in targets:
|
|
1077
|
+
predecessors.setdefault(to, set()).add(source)
|
|
1078
|
+
for guard in branches:
|
|
1079
|
+
if guard["comparison"] and predecessors.get(guard["site"], set()) != {guard["comparison"]["site"]}:
|
|
1080
|
+
guard["comparison"] = None
|
|
1081
|
+
cache, spent, capped = {}, 0, False
|
|
1082
|
+
def reachable(start, blocked=None, stops=frozenset()):
|
|
1083
|
+
nonlocal spent, capped
|
|
1084
|
+
key = (start, blocked, stops)
|
|
1085
|
+
if key in cache:
|
|
1086
|
+
return cache[key]
|
|
1087
|
+
todo, seen = [start], set()
|
|
1088
|
+
while todo:
|
|
1089
|
+
at = todo.pop()
|
|
1090
|
+
if at in seen or at not in instructions or at in stops:
|
|
1091
|
+
continue
|
|
1092
|
+
if spent >= analysis_limit:
|
|
1093
|
+
capped = True
|
|
1094
|
+
return set()
|
|
1095
|
+
spent += 1
|
|
1096
|
+
seen.add(at)
|
|
1097
|
+
todo.extend(to for to in successors[at] if (at, to) != blocked)
|
|
1098
|
+
cache[key] = seen
|
|
1099
|
+
return seen
|
|
1100
|
+
|
|
1101
|
+
def dominates(first, second):
|
|
1102
|
+
"""Whether every route from the entry to second passes first."""
|
|
1103
|
+
return second not in reachable(entry, stops=frozenset((first,)))
|
|
1104
|
+
|
|
1105
|
+
def must_follow(first, second, guards):
|
|
1106
|
+
"""Whether every route from first's continuation reaches second before an exit or a guard revisit."""
|
|
1107
|
+
start = first + instructions[first].size
|
|
1108
|
+
if start == second:
|
|
1109
|
+
return True
|
|
1110
|
+
if start in guards:
|
|
1111
|
+
return False
|
|
1112
|
+
seen = reachable(start, stops=guards | {second})
|
|
1113
|
+
return not seen & exits and not any(to in guards for at in seen for to in successors[at])
|
|
1114
|
+
necessary = {at: [] for at in selected}
|
|
1115
|
+
for guard in branches:
|
|
1116
|
+
remaining = reachable(entry, (guard["site"], guard["to"]))
|
|
1117
|
+
for at in selected:
|
|
1118
|
+
if at not in remaining:
|
|
1119
|
+
necessary[at].append({k: v for k, v in guard.items() if k != "to"})
|
|
1120
|
+
later = {at: reachable(at + instructions[at].size) & sites for at in selected}
|
|
1121
|
+
rows = []
|
|
1122
|
+
for at in selected:
|
|
1123
|
+
following = at + instructions[at].size
|
|
1124
|
+
amount = _stack_cleanup(instructions.get(following))
|
|
1125
|
+
rows.append({"site": at, "necessaryGuards": necessary[at] if not capped else [],
|
|
1126
|
+
"cleanup": {"continuation": following, "site": following if amount is not None else None,
|
|
1127
|
+
"argumentBytes": amount, "status": "observed after assumed return" if amount is not None else "unread cleanup",
|
|
1128
|
+
"assumption": "callee returns to the next instruction"},
|
|
1129
|
+
"calleeEffects": {"status": "unresolved", "reason": "caller CFG never proves callee return success or restored/preserved state"}})
|
|
1130
|
+
partitions = {}
|
|
1131
|
+
for row in rows:
|
|
1132
|
+
key = tuple((g["site"], g["taken"]) for g in row["necessaryGuards"])
|
|
1133
|
+
partitions.setdefault(key, []).append(row["site"])
|
|
1134
|
+
groups = []
|
|
1135
|
+
for members in partitions.values():
|
|
1136
|
+
shared_guards = next(r["necessaryGuards"] for r in rows if r["site"] == members[0]) if not capped else []
|
|
1137
|
+
guard_sites = frozenset(g["site"] for g in shared_guards)
|
|
1138
|
+
local_later = {at: reachable(at + instructions[at].size, stops=guard_sites) & sites for at in members}
|
|
1139
|
+
pairs = [(a, z) for index, a in enumerate(members) for z in members[index + 1:]]
|
|
1140
|
+
sequence = all((z in local_later[a]) != (a in local_later[z]) for a, z in pairs) and all(a not in local_later[a] for a in members)
|
|
1141
|
+
alternatives = len(members) > 1 and all(z not in local_later[a] and a not in local_later[z] for a, z in pairs)
|
|
1142
|
+
kind = "unread" if not usable or capped else "sequence" if sequence else "branchAlternatives" if alternatives else "unread"
|
|
1143
|
+
order = sorted(members, key=lambda a: sum(a in local_later[z] for z in members if z != a)) if kind == "sequence" else None
|
|
1144
|
+
# Equal single-edge guards do not make every member run: a later call may be reached around an
|
|
1145
|
+
# earlier one (two branches to it, a jump table) or be skipped after it. Each call must dominate
|
|
1146
|
+
# the next and the next must follow it on every route within the visit.
|
|
1147
|
+
if order and not all(dominates(a, z) and must_follow(a, z, guard_sites) for a, z in zip(order, order[1:])):
|
|
1148
|
+
kind, order = "unread", None
|
|
1149
|
+
groups.append({"kind": kind, "sites": members, "order": order,
|
|
1150
|
+
"sharedGuards": shared_guards,
|
|
1151
|
+
"scope": "one visit past the shared guard edges" if guard_sites else "caller CFG",
|
|
1152
|
+
"mayRepeatAcrossGuardVisits": any(a in later[a] for a in members)})
|
|
1153
|
+
if capped or not usable:
|
|
1154
|
+
# A cycle found in a partial read still may repeat; finding none there proves nothing.
|
|
1155
|
+
for group in groups:
|
|
1156
|
+
group["mayRepeatAcrossGuardVisits"] = group["mayRepeatAcrossGuardVisits"] or None
|
|
1157
|
+
if capped:
|
|
1158
|
+
for group in groups:
|
|
1159
|
+
group.update(kind="unread", order=None, sharedGuards=[])
|
|
1160
|
+
for row in rows:
|
|
1161
|
+
row["necessaryGuards"] = []
|
|
1162
|
+
relations = []
|
|
1163
|
+
for index, first in enumerate(selected):
|
|
1164
|
+
for second in selected[index + 1:]:
|
|
1165
|
+
forward, backward = second in later[first], first in later[second]
|
|
1166
|
+
kind = ("unread" if not usable or capped or forward and backward else
|
|
1167
|
+
"sequence" if forward or backward else "branchAlternatives")
|
|
1168
|
+
relations.append({"sites": [first, second], "kind": kind,
|
|
1169
|
+
"order": [first, second] if kind == "sequence" and forward else [second, first] if kind == "sequence" else None})
|
|
1170
|
+
if relations and all(r["kind"] == "branchAlternatives" for r in relations):
|
|
1171
|
+
common = [g for g in rows[0]["necessaryGuards"] if all(any(g["site"] == h["site"] and g["taken"] == h["taken"] for h in row["necessaryGuards"]) for row in rows)]
|
|
1172
|
+
groups = [{"kind": "branchAlternatives", "sites": selected, "order": None, "sharedGuards": common,
|
|
1173
|
+
"scope": "caller CFG", "mayRepeatAcrossGuardVisits": any(a in later[a] for a in selected)}]
|
|
1174
|
+
reports.append({"entry": entry, "ranges": b["intervals"], "boundaryUsable": usable, "orderingUsable": usable and not capped,
|
|
1175
|
+
"gaps": b["gaps"], "conflicts": conflicts.get(entry, []), "analysisCapped": capped, "analysisSteps": spent,
|
|
1176
|
+
"calls": rows, "groups": groups, "relations": relations, "assumptions": b["assumedContinuations"]})
|
|
1177
|
+
controls = config.get("orderControls", [])
|
|
1178
|
+
if not isinstance(controls, list) or len(controls) > 256:
|
|
1179
|
+
raise ValueError("Invalid call order controls")
|
|
1180
|
+
for control in controls:
|
|
1181
|
+
if not isinstance(control, dict) or control.get("kind") not in ("sequence", "branchAlternatives") or not isinstance(control.get("sites"), list) or not control["sites"] or len(control["sites"]) > 256 or any(type(at) is not int for at in control["sites"]) or type(control.get("entry")) is not int:
|
|
1182
|
+
raise ValueError("Invalid call order control")
|
|
1183
|
+
report = next((r for r in reports if r["entry"] == control.get("entry")), None)
|
|
1184
|
+
# Sequence sites are compared in order; branch alternatives have none.
|
|
1185
|
+
expected = control["sites"] if control["kind"] == "sequence" else sorted(control["sites"])
|
|
1186
|
+
group = next((g for g in report["groups"] if g["kind"] == control["kind"] and (g["order"] if g["kind"] == "sequence" else sorted(g["sites"])) == expected), None) if report else None
|
|
1187
|
+
if group is None:
|
|
1188
|
+
raise ValueError("Call order positive control missed")
|
|
1189
|
+
return {"incoming": flat, "callers": reports, "uncheckedEntries": unchecked, "orderControls": controls,
|
|
1190
|
+
"unownedSites": sorted(at for at in sites if not owners[at]),
|
|
1191
|
+
"interpretation": "Order and guards describe a conditional caller CFG. Cleanup follows an assumed return; callees' effects, successful return and state restoration remain unresolved."}
|
|
1192
|
+
|
|
1193
|
+
|
|
997
1194
|
def _body_report(b):
|
|
998
1195
|
return {k: v for k, v in b.items() if k != "instructions"} | {"instructionCount": len(b["instructions"])}
|
|
999
1196
|
|
|
@@ -1158,6 +1355,8 @@ def _run_report(image, config, command):
|
|
|
1158
1355
|
return owner(image, config)
|
|
1159
1356
|
if command == "callees":
|
|
1160
1357
|
return callees(image, config)
|
|
1358
|
+
if command == "call-order":
|
|
1359
|
+
return call_order(image, config)
|
|
1161
1360
|
if command == "incoming":
|
|
1162
1361
|
return incoming(image, config)
|
|
1163
1362
|
if command == "operand-candidates":
|
|
@@ -1180,6 +1379,8 @@ def _run_report(image, config, command):
|
|
|
1180
1379
|
report = near_pointer_provenance(report, config)
|
|
1181
1380
|
if command == "allocation":
|
|
1182
1381
|
return allocations(report, config)
|
|
1382
|
+
if command == "returns":
|
|
1383
|
+
report = return_flows(report, config)
|
|
1183
1384
|
if command != "trace":
|
|
1184
1385
|
kinds = {"arguments": ("address-formation", "read", "call", "call-return"), "effects": ("address-formation", "write", "call", "call-return", "return", "branch", "string-operation",
|
|
1185
1386
|
"flag-assumption", "flag-write", "flags-save", "flags-restore", "local-iret"),
|
|
@@ -1187,7 +1388,9 @@ def _run_report(image, config, command):
|
|
|
1187
1388
|
"guards": ("compare", "branch", "read", "write", "call", "call-return"),
|
|
1188
1389
|
"memory": ("read", "write", "address-formation")}[command]
|
|
1189
1390
|
for path in report["paths"]:
|
|
1190
|
-
|
|
1391
|
+
# Returns keep the transfers, conversions and reads that depend on a declared result.
|
|
1392
|
+
consumed = {c["order"] for f in path.get("returnFlows", {}).get("results", ()) for c in f["consumers"]}
|
|
1393
|
+
path["events"] = [e for e in path["events"] if e["kind"] in kinds or e["order"] in consumed or
|
|
1191
1394
|
(command == "effects" and e["kind"] == "read" and (e.get("nearPointerAccessCandidates") or e.get("nearPointerArgumentCandidates")))]
|
|
1192
1395
|
return report
|
|
1193
1396
|
|
|
@@ -0,0 +1,131 @@
|
|
|
1
|
+
"""Declared result roles and bounded producer dependencies across caller continuations."""
|
|
2
|
+
from .image import integer
|
|
3
|
+
from .machine import ALIASES, BRANCH_CONDITIONS
|
|
4
|
+
from .values import result_marker
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
def validate_contracts(config, image):
|
|
8
|
+
contracts = config.get("returnContracts", [])
|
|
9
|
+
if not isinstance(contracts, list) or len(contracts) > 256:
|
|
10
|
+
raise ValueError("At most 256 return contracts")
|
|
11
|
+
keys = set()
|
|
12
|
+
for c in contracts:
|
|
13
|
+
if not isinstance(c, dict):
|
|
14
|
+
raise ValueError("Return contract must be an object")
|
|
15
|
+
entry = integer(c.get("entry"), 0, len(image.data) - 1, "return contract entry")
|
|
16
|
+
register = c.get("register")
|
|
17
|
+
if not isinstance(register, str) or register not in ALIASES or not isinstance(c.get("evidence"), str) or not c["evidence"].strip():
|
|
18
|
+
raise ValueError("Return contract requires register and evidence")
|
|
19
|
+
if (entry, register) in keys:
|
|
20
|
+
raise ValueError("Return contract entry/register must be unique")
|
|
21
|
+
keys.add((entry, register))
|
|
22
|
+
bits = ALIASES[register][2]
|
|
23
|
+
failures = c.get("failures", [])
|
|
24
|
+
if not isinstance(failures, list) or len(failures) > 256 or any(type(n) is not int or not 0 <= n < 1 << bits for n in failures):
|
|
25
|
+
raise ValueError("Failure encodings must fit the consumed return width")
|
|
26
|
+
encodings = c.get("encodings", [])
|
|
27
|
+
if not isinstance(encodings, list) or len(encodings) > 256:
|
|
28
|
+
raise ValueError("At most 256 result encodings")
|
|
29
|
+
for encoding in encodings:
|
|
30
|
+
if not isinstance(encoding, dict) or type(encoding.get("value")) is not int or not 0 <= encoding["value"] < 1 << bits:
|
|
31
|
+
raise ValueError("Result encoding must fit the declared return width")
|
|
32
|
+
if not all(isinstance(encoding.get(k), str) and encoding[k].strip() for k in ("role", "evidence")):
|
|
33
|
+
raise ValueError("Result encoding requires role and evidence")
|
|
34
|
+
return contracts
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def result_contracts(state, contracts, entry):
|
|
38
|
+
rows = []
|
|
39
|
+
for c in contracts:
|
|
40
|
+
if c["entry"] != entry:
|
|
41
|
+
continue
|
|
42
|
+
register = c["register"]
|
|
43
|
+
# Tag the declared value with a marker unique to the return event about to be
|
|
44
|
+
# recorded. The instruction site would also tag the SP the return pops, every
|
|
45
|
+
# register a call model clobbers and every later execution of the same return.
|
|
46
|
+
# A producer dependency is still weaker than unchanged value identity.
|
|
47
|
+
state.setreg(register, state.reg(register), result_marker(len(state.events)))
|
|
48
|
+
value = state.reg(register)
|
|
49
|
+
rows.append({"entry": entry, "register": register, "value": value.report(),
|
|
50
|
+
"failureEncodings": c.get("failures", []), "encodings": c.get("encodings", []),
|
|
51
|
+
"matchesFailureEncoding": None if value.number is None else value.number in c.get("failures", []),
|
|
52
|
+
"matchingRoles": None if value.number is None else [e for e in c.get("encodings", []) if e["value"] == value.number],
|
|
53
|
+
"evidence": c["evidence"], "encodingsExhaustive": False, "successEstablished": False})
|
|
54
|
+
return rows
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
# Sign and overflow tests read the operand as a signed value; carry tests as unsigned.
|
|
58
|
+
DOMAINS = {"l": "signed", "le": "signed", "s": "signed", "o": "signed", "c": "unsigned", "be": "unsigned"}
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
VALUE_FIELDS = ("sourceValue", "resultValue", "destinationContainerValue", "result", "value", "left", "right", "carry", "count")
|
|
62
|
+
CONSUMER_KINDS = ("value-transfer", "conversion", "read", "write", "compare", "branch", "return")
|
|
63
|
+
ROW_FIELDS = ("kind", "site", "entry", "depth", "order", "operation", "source", "destination", "sourceBits", "destinationBits",
|
|
64
|
+
"destinationContainer", "conversion", "sourceRegister", "destinationRegister", "effectiveOperandBits", "decoderMnemonic",
|
|
65
|
+
"width", "segment", "offset", "effectiveSegmentRegister", "predicate", "taken", "flagProducer")
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def _event_values(event):
|
|
69
|
+
values = {k: event[k] for k in VALUE_FIELDS if isinstance(event.get(k), dict)}
|
|
70
|
+
if event["kind"] == "return":
|
|
71
|
+
values.update({"return:" + c["register"]: c["value"] for c in event.get("resultContracts", [])})
|
|
72
|
+
return values
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def return_flows(report, config):
|
|
76
|
+
limit = integer(config.get("returnFlowLimit", 128), 1, 1024, "return flow limit")
|
|
77
|
+
consumer_limit = integer(config.get("returnConsumerLimit", 256), 1, 10000, "return consumer limit")
|
|
78
|
+
analysis_limit = integer(config.get("returnFlowAnalysisLimit", 1000000), 1, 10000000, "return flow analysis limit")
|
|
79
|
+
steps, capped = 0, False
|
|
80
|
+
for path in report["paths"]:
|
|
81
|
+
events = path["events"]
|
|
82
|
+
event_values = {}
|
|
83
|
+
flows, omitted = [], 0
|
|
84
|
+
for index, event in enumerate(events):
|
|
85
|
+
if event["kind"] != "return" and not (event["kind"] == "call-return" and event.get("modeled")):
|
|
86
|
+
continue
|
|
87
|
+
for contract in event.get("resultContracts", []):
|
|
88
|
+
if len(flows) >= limit or capped:
|
|
89
|
+
omitted += 1
|
|
90
|
+
continue
|
|
91
|
+
consumers, dropped, scanned = [], 0, True
|
|
92
|
+
marker = event["order"]
|
|
93
|
+
for later_index in range(index + 1, len(events)):
|
|
94
|
+
if steps >= analysis_limit:
|
|
95
|
+
capped, scanned = True, False
|
|
96
|
+
break
|
|
97
|
+
steps += 1
|
|
98
|
+
later = events[later_index]
|
|
99
|
+
if later["kind"] not in CONSUMER_KINDS:
|
|
100
|
+
continue
|
|
101
|
+
if later_index not in event_values:
|
|
102
|
+
event_values[later_index] = _event_values(later)
|
|
103
|
+
values = event_values[later_index]
|
|
104
|
+
dependent = {k: v for k, v in values.items() if marker in v.get("resultOrigins", ())}
|
|
105
|
+
if not dependent:
|
|
106
|
+
continue
|
|
107
|
+
if len(consumers) >= consumer_limit:
|
|
108
|
+
dropped += 1
|
|
109
|
+
continue
|
|
110
|
+
row = {k: later[k] for k in ROW_FIELDS if k in later}
|
|
111
|
+
widths = {k: "narrower" if v["bits"] < contract["value"]["bits"] else
|
|
112
|
+
"wider" if v["bits"] > contract["value"]["bits"] else "sameWidth"
|
|
113
|
+
for k, v in dependent.items()}
|
|
114
|
+
row.update(values=values, dependentValueFields=list(dependent), returnWidthRelationships=widths,
|
|
115
|
+
relationship="producer dependency only; not value or storage identity")
|
|
116
|
+
if later["kind"] == "branch":
|
|
117
|
+
condition = BRANCH_CONDITIONS.get(later.get("predicate"), (None,))[0]
|
|
118
|
+
row["predicateDomain"] = DOMAINS.get(condition, "flags/equality")
|
|
119
|
+
consumers.append(row)
|
|
120
|
+
flows.append({"originOrder": event["order"], "originSite": event["site"], "calleeEntry": contract["entry"],
|
|
121
|
+
"callSite": event.get("callSite"), "callerEntry": event.get("callerEntry"),
|
|
122
|
+
"conditionalModel": event.get("modeled", False), "resultContract": contract,
|
|
123
|
+
"consumers": consumers, "consumersOmitted": dropped, "consumerScanComplete": scanned,
|
|
124
|
+
"successEstablished": False})
|
|
125
|
+
path["returnFlows"] = {"results": flows, "resultsOmitted": omitted, "resultLimit": limit,
|
|
126
|
+
"consumerLimit": consumer_limit,
|
|
127
|
+
"complete": not omitted and all(not f["consumersOmitted"] and f["consumerScanComplete"] for f in flows)
|
|
128
|
+
and path["returned"] and not report["gaps"],
|
|
129
|
+
"interpretation": "conditional static dependency paths; encodings are declared evidence, not live occurrence; a branch never establishes initialization, accepted contents or extent"}
|
|
130
|
+
report["returnFlowAnalysis"] = {"limit": analysis_limit, "steps": steps, "capped": capped}
|
|
131
|
+
return report
|
|
@@ -2,23 +2,10 @@
|
|
|
2
2
|
from copy import deepcopy
|
|
3
3
|
from capstone.x86 import X86_OP_IMM, X86_OP_REG, X86_OP_MEM
|
|
4
4
|
from .image import integer
|
|
5
|
-
from .machine import (State, StopPath, ordinary, predicate, REGISTERS, ALIASES,
|
|
6
|
-
string_effect, check_string_form)
|
|
5
|
+
from .machine import (State, StopPath, ordinary, predicate, REGISTERS, ALIASES, BRANCH_CONDITIONS, string_instruction,
|
|
6
|
+
string_count, string_effect, check_string_form)
|
|
7
7
|
from .values import const, unknown, sources, op, Value
|
|
8
|
-
|
|
9
|
-
# Synonymous and complementary branches on one flag producer share a single assumption.
|
|
10
|
-
BRANCH_CONDITIONS = {}
|
|
11
|
-
for names, condition in ((("je", "jz"), "z"), (("jb", "jc", "jnae"), "c"), (("jbe", "jna"), "be"),
|
|
12
|
-
(("jl", "jnge"), "l"), (("jle", "jng"), "le"), (("js",), "s"),
|
|
13
|
-
(("jo",), "o"), (("jp", "jpe"), "p")):
|
|
14
|
-
for name in names:
|
|
15
|
-
BRANCH_CONDITIONS[name] = (condition, False)
|
|
16
|
-
for names, condition in ((("jne", "jnz"), "z"), (("jae", "jnb", "jnc"), "c"), (("ja", "jnbe"), "be"),
|
|
17
|
-
(("jge", "jnl"), "l"), (("jg", "jnle"), "le"), (("jns",), "s"),
|
|
18
|
-
(("jno",), "o"), (("jnp", "jpo"), "p")):
|
|
19
|
-
for name in names:
|
|
20
|
-
BRANCH_CONDITIONS[name] = (condition, True)
|
|
21
|
-
|
|
8
|
+
from .result_flow import validate_contracts, result_contracts
|
|
22
9
|
|
|
23
10
|
def call_target(image, site, ins):
|
|
24
11
|
if ins.mnemonic in ("lcall", "ljmp"):
|
|
@@ -237,6 +224,7 @@ def trace(image, config):
|
|
|
237
224
|
integer(config.get("returnBytes", image.bits // 8), 2, 4, "returnBytes")
|
|
238
225
|
if config.get("returnBytes", image.bits // 8) not in ((4,) if image.flat else (2, 4)):
|
|
239
226
|
raise ValueError("returnBytes must agree with the selected near/far frame model")
|
|
227
|
+
contracts = validate_contracts(config, image)
|
|
240
228
|
models = config.get("callModels", [])
|
|
241
229
|
if not isinstance(models, list) or len(models) > 64:
|
|
242
230
|
raise ValueError("At most 64 explicit call models")
|
|
@@ -386,7 +374,7 @@ def trace(image, config):
|
|
|
386
374
|
created += 1
|
|
387
375
|
for r in REGISTERS:
|
|
388
376
|
if r not in model.get("preserves", []) and r not in ("esp", "cs"):
|
|
389
|
-
child.
|
|
377
|
+
child.setreg(r, unknown(f"modeled-call:{at}:{r}", ALIASES[r][2]), at)
|
|
390
378
|
child.clear_memory()
|
|
391
379
|
child.forget_flags()
|
|
392
380
|
child.direction_flag = unknown(f"modeled-call:{at}:DF:{child.flag_serial}", 1, at)
|
|
@@ -395,7 +383,8 @@ def trace(image, config):
|
|
|
395
383
|
child.setreg(r, const(n, ALIASES[r][2], at), at)
|
|
396
384
|
child.conditional.append({"site": at, "evidence": model["evidence"],
|
|
397
385
|
"assumption": "call returns with balanced stack; memory effects unresolved"})
|
|
398
|
-
child.event("call-return", callSite=at,
|
|
386
|
+
child.event("call-return", callSite=at, callerEntry=state.frames[-1]["entry"],
|
|
387
|
+
resultContracts=result_contracts(child, contracts, target), registers=snapshot(child), modeled=True,
|
|
399
388
|
unknownMemoryEffects=True)
|
|
400
389
|
child.at = following
|
|
401
390
|
pending.append(child)
|
|
@@ -451,21 +440,10 @@ def trace(image, config):
|
|
|
451
440
|
if image.flat and m == "retf":
|
|
452
441
|
raise StopPath("Far return is outside the PE32 flat model")
|
|
453
442
|
frame = state.frames[-1]
|
|
454
|
-
roles = []
|
|
455
|
-
|
|
456
|
-
|
|
457
|
-
|
|
458
|
-
register = contract.get("register")
|
|
459
|
-
if register not in ALIASES or not contract.get("evidence"):
|
|
460
|
-
raise ValueError("Return contract requires register and evidence")
|
|
461
|
-
value = state.reg(register)
|
|
462
|
-
failures = contract.get("failures", [])
|
|
463
|
-
if not isinstance(failures, list) or len(failures) > 256 or any(type(n) is not int or not 0 <= n < 1 << value.bits for n in failures):
|
|
464
|
-
raise ValueError("Failure encodings must fit the consumed return width")
|
|
465
|
-
roles.append({"register": register, "value": value.report(), "failureEncodings": failures,
|
|
466
|
-
"matchesFailureEncoding": None if value.number is None else value.number in failures,
|
|
467
|
-
"evidence": contract["evidence"]})
|
|
468
|
-
state.event("return", registers=snapshot(state), cleanupBytes=ins.operands[0].imm if ins.operands else 0, resultContracts=roles)
|
|
443
|
+
roles = result_contracts(state, contracts, frame["entry"])
|
|
444
|
+
state.event("return", registers=snapshot(state), cleanupBytes=ins.operands[0].imm if ins.operands else 0,
|
|
445
|
+
resultContracts=roles, callSite=frame.get("callSite"),
|
|
446
|
+
callerEntry=state.frames[-2]["entry"] if len(state.frames) > 1 else None)
|
|
469
447
|
# The entry frame gets the same width and balance checks as a traced call.
|
|
470
448
|
expected = 4 if m == "retf" else image.bits // 8
|
|
471
449
|
if expected != frame["returnBytes"] or state.reg(state.sp).term != frame["sp"].term:
|
|
@@ -22,8 +22,26 @@ class Value:
|
|
|
22
22
|
return self.term[1] if self.term[0] == "constant" else None
|
|
23
23
|
|
|
24
24
|
def report(self):
|
|
25
|
-
|
|
26
|
-
|
|
25
|
+
row = {"bits": self.bits, "expression": self.term, "value": self.number, "producers": producers(self)}
|
|
26
|
+
origins = result_origins(self)
|
|
27
|
+
if origins:
|
|
28
|
+
row["resultOrigins"] = origins
|
|
29
|
+
return row
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def result_marker(order):
|
|
33
|
+
"""A source tag naming the return event at ``order``; instruction sites are never negative."""
|
|
34
|
+
return -order - 1
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def producers(v):
|
|
38
|
+
"""The instruction sites among a value's sources, without return-result markers."""
|
|
39
|
+
return [s for s in v.sources if s >= 0]
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def result_origins(v):
|
|
43
|
+
"""The event orders of the declared return results a value depends on."""
|
|
44
|
+
return sorted(-s - 1 for s in v.sources if s < 0)
|
|
27
45
|
|
|
28
46
|
|
|
29
47
|
def const(n, bits, site=None):
|
|
@@ -62,6 +62,163 @@ def events(result, kind):
|
|
|
62
62
|
return [e for path in result["paths"] for e in path["events"] if e["kind"] == kind]
|
|
63
63
|
|
|
64
64
|
|
|
65
|
+
class ReturnFlowTests(unittest.TestCase):
|
|
66
|
+
def flow(self, code, contracts, **extra):
|
|
67
|
+
return report(code, "returns", returnContracts=contracts, **extra)
|
|
68
|
+
|
|
69
|
+
def contract(self, entry, register="ax", **extra):
|
|
70
|
+
return {"entry": entry, "register": register, "evidence": "synthetic result contract", **extra}
|
|
71
|
+
|
|
72
|
+
def test_nested_failure_word_byte_store_zero_extension_and_nonzero_gate(self):
|
|
73
|
+
c = Code().branch("e8", "wrapper").emit("0f b6 c0 85 c0").branch("74", "zero").emit("c3")
|
|
74
|
+
c.label("zero").emit("c3").label("wrapper").branch("e8", "callee").emit("a2 20 00 a0 20 00 c3")
|
|
75
|
+
c.label("callee").emit("b8 ff ff c3")
|
|
76
|
+
r = self.flow(c, [self.contract(c.labels["callee"], failures=[65535]), self.contract(c.labels["wrapper"], "al")], registers={"ds": 8192, "ss": 8192, "sp": 32768})
|
|
77
|
+
self.assertTrue(r["completeWithinModel"])
|
|
78
|
+
rows = r["paths"][0]["returnFlows"]["results"]
|
|
79
|
+
failure = next(f for f in rows if f["calleeEntry"] == c.labels["callee"])
|
|
80
|
+
self.assertTrue(failure["resultContract"]["matchesFailureEncoding"])
|
|
81
|
+
self.assertEqual(failure["callerEntry"], c.labels["wrapper"])
|
|
82
|
+
store = next(e for e in failure["consumers"] if e["kind"] == "write")
|
|
83
|
+
self.assertEqual(store["width"], 1)
|
|
84
|
+
self.assertEqual(store["values"]["value"]["value"], 255)
|
|
85
|
+
self.assertEqual(store["returnWidthRelationships"]["value"], "narrower")
|
|
86
|
+
extension = next(e for e in failure["consumers"] if e.get("conversion") == "zeroExtend")
|
|
87
|
+
self.assertEqual((extension["sourceBits"], extension["destinationBits"]), (8, 16))
|
|
88
|
+
branch = next(e for e in failure["consumers"] if e["kind"] == "branch")
|
|
89
|
+
self.assertEqual((branch["predicate"], branch["taken"], branch["values"]["left"]["value"]), ("je", False, 255))
|
|
90
|
+
self.assertFalse(failure["successEstablished"])
|
|
91
|
+
|
|
92
|
+
def test_full_word_failure_normalization(self):
|
|
93
|
+
c = Code().branch("e8", "callee").emit("83 f8 ff").branch("74", "bad").emit("b0 01 c3")
|
|
94
|
+
c.label("bad").emit("b0 00 c3").label("callee").emit("b8 ff ff c3")
|
|
95
|
+
r = self.flow(c, [self.contract(c.labels["callee"], failures=[65535]), self.contract(0, "al")])
|
|
96
|
+
f = r["paths"][0]["returnFlows"]["results"][0]
|
|
97
|
+
branch = next(e for e in f["consumers"] if e["kind"] == "branch")
|
|
98
|
+
self.assertEqual((branch["values"]["left"]["bits"], branch["taken"]), (16, True))
|
|
99
|
+
self.assertEqual(branch["values"]["right"]["value"], 65535)
|
|
100
|
+
self.assertNotIn("right", branch["dependentValueFields"])
|
|
101
|
+
self.assertEqual(r["paths"][0]["registers"]["al"]["value"], 0)
|
|
102
|
+
|
|
103
|
+
def test_signed_consumer_and_raw_field_role_remain_distinct(self):
|
|
104
|
+
c = Code().branch("e8", "callee").emit("83 f8 01").branch("7c", "reject").emit("c3")
|
|
105
|
+
c.label("reject").emit("c3").label("callee").emit("b8 00 80 c3")
|
|
106
|
+
encoding = {"value": 32768, "role": "raw field", "evidence": "synthetic field contract"}
|
|
107
|
+
r = self.flow(c, [self.contract(c.labels["callee"], failures=[65535], encodings=[encoding])])
|
|
108
|
+
f = r["paths"][0]["returnFlows"]["results"][0]
|
|
109
|
+
self.assertFalse(f["resultContract"]["matchesFailureEncoding"])
|
|
110
|
+
self.assertEqual(f["resultContract"]["matchingRoles"], [encoding])
|
|
111
|
+
branch = next(e for e in f["consumers"] if e["kind"] == "branch")
|
|
112
|
+
self.assertEqual((branch["predicateDomain"], branch["taken"]), ("signed", True))
|
|
113
|
+
|
|
114
|
+
def test_sign_extension_and_unrelated_equal_constant(self):
|
|
115
|
+
c = Code().branch("e8", "callee").emit("0f be d0 bb ff ff 83 fb ff c3")
|
|
116
|
+
c.label("callee").emit("b0 ff c3")
|
|
117
|
+
r = self.flow(c, [self.contract(c.labels["callee"], "al", failures=[255])])
|
|
118
|
+
consumers = r["paths"][0]["returnFlows"]["results"][0]["consumers"]
|
|
119
|
+
extension = next(e for e in consumers if e.get("conversion") == "signExtend")
|
|
120
|
+
self.assertEqual(extension["values"]["resultValue"]["value"], 65535)
|
|
121
|
+
self.assertFalse(any(e["kind"] == "compare" for e in consumers))
|
|
122
|
+
|
|
123
|
+
def test_implicit_sign_extension_retains_effective_widths(self):
|
|
124
|
+
c = Code().branch("e8", "callee").emit("98 99 c3").label("callee").emit("b0 ff c3")
|
|
125
|
+
r = self.flow(c, [self.contract(c.labels["callee"], "al", failures=[255])])
|
|
126
|
+
rows = r["paths"][0]["returnFlows"]["results"][0]["consumers"]
|
|
127
|
+
conversions = [e for e in rows if e["kind"] == "conversion"]
|
|
128
|
+
self.assertEqual([e["conversion"] for e in conversions], ["signExtend", "signFillHighHalf"])
|
|
129
|
+
self.assertEqual((conversions[0]["sourceBits"], conversions[0]["destinationBits"]), (8, 16))
|
|
130
|
+
|
|
131
|
+
def test_sibling_high_byte_zeroing_is_retained_before_word_zero_gate(self):
|
|
132
|
+
c = Code().branch("e8", "callee").emit("b4 00 09 c0").branch("75", "nonzero").emit("c3")
|
|
133
|
+
c.label("nonzero").emit("c3").label("callee").emit("b0 ff c3")
|
|
134
|
+
r = self.flow(c, [self.contract(c.labels["callee"], "al", failures=[255])])
|
|
135
|
+
rows = r["paths"][0]["returnFlows"]["results"][0]["consumers"]
|
|
136
|
+
write = next(e for e in rows if e["kind"] == "value-transfer")
|
|
137
|
+
self.assertEqual(write["destination"]["register"], "ah")
|
|
138
|
+
self.assertEqual(write["values"]["sourceValue"]["value"], 0)
|
|
139
|
+
self.assertIn("destinationContainerValue", write["dependentValueFields"])
|
|
140
|
+
branch = next(e for e in rows if e["kind"] == "branch")
|
|
141
|
+
self.assertEqual((branch["operation"], branch["predicate"], branch["taken"], branch["values"]["left"]["value"]), ("or", "jne", True, 255))
|
|
142
|
+
|
|
143
|
+
def test_models_stops_and_all_summary_caps_remain_explicit(self):
|
|
144
|
+
c = Code().branch("e8", "callee").emit("a2 20 00 85 c0").branch("74", "end").emit("90")
|
|
145
|
+
c.label("end").emit("c3").label("callee").emit("b8 ff ff c3")
|
|
146
|
+
contracts = [self.contract(c.labels["callee"], failures=[65535]), self.contract(0)]
|
|
147
|
+
for extra in ({"returnFlowLimit": 1}, {"returnConsumerLimit": 1}, {"returnFlowAnalysisLimit": 1}, {"maxSteps": 2}, {"maxDepth": 1}):
|
|
148
|
+
r = self.flow(c, contracts, **extra)
|
|
149
|
+
self.assertFalse(r["paths"][0]["returnFlows"]["complete"])
|
|
150
|
+
r = self.flow(c, contracts, callModels=[{"site": 0, "evidence": "synthetic conditional case", "cases": [{"registers": {"ax": 65535}}]}])
|
|
151
|
+
f = r["paths"][0]["returnFlows"]["results"][0]
|
|
152
|
+
self.assertTrue(f["conditionalModel"])
|
|
153
|
+
self.assertTrue(r["paths"][0]["conditionalModels"])
|
|
154
|
+
|
|
155
|
+
def test_values_sharing_only_the_return_or_call_site_are_not_dependent(self):
|
|
156
|
+
# The return pops SP at its own site; a call model clobbers BX at the call site.
|
|
157
|
+
c = Code().branch("e8", "callee").emit("89 e5 39 e9 89 d9 c3").label("callee").emit("b8 ff ff c3")
|
|
158
|
+
contracts = [self.contract(c.labels["callee"])]
|
|
159
|
+
r = self.flow(c, contracts, registers={"ss": 8192, "sp": 32768})
|
|
160
|
+
self.assertEqual(r["paths"][0]["returnFlows"]["results"][0]["consumers"], [])
|
|
161
|
+
r = self.flow(c, contracts, callModels=[{"site": 0, "evidence": "synthetic conditional case", "cases": [{"registers": {"ax": 1}}]}])
|
|
162
|
+
f = r["paths"][0]["returnFlows"]["results"][0]
|
|
163
|
+
self.assertTrue(f["conditionalModel"])
|
|
164
|
+
self.assertEqual(f["consumers"], [])
|
|
165
|
+
self.assertNotIn(0, r["paths"][0]["registers"]["sp"].get("resultOrigins", []))
|
|
166
|
+
|
|
167
|
+
def test_each_execution_of_one_return_is_a_separate_origin(self):
|
|
168
|
+
c = Code().branch("e8", "callee").emit("89 c3").branch("e8", "callee").emit("89 c1 c3")
|
|
169
|
+
c.label("callee").emit("66 b8 ff ff 00 00 c3")
|
|
170
|
+
rows = self.flow(c, [self.contract(c.labels["callee"])])["paths"][0]["returnFlows"]["results"]
|
|
171
|
+
self.assertEqual(len(rows), 2)
|
|
172
|
+
self.assertEqual([[e["site"] for e in f["consumers"]] for f in rows], [[3], [8]])
|
|
173
|
+
self.assertEqual(rows[0]["resultContract"]["value"]["resultOrigins"], [rows[0]["originOrder"]])
|
|
174
|
+
self.assertTrue(all(isinstance(p, int) and p >= 0 for p in rows[0]["resultContract"]["value"]["producers"]))
|
|
175
|
+
|
|
176
|
+
def test_sign_flag_gate_is_a_signed_predicate(self):
|
|
177
|
+
c = Code().branch("e8", "callee").emit("85 c0").branch("78", "negative").emit("c3")
|
|
178
|
+
c.label("negative").emit("c3").label("callee").emit("b8 ff ff c3")
|
|
179
|
+
rows = self.flow(c, [self.contract(c.labels["callee"], failures=[65535])])["paths"][0]["returnFlows"]["results"][0]["consumers"]
|
|
180
|
+
branch = next(e for e in rows if e["kind"] == "branch")
|
|
181
|
+
self.assertEqual((branch["predicate"], branch["predicateDomain"], branch["taken"]), ("js", "signed", True))
|
|
182
|
+
|
|
183
|
+
def test_analysis_cap_marks_the_truncated_flow(self):
|
|
184
|
+
c = Code().branch("e8", "callee").emit("a2 20 00 85 c0").branch("74", "end").emit("90")
|
|
185
|
+
c.label("end").emit("c3").label("callee").emit("b8 ff ff c3")
|
|
186
|
+
r = self.flow(c, [self.contract(c.labels["callee"], failures=[65535]), self.contract(0)], returnFlowAnalysisLimit=1)
|
|
187
|
+
flows = r["paths"][0]["returnFlows"]
|
|
188
|
+
self.assertTrue(r["returnFlowAnalysis"]["capped"])
|
|
189
|
+
self.assertFalse(flows["results"][0]["consumerScanComplete"])
|
|
190
|
+
self.assertEqual(flows["resultsOmitted"], 1)
|
|
191
|
+
self.assertFalse(flows["complete"])
|
|
192
|
+
r = self.flow(c, [self.contract(c.labels["callee"], failures=[65535]), self.contract(0)])
|
|
193
|
+
self.assertTrue(all(f["consumerScanComplete"] for f in r["paths"][0]["returnFlows"]["results"]))
|
|
194
|
+
self.assertTrue(r["paths"][0]["returnFlows"]["complete"])
|
|
195
|
+
|
|
196
|
+
def test_overwriting_the_result_register_ends_the_dependency_byte_by_byte(self):
|
|
197
|
+
# mov ax,5 replaces both result bytes; mov al,5 leaves AH, so the word compare still depends on it.
|
|
198
|
+
for overwrite, dependent in (("b8 05 00", False), ("b0 05", True)):
|
|
199
|
+
c = Code().branch("e8", "callee").emit(overwrite + " 83 f8 03 c3").label("callee").emit("b8 ff ff c3")
|
|
200
|
+
rows = self.flow(c, [self.contract(c.labels["callee"], failures=[65535])])["paths"][0]["returnFlows"]["results"][0]["consumers"]
|
|
201
|
+
self.assertEqual(any(e["kind"] == "compare" for e in rows), dependent)
|
|
202
|
+
r = report("b8 01 00 b0 02 c3")
|
|
203
|
+
self.assertEqual(r["paths"][0]["registers"]["ah"]["producers"], [0])
|
|
204
|
+
self.assertEqual(r["paths"][0]["registers"]["ax"]["producers"], [0, 3])
|
|
205
|
+
|
|
206
|
+
def test_value_transfers_are_recorded_only_for_declared_results(self):
|
|
207
|
+
c = Code().branch("e8", "callee").emit("89 c3 89 d1 8b 16 20 00 c3").label("callee").emit("b8 ff ff c3")
|
|
208
|
+
self.assertEqual(events(report(c), "value-transfer"), [])
|
|
209
|
+
r = self.flow(c, [self.contract(c.labels["callee"])], registers={"ds": 8192})
|
|
210
|
+
consumers = r["paths"][0]["returnFlows"]["results"][0]["consumers"]
|
|
211
|
+
self.assertEqual([e["site"] for e in consumers if e["kind"] == "value-transfer"], [3])
|
|
212
|
+
# Transfers and reads that do not depend on the result stay out of the returns events.
|
|
213
|
+
self.assertEqual([e["site"] for e in r["paths"][0]["events"] if e["kind"] in ("value-transfer", "read")], [3])
|
|
214
|
+
|
|
215
|
+
def test_invalid_unreachable_contracts_are_rejected(self):
|
|
216
|
+
for bad in (self.contract(0, failures=[65536]), self.contract(0, encodings=[{"value": 1, "role": "failure"}]),
|
|
217
|
+
self.contract(0, "fpu"), self.contract(0, ["ax"]), self.contract(0, evidence="")):
|
|
218
|
+
with self.assertRaises(ValueError):
|
|
219
|
+
self.flow(Code().emit("eb fe c3"), [bad], maxSteps=1)
|
|
220
|
+
|
|
221
|
+
|
|
65
222
|
class CalleeGraphTests(unittest.TestCase):
|
|
66
223
|
def graph(self, code, names, **extra):
|
|
67
224
|
data = code.bytes()
|
|
@@ -349,6 +506,137 @@ class NearPointerSegmentTests(unittest.TestCase):
|
|
|
349
506
|
self.assertTrue(all(p["dereferenceSegmentRegister"] == "es" and p["segmentRelationship"] == "sameWithinModel" for p in links))
|
|
350
507
|
|
|
351
508
|
|
|
509
|
+
class CallOrderTests(unittest.TestCase):
|
|
510
|
+
def run_order(self, code, **extra):
|
|
511
|
+
data = code.bytes()
|
|
512
|
+
cfg = configuration(data, target=code.labels["helper"], **extra)
|
|
513
|
+
cfg["regions"][0]["entries"] = [0, code.labels["helper"]]
|
|
514
|
+
return run_report(data, cfg, "call-order"), cfg
|
|
515
|
+
|
|
516
|
+
def test_guarded_sequence_retains_shared_cmp_cleanup_and_unread_effects(self):
|
|
517
|
+
c = Code().emit("83 7e fa 05").branch("7c", "end")
|
|
518
|
+
c.label("one").branch("e8", "helper").emit("83 c4 08")
|
|
519
|
+
c.label("two").branch("e8", "helper").emit("83 c4 08")
|
|
520
|
+
c.label("end").emit("c3").label("helper").emit("c3")
|
|
521
|
+
r, cfg = self.run_order(c, controls=[c.labels["one"], c.labels["two"]], orderControls=[{"entry": 0, "kind": "sequence", "sites": [c.labels["one"], c.labels["two"]]}])
|
|
522
|
+
caller = r["callers"][0]
|
|
523
|
+
self.assertEqual(caller["groups"][0]["kind"], "sequence")
|
|
524
|
+
self.assertEqual(caller["groups"][0]["sharedGuards"][0]["comparison"]["operands"][0]["segmentRegister"], "ss")
|
|
525
|
+
self.assertTrue(all(row["cleanup"]["argumentBytes"] == 8 and row["calleeEffects"]["status"] == "unresolved" for row in caller["calls"]))
|
|
526
|
+
self.assertEqual(len(r["incoming"]["confirmed"]), 2)
|
|
527
|
+
|
|
528
|
+
def test_adjacent_comparison_is_not_claimed_when_another_edge_enters_the_branch(self):
|
|
529
|
+
c = Code().emit("85 c0").branch("74", "compare").branch("e9", "condition")
|
|
530
|
+
c.label("compare").emit("83 fb 05").label("condition").branch("7c", "end").branch("e8", "helper")
|
|
531
|
+
c.label("end").emit("c3").label("helper").emit("c3")
|
|
532
|
+
r, _ = self.run_order(c)
|
|
533
|
+
guards = r["callers"][0]["groups"][0]["sharedGuards"]
|
|
534
|
+
conditional = next(g for g in guards if g["site"] == c.labels["condition"])
|
|
535
|
+
self.assertIsNone(conditional["comparison"])
|
|
536
|
+
|
|
537
|
+
def test_sequence_order_comes_from_flow_not_ascending_addresses(self):
|
|
538
|
+
c = Code().branch("e9", "first").label("second").branch("e8", "helper").emit("c3")
|
|
539
|
+
c.label("first").branch("e8", "helper").branch("e9", "second").label("helper").emit("c3")
|
|
540
|
+
r, _ = self.run_order(c)
|
|
541
|
+
self.assertEqual(r["callers"][0]["groups"][0]["order"], [c.labels["first"], c.labels["second"]])
|
|
542
|
+
|
|
543
|
+
def test_branch_alternatives_do_not_become_a_sequence(self):
|
|
544
|
+
c = Code().emit("85 c0").branch("74", "other").branch("e8", "helper").emit("c3")
|
|
545
|
+
c.label("other").branch("e8", "helper").emit("c3").label("helper").emit("c3")
|
|
546
|
+
r, cfg = self.run_order(c)
|
|
547
|
+
self.assertEqual(r["callers"][0]["groups"][0]["kind"], "branchAlternatives")
|
|
548
|
+
cfg["orderControls"] = [{"entry": 0, "kind": "sequence", "sites": r["callers"][0]["groups"][0]["sites"]}]
|
|
549
|
+
with self.assertRaisesRegex(ValueError, "positive control"):
|
|
550
|
+
run_report(c.bytes(), cfg, "call-order")
|
|
551
|
+
|
|
552
|
+
def test_outer_loop_keeps_sequence_within_a_guard_visit_and_reports_recurrence(self):
|
|
553
|
+
c = Code().label("guard").emit("83 f8 05").branch("7c", "end")
|
|
554
|
+
c.label("one").branch("e8", "helper").label("two").branch("e8", "helper").branch("eb", "guard")
|
|
555
|
+
c.label("end").emit("c3").label("helper").emit("c3")
|
|
556
|
+
r, _ = self.run_order(c)
|
|
557
|
+
g = r["callers"][0]["groups"][0]
|
|
558
|
+
self.assertEqual(g["kind"], "sequence")
|
|
559
|
+
self.assertEqual(g["order"], [c.labels["one"], c.labels["two"]])
|
|
560
|
+
self.assertTrue(g["mayRepeatAcrossGuardVisits"])
|
|
561
|
+
self.assertEqual(g["scope"], "one visit past the shared guard edges")
|
|
562
|
+
|
|
563
|
+
def test_shared_ownership_never_verifies_order(self):
|
|
564
|
+
c = Code().emit("90").branch("e8", "helper").emit("c3").label("helper").emit("c3")
|
|
565
|
+
cfg = configuration(c.bytes(), target=c.labels["helper"])
|
|
566
|
+
cfg["regions"][0]["entries"] = [0, 1, c.labels["helper"]]
|
|
567
|
+
r = run_report(c.bytes(), cfg, "call-order")
|
|
568
|
+
self.assertTrue(r["callers"])
|
|
569
|
+
self.assertTrue(all(not caller["orderingUsable"] for caller in r["callers"]))
|
|
570
|
+
self.assertTrue(all(g["kind"] == "unread" for caller in r["callers"] for g in caller["groups"]))
|
|
571
|
+
|
|
572
|
+
def test_overlapping_entries_are_rejected_by_the_flat_boundary_inventory(self):
|
|
573
|
+
c = Code().emit("66 90").branch("e8", "helper").emit("c3").label("helper").emit("c3")
|
|
574
|
+
cfg = configuration(c.bytes(), target=c.labels["helper"])
|
|
575
|
+
cfg["regions"][0]["entries"] = [0, 1, c.labels["helper"]]
|
|
576
|
+
r = run_report(c.bytes(), cfg, "call-order")
|
|
577
|
+
self.assertFalse(r["incoming"]["confirmed"])
|
|
578
|
+
self.assertGreater(sum(r["incoming"]["counts"].values()), 0)
|
|
579
|
+
self.assertFalse(r["callers"])
|
|
580
|
+
|
|
581
|
+
def test_caps_and_recurring_calls_leave_order_unread(self):
|
|
582
|
+
c = Code().branch("e8", "helper").branch("e8", "helper").emit("c3").label("helper").emit("c3")
|
|
583
|
+
for options in ({"entryLimit": 1}, {"limit": 1}, {"instructionLimit": 1}, {"analysisLimit": 1}):
|
|
584
|
+
r, _ = self.run_order(c, **options)
|
|
585
|
+
self.assertTrue(all(g["kind"] == "unread" for caller in r["callers"] for g in caller["groups"]))
|
|
586
|
+
loop = Code().label("again").branch("e8", "helper").branch("eb", "again").label("helper").emit("c3")
|
|
587
|
+
r, _ = self.run_order(loop)
|
|
588
|
+
self.assertTrue(all(g["kind"] == "unread" for caller in r["callers"] for g in caller["groups"]))
|
|
589
|
+
|
|
590
|
+
def test_capped_analysis_does_not_deny_recurrence(self):
|
|
591
|
+
c = Code().branch("e8", "helper").emit("90 c3").label("helper").emit("c3")
|
|
592
|
+
r, _ = self.run_order(c, analysisLimit=1)
|
|
593
|
+
self.assertTrue(r["callers"][0]["analysisCapped"])
|
|
594
|
+
self.assertIsNone(r["callers"][0]["groups"][0]["mayRepeatAcrossGuardVisits"])
|
|
595
|
+
|
|
596
|
+
def test_a_call_reached_around_another_is_not_sequenced_after_it(self):
|
|
597
|
+
# Two branches reach the first call, so neither edge alone is necessary, and the JMP skips it.
|
|
598
|
+
c = Code().emit("85 c0").branch("74", "first").branch("75", "first").branch("eb", "second")
|
|
599
|
+
c.label("first").branch("e8", "helper").label("second").branch("e8", "helper").emit("c3").label("helper").emit("c3")
|
|
600
|
+
r, cfg = self.run_order(c)
|
|
601
|
+
self.assertEqual(r["callers"][0]["groups"][0]["kind"], "unread")
|
|
602
|
+
cfg["orderControls"] = [{"entry": 0, "kind": "sequence", "sites": [c.labels["first"], c.labels["second"]]}]
|
|
603
|
+
with self.assertRaisesRegex(ValueError, "positive control"):
|
|
604
|
+
run_report(c.bytes(), cfg, "call-order")
|
|
605
|
+
|
|
606
|
+
def test_a_call_that_can_be_skipped_after_another_is_not_sequenced_after_it(self):
|
|
607
|
+
c = Code().label("first").branch("e8", "helper").emit("85 c0").branch("74", "second").branch("75", "second").emit("c3")
|
|
608
|
+
c.label("second").branch("e8", "helper").emit("c3").label("helper").emit("c3")
|
|
609
|
+
r, _ = self.run_order(c)
|
|
610
|
+
self.assertEqual(r["callers"][0]["groups"][0]["kind"], "unread")
|
|
611
|
+
|
|
612
|
+
def test_alternatives_have_the_full_group_shape_and_unordered_controls(self):
|
|
613
|
+
c = Code().emit("85 c0").branch("74", "other").label("one").branch("e8", "helper").emit("c3")
|
|
614
|
+
c.label("other").branch("e8", "helper").emit("c3").label("helper").emit("c3")
|
|
615
|
+
r, _ = self.run_order(c, orderControls=[{"entry": 0, "kind": "branchAlternatives", "sites": [c.labels["other"], c.labels["one"]]}])
|
|
616
|
+
g = r["callers"][0]["groups"][0]
|
|
617
|
+
self.assertEqual(g["kind"], "branchAlternatives")
|
|
618
|
+
self.assertEqual((g["scope"], g["mayRepeatAcrossGuardVisits"]), ("caller CFG", False))
|
|
619
|
+
|
|
620
|
+
def test_count_branches_and_entry_branches_take_no_adjacent_comparison(self):
|
|
621
|
+
c = Code().emit("83 f8 05").branch("e3", "end").branch("e8", "helper").label("end").emit("c3").label("helper").emit("c3")
|
|
622
|
+
r, _ = self.run_order(c)
|
|
623
|
+
self.assertIsNone(r["callers"][0]["calls"][0]["necessaryGuards"][0]["comparison"])
|
|
624
|
+
# The CMP before the entry is its only CFG predecessor, but the entry is also entered from outside.
|
|
625
|
+
c = Code().label("compare").emit("83 f8 05").label("entry").branch("7c", "end").branch("e8", "helper").branch("eb", "compare")
|
|
626
|
+
c.label("end").emit("c3").label("helper").emit("c3")
|
|
627
|
+
data = c.bytes()
|
|
628
|
+
cfg = configuration(data, target=c.labels["helper"])
|
|
629
|
+
cfg["regions"][0]["entries"] = [c.labels["entry"], c.labels["helper"]]
|
|
630
|
+
r = run_report(data, cfg, "call-order")
|
|
631
|
+
self.assertIsNone(r["callers"][0]["calls"][0]["necessaryGuards"][0]["comparison"])
|
|
632
|
+
|
|
633
|
+
def test_a_negative_stack_adjustment_is_not_cleanup(self):
|
|
634
|
+
c = Code().branch("e8", "helper").emit("81 c4 00 80 c3").label("helper").emit("c3")
|
|
635
|
+
r, _ = self.run_order(c)
|
|
636
|
+
self.assertEqual(r["callers"][0]["calls"][0]["cleanup"]["status"], "unread cleanup")
|
|
637
|
+
self.assertIsNone(r["callers"][0]["calls"][0]["cleanup"]["argumentBytes"])
|
|
638
|
+
|
|
639
|
+
|
|
352
640
|
class ReporterTests(unittest.TestCase):
|
|
353
641
|
def test_register_parts_preserve_neighbor(self):
|
|
354
642
|
r = report("b8 34 12 b0 00 c3")
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|