scientific-method-engine 0.6.0__tar.gz → 0.8.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (61) hide show
  1. {scientific_method_engine-0.6.0 → scientific_method_engine-0.8.0}/PKG-INFO +1 -1
  2. {scientific_method_engine-0.6.0 → scientific_method_engine-0.8.0}/pyproject.toml +1 -1
  3. {scientific_method_engine-0.6.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/cli.py +8 -1
  4. {scientific_method_engine-0.6.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/x86/machine.py +13 -14
  5. {scientific_method_engine-0.6.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/x86/pcode.py +5 -2
  6. {scientific_method_engine-0.6.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/x86/pcode_backend.py +150 -60
  7. {scientific_method_engine-0.6.0 → scientific_method_engine-0.8.0}/tests/oracle.py +6 -6
  8. {scientific_method_engine-0.6.0 → scientific_method_engine-0.8.0}/tests/test_dispatch.py +1 -1
  9. {scientific_method_engine-0.6.0 → scientific_method_engine-0.8.0}/tests/test_effect_order.py +1 -3
  10. {scientific_method_engine-0.6.0 → scientific_method_engine-0.8.0}/tests/test_oracle.py +93 -13
  11. {scientific_method_engine-0.6.0 → scientific_method_engine-0.8.0}/tests/test_pe.py +2 -3
  12. {scientific_method_engine-0.6.0 → scientific_method_engine-0.8.0}/tests/test_x86.py +45 -28
  13. scientific_method_engine-0.6.0/src/scientific_method_engine/x86/handwritten.py +0 -391
  14. scientific_method_engine-0.6.0/src/scientific_method_engine/x86/semantics.py +0 -77
  15. scientific_method_engine-0.6.0/tests/differential.py +0 -120
  16. scientific_method_engine-0.6.0/tests/test_differential.py +0 -71
  17. {scientific_method_engine-0.6.0 → scientific_method_engine-0.8.0}/.gitignore +0 -0
  18. {scientific_method_engine-0.6.0 → scientific_method_engine-0.8.0}/LICENSE +0 -0
  19. {scientific_method_engine-0.6.0 → scientific_method_engine-0.8.0}/README.md +0 -0
  20. {scientific_method_engine-0.6.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/__init__.py +0 -0
  21. {scientific_method_engine-0.6.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/__main__.py +0 -0
  22. {scientific_method_engine-0.6.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/ghidra/ClearNoReturnFunctions.java +0 -0
  23. {scientific_method_engine-0.6.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/ghidra/CreateFunctions.java +0 -0
  24. {scientific_method_engine-0.6.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/ghidra/ExportBoundedFlow.java +0 -0
  25. {scientific_method_engine-0.6.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/ghidra/ExportFunctionFingerprints.java +0 -0
  26. {scientific_method_engine-0.6.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/ghidra/ExportFunctionInventory.java +0 -0
  27. {scientific_method_engine-0.6.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/ghidra/MergeFallThroughFragment.java +0 -0
  28. {scientific_method_engine-0.6.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/ghidra/RecoverCitedFunctions.java +0 -0
  29. {scientific_method_engine-0.6.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/ghidra/RepairReturningCallers.java +0 -0
  30. {scientific_method_engine-0.6.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/ghidra/ReportCallArguments.java +0 -0
  31. {scientific_method_engine-0.6.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/ghidra/ReportCallPaths.java +0 -0
  32. {scientific_method_engine-0.6.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/ghidra/ReportCallSitesWithScalars.java +0 -0
  33. {scientific_method_engine-0.6.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/ghidra/ReportCallsToRange.java +0 -0
  34. {scientific_method_engine-0.6.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/ghidra/ReportConstantFirstArgumentCalls.java +0 -0
  35. {scientific_method_engine-0.6.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/ghidra/ReportDataBytes.java +0 -0
  36. {scientific_method_engine-0.6.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/ghidra/ReportDecompileMatches.java +0 -0
  37. {scientific_method_engine-0.6.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/ghidra/ReportDecompileWindow.java +0 -0
  38. {scientific_method_engine-0.6.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/ghidra/ReportFilePatternInMemory.java +0 -0
  39. {scientific_method_engine-0.6.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/ghidra/ReportFirstArgumentCallSummary.java +0 -0
  40. {scientific_method_engine-0.6.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/ghidra/ReportFunctionScalarConstants.java +0 -0
  41. {scientific_method_engine-0.6.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/ghidra/ReportFunctionSummary.java +0 -0
  42. {scientific_method_engine-0.6.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/ghidra/ReportInstructionContext.java +0 -0
  43. {scientific_method_engine-0.6.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/ghidra/ReportInstructionWindow.java +0 -0
  44. {scientific_method_engine-0.6.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/ghidra/ReportMemoryBlockForFileOffset.java +0 -0
  45. {scientific_method_engine-0.6.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/ghidra/ReportMemoryBlocks.java +0 -0
  46. {scientific_method_engine-0.6.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/ghidra/ReportRandomnessCandidates.java +0 -0
  47. {scientific_method_engine-0.6.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/ghidra/ReportReferences.java +0 -0
  48. {scientific_method_engine-0.6.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/ghidra/ReportScalarConstants.java +0 -0
  49. {scientific_method_engine-0.6.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/ghidra/ReportStringReferences.java +0 -0
  50. {scientific_method_engine-0.6.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/ghidra/ReportSymbolReferences.java +0 -0
  51. {scientific_method_engine-0.6.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/x86/__init__.py +0 -0
  52. {scientific_method_engine-0.6.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/x86/dispatch.py +0 -0
  53. {scientific_method_engine-0.6.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/x86/effect_order.py +0 -0
  54. {scientific_method_engine-0.6.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/x86/image.py +0 -0
  55. {scientific_method_engine-0.6.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/x86/pe.py +0 -0
  56. {scientific_method_engine-0.6.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/x86/reports.py +0 -0
  57. {scientific_method_engine-0.6.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/x86/result_flow.py +0 -0
  58. {scientific_method_engine-0.6.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/x86/trace.py +0 -0
  59. {scientific_method_engine-0.6.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/x86/values.py +0 -0
  60. {scientific_method_engine-0.6.0 → scientific_method_engine-0.8.0}/tests/test_nested_frame_request.py +0 -0
  61. {scientific_method_engine-0.6.0 → scientific_method_engine-0.8.0}/tests/test_table_continuations.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: scientific-method-engine
3
- Version: 0.6.0
3
+ Version: 0.8.0
4
4
  Summary: Bounded instruction-derived x86 evidence reports for segmented MZ/FBOV and PE32/i386 code.
5
5
  Project-URL: Source, https://github.com/kibertoad/refurbished-dinosaurs-toolkit/tree/main/packages/scientific-method-engine
6
6
  Author: kibertoad
@@ -5,7 +5,7 @@ build-backend = "hatchling.build"
5
5
  [project]
6
6
  name = "scientific-method-engine"
7
7
  # The release workflow writes the published version from the package's release tag.
8
- version = "0.6.0"
8
+ version = "0.8.0"
9
9
  description = "Bounded instruction-derived x86 evidence reports for segmented MZ/FBOV and PE32/i386 code."
10
10
  readme = "README.md"
11
11
  requires-python = ">=3.12"
@@ -3,10 +3,16 @@ import json
3
3
  import sys
4
4
  from pathlib import Path
5
5
 
6
+ import capstone
7
+ import pypcode
8
+
6
9
  from . import PREPARED_PROTOCOL
7
10
 
8
11
  CONFIG_LIMIT = 1024 * 1024
9
12
  PREPARED_CONFIG_LIMIT = 16 * 1024 * 1024
13
+ # The header names the decoder and instruction semantics that actually ran, not the pins in pyproject.toml.
14
+ DECODER = "capstone " + capstone.__version__
15
+ INSTRUCTION_SEMANTICS = f"pypcode {pypcode.__version__} (Ghidra SLEIGH x86)"
10
16
  USAGE = ("Usage: scientific-method-engine <operand|operand-candidates|target|bounds|owner|callees|trace|uses|arguments|"
11
17
  "effects|returns|memory|incoming|call-order|guards|allocation|dispatch> <config.json|->\n"
12
18
  "effects includes ordered path writes/calls and local restoration witnesses; transactionality remains unestablished.\n"
@@ -60,7 +66,8 @@ def main(argv):
60
66
  raise ValueError("preparedProtocol is set by the reader and cannot be supplied")
61
67
  data, identity = read_source(config, base)
62
68
  result = run_report(data, config, command)
63
- print(json.dumps({"schema": "bounded-x86-v1", "decoder": "capstone 5.0.7", "sourceIdentity": identity,
69
+ print(json.dumps({"schema": "bounded-x86-v1", "decoder": DECODER,
70
+ "instructionSemantics": INSTRUCTION_SEMANTICS, "sourceIdentity": identity,
64
71
  "status": "Conditional static report; never promotes an evidence entry", **result}, indent=2))
65
72
 
66
73
 
@@ -1,8 +1,7 @@
1
- """Path state for the evidence layer. Instruction semantics come from the backend a State holds (semantics.py)."""
1
+ """Path state for the evidence layer. Instruction semantics come from pypcode (pcode_backend.py)."""
2
2
  from copy import deepcopy
3
3
  from capstone.x86 import X86_OP_REG, X86_OP_IMM, X86_OP_MEM
4
4
  from .values import Value, const, unknown, op, extract, join, resize, sources, address_parts, producers
5
- from . import semantics
6
5
 
7
6
  REGISTERS = ("eax", "ebx", "ecx", "edx", "esi", "edi", "ebp", "esp", "cs", "ds", "es", "ss", "fs", "gs")
8
7
  ALIASES = {}
@@ -39,7 +38,10 @@ def alias(name):
39
38
  class State:
40
39
  def __init__(self, entry, image, config):
41
40
  self.bits, self.flat, self.mask = image.bits, image.flat, image.mask
42
- self.semantics = semantics.current()
41
+ # Imported here: the backend imports this module, so a module-level import would make the
42
+ # import order matter.
43
+ from .pcode_backend import BACKEND
44
+ self.semantics = BACKEND
43
45
  self.sp, self.bp = ("esp", "ebp") if self.flat else ("sp", "bp")
44
46
  self.at = entry
45
47
  self.regs = {r: unknown("initial:" + r, ALIASES[r][2]) for r in REGISTERS}
@@ -64,7 +66,8 @@ class State:
64
66
  self.guards = []
65
67
  self.assumptions = {}
66
68
  self.flags = None
67
- # Arithmetic flags as values when the last flag-writing instruction ran on p-code; None otherwise.
69
+ # Arithmetic flags as values from the last flag-writing instruction's p-code; None when no
70
+ # instruction computed them yet or the last one forgot them.
68
71
  self.flag_values = None
69
72
  # CF when an instruction sets it without leaving a comparable flag producer; None defers to flags.
70
73
  self.carry = None
@@ -123,9 +126,10 @@ class State:
123
126
  def carry_value(self):
124
127
  """CF as a one-bit value: from the last comparable flag producer, an explicit carry, or unknown."""
125
128
  if self.flags is not None:
126
- answer, _ = self.semantics.condition(self, "jb")
127
- if answer is not None:
128
- return const(int(answer), 1, self.flags[3])
129
+ if self.flag_values is not None and "CF" in self.flag_values:
130
+ answer, _ = self.semantics.condition(self, "jb")
131
+ if answer is not None:
132
+ return const(int(answer), 1, self.flags[3])
129
133
  # Name the carry by its producer's operands, so every reading of one comparison shares an assumption.
130
134
  a, b, operation, site = self.flags
131
135
  return unknown(f"carry:{site}:{(operation, a.term, b.term)!r}", 1, site)
@@ -367,11 +371,10 @@ def string_effect(state, ins, count, remaining, charge=None):
367
371
  # any segment override.
368
372
  source_name = (segment_register(ins, ins.operands[1].mem) if operation in ("movs", "lods") else
369
373
  segment_register(ins, ins.operands[0].mem) if operation == "cmps" else None)
370
- delta = -width if state.direction_flag.number else width
371
374
  counter = "ecx" if state.flat else "cx"
372
375
  if not compare or not repeated(ins):
373
376
  for _ in range(count.number):
374
- state.semantics.string_iteration(state, ins, operation, width, source_name, delta)
377
+ state.semantics.string_iteration(state, ins, operation, width, source_name)
375
378
  if repeated(ins):
376
379
  state.setreg(counter, const(0, state.bits, state.at), state.at)
377
380
  return count.number
@@ -381,7 +384,7 @@ def string_effect(state, ins, count, remaining, charge=None):
381
384
  raise StopPath("String iteration budget exhausted; remaining effects unresolved")
382
385
  if charge is not None:
383
386
  charge(1)
384
- holds = state.semantics.string_iteration(state, ins, operation, width, source_name, delta)
387
+ holds = state.semantics.string_iteration(state, ins, operation, width, source_name)
385
388
  iterations += 1
386
389
  outcomes.append(holds)
387
390
  if iterations == count.number:
@@ -398,7 +401,3 @@ def string_effect(state, ins, count, remaining, charge=None):
398
401
  state.event("string-compare-exit", iterations=iterations, exit=reason, counter=state.reg(counter).report())
399
402
  return iterations
400
403
 
401
-
402
- # The handwritten backend registers itself as the default; it imports names defined above.
403
- from . import handwritten # noqa: E402,F401
404
- from . import pcode_backend # noqa: E402,F401
@@ -300,8 +300,11 @@ def evaluate(code, args, bits, site):
300
300
  # Identities that keep flag expressions small; constants fold in values.op.
301
301
  mask = (1 << bits) - 1
302
302
  for x, y in ((a, b), (b, a)):
303
- if x.number == 0:
304
- return Value(bits, ("constant", 0), origin) if name == "and" else Value(bits, y.term, origin)
303
+ if x.number == 0 and name == "and":
304
+ return Value(bits, ("constant", 0), origin)
305
+ if x.number == 0 and boolean(y):
306
+ # Flag selections OR a zero arm in; a value operand keeps its OR, as reports write it.
307
+ return Value(bits, y.term, origin)
305
308
  if name == "and" and (x.number == mask or (x.number == 1 and boolean(y))):
306
309
  return Value(bits, y.term, origin)
307
310
  if name == "or" and (x.number == mask):
@@ -1,31 +1,30 @@
1
- """The pypcode semantics backend (ADR 0003, phase 3).
2
-
3
- Values come from the p-code pypcode lifts. The evidence layer keeps what it always decided: which
4
- segment register an access uses (from Capstone, decision 3), access roles, the model's acceptance
5
- rules, flag-producer records and the events each instruction reports. Groups of mnemonics move to
6
- this backend one at a time; the others still run on the handwritten backend.
1
+ """Instruction semantics from pypcode (ADR 0003).
2
+
3
+ Values, flags and branch conditions come from the p-code pypcode lifts from Ghidra's SLEIGH
4
+ specification. The evidence layer keeps what it always decided: which segment register an access
5
+ uses (from Capstone, decision 3), access roles, the model's acceptance rules, flag-producer records
6
+ and the events each instruction reports. A ``State`` reaches this module through ``BACKEND``.
7
+
8
+ What this module and ``pcode.Run`` use of a ``State``:
9
+
10
+ - Methods: ``get``, ``put``, ``reg``, ``setreg``, ``segment``, ``access``, ``address``, ``event``,
11
+ ``set_flags``, ``forget_flags``, ``carry_value``, ``save_flags`` and ``restore_flags``.
12
+ - Read only: ``at``, ``bits``, ``flat``, ``sp``, ``flags``, ``flag_serial``, ``flag_epoch``,
13
+ ``unknown_flag_site``, ``segment_bases`` and ``value_transfers``.
14
+ - Read and assigned: ``flag_values`` (the arithmetic flags p-code computed, or None), ``carry``,
15
+ ``direction_flag`` and ``interrupt_flag``. ``conditional`` is appended to (the divide-error
16
+ assumption).
17
+
18
+ Every member above belongs to one path, and ``trace`` copies a path with ``deepcopy`` when a branch
19
+ splits it. The backend keeps no path state of its own: ``Pypcode.__deepcopy__`` returns the same
20
+ object, so a value this module stores must live on the ``State`` to be copied with its path.
7
21
  """
8
22
  from capstone.x86 import X86_OP_REG, X86_OP_IMM, X86_OP_MEM
9
23
 
10
- from . import handwritten, semantics
11
24
  from .machine import ALIASES, StopPath
12
25
  from .pcode import LIFTER, Address, Run, FLAGS, SEGMENT_BASES, segment_base
13
26
  from .values import Value, const, unknown, op, extract, join, resize, sources
14
27
 
15
- GROUPS = {
16
- "data movement": ("mov", "movzx", "movsx", "xchg", "nop"),
17
- "address forms": ("lea", "lds", "les"),
18
- "stack": ("push", "pop", "leave", "pushf", "pushfd", "popf", "popfd"),
19
- "compare": ("cmp", "test"),
20
- "arithmetic and logic": ("add", "sub", "and", "or", "xor", "inc", "dec", "not", "neg"),
21
- "carry chain": ("adc", "sbb", "clc", "stc", "cmc"),
22
- "shifts and rotates": ("shl", "sal", "shr", "sar", "rol", "ror", "rcl", "rcr"),
23
- "multiply and divide": ("mul", "imul", "div", "idiv"),
24
- "conversions": ("cbw", "cwde", "cwd", "cdq"),
25
- "flags and direction": ("cld", "std", "cli", "sti"),
26
- "string operations": ("movs", "stos", "lods", "cmps", "scas"),
27
- }
28
-
29
28
  # Conditional branches by the condition code SLEIGH decodes from 0x70 + code.
30
29
  CONDITION_CODES = {}
31
30
  for code, names in enumerate((("jo",), ("jno",), ("jb", "jc", "jnae"), ("jae", "jnb", "jnc"), ("je", "jz"),
@@ -77,7 +76,7 @@ def bit(value, site):
77
76
  # SLESS(x, 0) is the sign bit of x; a SLEIGH shift writes it as SLESS(x << k, 0).
78
77
  if term[0] == "sless" and term[2] == ("constant", 0):
79
78
  inner, shift = term[1], 0
80
- if inner[0] == "shl" and inner[2][0] == "constant":
79
+ if inner[0] == "shl" and inner[2][0] == "constant" and inner not in LEAVES:
81
80
  inner, shift = inner[1], inner[2][1]
82
81
  width = width_of(inner, value)
83
82
  if width is not None and shift < width:
@@ -90,7 +89,7 @@ def bit(value, site):
90
89
  inner, mask = term[1][1], term[1][2]
91
90
  if mask[0] == "constant" and mask[1] and mask[1] & (mask[1] - 1) == 0:
92
91
  position = mask[1].bit_length() - 1
93
- if inner[0] in ("shr", "sar") and inner[2][0] == "constant":
92
+ if inner[0] in ("shr", "sar") and inner[2][0] == "constant" and inner not in LEAVES:
94
93
  position += inner[2][1]
95
94
  inner = inner[1]
96
95
  width = width_of(inner, value)
@@ -121,6 +120,9 @@ def width_of(term, value):
121
120
 
122
121
  # Widths of the terms the current instruction read or computed; set per frame.
123
122
  WIDTHS = {}
123
+ # The current instruction's operand values. Field decomposition stops at them, because reports
124
+ # treat each operand as one value even when an earlier instruction built it from fields.
125
+ LEAVES = set()
124
126
 
125
127
 
126
128
  def fields(term, width):
@@ -130,6 +132,8 @@ def fields(term, width):
130
132
  (right when negative). Rotates and shifts of joined values take this form in p-code.
131
133
  """
132
134
  head = term[0]
135
+ if term in LEAVES:
136
+ return [(term, width, 0)]
133
137
  top = top_bit(term)
134
138
  if top is not None:
135
139
  return [top]
@@ -252,6 +256,7 @@ class Frame:
252
256
 
253
257
  def __init__(self, state, ins, image, widths=None, roles=None, present=None):
254
258
  self.state, self.ins, self.image = state, ins, image
259
+ # present(value, frame) may rewrite a written value into the equal term reports use.
255
260
  self.present = present
256
261
  self.widths = widths or {}
257
262
  self.roles = roles or {}
@@ -275,7 +280,7 @@ class Frame:
275
280
  self.addresses = {}
276
281
  if ins.mnemonic != "pop":
277
282
  # Operands address with the registers the instruction started with. POP's destination
278
- # is addressed after the stack pointer moves, as the handwritten backend does.
283
+ # is addressed after the stack pointer moves, as the CPU addresses it.
279
284
  for index, operand in enumerate(ins.operands):
280
285
  if operand.type == X86_OP_MEM:
281
286
  self.address(index)
@@ -325,6 +330,7 @@ class Frame:
325
330
  addressing_register=register)
326
331
  self.loaded[index] = self.cache[index]
327
332
  WIDTHS.setdefault(self.cache[index].term, self.cache[index].bits)
333
+ LEAVES.add(self.cache[index].term)
328
334
  return extract(self.cache[index], d * 8, size * 8)
329
335
  location = evidence if d == 0 else op("add", evidence, const(d, evidence.bits), state.at)
330
336
  state.access(segment, location, size, value, addressing_register=register)
@@ -345,9 +351,12 @@ class Frame:
345
351
 
346
352
  def execute(self):
347
353
  WIDTHS.clear()
354
+ LEAVES.clear()
348
355
  for index, value in self.before.items():
349
356
  WIDTHS[value.term] = value.bits
350
- self.run = Run(self.state, self.ops, self.memory, self.constant, present=self.present)
357
+ LEAVES.add(value.term)
358
+ present = (lambda value: self.present(value, self)) if self.present else None
359
+ self.run = Run(self.state, self.ops, self.memory, self.constant, present=present)
351
360
  self.run.execute()
352
361
  for value in [*self.loaded.values(), *self.run.temps.values(), *self.run.registers.values()]:
353
362
  if isinstance(value, Value): # Real-mode temporaries may hold a segmented Address.
@@ -412,41 +421,59 @@ def carry_out(state, run, site):
412
421
 
413
422
 
414
423
  class Pypcode:
415
- """Instruction semantics from pypcode for the moved groups; the handwritten backend for the rest."""
416
-
417
- name = "pypcode"
418
-
419
- def __init__(self, groups):
420
- self.groups = tuple(groups)
421
- self.mnemonics = {m for group in self.groups for m in GROUPS[group]}
424
+ """The engine's instruction semantics: ordinary instructions, branch conditions and string bodies."""
422
425
 
423
426
  def __deepcopy__(self, memo):
427
+ # The backend holds no path state, so every copied path shares it.
424
428
  return self
425
429
 
426
430
  def ordinary(self, state, ins, image):
427
- m = ins.mnemonic
428
- if m not in self.mnemonics:
429
- return handwritten.ordinary(state, ins, image)
430
- HANDLERS[m](state, ins, image)
431
+ """Apply one instruction that is neither a control transfer nor a string operation."""
432
+ handler = HANDLERS.get(ins.mnemonic)
433
+ if handler is None:
434
+ raise StopPath("Unsupported instruction semantics: " + ins.mnemonic)
435
+ handler(state, ins, image)
431
436
 
432
437
  def condition(self, state, mnemonic):
433
- answer, info = handwritten.predicate(state, mnemonic)
434
- if "compare" not in self.groups or mnemonic not in CONDITION_CODES:
435
- return answer, info
436
- flags = state.flag_values
437
- if flags is None:
438
- return answer, info
438
+ """Evaluate a conditional branch on the current flags.
439
+
440
+ Returns ``(answer, info)``: True, False or None when unresolved, and the ``branch`` event
441
+ fields. They describe the evidence layer's record of the flag producer, then either
442
+ ``decidedBy: "p-code flags"`` for a decided branch or a ``reason`` for an undecided one.
443
+ """
439
444
  ops, _ = LIFTER.ops(state.flat, bytes((0x70 + CONDITION_CODES[mnemonic], 0)), 0x100)
440
445
  needed = {LIFTER.register(state.flat, v[1], v[2]) for o in ops for v in o.inputs if v[0] == "register"}
441
- if not needed <= set(flags):
442
- # A handwritten instruction produced some of these flags; its predicate decides.
443
- return answer, info
444
- condition = Run(state, ops, None, flags=flags).execute(stop_at_branch=True)
445
- return (None if condition.number is None else bool(condition.number)), info
446
-
447
- def string_iteration(self, state, ins, operation, width, source_segment, delta_step):
448
- if operation not in self.mnemonics:
449
- return handwritten.string_iteration(state, ins, operation, width, source_segment, delta_step)
446
+ carry_only = needed == {"CF"} and state.flags is None and state.carry is not None
447
+ if carry_only:
448
+ info = {"predicate": mnemonic, "flag": "CF", "carry": state.carry.report()}
449
+ elif state.flags is None:
450
+ info = {"predicate": mnemonic, "reason": "flag producer unresolved",
451
+ "flagProducer": state.unknown_flag_site, "flagGeneration": state.flag_epoch}
452
+ else:
453
+ a, b, operation, site = state.flags
454
+ info = {"predicate": mnemonic, "flagProducer": site, "operation": operation,
455
+ "left": a.report(), "right": b.report()}
456
+ # CF is always readable: the evidence layer names it when no instruction resolved it.
457
+ if not needed <= set(state.flag_values or ()) | {"CF"}:
458
+ info.setdefault("reason", "flags unresolved")
459
+ return None, info
460
+ condition = Run(state, ops, None, flags=state.flag_values or {}).execute(stop_at_branch=True)
461
+ if condition.number is None:
462
+ # Every undecided branch says why: the carry, the producer or its flags are unknown.
463
+ info.setdefault("reason", "carry unresolved" if carry_only else "flags unresolved")
464
+ return None, info
465
+ # p-code decides every branch the engine resolves; the record above only describes the producer.
466
+ info.pop("reason", None)
467
+ info["decidedBy"] = "p-code flags"
468
+ return bool(condition.number), info
469
+
470
+ def string_iteration(self, state, ins, operation, width, source_segment):
471
+ """Apply one iteration of an accepted string form; see ``machine.string_effect``.
472
+
473
+ ``source_segment`` names the source operand's segment register (MOVS, LODS and CMPS only).
474
+ p-code steps SI and DI by the direction flag. For a repeated CMPS or SCAS, returns whether
475
+ the repeat condition holds afterwards (1, 0 or unknown); otherwise None.
476
+ """
450
477
  si, di = ("esi", "edi") if state.flat else ("si", "di")
451
478
  source, destination = state.reg(si), state.reg(di)
452
479
  loaded = {}
@@ -605,9 +632,25 @@ def compare(state, ins, image):
605
632
  state.event("compare", operation=m, left=a.report(), right=b.report())
606
633
 
607
634
 
635
+ def unfolded(name, value, f, site):
636
+ """A logic result p-code folded to one operand, written as the operation reports use.
637
+
638
+ ``x | 0``, ``x ^ 0`` and ``x & ~0`` equal ``x``; the interpreter folds them, and reports
639
+ keep the instruction's operation in the result's expression.
640
+ """
641
+ if name not in ("and", "or", "xor") or value.number is not None:
642
+ return value
643
+ a = f.value(0)
644
+ b = resize(f.value(1), a.bits)
645
+ neutral = (1 << a.bits) - 1 if name == "and" else 0
646
+ if value.bits == a.bits and (b.number == neutral and value.term == a.term or a.number == neutral and value.term == b.term):
647
+ return Value(value.bits, op(name, a, b, site).term, value.sources)
648
+ return value
649
+
650
+
608
651
  def arithmetic(state, ins, image):
609
652
  m = ins.mnemonic
610
- f = run_plain(state, ins, image)
653
+ f = run_plain(state, ins, image, present=lambda v, frame: unfolded(m, v, frame, state.at))
611
654
  a = f.value(0)
612
655
  b = resize(f.value(1), a.bits)
613
656
  result = f.result(0)
@@ -705,7 +748,7 @@ def rotate(state, ins, image):
705
748
  state.forget_flags(keep_carry=True)
706
749
  return
707
750
  # Reports write a rotate as its shifted copies, from the leftmost copy for ROL and RCL.
708
- f = run_plain(state, ins, image, present=lambda v: arranged(v, m in ("rol", "rcl"), state.at))
751
+ f = run_plain(state, ins, image, present=lambda v, _: arranged(v, m in ("rol", "rcl"), state.at))
709
752
  a = f.value(0)
710
753
  value = f.result(0)
711
754
  carry = bit(f.run.flags["CF"], state.at)
@@ -733,10 +776,33 @@ def temporary(f, code):
733
776
  return f.run.temps[o.output[1]]
734
777
 
735
778
 
779
+ def low_product(value, site):
780
+ """The low half of a product of two extended values, as the product at the operands' width.
781
+
782
+ SLEIGH writes a two- or three-operand IMUL as the double-width product of the sign-extended
783
+ operands, truncated; the low half does not depend on the extension, and reports write it as
784
+ the operand-width product. An operand that already holds a sign extension (after CBW or MOVSX)
785
+ reaches the product as one extension of the narrower value, which is extended back to the
786
+ operand's width here.
787
+ """
788
+ term = value.term
789
+ if term[0] != "extract" or term[2] != 0 or term[1][0] != "mul":
790
+ return value
791
+ factors = []
792
+ for factor in term[1][1:]:
793
+ if factor[0] in ("signExtend", "zeroExtend") and factor[2] <= value.bits:
794
+ factors.append(resize(Value(factor[2], factor[1]), value.bits, signed=factor[0] == "signExtend"))
795
+ elif factor[0] == "constant":
796
+ factors.append(const(factor[1], value.bits))
797
+ else:
798
+ return value
799
+ return Value(value.bits, op("mul", *factors, site).term, value.sources)
800
+
801
+
736
802
  def multiply(state, ins, image):
737
803
  m = ins.mnemonic
738
804
  if m == "imul" and len(ins.operands) in (2, 3):
739
- f = run_plain(state, ins, image)
805
+ f = run_plain(state, ins, image, present=lambda v, _: low_product(v, state.at))
740
806
  left = resize(f.value(0 if len(ins.operands) == 2 else 1), ins.operands[0].size * 8)
741
807
  right = resize(f.value(1 if len(ins.operands) == 2 else 2), left.bits, signed=True)
742
808
  result = f.result(0)
@@ -786,10 +852,37 @@ def divide(state, ins, image):
786
852
  quotient=quotient.report(), remainder=remainder.report(), fault=fault)
787
853
 
788
854
 
855
+ def through_extension(value, source, form):
856
+ """``value`` as ``form(source)`` when p-code wrote ``form`` of the value ``source`` sign-extends.
857
+
858
+ When the source register holds a sign extension (after CBW, CWDE or MOVSX), p-code reads
859
+ through it to the narrower value it extended, because ``pcode.evaluate`` folds an extension
860
+ of an extension. Both name the same bits; reports name them in the register the instruction
861
+ reads.
862
+ """
863
+ if value.number is not None or source.term[0] != "signExtend":
864
+ return value
865
+ inner = Value(source.term[2], source.term[1])
866
+ if value.term == form(inner).term:
867
+ return Value(value.bits, form(source).term, value.sources)
868
+ return value
869
+
870
+
789
871
  def conversion(state, ins, image):
790
872
  m = ins.mnemonic
791
873
  before = {name: state.reg(name) for name in ("al", "ax", "eax")}
792
- f = run_plain(state, ins, image)
874
+
875
+ def present(value, _):
876
+ if value.bits not in (16, 32):
877
+ return value
878
+ if m in ("cwd", "cdq"):
879
+ # The high half is the sign bit of AX or EAX.
880
+ return through_extension(value, before["ax" if value.bits == 16 else "eax"],
881
+ lambda v: resize(extract(v, v.bits - 1, 1), value.bits, signed=True))
882
+ # CBW and CWDE write the sign extension of AL or AX.
883
+ return through_extension(value, before["al" if value.bits == 16 else "ax"],
884
+ lambda v: resize(v, value.bits, signed=True))
885
+ f = run_plain(state, ins, image, present=present)
793
886
  (destination, value), = f.run.registers.items()
794
887
  wide = value.bits == 32
795
888
  if m in ("cbw", "cwde"):
@@ -828,8 +921,5 @@ for names, handler in ((("mov", "movzx", "movsx", "xchg"), move), (("nop",), nop
828
921
  for name in names:
829
922
  HANDLERS[name] = handler
830
923
 
831
- MOVED = ("data movement", "address forms", "stack", "compare", "arithmetic and logic", "carry chain",
832
- "shifts and rotates", "multiply and divide", "conversions", "flags and direction",
833
- "string operations")
834
-
835
- semantics.register(Pypcode(MOVED), default=bool(MOVED))
924
+ # The backend every State uses.
925
+ BACKEND = Pypcode()
@@ -4,12 +4,14 @@ Unicorn is a test dependency only. ``check`` runs one synthetic routine on the e
4
4
  Unicorn with the same segment layout and concrete registers, and compares every register the
5
5
  engine resolved to a constant at the routine's return with Unicorn's value there.
6
6
  """
7
- import contextlib
7
+ import sys
8
+ from pathlib import Path
8
9
 
9
10
  from unicorn import UC_ARCH_X86, UC_MODE_16, Uc
10
11
  from unicorn import x86_const as U
11
12
 
12
- from differential import accepted, run_report
13
+ sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "src"))
14
+ from scientific_method_engine.x86.reports import run_report # noqa: E402
13
15
 
14
16
  SEGMENT = 0x1000
15
17
  STACK = {"ss": 0x2000, "sp": 0xFFF0}
@@ -47,18 +49,16 @@ def unicorn(data, registers, direction=None):
47
49
  return stopped.get("at"), {name: uc.reg_read(getattr(U, "UC_X86_REG_" + name.upper())) for name in names}
48
50
 
49
51
 
50
- def check(test, code, registers=None, direction=None, resolved=(), extended=None):
52
+ def check(test, code, registers=None, direction=None, resolved=()):
51
53
  """Compare the engine's resolved registers with Unicorn's for one synthetic routine.
52
54
 
53
55
  Routines write any memory they read, because the engine starts with memory unknown and
54
56
  Unicorn with zeros. ``resolved`` names registers the engine must resolve.
55
- ``extended`` states why the default backend may resolve more than the handwritten one here.
56
57
  """
57
58
  data = bytes.fromhex(code)
58
59
  registers = {**STACK, **(registers or {})}
59
60
  flags = {} if direction is None else {"flags": {"direction": direction}}
60
- with accepted("extended", extended) if extended else contextlib.nullcontext():
61
- result = run_report(data, configuration(data, registers, **flags), "trace")
61
+ result = run_report(data, configuration(data, registers, **flags), "trace")
62
62
  test.assertEqual(len(result["paths"]), 1, "the oracle compares one resolved path")
63
63
  path = result["paths"][0]
64
64
  test.assertTrue(path["returned"], path["stop"])
@@ -5,7 +5,7 @@ import sys
5
5
  import unittest
6
6
  SRC = Path(__file__).resolve().parents[1] / "src"
7
7
  sys.path.insert(0, str(SRC))
8
- from differential import run_report
8
+ from scientific_method_engine.x86.reports import run_report
9
9
  from scientific_method_engine.x86.image import Image
10
10
  from scientific_method_engine.x86.trace import walk, OVERLAP_REASON
11
11
 
@@ -1,6 +1,5 @@
1
1
  """Synthetic path evidence; no original bytes or claims."""
2
2
  import unittest
3
- from differential import accepted
4
3
  from test_x86 import Code, report
5
4
 
6
5
 
@@ -73,8 +72,7 @@ class EffectOrderTests(unittest.TestCase):
73
72
  # A loop re-pushes the same return address into a stack slot another call overwrote.
74
73
  c = Code().emit("b9 02 00").label("top").branch("e8", "a").branch("e8", "b").emit("49").branch("75", "top").emit("c3")
75
74
  c.label("a").emit("c3").label("b").emit("c3")
76
- with accepted("extended", "DEC flags from p-code resolve the loop exit; test_oracle checks the count"):
77
- r = report(c, "effects", registers={"ss": 0x3000, "sp": 0xff00})
75
+ r = report(c, "effects", registers={"ss": 0x3000, "sp": 0xff00})
78
76
  self.assertTrue(any(p["returned"] for p in self.paths(r)))
79
77
  self.assertFalse([w for p in self.paths(r) for w in p["localRestorationWitnesses"]])
80
78