scientific-method-engine 0.7.0__tar.gz → 0.8.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (61) hide show
  1. {scientific_method_engine-0.7.0 → scientific_method_engine-0.8.0}/PKG-INFO +1 -1
  2. {scientific_method_engine-0.7.0 → scientific_method_engine-0.8.0}/pyproject.toml +1 -1
  3. {scientific_method_engine-0.7.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/cli.py +8 -1
  4. {scientific_method_engine-0.7.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/x86/machine.py +13 -14
  5. {scientific_method_engine-0.7.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/x86/pcode_backend.py +63 -55
  6. {scientific_method_engine-0.7.0 → scientific_method_engine-0.8.0}/tests/oracle.py +6 -6
  7. {scientific_method_engine-0.7.0 → scientific_method_engine-0.8.0}/tests/test_dispatch.py +1 -1
  8. {scientific_method_engine-0.7.0 → scientific_method_engine-0.8.0}/tests/test_effect_order.py +1 -3
  9. {scientific_method_engine-0.7.0 → scientific_method_engine-0.8.0}/tests/test_oracle.py +29 -16
  10. {scientific_method_engine-0.7.0 → scientific_method_engine-0.8.0}/tests/test_pe.py +2 -3
  11. {scientific_method_engine-0.7.0 → scientific_method_engine-0.8.0}/tests/test_x86.py +45 -28
  12. scientific_method_engine-0.7.0/src/scientific_method_engine/x86/handwritten.py +0 -391
  13. scientific_method_engine-0.7.0/src/scientific_method_engine/x86/semantics.py +0 -77
  14. scientific_method_engine-0.7.0/tests/differential.py +0 -120
  15. scientific_method_engine-0.7.0/tests/test_differential.py +0 -71
  16. {scientific_method_engine-0.7.0 → scientific_method_engine-0.8.0}/.gitignore +0 -0
  17. {scientific_method_engine-0.7.0 → scientific_method_engine-0.8.0}/LICENSE +0 -0
  18. {scientific_method_engine-0.7.0 → scientific_method_engine-0.8.0}/README.md +0 -0
  19. {scientific_method_engine-0.7.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/__init__.py +0 -0
  20. {scientific_method_engine-0.7.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/__main__.py +0 -0
  21. {scientific_method_engine-0.7.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/ghidra/ClearNoReturnFunctions.java +0 -0
  22. {scientific_method_engine-0.7.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/ghidra/CreateFunctions.java +0 -0
  23. {scientific_method_engine-0.7.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/ghidra/ExportBoundedFlow.java +0 -0
  24. {scientific_method_engine-0.7.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/ghidra/ExportFunctionFingerprints.java +0 -0
  25. {scientific_method_engine-0.7.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/ghidra/ExportFunctionInventory.java +0 -0
  26. {scientific_method_engine-0.7.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/ghidra/MergeFallThroughFragment.java +0 -0
  27. {scientific_method_engine-0.7.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/ghidra/RecoverCitedFunctions.java +0 -0
  28. {scientific_method_engine-0.7.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/ghidra/RepairReturningCallers.java +0 -0
  29. {scientific_method_engine-0.7.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/ghidra/ReportCallArguments.java +0 -0
  30. {scientific_method_engine-0.7.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/ghidra/ReportCallPaths.java +0 -0
  31. {scientific_method_engine-0.7.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/ghidra/ReportCallSitesWithScalars.java +0 -0
  32. {scientific_method_engine-0.7.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/ghidra/ReportCallsToRange.java +0 -0
  33. {scientific_method_engine-0.7.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/ghidra/ReportConstantFirstArgumentCalls.java +0 -0
  34. {scientific_method_engine-0.7.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/ghidra/ReportDataBytes.java +0 -0
  35. {scientific_method_engine-0.7.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/ghidra/ReportDecompileMatches.java +0 -0
  36. {scientific_method_engine-0.7.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/ghidra/ReportDecompileWindow.java +0 -0
  37. {scientific_method_engine-0.7.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/ghidra/ReportFilePatternInMemory.java +0 -0
  38. {scientific_method_engine-0.7.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/ghidra/ReportFirstArgumentCallSummary.java +0 -0
  39. {scientific_method_engine-0.7.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/ghidra/ReportFunctionScalarConstants.java +0 -0
  40. {scientific_method_engine-0.7.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/ghidra/ReportFunctionSummary.java +0 -0
  41. {scientific_method_engine-0.7.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/ghidra/ReportInstructionContext.java +0 -0
  42. {scientific_method_engine-0.7.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/ghidra/ReportInstructionWindow.java +0 -0
  43. {scientific_method_engine-0.7.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/ghidra/ReportMemoryBlockForFileOffset.java +0 -0
  44. {scientific_method_engine-0.7.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/ghidra/ReportMemoryBlocks.java +0 -0
  45. {scientific_method_engine-0.7.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/ghidra/ReportRandomnessCandidates.java +0 -0
  46. {scientific_method_engine-0.7.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/ghidra/ReportReferences.java +0 -0
  47. {scientific_method_engine-0.7.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/ghidra/ReportScalarConstants.java +0 -0
  48. {scientific_method_engine-0.7.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/ghidra/ReportStringReferences.java +0 -0
  49. {scientific_method_engine-0.7.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/ghidra/ReportSymbolReferences.java +0 -0
  50. {scientific_method_engine-0.7.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/x86/__init__.py +0 -0
  51. {scientific_method_engine-0.7.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/x86/dispatch.py +0 -0
  52. {scientific_method_engine-0.7.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/x86/effect_order.py +0 -0
  53. {scientific_method_engine-0.7.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/x86/image.py +0 -0
  54. {scientific_method_engine-0.7.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/x86/pcode.py +0 -0
  55. {scientific_method_engine-0.7.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/x86/pe.py +0 -0
  56. {scientific_method_engine-0.7.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/x86/reports.py +0 -0
  57. {scientific_method_engine-0.7.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/x86/result_flow.py +0 -0
  58. {scientific_method_engine-0.7.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/x86/trace.py +0 -0
  59. {scientific_method_engine-0.7.0 → scientific_method_engine-0.8.0}/src/scientific_method_engine/x86/values.py +0 -0
  60. {scientific_method_engine-0.7.0 → scientific_method_engine-0.8.0}/tests/test_nested_frame_request.py +0 -0
  61. {scientific_method_engine-0.7.0 → scientific_method_engine-0.8.0}/tests/test_table_continuations.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: scientific-method-engine
3
- Version: 0.7.0
3
+ Version: 0.8.0
4
4
  Summary: Bounded instruction-derived x86 evidence reports for segmented MZ/FBOV and PE32/i386 code.
5
5
  Project-URL: Source, https://github.com/kibertoad/refurbished-dinosaurs-toolkit/tree/main/packages/scientific-method-engine
6
6
  Author: kibertoad
@@ -5,7 +5,7 @@ build-backend = "hatchling.build"
5
5
  [project]
6
6
  name = "scientific-method-engine"
7
7
  # The release workflow writes the published version from the package's release tag.
8
- version = "0.7.0"
8
+ version = "0.8.0"
9
9
  description = "Bounded instruction-derived x86 evidence reports for segmented MZ/FBOV and PE32/i386 code."
10
10
  readme = "README.md"
11
11
  requires-python = ">=3.12"
@@ -3,10 +3,16 @@ import json
3
3
  import sys
4
4
  from pathlib import Path
5
5
 
6
+ import capstone
7
+ import pypcode
8
+
6
9
  from . import PREPARED_PROTOCOL
7
10
 
8
11
  CONFIG_LIMIT = 1024 * 1024
9
12
  PREPARED_CONFIG_LIMIT = 16 * 1024 * 1024
13
+ # The header names the decoder and instruction semantics that actually ran, not the pins in pyproject.toml.
14
+ DECODER = "capstone " + capstone.__version__
15
+ INSTRUCTION_SEMANTICS = f"pypcode {pypcode.__version__} (Ghidra SLEIGH x86)"
10
16
  USAGE = ("Usage: scientific-method-engine <operand|operand-candidates|target|bounds|owner|callees|trace|uses|arguments|"
11
17
  "effects|returns|memory|incoming|call-order|guards|allocation|dispatch> <config.json|->\n"
12
18
  "effects includes ordered path writes/calls and local restoration witnesses; transactionality remains unestablished.\n"
@@ -60,7 +66,8 @@ def main(argv):
60
66
  raise ValueError("preparedProtocol is set by the reader and cannot be supplied")
61
67
  data, identity = read_source(config, base)
62
68
  result = run_report(data, config, command)
63
- print(json.dumps({"schema": "bounded-x86-v1", "decoder": "capstone 5.0.7", "sourceIdentity": identity,
69
+ print(json.dumps({"schema": "bounded-x86-v1", "decoder": DECODER,
70
+ "instructionSemantics": INSTRUCTION_SEMANTICS, "sourceIdentity": identity,
64
71
  "status": "Conditional static report; never promotes an evidence entry", **result}, indent=2))
65
72
 
66
73
 
@@ -1,8 +1,7 @@
1
- """Path state for the evidence layer. Instruction semantics come from the backend a State holds (semantics.py)."""
1
+ """Path state for the evidence layer. Instruction semantics come from pypcode (pcode_backend.py)."""
2
2
  from copy import deepcopy
3
3
  from capstone.x86 import X86_OP_REG, X86_OP_IMM, X86_OP_MEM
4
4
  from .values import Value, const, unknown, op, extract, join, resize, sources, address_parts, producers
5
- from . import semantics
6
5
 
7
6
  REGISTERS = ("eax", "ebx", "ecx", "edx", "esi", "edi", "ebp", "esp", "cs", "ds", "es", "ss", "fs", "gs")
8
7
  ALIASES = {}
@@ -39,7 +38,10 @@ def alias(name):
39
38
  class State:
40
39
  def __init__(self, entry, image, config):
41
40
  self.bits, self.flat, self.mask = image.bits, image.flat, image.mask
42
- self.semantics = semantics.current()
41
+ # Imported here: the backend imports this module, so a module-level import would make the
42
+ # import order matter.
43
+ from .pcode_backend import BACKEND
44
+ self.semantics = BACKEND
43
45
  self.sp, self.bp = ("esp", "ebp") if self.flat else ("sp", "bp")
44
46
  self.at = entry
45
47
  self.regs = {r: unknown("initial:" + r, ALIASES[r][2]) for r in REGISTERS}
@@ -64,7 +66,8 @@ class State:
64
66
  self.guards = []
65
67
  self.assumptions = {}
66
68
  self.flags = None
67
- # Arithmetic flags as values when the last flag-writing instruction ran on p-code; None otherwise.
69
+ # Arithmetic flags as values from the last flag-writing instruction's p-code; None when no
70
+ # instruction computed them yet or the last one forgot them.
68
71
  self.flag_values = None
69
72
  # CF when an instruction sets it without leaving a comparable flag producer; None defers to flags.
70
73
  self.carry = None
@@ -123,9 +126,10 @@ class State:
123
126
  def carry_value(self):
124
127
  """CF as a one-bit value: from the last comparable flag producer, an explicit carry, or unknown."""
125
128
  if self.flags is not None:
126
- answer, _ = self.semantics.condition(self, "jb")
127
- if answer is not None:
128
- return const(int(answer), 1, self.flags[3])
129
+ if self.flag_values is not None and "CF" in self.flag_values:
130
+ answer, _ = self.semantics.condition(self, "jb")
131
+ if answer is not None:
132
+ return const(int(answer), 1, self.flags[3])
129
133
  # Name the carry by its producer's operands, so every reading of one comparison shares an assumption.
130
134
  a, b, operation, site = self.flags
131
135
  return unknown(f"carry:{site}:{(operation, a.term, b.term)!r}", 1, site)
@@ -367,11 +371,10 @@ def string_effect(state, ins, count, remaining, charge=None):
367
371
  # any segment override.
368
372
  source_name = (segment_register(ins, ins.operands[1].mem) if operation in ("movs", "lods") else
369
373
  segment_register(ins, ins.operands[0].mem) if operation == "cmps" else None)
370
- delta = -width if state.direction_flag.number else width
371
374
  counter = "ecx" if state.flat else "cx"
372
375
  if not compare or not repeated(ins):
373
376
  for _ in range(count.number):
374
- state.semantics.string_iteration(state, ins, operation, width, source_name, delta)
377
+ state.semantics.string_iteration(state, ins, operation, width, source_name)
375
378
  if repeated(ins):
376
379
  state.setreg(counter, const(0, state.bits, state.at), state.at)
377
380
  return count.number
@@ -381,7 +384,7 @@ def string_effect(state, ins, count, remaining, charge=None):
381
384
  raise StopPath("String iteration budget exhausted; remaining effects unresolved")
382
385
  if charge is not None:
383
386
  charge(1)
384
- holds = state.semantics.string_iteration(state, ins, operation, width, source_name, delta)
387
+ holds = state.semantics.string_iteration(state, ins, operation, width, source_name)
385
388
  iterations += 1
386
389
  outcomes.append(holds)
387
390
  if iterations == count.number:
@@ -398,7 +401,3 @@ def string_effect(state, ins, count, remaining, charge=None):
398
401
  state.event("string-compare-exit", iterations=iterations, exit=reason, counter=state.reg(counter).report())
399
402
  return iterations
400
403
 
401
-
402
- # The handwritten backend registers itself as the default; it imports names defined above.
403
- from . import handwritten # noqa: E402,F401
404
- from . import pcode_backend # noqa: E402,F401
@@ -1,31 +1,30 @@
1
- """The pypcode semantics backend (ADR 0003, phase 3).
2
-
3
- Values come from the p-code pypcode lifts. The evidence layer keeps what it always decided: which
4
- segment register an access uses (from Capstone, decision 3), access roles, the model's acceptance
5
- rules, flag-producer records and the events each instruction reports. Groups of mnemonics move to
6
- this backend one at a time; the others still run on the handwritten backend.
1
+ """Instruction semantics from pypcode (ADR 0003).
2
+
3
+ Values, flags and branch conditions come from the p-code pypcode lifts from Ghidra's SLEIGH
4
+ specification. The evidence layer keeps what it always decided: which segment register an access
5
+ uses (from Capstone, decision 3), access roles, the model's acceptance rules, flag-producer records
6
+ and the events each instruction reports. A ``State`` reaches this module through ``BACKEND``.
7
+
8
+ What this module and ``pcode.Run`` use of a ``State``:
9
+
10
+ - Methods: ``get``, ``put``, ``reg``, ``setreg``, ``segment``, ``access``, ``address``, ``event``,
11
+ ``set_flags``, ``forget_flags``, ``carry_value``, ``save_flags`` and ``restore_flags``.
12
+ - Read only: ``at``, ``bits``, ``flat``, ``sp``, ``flags``, ``flag_serial``, ``flag_epoch``,
13
+ ``unknown_flag_site``, ``segment_bases`` and ``value_transfers``.
14
+ - Read and assigned: ``flag_values`` (the arithmetic flags p-code computed, or None), ``carry``,
15
+ ``direction_flag`` and ``interrupt_flag``. ``conditional`` is appended to (the divide-error
16
+ assumption).
17
+
18
+ Every member above belongs to one path, and ``trace`` copies a path with ``deepcopy`` when a branch
19
+ splits it. The backend keeps no path state of its own: ``Pypcode.__deepcopy__`` returns the same
20
+ object, so a value this module stores must live on the ``State`` to be copied with its path.
7
21
  """
8
22
  from capstone.x86 import X86_OP_REG, X86_OP_IMM, X86_OP_MEM
9
23
 
10
- from . import handwritten, semantics
11
24
  from .machine import ALIASES, StopPath
12
25
  from .pcode import LIFTER, Address, Run, FLAGS, SEGMENT_BASES, segment_base
13
26
  from .values import Value, const, unknown, op, extract, join, resize, sources
14
27
 
15
- GROUPS = {
16
- "data movement": ("mov", "movzx", "movsx", "xchg", "nop"),
17
- "address forms": ("lea", "lds", "les"),
18
- "stack": ("push", "pop", "leave", "pushf", "pushfd", "popf", "popfd"),
19
- "compare": ("cmp", "test"),
20
- "arithmetic and logic": ("add", "sub", "and", "or", "xor", "inc", "dec", "not", "neg"),
21
- "carry chain": ("adc", "sbb", "clc", "stc", "cmc"),
22
- "shifts and rotates": ("shl", "sal", "shr", "sar", "rol", "ror", "rcl", "rcr"),
23
- "multiply and divide": ("mul", "imul", "div", "idiv"),
24
- "conversions": ("cbw", "cwde", "cwd", "cdq"),
25
- "flags and direction": ("cld", "std", "cli", "sti"),
26
- "string operations": ("movs", "stos", "lods", "cmps", "scas"),
27
- }
28
-
29
28
  # Conditional branches by the condition code SLEIGH decodes from 0x70 + code.
30
29
  CONDITION_CODES = {}
31
30
  for code, names in enumerate((("jo",), ("jno",), ("jb", "jc", "jnae"), ("jae", "jnb", "jnc"), ("je", "jz"),
@@ -281,7 +280,7 @@ class Frame:
281
280
  self.addresses = {}
282
281
  if ins.mnemonic != "pop":
283
282
  # Operands address with the registers the instruction started with. POP's destination
284
- # is addressed after the stack pointer moves, as the handwritten backend does.
283
+ # is addressed after the stack pointer moves, as the CPU addresses it.
285
284
  for index, operand in enumerate(ins.operands):
286
285
  if operand.type == X86_OP_MEM:
287
286
  self.address(index)
@@ -422,47 +421,59 @@ def carry_out(state, run, site):
422
421
 
423
422
 
424
423
  class Pypcode:
425
- """Instruction semantics from pypcode for the moved groups; the handwritten backend for the rest."""
426
-
427
- name = "pypcode"
428
-
429
- def __init__(self, groups):
430
- self.groups = tuple(groups)
431
- self.mnemonics = {m for group in self.groups for m in GROUPS[group]}
424
+ """The engine's instruction semantics: ordinary instructions, branch conditions and string bodies."""
432
425
 
433
426
  def __deepcopy__(self, memo):
427
+ # The backend holds no path state, so every copied path shares it.
434
428
  return self
435
429
 
436
430
  def ordinary(self, state, ins, image):
437
- m = ins.mnemonic
438
- if m not in self.mnemonics:
439
- return handwritten.ordinary(state, ins, image)
440
- HANDLERS[m](state, ins, image)
431
+ """Apply one instruction that is neither a control transfer nor a string operation."""
432
+ handler = HANDLERS.get(ins.mnemonic)
433
+ if handler is None:
434
+ raise StopPath("Unsupported instruction semantics: " + ins.mnemonic)
435
+ handler(state, ins, image)
441
436
 
442
437
  def condition(self, state, mnemonic):
443
- answer, info = handwritten.predicate(state, mnemonic)
444
- if "compare" not in self.groups or mnemonic not in CONDITION_CODES:
445
- return answer, info
446
- flags = state.flag_values
447
- if flags is None:
448
- return answer, info
438
+ """Evaluate a conditional branch on the current flags.
439
+
440
+ Returns ``(answer, info)``: True, False or None when unresolved, and the ``branch`` event
441
+ fields. They describe the evidence layer's record of the flag producer, then either
442
+ ``decidedBy: "p-code flags"`` for a decided branch or a ``reason`` for an undecided one.
443
+ """
449
444
  ops, _ = LIFTER.ops(state.flat, bytes((0x70 + CONDITION_CODES[mnemonic], 0)), 0x100)
450
445
  needed = {LIFTER.register(state.flat, v[1], v[2]) for o in ops for v in o.inputs if v[0] == "register"}
451
- if not needed <= set(flags):
452
- # A handwritten instruction produced some of these flags; its predicate decides.
453
- return answer, info
454
- condition = Run(state, ops, None, flags=flags).execute(stop_at_branch=True)
446
+ carry_only = needed == {"CF"} and state.flags is None and state.carry is not None
447
+ if carry_only:
448
+ info = {"predicate": mnemonic, "flag": "CF", "carry": state.carry.report()}
449
+ elif state.flags is None:
450
+ info = {"predicate": mnemonic, "reason": "flag producer unresolved",
451
+ "flagProducer": state.unknown_flag_site, "flagGeneration": state.flag_epoch}
452
+ else:
453
+ a, b, operation, site = state.flags
454
+ info = {"predicate": mnemonic, "flagProducer": site, "operation": operation,
455
+ "left": a.report(), "right": b.report()}
456
+ # CF is always readable: the evidence layer names it when no instruction resolved it.
457
+ if not needed <= set(state.flag_values or ()) | {"CF"}:
458
+ info.setdefault("reason", "flags unresolved")
459
+ return None, info
460
+ condition = Run(state, ops, None, flags=state.flag_values or {}).execute(stop_at_branch=True)
455
461
  if condition.number is None:
462
+ # Every undecided branch says why: the carry, the producer or its flags are unknown.
463
+ info.setdefault("reason", "carry unresolved" if carry_only else "flags unresolved")
456
464
  return None, info
457
- if answer is None:
458
- # The handwritten record says why it could not decide; p-code flags decided it.
459
- info = {key: value for key, value in info.items() if key != "reason"}
460
- info["decidedBy"] = "p-code flags"
465
+ # p-code decides every branch the engine resolves; the record above only describes the producer.
466
+ info.pop("reason", None)
467
+ info["decidedBy"] = "p-code flags"
461
468
  return bool(condition.number), info
462
469
 
463
- def string_iteration(self, state, ins, operation, width, source_segment, delta_step):
464
- if operation not in self.mnemonics:
465
- return handwritten.string_iteration(state, ins, operation, width, source_segment, delta_step)
470
+ def string_iteration(self, state, ins, operation, width, source_segment):
471
+ """Apply one iteration of an accepted string form; see ``machine.string_effect``.
472
+
473
+ ``source_segment`` names the source operand's segment register (MOVS, LODS and CMPS only).
474
+ p-code steps SI and DI by the direction flag. For a repeated CMPS or SCAS, returns whether
475
+ the repeat condition holds afterwards (1, 0 or unknown); otherwise None.
476
+ """
466
477
  si, di = ("esi", "edi") if state.flat else ("si", "di")
467
478
  source, destination = state.reg(si), state.reg(di)
468
479
  loaded = {}
@@ -910,8 +921,5 @@ for names, handler in ((("mov", "movzx", "movsx", "xchg"), move), (("nop",), nop
910
921
  for name in names:
911
922
  HANDLERS[name] = handler
912
923
 
913
- MOVED = ("data movement", "address forms", "stack", "compare", "arithmetic and logic", "carry chain",
914
- "shifts and rotates", "multiply and divide", "conversions", "flags and direction",
915
- "string operations")
916
-
917
- semantics.register(Pypcode(MOVED), default=bool(MOVED))
924
+ # The backend every State uses.
925
+ BACKEND = Pypcode()
@@ -4,12 +4,14 @@ Unicorn is a test dependency only. ``check`` runs one synthetic routine on the e
4
4
  Unicorn with the same segment layout and concrete registers, and compares every register the
5
5
  engine resolved to a constant at the routine's return with Unicorn's value there.
6
6
  """
7
- import contextlib
7
+ import sys
8
+ from pathlib import Path
8
9
 
9
10
  from unicorn import UC_ARCH_X86, UC_MODE_16, Uc
10
11
  from unicorn import x86_const as U
11
12
 
12
- from differential import accepted, run_report
13
+ sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "src"))
14
+ from scientific_method_engine.x86.reports import run_report # noqa: E402
13
15
 
14
16
  SEGMENT = 0x1000
15
17
  STACK = {"ss": 0x2000, "sp": 0xFFF0}
@@ -47,18 +49,16 @@ def unicorn(data, registers, direction=None):
47
49
  return stopped.get("at"), {name: uc.reg_read(getattr(U, "UC_X86_REG_" + name.upper())) for name in names}
48
50
 
49
51
 
50
- def check(test, code, registers=None, direction=None, resolved=(), extended=None):
52
+ def check(test, code, registers=None, direction=None, resolved=()):
51
53
  """Compare the engine's resolved registers with Unicorn's for one synthetic routine.
52
54
 
53
55
  Routines write any memory they read, because the engine starts with memory unknown and
54
56
  Unicorn with zeros. ``resolved`` names registers the engine must resolve.
55
- ``extended`` states why the default backend may resolve more than the handwritten one here.
56
57
  """
57
58
  data = bytes.fromhex(code)
58
59
  registers = {**STACK, **(registers or {})}
59
60
  flags = {} if direction is None else {"flags": {"direction": direction}}
60
- with accepted("extended", extended) if extended else contextlib.nullcontext():
61
- result = run_report(data, configuration(data, registers, **flags), "trace")
61
+ result = run_report(data, configuration(data, registers, **flags), "trace")
62
62
  test.assertEqual(len(result["paths"]), 1, "the oracle compares one resolved path")
63
63
  path = result["paths"][0]
64
64
  test.assertTrue(path["returned"], path["stop"])
@@ -5,7 +5,7 @@ import sys
5
5
  import unittest
6
6
  SRC = Path(__file__).resolve().parents[1] / "src"
7
7
  sys.path.insert(0, str(SRC))
8
- from differential import run_report
8
+ from scientific_method_engine.x86.reports import run_report
9
9
  from scientific_method_engine.x86.image import Image
10
10
  from scientific_method_engine.x86.trace import walk, OVERLAP_REASON
11
11
 
@@ -1,6 +1,5 @@
1
1
  """Synthetic path evidence; no original bytes or claims."""
2
2
  import unittest
3
- from differential import accepted
4
3
  from test_x86 import Code, report
5
4
 
6
5
 
@@ -73,8 +72,7 @@ class EffectOrderTests(unittest.TestCase):
73
72
  # A loop re-pushes the same return address into a stack slot another call overwrote.
74
73
  c = Code().emit("b9 02 00").label("top").branch("e8", "a").branch("e8", "b").emit("49").branch("75", "top").emit("c3")
75
74
  c.label("a").emit("c3").label("b").emit("c3")
76
- with accepted("extended", "DEC flags from p-code resolve the loop exit; test_oracle checks the count"):
77
- r = report(c, "effects", registers={"ss": 0x3000, "sp": 0xff00})
75
+ r = report(c, "effects", registers={"ss": 0x3000, "sp": 0xff00})
78
76
  self.assertTrue(any(p["returned"] for p in self.paths(r)))
79
77
  self.assertFalse([w for p in self.paths(r) for w in p["localRestorationWitnesses"]])
80
78
 
@@ -5,7 +5,6 @@ from oracle import STACK, check, configuration, run_report
5
5
 
6
6
  DATA = {"ds": 0x3000, "es": 0x4000}
7
7
  SAME = {"ds": 0x4000, "es": 0x4000}
8
- STRING_COMPARE = "the handwritten backend stops on CMPS and SCAS"
9
8
 
10
9
 
11
10
  class DataMovement(unittest.TestCase):
@@ -40,9 +39,9 @@ class Stack(unittest.TestCase):
40
39
 
41
40
 
42
41
  class Compare(unittest.TestCase):
43
- def branch(self, setup, jcc, extended=None):
42
+ def branch(self, setup, jcc):
44
43
  # cx = 2 when the branch is taken, 1 when not.
45
- check(self, setup + jcc + "04 b90100 c3 b90200 c3", resolved=("cx",), extended=extended)
44
+ check(self, setup + jcc + "04 b90100 c3 b90200 c3", resolved=("cx",))
46
45
 
47
46
  def test_signed_and_unsigned_conditions(self):
48
47
  for setup in ("b80300 bb0500 39d8", "b8fdff bb0500 39d8", "b80500 bb0500 39d8", "b80080 bbff7f 39d8"):
@@ -54,14 +53,30 @@ class Compare(unittest.TestCase):
54
53
  for setup in ("b80f00 a80f", "b80300 3d0300", "b80100 3d0300"):
55
54
  for jcc in ("74", "75", "7a", "7b", "72", "70"):
56
55
  with self.subTest(setup=setup, jcc=jcc):
57
- # The handwritten predicate never resolves PF; p-code computes it.
58
- self.branch(setup, jcc, extended="parity flag from p-code" if jcc in ("7a", "7b") else None)
56
+ self.branch(setup, jcc)
59
57
 
60
58
  def test_same_register_compare_resolves_with_unknown_input(self):
61
59
  for jcc in ("74", "72", "7c", "7f", "70"):
62
60
  with self.subTest(jcc=jcc):
63
61
  check(self, "39c0" + jcc + "04 b90100 c3 b90200 c3", registers={"ax": 0x1234}, resolved=("cx",))
64
62
 
63
+ def test_branches_after_a_comparison_name_how_they_were_decided(self):
64
+ # cmp ax, 3 with AX = 3: p-code decides JE and JP, and the event keeps the comparison record.
65
+ for jcc in ("74", "7a"):
66
+ with self.subTest(jcc=jcc):
67
+ result = check(self, "b80300 3d0300" + jcc + "04 b90100 c3 b90200 c3", resolved=("cx",))
68
+ branch, = [e for e in result["paths"][0]["events"] if e["kind"] == "branch"]
69
+ self.assertEqual((branch["operation"], branch["decidedBy"]), ("cmp", "p-code flags"))
70
+ self.assertNotIn("reason", branch)
71
+ # With AX unknown the comparison decides nothing: both arms run and each says why.
72
+ data = bytes.fromhex("3d0300 7401 90 c3".replace(" ", ""))
73
+ result = run_report(data, configuration(data, dict(STACK)), "trace")
74
+ branches = [e for path in result["paths"] for e in path["events"] if e["kind"] == "branch"]
75
+ self.assertEqual(sorted(e["taken"] for e in branches), [False, True])
76
+ for e in branches:
77
+ self.assertEqual((e["operation"], e["reason"]), ("cmp", "flags unresolved"))
78
+ self.assertNotIn("decidedBy", e)
79
+
65
80
 
66
81
  class ArithmeticAndLogic(unittest.TestCase):
67
82
  def test_values(self):
@@ -69,18 +84,17 @@ class ArithmeticAndLogic(unittest.TestCase):
69
84
  resolved=("ax", "bx", "cx", "dx"))
70
85
 
71
86
  def test_counted_loop_exits_on_decrement_flags(self):
72
- # inc ax; dec cx; jnz back: the handwritten backend forgets DEC's flags and splits.
73
- result = check(self, "b90300 b80000 40 49 75fc c3", resolved=("ax", "cx"),
74
- extended="DEC flags from p-code resolve the loop exit")
87
+ # inc ax; dec cx; jnz back: DEC's flags decide each exit, so the loop runs as one path.
88
+ result = check(self, "b90300 b80000 40 49 75fc c3", resolved=("ax", "cx"))
75
89
  branches = [e for e in result["paths"][0]["events"] if e["kind"] == "branch"]
76
90
  self.assertEqual([e["taken"] for e in branches], [True, True, False])
77
- # A branch p-code decided does not keep the handwritten backend's reason for not deciding it.
91
+ # No comparison record exists, so the branch says p-code flags decided it.
78
92
  for e in branches:
79
93
  self.assertNotIn("reason", e)
80
94
  self.assertEqual((e["decidedBy"], e["flagProducer"]), ("p-code flags", 7))
81
95
 
82
96
  def test_undecided_branch_keeps_the_reason(self):
83
- # inc ax; jnz: AX is unknown, so neither backend decides the branch and both arms run.
97
+ # inc ax; jnz: AX is unknown, so the engine does not decide the branch and both arms run.
84
98
  data = bytes.fromhex("40 7501 90 c3".replace(" ", ""))
85
99
  result = run_report(data, configuration(data, dict(STACK)), "trace")
86
100
  branches = [e for path in result["paths"] for e in path["events"] if e["kind"] == "branch"]
@@ -129,8 +143,7 @@ class ShiftsAndRotates(unittest.TestCase):
129
143
  def test_double_word_shift_from_a_zero_high_half(self):
130
144
  # DX starts at zero; RCL carries AX's unknown top bits into it. The second RCL shifts out
131
145
  # DX's bit 15, which is still zero; ADC moves that carry into BX for Unicorn to check.
132
- result = check(self, "b90100 31d2 d1e0 d1d2 d1e0 d1d2 bb0000 11db c3", resolved=("cx", "bx"),
133
- extended="a rotate's carry out folded from the operand's known bits")
146
+ result = check(self, "b90100 31d2 d1e0 d1d2 d1e0 d1d2 bb0000 11db c3", resolved=("cx", "bx"))
134
147
  self.assertEqual(result["paths"][0]["registers"]["dx"]["expression"][0], "or")
135
148
 
136
149
  def test_memory_operands(self):
@@ -194,16 +207,16 @@ class StringOperations(unittest.TestCase):
194
207
 
195
208
  def test_repne_scas_finds_a_terminator(self):
196
209
  check(self, "fc bf1000 c7056162 c6450200 b000 b9ffff f2ae c3", registers=SAME,
197
- resolved=("cx", "di"), extended=STRING_COMPARE)
210
+ resolved=("cx", "di"))
198
211
 
199
212
  def test_repe_cmps_stops_at_the_first_difference(self):
200
213
  # SI's byte is below DI's at the difference, so JB takes the branch: BX = 2.
201
214
  check(self, "fc be1000 bf2000 c7046162 c6440263 c7056162 c6450264 b90500 f3a6 7204 bb0100 c3 bb0200 c3",
202
- registers=SAME, resolved=("cx", "si", "di", "bx"), extended=STRING_COMPARE)
215
+ registers=SAME, resolved=("cx", "si", "di", "bx"))
203
216
 
204
217
  def test_repe_cmps_runs_out_of_count_and_single_forms(self):
205
218
  check(self, "fc be1000 bf2000 c7046162 c7056162 b90200 f3a6 be1000 bf2000 a7 b86162 bf2000 af 7504 bb0100 c3 bb0200 c3",
206
- registers=SAME, resolved=("cx", "si", "di", "bx"), extended=STRING_COMPARE)
219
+ registers=SAME, resolved=("cx", "si", "di", "bx"))
207
220
 
208
221
  def test_cmps_with_equal_source_and_destination_offsets(self):
209
222
  # SI = DI = 0x10. In different segments 'a' < 'b' takes JB (BX = 2); in one segment the second
@@ -211,7 +224,7 @@ class StringOperations(unittest.TestCase):
211
224
  code = "fc be1000 bf1000 c60461 26c60562 a6 7204 bb0100 c3 bb0200 c3"
212
225
  for registers in (DATA, SAME):
213
226
  with self.subTest(registers=registers):
214
- check(self, code, registers=registers, resolved=("si", "di", "bx"), extended=STRING_COMPARE)
227
+ check(self, code, registers=registers, resolved=("si", "di", "bx"))
215
228
 
216
229
  def test_backward_steps(self):
217
230
  check(self, "fd bf0a00 b0aa b90400 f3aa be0700 ac c3", registers={**DATA, "ds": 0x4000},
@@ -18,7 +18,7 @@ ENGINE_ENV = {**os.environ, "PYTHONPATH": os.pathsep.join(filter(None, [str(SRC)
18
18
  READER = SRC.parents[1] / "executable-reader" / "bin" / "scientific-method.ts"
19
19
  from scientific_method_engine.x86.image import Image
20
20
  from scientific_method_engine.x86.pe import pe32
21
- from differential import accepted, run_report
21
+ from scientific_method_engine.x86.reports import run_report
22
22
 
23
23
  BASE, CODE_VA, DATA_VA = 0x400000, 0x401000, 0x402000
24
24
  CODE_RAW, DATA_RAW = 0x200, 0x400
@@ -80,8 +80,7 @@ def events(result, kind):
80
80
  class PEReporterTests(unittest.TestCase):
81
81
  def test_cmps_with_one_address_for_both_operands(self):
82
82
  # ESI = EDI: both CMPSB loads match both operands, and each operand takes one of them.
83
- with accepted('extended', 'the handwritten backend stops on CMPS'):
84
- r = report('fc be 00 20 40 00 bf 00 20 40 00 c6 06 61 a6 74 01 c3 c3')
83
+ r = report('fc be 00 20 40 00 bf 00 20 40 00 c6 06 61 a6 74 01 c3 c3')
85
84
  path, = r['paths']
86
85
  self.assertTrue(path['returned'], path['stop'])
87
86
  self.assertEqual((path['registers']['esi']['value'], path['registers']['edi']['value']), (DATA_VA + 1,) * 2)
@@ -13,9 +13,10 @@ sys.path.insert(0, str(SRC))
13
13
  # The engine CLI runs from this checkout's source whether or not the package is installed.
14
14
  ENGINE = [sys.executable, "-B", "-m", "scientific_method_engine"]
15
15
  ENGINE_ENV = {**os.environ, "PYTHONPATH": os.pathsep.join(filter(None, [str(SRC), os.environ.get("PYTHONPATH")]))}
16
+ import capstone
17
+ import pypcode
16
18
  from scientific_method_engine.x86.image import Image
17
- from scientific_method_engine.x86.reports import run_report as engine_report
18
- from differential import accepted, run_report
19
+ from scientific_method_engine.x86.reports import run_report
19
20
  from scientific_method_engine.x86.trace import trace, walk, OVERLAP_REASON, CONTESTED_REASON
20
21
  from scientific_method_engine.x86.values import const, unknown, op, extract, resize
21
22
 
@@ -1404,6 +1405,14 @@ class ReporterTests(unittest.TestCase):
1404
1405
  self.assertFalse(calls[0]["guards"][0]["sameTargetValue"])
1405
1406
  self.assertFalse(result["completeWithinModel"])
1406
1407
 
1408
+ def test_every_engine_module_imports_first(self):
1409
+ modules = sorted(p.stem for p in (SRC / "scientific_method_engine" / "x86").glob("*.py") if p.stem != "__init__")
1410
+ for module in modules:
1411
+ with self.subTest(module=module):
1412
+ code = f"import scientific_method_engine.x86.{module}"
1413
+ result = subprocess.run([sys.executable, "-B", "-c", code], capture_output=True, text=True, env=ENGINE_ENV)
1414
+ self.assertEqual(result.returncode, 0, result.stderr)
1415
+
1407
1416
  def test_cli_identity_and_errors(self):
1408
1417
  with tempfile.TemporaryDirectory() as folder:
1409
1418
  root = Path(folder); data = bytes.fromhex("b8 01 00 c3")
@@ -1413,7 +1422,10 @@ class ReporterTests(unittest.TestCase):
1413
1422
  args = [*ENGINE, "trace", str(path)]
1414
1423
  result = subprocess.run(args, capture_output=True, text=True, env=ENGINE_ENV)
1415
1424
  self.assertEqual(result.returncode, 0, result.stderr)
1416
- self.assertEqual(json.loads(result.stdout)["sourceIdentity"]["size"], 4)
1425
+ header = json.loads(result.stdout)
1426
+ self.assertEqual(header["sourceIdentity"]["size"], 4)
1427
+ self.assertEqual((header["decoder"], header["instructionSemantics"]),
1428
+ ("capstone " + capstone.__version__, f"pypcode {pypcode.__version__} (Ghidra SLEIGH x86)"))
1417
1429
  cfg["sha256"] = "0" * 64; path.write_text(json.dumps(cfg))
1418
1430
  result = subprocess.run(args, capture_output=True, text=True, env=ENGINE_ENV)
1419
1431
  self.assertEqual(result.returncode, 1)
@@ -1470,54 +1482,59 @@ class ReporterTests(unittest.TestCase):
1470
1482
 
1471
1483
  def test_repeated_string_comparisons(self):
1472
1484
  es = {"es": 0x2000, "ds": 0x2000}
1473
- with accepted("extended", "the handwritten backend stops on CMPS and SCAS"):
1474
- # Positive control: REPNE SCASB stops at the terminator it compared, after two iterations.
1475
- result = report("bf 00 01 c6 05 61 c6 45 01 00 b0 00 b9 10 00 f2 ae c3", flags={"direction": 0}, registers=es)
1476
- path = result["paths"][0]
1477
- self.assertTrue(path["returned"], path["stop"])
1478
- exit_event, = events(result, "string-compare-exit")
1479
- self.assertEqual((exit_event["iterations"], exit_event["exit"]), (2, "condition"))
1480
- self.assertEqual((path["registers"]["cx"]["value"], path["registers"]["di"]["value"]), (14, 0x102))
1481
- self.assertEqual(result["stringIterationsUsed"], 2)
1482
- # Unknown memory leaves the repeat condition unresolved after the first iteration.
1483
- result = report("bf 00 01 b0 00 b9 10 00 f2 ae c3", flags={"direction": 0}, registers=es)
1484
- self.assertIn("comparison outcome unresolved", result["paths"][0]["stop"])
1485
- self.assertEqual(len(events(result, "read")), 1)
1486
- # A repeated comparison pays per iteration, so the budget stops it mid-loop.
1487
- result = report("bf 00 01 c7 05 00 00 b0 01 b9 10 00 f2 ae c3", flags={"direction": 0}, registers=es,
1488
- stringIterations=1)
1489
- self.assertIn("budget exhausted", result["paths"][0]["stop"])
1490
- self.assertEqual(result["stringIterationsUsed"], 1)
1485
+ # Positive control: REPNE SCASB stops at the terminator it compared, after two iterations.
1486
+ result = report("bf 00 01 c6 05 61 c6 45 01 00 b0 00 b9 10 00 f2 ae c3", flags={"direction": 0}, registers=es)
1487
+ path = result["paths"][0]
1488
+ self.assertTrue(path["returned"], path["stop"])
1489
+ exit_event, = events(result, "string-compare-exit")
1490
+ self.assertEqual((exit_event["iterations"], exit_event["exit"]), (2, "condition"))
1491
+ self.assertEqual((path["registers"]["cx"]["value"], path["registers"]["di"]["value"]), (14, 0x102))
1492
+ self.assertEqual(result["stringIterationsUsed"], 2)
1493
+ # Unknown memory leaves the repeat condition unresolved after the first iteration.
1494
+ result = report("bf 00 01 b0 00 b9 10 00 f2 ae c3", flags={"direction": 0}, registers=es)
1495
+ self.assertIn("comparison outcome unresolved", result["paths"][0]["stop"])
1496
+ self.assertEqual(len(events(result, "read")), 1)
1497
+ # A repeated comparison pays per iteration, so the budget stops it mid-loop.
1498
+ result = report("bf 00 01 c7 05 00 00 b0 01 b9 10 00 f2 ae c3", flags={"direction": 0}, registers=es,
1499
+ stringIterations=1)
1500
+ self.assertIn("budget exhausted", result["paths"][0]["stop"])
1501
+ self.assertEqual(result["stringIterationsUsed"], 1)
1491
1502
  # REPNE stays rejected on the forms that do not compare.
1492
1503
  result = report("f2 a4 c3", flags={"direction": 0})
1493
1504
  self.assertIn("REPNE", result["paths"][0]["stop"])
1494
1505
 
1495
1506
  def test_repeated_comparison_splits_an_unknown_direction_when_its_count_exceeds_the_budget(self):
1496
1507
  es = {"es": 0x2000, "ds": 0x2000}
1497
- with accepted("extended", "the handwritten backend stops on CMPS and SCAS"):
1498
- # CX = 0xFFFF exceeds the budget, but the terminator ends the scan after one iteration either way.
1499
- result = report("bf 00 01 c6 05 00 b0 00 b9 ff ff f2 ae c3", registers=es)
1508
+ # CX = 0xFFFF exceeds the budget, but the terminator ends the scan after one iteration either way.
1509
+ result = report("bf 00 01 c6 05 00 b0 00 b9 ff ff f2 ae c3", registers=es)
1500
1510
  self.assertTrue(all(p["returned"] for p in result["paths"]), [p["stop"] for p in result["paths"]])
1501
1511
  self.assertEqual(sorted(p["registers"]["di"]["value"] for p in result["paths"]), [0xff, 0x101])
1502
1512
  self.assertEqual(result["stringIterationsUsed"], 2)
1503
1513
 
1504
1514
  def test_rotates_and_sal_by_one_on_unknown_operands_keep_the_reported_forms(self):
1505
- # The differential run checks that both backends report the same result and operation.
1506
1515
  for code in ("d1 c0", "d1 c8", "d0 c0", "d0 c8", "d0 cc", "d1 f0"):
1507
1516
  with self.subTest(code=code):
1508
1517
  path = report(code + " c3")["paths"][0]
1509
1518
  self.assertTrue(path["returned"], path["stop"])
1519
+ event, = [e for e in path["events"] if e["kind"] == "arithmetic"]
1520
+ operation = {"c0": "rol", "c8": "ror", "cc": "ror", "f0": "sal"}[code[-2:]]
1521
+ self.assertEqual(event["operation"], operation)
1522
+ # A rotate reports the OR of its two shifted halves; SAL by one reports one shift.
1523
+ self.assertEqual(event["result"]["expression"][0], "shl" if operation == "sal" else "or")
1524
+ if operation != "sal":
1525
+ # The carry out is bit 0 (ROR) or the top bit (ROL) of the rotated operand.
1526
+ bits = event["left"]["bits"]
1527
+ low = event["left"]["expression"][2]
1528
+ self.assertEqual(event["carryOut"]["expression"][2], low + (bits - 1 if operation == "rol" else 0))
1510
1529
 
1511
1530
  def test_rotate_through_unknown_carry_resolves_a_carry_out_from_a_known_operand(self):
1512
1531
  # RCL by n carries out bit 16 - n of a 16-bit operand, RCR by n bit n - 1; CF starts unknown.
1513
- # The result term still differs in form from the handwritten backend's, so this runs the
1514
- # engine's default backend alone.
1515
1532
  cases = (("bb 10 00 c1 d3 05", 0), ("bb 00 08 c1 d3 05", 1), ("bb 10 00 c1 db 05", 1),
1516
1533
  ("bb 08 00 c1 db 05", 0), ("c1 d3 05", None))
1517
1534
  for code, carry in cases:
1518
1535
  with self.subTest(code=code):
1519
1536
  data = bytes.fromhex(code + " c3")
1520
- result = engine_report(data, configuration(data), "trace")
1537
+ result = run_report(data, configuration(data), "trace")
1521
1538
  event, = events(result, "arithmetic")
1522
1539
  self.assertEqual(event["carryOut"]["value"], carry)
1523
1540