scientific-method-engine 0.4.0__tar.gz → 0.6.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (62) hide show
  1. scientific_method_engine-0.4.0/README.md → scientific_method_engine-0.6.0/PKG-INFO +27 -2
  2. scientific_method_engine-0.4.0/PKG-INFO → scientific_method_engine-0.6.0/README.md +12 -14
  3. {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/pyproject.toml +8 -3
  4. {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/cli.py +1 -0
  5. {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/x86/dispatch.py +1 -1
  6. {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/x86/effect_order.py +9 -5
  7. scientific_method_engine-0.6.0/src/scientific_method_engine/x86/handwritten.py +391 -0
  8. scientific_method_engine-0.6.0/src/scientific_method_engine/x86/machine.py +404 -0
  9. scientific_method_engine-0.6.0/src/scientific_method_engine/x86/pcode.py +376 -0
  10. scientific_method_engine-0.6.0/src/scientific_method_engine/x86/pcode_backend.py +835 -0
  11. {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/x86/reports.py +5 -3
  12. scientific_method_engine-0.6.0/src/scientific_method_engine/x86/semantics.py +77 -0
  13. {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/x86/trace.py +114 -14
  14. scientific_method_engine-0.6.0/tests/differential.py +120 -0
  15. scientific_method_engine-0.6.0/tests/oracle.py +75 -0
  16. scientific_method_engine-0.6.0/tests/test_differential.py +71 -0
  17. {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/tests/test_dispatch.py +1 -1
  18. {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/tests/test_effect_order.py +3 -1
  19. scientific_method_engine-0.6.0/tests/test_nested_frame_request.py +61 -0
  20. scientific_method_engine-0.6.0/tests/test_oracle.py +155 -0
  21. {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/tests/test_pe.py +16 -1
  22. scientific_method_engine-0.6.0/tests/test_table_continuations.py +161 -0
  23. {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/tests/test_x86.py +55 -1
  24. scientific_method_engine-0.4.0/src/scientific_method_engine/x86/machine.py +0 -693
  25. {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/.gitignore +0 -0
  26. {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/LICENSE +0 -0
  27. {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/__init__.py +0 -0
  28. {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/__main__.py +0 -0
  29. {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/ghidra/ClearNoReturnFunctions.java +0 -0
  30. {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/ghidra/CreateFunctions.java +0 -0
  31. {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/ghidra/ExportBoundedFlow.java +0 -0
  32. {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/ghidra/ExportFunctionFingerprints.java +0 -0
  33. {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/ghidra/ExportFunctionInventory.java +0 -0
  34. {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/ghidra/MergeFallThroughFragment.java +0 -0
  35. {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/ghidra/RecoverCitedFunctions.java +0 -0
  36. {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/ghidra/RepairReturningCallers.java +0 -0
  37. {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/ghidra/ReportCallArguments.java +0 -0
  38. {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/ghidra/ReportCallPaths.java +0 -0
  39. {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/ghidra/ReportCallSitesWithScalars.java +0 -0
  40. {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/ghidra/ReportCallsToRange.java +0 -0
  41. {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/ghidra/ReportConstantFirstArgumentCalls.java +0 -0
  42. {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/ghidra/ReportDataBytes.java +0 -0
  43. {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/ghidra/ReportDecompileMatches.java +0 -0
  44. {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/ghidra/ReportDecompileWindow.java +0 -0
  45. {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/ghidra/ReportFilePatternInMemory.java +0 -0
  46. {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/ghidra/ReportFirstArgumentCallSummary.java +0 -0
  47. {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/ghidra/ReportFunctionScalarConstants.java +0 -0
  48. {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/ghidra/ReportFunctionSummary.java +0 -0
  49. {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/ghidra/ReportInstructionContext.java +0 -0
  50. {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/ghidra/ReportInstructionWindow.java +0 -0
  51. {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/ghidra/ReportMemoryBlockForFileOffset.java +0 -0
  52. {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/ghidra/ReportMemoryBlocks.java +0 -0
  53. {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/ghidra/ReportRandomnessCandidates.java +0 -0
  54. {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/ghidra/ReportReferences.java +0 -0
  55. {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/ghidra/ReportScalarConstants.java +0 -0
  56. {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/ghidra/ReportStringReferences.java +0 -0
  57. {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/ghidra/ReportSymbolReferences.java +0 -0
  58. {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/x86/__init__.py +0 -0
  59. {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/x86/image.py +0 -0
  60. {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/x86/pe.py +0 -0
  61. {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/x86/result_flow.py +0 -0
  62. {scientific_method_engine-0.4.0 → scientific_method_engine-0.6.0}/src/scientific_method_engine/x86/values.py +0 -0
@@ -1,8 +1,24 @@
1
+ Metadata-Version: 2.5
2
+ Name: scientific-method-engine
3
+ Version: 0.6.0
4
+ Summary: Bounded instruction-derived x86 evidence reports for segmented MZ/FBOV and PE32/i386 code.
5
+ Project-URL: Source, https://github.com/kibertoad/refurbished-dinosaurs-toolkit/tree/main/packages/scientific-method-engine
6
+ Author: kibertoad
7
+ License-Expression: MIT
8
+ License-File: LICENSE
9
+ Requires-Python: >=3.12
10
+ Requires-Dist: capstone==5.0.7
11
+ Requires-Dist: pypcode==4.0.0
12
+ Provides-Extra: test
13
+ Requires-Dist: unicorn==2.1.4; extra == 'test'
14
+ Description-Content-Type: text/markdown
15
+
1
16
  # scientific-method-engine
2
17
 
3
18
  Bounded instruction-derived x86 evidence reports for segmented 16-bit MZ/FBOV code and
4
- PE32/i386 code. The engine decodes instructions with Capstone, follows bounded paths and emits
5
- `bounded-x86-v1` JSON. It never runs the original program.
19
+ PE32/i386 code. The engine decodes instructions with Capstone, takes their values, flags and
20
+ branch conditions from Ghidra's SLEIGH specification through pypcode, follows bounded paths and
21
+ emits `bounded-x86-v1` JSON. It never runs the original program. It needs Python 3.12 or later.
6
22
 
7
23
  ```sh
8
24
  uv add --group research scientific-method-engine # or: pip install scientific-method-engine
@@ -101,3 +117,12 @@ The `effects` command additionally emits path-local `effectOrdering` timelines,
101
117
  pre-call write prefixes and bounded local restoration witnesses. Unknown modeled
102
118
  or nested service effects stay separate; no return code establishes rollback or
103
119
  transactionality. See [the full contract](https://github.com/kibertoad/refurbished-dinosaurs-toolkit/blob/main/docs/bounded-evidence-reporters.md#ordered-effect-path-summaries).
120
+
121
+ Evidenced segmented16 `indirectJumps` declarations also expose separate
122
+ `declaredContinuationPaths`. Ordinary paths retain their unresolved transfer;
123
+ conditional routes preserve prefix/child effects with target-choice, live-table
124
+ and selector assumptions. Effects summaries never mark those routes complete.
125
+ Concrete values/field addresses reject inconsistent rows; overlap and shared
126
+ path/step/visit/total/boundary limits remain explicit. Partial-table splits are
127
+ partial evidence, never complete dispatch or native-reachability claims. See the
128
+ [table contract](https://github.com/kibertoad/refurbished-dinosaurs-toolkit/blob/main/docs/bounded-evidence-reporters.md#evidenced-indirect-jump-tables).
@@ -1,20 +1,9 @@
1
- Metadata-Version: 2.5
2
- Name: scientific-method-engine
3
- Version: 0.4.0
4
- Summary: Bounded instruction-derived x86 evidence reports for segmented MZ/FBOV and PE32/i386 code.
5
- Project-URL: Source, https://github.com/kibertoad/refurbished-dinosaurs-toolkit/tree/main/packages/scientific-method-engine
6
- Author: kibertoad
7
- License-Expression: MIT
8
- License-File: LICENSE
9
- Requires-Python: >=3.10
10
- Requires-Dist: capstone==5.0.7
11
- Description-Content-Type: text/markdown
12
-
13
1
  # scientific-method-engine
14
2
 
15
3
  Bounded instruction-derived x86 evidence reports for segmented 16-bit MZ/FBOV code and
16
- PE32/i386 code. The engine decodes instructions with Capstone, follows bounded paths and emits
17
- `bounded-x86-v1` JSON. It never runs the original program.
4
+ PE32/i386 code. The engine decodes instructions with Capstone, takes their values, flags and
5
+ branch conditions from Ghidra's SLEIGH specification through pypcode, follows bounded paths and
6
+ emits `bounded-x86-v1` JSON. It never runs the original program. It needs Python 3.12 or later.
18
7
 
19
8
  ```sh
20
9
  uv add --group research scientific-method-engine # or: pip install scientific-method-engine
@@ -113,3 +102,12 @@ The `effects` command additionally emits path-local `effectOrdering` timelines,
113
102
  pre-call write prefixes and bounded local restoration witnesses. Unknown modeled
114
103
  or nested service effects stay separate; no return code establishes rollback or
115
104
  transactionality. See [the full contract](https://github.com/kibertoad/refurbished-dinosaurs-toolkit/blob/main/docs/bounded-evidence-reporters.md#ordered-effect-path-summaries).
105
+
106
+ Evidenced segmented16 `indirectJumps` declarations also expose separate
107
+ `declaredContinuationPaths`. Ordinary paths retain their unresolved transfer;
108
+ conditional routes preserve prefix/child effects with target-choice, live-table
109
+ and selector assumptions. Effects summaries never mark those routes complete.
110
+ Concrete values/field addresses reject inconsistent rows; overlap and shared
111
+ path/step/visit/total/boundary limits remain explicit. Partial-table splits are
112
+ partial evidence, never complete dispatch or native-reachability claims. See the
113
+ [table contract](https://github.com/kibertoad/refurbished-dinosaurs-toolkit/blob/main/docs/bounded-evidence-reporters.md#evidenced-indirect-jump-tables).
@@ -5,13 +5,18 @@ build-backend = "hatchling.build"
5
5
  [project]
6
6
  name = "scientific-method-engine"
7
7
  # The release workflow writes the published version from the package's release tag.
8
- version = "0.4.0"
8
+ version = "0.6.0"
9
9
  description = "Bounded instruction-derived x86 evidence reports for segmented MZ/FBOV and PE32/i386 code."
10
10
  readme = "README.md"
11
- requires-python = ">=3.10"
11
+ requires-python = ">=3.12"
12
12
  license = "MIT"
13
13
  authors = [{ name = "kibertoad" }]
14
- dependencies = ["capstone==5.0.7"]
14
+ dependencies = ["capstone==5.0.7", "pypcode==4.0.0"]
15
+
16
+ [project.optional-dependencies]
17
+ # Unicorn is the tests' concrete oracle (ADR 0003, decision 4). Its core is GPLv2, so it never
18
+ # becomes a runtime dependency.
19
+ test = ["unicorn==2.1.4"]
15
20
 
16
21
  [project.scripts]
17
22
  scientific-method-engine = "scientific_method_engine.cli:run"
@@ -10,6 +10,7 @@ PREPARED_CONFIG_LIMIT = 16 * 1024 * 1024
10
10
  USAGE = ("Usage: scientific-method-engine <operand|operand-candidates|target|bounds|owner|callees|trace|uses|arguments|"
11
11
  "effects|returns|memory|incoming|call-order|guards|allocation|dispatch> <config.json|->\n"
12
12
  "effects includes ordered path writes/calls and local restoration witnesses; transactionality remains unestablished.\n"
13
+ "Declared table continuations are separate conditional paths; ordinary computed transfers remain stopped.\n"
13
14
  " scientific-method-engine ghidra-scripts")
14
15
 
15
16
 
@@ -1,4 +1,4 @@
1
- """Explicit, source-derived indirect jump tables for CFG discovery only."""
1
+ """Explicit, source-derived indirect jump tables for CFG discovery and explicitly conditional path continuations."""
2
2
  from capstone.x86 import X86_OP_IMM
3
3
  from .image import integer
4
4
 
@@ -3,7 +3,7 @@ from bisect import bisect_right
3
3
 
4
4
 
5
5
  KINDS = {"read", "write", "call", "call-return", "return", "branch", "compare", "flag-assumption",
6
- "arithmetic", "value-transfer", "conversion", "flag-write", "flags-save", "flags-restore", "local-iret", "string-operation"}
6
+ "arithmetic", "value-transfer", "conversion", "flag-write", "flags-save", "flags-restore", "local-iret", "string-operation", "declared-jump-continuation"}
7
7
 
8
8
 
9
9
  def _storage(event):
@@ -25,7 +25,10 @@ def effect_ordering(report):
25
25
  aliases, external effects, resources or transactionality.
26
26
  """
27
27
  summaries = []
28
- for index, path in enumerate(report["paths"]):
28
+ conditional_summaries = []
29
+ combined = [(False, i, p) for i, p in enumerate(report["paths"])]
30
+ combined += [(True, i, p) for i, p in enumerate(report.get("declaredContinuationPaths", []))]
31
+ for conditional, index, path in combined:
29
32
  timeline, writes, calls, witnesses = [], [], [], []
30
33
  snapshots = {}
31
34
  pending = {}
@@ -78,13 +81,14 @@ def effect_ordering(report):
78
81
  if path.get("stop"):
79
82
  boundary = {"site": path.get("stopSite"), "reason": path["stop"],
80
83
  "writesBeforeCount": len(writes), "meaning": "later effects are not read"}
81
- summaries.append({"path": index, "returned": path["returned"], "stop": boundary,
84
+ destination = conditional_summaries if conditional else summaries
85
+ destination.append({"path": index, "declaredJumpAssumptions": path.get("declaredJumpAssumptions", []), "returned": path["returned"], "stop": boundary,
82
86
  "guards": path["guards"], "timeline": timeline,
83
87
  "writeOrders": [w["order"] for w in writes], "calls": calls,
84
88
  "localRestorationWitnesses": witnesses,
85
- "effectCompleteWithinModel": bool(path["returned"] and not unknown_orders),
89
+ "effectCompleteWithinModel": bool(path["returned"] and not unknown_orders and not conditional),
86
90
  "transactionality": "not established; local writes and result codes cannot prove external rollback"})
87
- report["effectOrdering"] = {"paths": summaries,
91
+ report["effectOrdering"] = {"paths": summaries, "declaredContinuationPaths": conditional_summaries,
88
92
  "allPathsRead": report["completeWithinModel"],
89
93
  "nativeReachability": "unconfirmed",
90
94
  "meaning": "separate conditional paths; write prefixes index each path's writeOrders"}
@@ -0,0 +1,391 @@
1
+ """The handwritten semantics backend (ADR 0003): value computation, flag predicates and string bodies.
2
+
3
+ ADR 0003 decision 6 freezes this module. No mnemonic, flag rule or value computation is added here;
4
+ the pypcode backend replaces it group by group, and phase 5 deletes it.
5
+ """
6
+ from capstone.x86 import X86_OP_IMM, X86_OP_REG, X86_OP_MEM
7
+ from . import semantics
8
+ from .machine import ALIASES, StopPath
9
+ from .values import Value, const, unknown, op, extract, join, resize, sources
10
+
11
+ CARRY_BRANCHES = {"jb": True, "jc": True, "jnae": True, "jae": False, "jnb": False, "jnc": False}
12
+ # Branches taken when CF or OF is set; logic operations clear both whatever their operands.
13
+ CLEARED_BY_LOGIC = {**CARRY_BRANCHES, "jo": True, "jno": False}
14
+
15
+
16
+ def predicate(state, mnemonic):
17
+ flags = state.flags
18
+ if flags is None and state.carry is not None and mnemonic in CARRY_BRANCHES:
19
+ info = {"predicate": mnemonic, "flag": "CF", "carry": state.carry.report()}
20
+ if state.carry.number is None:
21
+ return None, {**info, "reason": "carry unresolved"}
22
+ return bool(state.carry.number) == CARRY_BRANCHES[mnemonic], info
23
+ if flags is None:
24
+ return None, {"predicate": mnemonic, "reason": "flag producer unresolved",
25
+ "flagProducer": state.unknown_flag_site, "flagGeneration": state.flag_epoch}
26
+ a, b, operation, site = flags
27
+ info = {"predicate": mnemonic, "flagProducer": site, "operation": operation,
28
+ "left": a.report(), "right": b.report()}
29
+ if a.number is not None and b.number is not None:
30
+ x, y = a.number, b.number
31
+ elif operation in ("cmp", "sub", "xor") and a.term == b.term:
32
+ # Any value compared with, subtracted from or XORed with itself yields zero.
33
+ x = y = 0
34
+ elif operation in ("test", "and", "or", "xor") and mnemonic in CLEARED_BY_LOGIC:
35
+ return not CLEARED_BY_LOGIC[mnemonic], info
36
+ else:
37
+ return None, info
38
+ bits = a.bits
39
+ if operation in ("cmp", "sub"):
40
+ raw = x - y
41
+ result = raw % (1 << bits)
42
+ cf = x < y
43
+ of = bool(((x ^ y) & (x ^ result)) & (1 << (bits - 1)))
44
+ elif operation == "add":
45
+ raw = x + y
46
+ result = raw % (1 << bits)
47
+ cf = raw >= 1 << bits
48
+ of = bool((~(x ^ y) & (x ^ result)) & (1 << (bits - 1)))
49
+ elif operation in ("test", "and", "or", "xor"):
50
+ result = {"test": x & y, "and": x & y, "or": x | y, "xor": x ^ y}[operation]
51
+ cf = of = False
52
+ else:
53
+ return None, info
54
+ zf, sf = result == 0, bool(result & (1 << (bits - 1)))
55
+ conditions = {"je": zf, "jz": zf, "jne": not zf, "jnz": not zf,
56
+ "jb": cf, "jc": cf, "jnae": cf, "jae": not cf, "jnb": not cf, "jnc": not cf,
57
+ "jbe": cf or zf, "jna": cf or zf, "ja": not cf and not zf, "jnbe": not cf and not zf,
58
+ "jl": sf != of, "jnge": sf != of, "jge": sf == of, "jnl": sf == of,
59
+ "jle": zf or sf != of, "jng": zf or sf != of, "jg": not zf and sf == of,
60
+ "jnle": not zf and sf == of, "js": sf, "jns": not sf, "jo": of, "jno": not of}
61
+ return conditions.get(mnemonic), info
62
+
63
+
64
+ def ordinary(state, ins, image):
65
+ m, operands = ins.mnemonic, ins.operands
66
+ if m in ("cld", "std", "cli", "sti"):
67
+ value = const(1 if m in ("std", "sti") else 0, 1, state.at)
68
+ flag = "DF" if m in ("cld", "std") else "IF"
69
+ if flag == "DF": state.direction_flag = value
70
+ else: state.interrupt_flag = value
71
+ state.event("flag-write", flag=flag, value=value.report(),
72
+ interpretation="local flag effect only; interrupts and timing are not simulated")
73
+ return
74
+ if m in ("pushf", "pushfd", "popf", "popfd"):
75
+ bits = 32 if (0x66 in ins.prefix) != state.flat else 16
76
+ if m.startswith("push"): state.save_flags(bits)
77
+ else: state.restore_flags(bits)
78
+ return
79
+ if m == "nop":
80
+ return
81
+ if m in ("mov", "movzx", "movsx"):
82
+ if state.flat and operands[0].type == X86_OP_REG and ins.reg_name(operands[0].reg) in state.segment_bases:
83
+ raise StopPath("Segment selector assignment requires a descriptor model")
84
+ value = state.get(ins, operands[1], image)
85
+ result = resize(value, operands[0].size * 8, signed=m == "movsx")
86
+ state.put(ins, operands[0], result)
87
+ if not state.value_transfers:
88
+ return
89
+
90
+ def location(operand):
91
+ return {"kind": "register", "register": ins.reg_name(operand.reg)} if operand.type == X86_OP_REG else {"kind": "memory"} if operand.type == X86_OP_MEM else {"kind": "immediate"}
92
+ destination_container = ALIASES[ins.reg_name(operands[0].reg)][0] if operands[0].type == X86_OP_REG else None
93
+ state.event("value-transfer", operation=m, source=location(operands[1]), destination=location(operands[0]),
94
+ destinationContainer=destination_container, destinationContainerValue=state.reg(destination_container).report() if destination_container else None,
95
+ sourceBits=value.bits, destinationBits=result.bits, sourceValue=value.report(), resultValue=result.report(),
96
+ conversion="truncate" if result.bits < value.bits else "signExtend" if m == "movsx" else "zeroExtend" if result.bits > value.bits else "sameWidth")
97
+ return
98
+ if m == "xchg":
99
+ values = [state.get(ins, operand, image) for operand in operands]
100
+ addresses = [state.address(ins, operand) if operand.type == X86_OP_MEM else None for operand in operands]
101
+ for index, operand in enumerate(operands):
102
+ value = values[1-index]
103
+ if addresses[index] is None:
104
+ state.put(ins, operand, value)
105
+ else:
106
+ segment, offset, register = addresses[index]
107
+ state.access(segment, offset, operand.size, resize(value, operand.size * 8), addressing_register=register)
108
+ return
109
+ if m == "imul" and len(operands) in (2, 3):
110
+ left, right = (state.get(ins, operand, image) for operand in (operands if len(operands) == 2 else operands[1:]))
111
+ left = resize(left, operands[0].size * 8)
112
+ right = resize(right, left.bits, signed=True)
113
+ result = op("mul", left, right, state.at)
114
+ state.put(ins, operands[0], result)
115
+ state.forget_flags() # CF/OF require the full signed product; other flags are undefined.
116
+ state.event("arithmetic", operation="imul", left=left.report(), right=right.report(),
117
+ result=result.report(), modulus=1 << left.bits, flags="unresolved signed-product overflow")
118
+ return
119
+ if m == "lea":
120
+ segment, offset, register = state.address(ins, operands[1])
121
+ state.put(ins, operands[0], offset)
122
+ state.event("address-formation", value=offset.report(), addressingSegment=segment.report(),
123
+ addressingSegmentRegister=register, destinationRegister=ins.reg_name(operands[0].reg),
124
+ note="LEA does not access memory; this addressing default does not bind a later dereference")
125
+ return
126
+ if m in ("lds", "les"):
127
+ if state.flat:
128
+ raise StopPath("Descriptor loads are outside the PE32 flat model")
129
+ segment, offset, register = state.address(ins, operands[1])
130
+ if operands[0].size != 2:
131
+ raise StopPath("Only 16:16 pointer loads are supported")
132
+ value = state.access(segment, offset, 4, role="far-pointer", addressing_register=register)
133
+ state.put(ins, operands[0], extract(value, 0, 16))
134
+ state.setreg("ds" if m == "lds" else "es", extract(value, 16, 16), state.at)
135
+ return
136
+ if m == "push":
137
+ if state.flat and operands[0].type == X86_OP_REG and ins.reg_name(operands[0].reg) in state.segment_bases:
138
+ raise StopPath("Segment stack operations require a descriptor model")
139
+ state.push(state.get(ins, operands[0], image))
140
+ return
141
+ if m == "pop":
142
+ if state.flat and operands[0].type == X86_OP_REG and ins.reg_name(operands[0].reg) in state.segment_bases:
143
+ raise StopPath("Segment selector assignment requires a descriptor model")
144
+ state.put(ins, operands[0], state.pop(operands[0].size))
145
+ return
146
+ if m == "leave":
147
+ if 0x66 in ins.prefix:
148
+ raise StopPath("Operand-size override on LEAVE is unsupported")
149
+ state.setreg(state.sp, state.reg(state.bp), state.at)
150
+ state.setreg(state.bp, state.pop(state.bits // 8), state.at)
151
+ return
152
+ if m in ("cmp", "test"):
153
+ a, b = (state.get(ins, o, image) for o in operands)
154
+ state.set_flags(a, resize(b, a.bits), m)
155
+ state.event("compare", operation=m, left=a.report(), right=b.report())
156
+ return
157
+ if m in ("add", "sub", "and", "or", "xor", "shl", "sal", "shr", "sar"):
158
+ a, b = (state.get(ins, o, image) for o in operands)
159
+ b = resize(b, a.bits)
160
+ result = op("shl" if m == "sal" else m, a, b, state.at)
161
+ state.put(ins, operands[0], result)
162
+ if m in ("add", "sub", "and", "or", "xor"):
163
+ state.set_flags(a, b, m)
164
+ else:
165
+ shift_carry(state, m, a, b)
166
+ state.event("arithmetic", operation=m, left=a.report(), right=b.report(), result=result.report(), modulus=1 << a.bits)
167
+ return
168
+ if m in ("inc", "dec"):
169
+ a = state.get(ins, operands[0], image)
170
+ result = op("add" if m == "inc" else "sub", a, const(1, a.bits), state.at)
171
+ state.put(ins, operands[0], result)
172
+ # Carry is preserved; the other flags are not modeled for INC/DEC.
173
+ state.forget_flags(keep_carry=True)
174
+ return
175
+ if m in ("cbw", "cwde"):
176
+ # Capstone 5 names these inconsistently in 16-bit mode. Use effective size.
177
+ # The prefix toggles the mode's default operand size (16-bit real mode, 32-bit flat).
178
+ wide = (0x66 in ins.prefix) != state.flat
179
+ source, destination = ("ax", "eax") if wide else ("al", "ax")
180
+ source_value = state.reg(source)
181
+ value = resize(source_value, 32 if wide else 16, True)
182
+ state.setreg(destination, value, state.at)
183
+ state.event("conversion", sourceRegister=source, destinationRegister=destination,
184
+ effectiveOperandBits=32 if wide else 16, decoderMnemonic=m,
185
+ mnemonicWidthMismatch=m != ("cwde" if wide else "cbw"), result=value.report(),
186
+ sourceValue=source_value.report(), sourceBits=source_value.bits, destinationBits=value.bits, conversion="signExtend")
187
+ return
188
+ if m in ("cwd", "cdq"):
189
+ wide = (0x66 in ins.prefix) != state.flat
190
+ source, destination = ("eax", "edx") if wide else ("ax", "dx")
191
+ bits = 32 if wide else 16
192
+ source_value = state.reg(source)
193
+ value = resize(extract(source_value, bits-1, 1), bits, signed=True)
194
+ state.setreg(destination, value, state.at)
195
+ state.event("conversion", sourceRegister=source, destinationRegister=destination,
196
+ effectiveOperandBits=bits, decoderMnemonic=m,
197
+ mnemonicWidthMismatch=m != ("cdq" if wide else "cwd"), result=value.report(),
198
+ sourceValue=source_value.report(), sourceBits=source_value.bits, destinationBits=value.bits, conversion="signFillHighHalf")
199
+ return
200
+ if m in ("clc", "stc", "cmc"):
201
+ if m == "cmc":
202
+ value = op("xor", state.carry_value(), const(1, 1), state.at)
203
+ else:
204
+ value = const(int(m == "stc"), 1, state.at)
205
+ state.forget_flags()
206
+ state.carry = value
207
+ state.event("flag-write", flag="CF", value=value.report(), interpretation="local carry effect")
208
+ return
209
+ if m in ("not", "neg"):
210
+ a = state.get(ins, operands[0], image)
211
+ if m == "not":
212
+ state.put(ins, operands[0], op("xor", a, const((1 << a.bits) - 1, a.bits), state.at))
213
+ return
214
+ result = op("sub", const(0, a.bits), a, state.at)
215
+ state.put(ins, operands[0], result)
216
+ state.set_flags(const(0, a.bits, state.at), a, "sub")
217
+ state.event("arithmetic", operation="neg", left=a.report(), result=result.report(), modulus=1 << a.bits)
218
+ return
219
+ if m in ("adc", "sbb"):
220
+ a, b = (state.get(ins, o, image) for o in operands)
221
+ b = resize(b, a.bits)
222
+ carry = state.carry_value()
223
+ name = "add" if m == "adc" else "sub"
224
+ result = op(name, op(name, a, b, state.at), resize(carry, a.bits), state.at)
225
+ state.put(ins, operands[0], result)
226
+ state.forget_flags()
227
+ if None not in (a.number, b.number, carry.number):
228
+ raw = a.number + b.number + carry.number if m == "adc" else a.number - b.number - carry.number
229
+ state.carry = const(int(raw < 0 or raw >= 1 << a.bits), 1, state.at)
230
+ else:
231
+ state.carry = unknown(f"carry:{state.at}:{state.flag_serial}", 1, state.at)
232
+ state.event("arithmetic", operation=m, left=a.report(), right=b.report(), carryIn=carry.report(),
233
+ result=result.report(), carryOut=state.carry.report(), modulus=1 << a.bits)
234
+ return
235
+ if m in ("rol", "ror", "rcl", "rcr"):
236
+ a = state.get(ins, operands[0], image)
237
+ count = state.get(ins, operands[1], image) if len(operands) > 1 else const(1, 8)
238
+ if len(operands) > 1 and operands[1].type == X86_OP_IMM:
239
+ # Capstone gives RCL's implicit count of 1 on a memory operand a size of 0.
240
+ count = const(operands[1].imm, 8, state.at)
241
+ if count.number is None:
242
+ raise StopPath("rotate count unresolved")
243
+ masked = count.number & 31
244
+ if masked == 0:
245
+ return # The value and flags are unchanged.
246
+ bits = a.bits
247
+ through = m in ("rcl", "rcr")
248
+ n = masked % (bits + 1 if through else bits)
249
+ if through and n == 0:
250
+ # A full rotation through CF restores the value and CF; OF is undefined.
251
+ state.forget_flags(keep_carry=True)
252
+ return
253
+ carry_in = resize(state.carry_value(), bits)
254
+ # Each form is an OR of shifted copies (positive shifts left), built once so the
255
+ # expression does not repeat the operand for every bit rotated.
256
+ parts = {"rol": [(a, n), (a, n - bits)],
257
+ "ror": [(a, -n), (a, bits - n)],
258
+ "rcl": [(a, n), (carry_in, n - 1), (a, n - bits - 1)],
259
+ "rcr": [(a, -n), (carry_in, bits - n), (a, bits + 1 - n)]}[m]
260
+ value = None
261
+ for part, shift in parts:
262
+ if abs(shift) >= bits:
263
+ continue # Every bit leaves the operand; op() would mask the count instead.
264
+ if shift:
265
+ part = op("shl" if shift > 0 else "shr", part, const(abs(shift), bits), state.at)
266
+ value = part if value is None else op("or", value, part, state.at)
267
+ # CF is the last bit rotated out: the result's low bit for ROL/RCL and its high bit for ROR/RCR.
268
+ out = {"rol": (bits - n) % bits, "ror": (n - 1) % bits, "rcl": bits - n, "rcr": n - 1}[m]
269
+ carry = extract(a, out, 1)
270
+ # A count read from CL is an input of the result and of CF.
271
+ value = Value(value.bits, value.term, sources(value, count, site=state.at))
272
+ state.put(ins, operands[0], value)
273
+ state.forget_flags()
274
+ state.carry = Value(1, carry.term, sources(carry, count, site=state.at))
275
+ state.event("arithmetic", operation=m, left=a.report(), count=n, result=value.report(),
276
+ carryOut=state.carry.report(), modulus=1 << bits)
277
+ return
278
+ if m in ("mul", "imul") and len(operands) == 1:
279
+ source = state.get(ins, operands[0], image)
280
+ bits = source.bits
281
+ low_reg, high_reg = {8: ("al", "ah"), 16: ("ax", "dx"), 32: ("eax", "edx")}[bits]
282
+ signed = m == "imul"
283
+ multiplicand = state.reg(low_reg)
284
+ product = op("mul", resize(multiplicand, 2 * bits, signed), resize(source, 2 * bits, signed), state.at)
285
+ low = extract(product, 0, bits)
286
+ if bits == 8:
287
+ state.setreg("ax", product, state.at)
288
+ else:
289
+ state.setreg(low_reg, low, state.at)
290
+ state.setreg(high_reg, extract(product, bits, bits), state.at)
291
+ state.forget_flags()
292
+ if product.number is None:
293
+ state.carry = unknown(f"carry:{state.at}:{state.flag_serial}", 1, state.at)
294
+ else:
295
+ # CF and OF say whether the high half carries information beyond the low half.
296
+ state.carry = const(int(resize(low, 2 * bits, signed).number != product.number), 1, state.at)
297
+ state.event("arithmetic", operation=m, left=multiplicand.report(), right=source.report(),
298
+ result=product.report(), resultBits=2 * bits, carryOut=state.carry.report())
299
+ return
300
+ if m in ("div", "idiv"):
301
+ divisor = state.get(ins, operands[0], image)
302
+ bits = divisor.bits
303
+ signed = m == "idiv"
304
+ if bits == 8:
305
+ dividend = state.reg("ax")
306
+ else:
307
+ dividend = join([state.reg({16: "ax", 32: "eax"}[bits]), state.reg({16: "dx", 32: "edx"}[bits])])
308
+ quotient_reg, remainder_reg = {8: ("al", "ah"), 16: ("ax", "dx"), 32: ("eax", "edx")}[bits]
309
+ fault = None
310
+ if divisor.number == 0:
311
+ raise StopPath("divide by zero raises interrupt 0; its handler is not modeled")
312
+ if None not in (dividend.number, divisor.number):
313
+ x, y = dividend.number, divisor.number
314
+ if signed:
315
+ x -= (x >> (2 * bits - 1)) << (2 * bits)
316
+ y -= (y >> (bits - 1)) << bits
317
+ q = abs(x) // abs(y) * (1 if (x < 0) == (y < 0) else -1)
318
+ r = x - q * y
319
+ if not ((-(1 << (bits - 1)) <= q < 1 << (bits - 1)) if signed else q < 1 << bits):
320
+ raise StopPath("divide overflow raises interrupt 0; its handler is not modeled")
321
+ # Both results are computed from the dividend and divisor, so they name their producers.
322
+ origin = sources(dividend, divisor, site=state.at)
323
+ quotient, remainder = Value(bits, const(q, bits).term, origin), Value(bits, const(r, bits).term, origin)
324
+ else:
325
+ wide = resize(divisor, 2 * bits, signed)
326
+ origin = sources(dividend, divisor, site=state.at)
327
+ quotient = extract(Value(2 * bits, ("sdiv" if signed else "udiv", dividend.term, wide.term), origin), 0, bits)
328
+ remainder = extract(Value(2 * bits, ("smod" if signed else "umod", dividend.term, wide.term), origin), 0, bits)
329
+ fault = "possible divide error (interrupt 0) unresolved; this path assumes none"
330
+ state.conditional.append({"site": state.at, "assumption": "no divide error"})
331
+ state.setreg(quotient_reg, quotient, state.at)
332
+ state.setreg(remainder_reg, remainder, state.at)
333
+ state.forget_flags()
334
+ state.event("arithmetic", operation=m, dividend=dividend.report(), divisor=divisor.report(),
335
+ quotient=quotient.report(), remainder=remainder.report(), fault=fault)
336
+ return
337
+ raise StopPath("Unsupported instruction semantics: " + m)
338
+
339
+
340
+ def shift_carry(state, m, a, count):
341
+ """CF after SHL/SHR/SAR: the last bit shifted out, when the count is known."""
342
+ n = None if count.number is None else count.number & 31
343
+ if n == 0:
344
+ return # A zero count leaves every flag unchanged.
345
+ state.forget_flags()
346
+ if n is None or n > a.bits:
347
+ state.carry = unknown(f"carry:{state.at}:{state.flag_serial}", 1, state.at)
348
+ elif m in ("shl", "sal"):
349
+ state.carry = Value(1, extract(a, a.bits - n, 1).term, sources(a, site=state.at))
350
+ else:
351
+ state.carry = Value(1, extract(a, n - 1, 1).term, sources(a, site=state.at))
352
+
353
+
354
+ def string_iteration(state, ins, operation, width, source_name, delta):
355
+ if operation in ("cmps", "scas"):
356
+ raise StopPath("Unsupported instruction semantics: " + ins.mnemonic)
357
+ si, di = ("esi", "edi") if state.flat else ("si", "di")
358
+ # DF chooses the step's sign, so the instruction that set DF is one of the step's producers.
359
+ step = Value(state.bits, const(delta, state.bits).term, state.direction_flag.sources)
360
+ if operation in ("movs", "lods"):
361
+ value = state.access(state.segment(source_name), state.reg(si), width, role="string-source", addressing_register=source_name)
362
+ state.setreg(si, op("add", state.reg(si), step, state.at), state.at)
363
+ else:
364
+ value = state.reg({1:"al",2:"ax",4:"eax"}[width])
365
+ if operation in ("movs", "stos"):
366
+ state.access(state.segment("es"), state.reg(di), width, value, role="string-destination", addressing_register="es")
367
+ state.setreg(di, op("add", state.reg(di), step, state.at), state.at)
368
+ else:
369
+ state.setreg({1:"al",2:"ax",4:"eax"}[width], value, state.at)
370
+
371
+
372
+ class Handwritten:
373
+ """The handwritten semantics backend: ``ordinary``, ``predicate`` and ``string_iteration``."""
374
+
375
+ name = "handwritten"
376
+
377
+ def ordinary(self, state, ins, image):
378
+ ordinary(state, ins, image)
379
+
380
+ def condition(self, state, mnemonic):
381
+ return predicate(state, mnemonic)
382
+
383
+ def string_iteration(self, state, ins, operation, width, source_segment, delta):
384
+ string_iteration(state, ins, operation, width, source_segment, delta)
385
+
386
+ def __deepcopy__(self, memo):
387
+ # Backends hold no path state, so every copied path shares one.
388
+ return self
389
+
390
+
391
+ semantics.register(Handwritten(), default=True)