scientific-method-engine 0.5.0__tar.gz → 0.7.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (62) hide show
  1. {scientific_method_engine-0.5.0 → scientific_method_engine-0.7.0}/PKG-INFO +8 -4
  2. {scientific_method_engine-0.5.0 → scientific_method_engine-0.7.0}/README.md +3 -2
  3. {scientific_method_engine-0.5.0 → scientific_method_engine-0.7.0}/pyproject.toml +8 -3
  4. scientific_method_engine-0.7.0/src/scientific_method_engine/x86/handwritten.py +391 -0
  5. scientific_method_engine-0.7.0/src/scientific_method_engine/x86/machine.py +404 -0
  6. scientific_method_engine-0.7.0/src/scientific_method_engine/x86/pcode.py +379 -0
  7. scientific_method_engine-0.7.0/src/scientific_method_engine/x86/pcode_backend.py +917 -0
  8. scientific_method_engine-0.7.0/src/scientific_method_engine/x86/semantics.py +77 -0
  9. {scientific_method_engine-0.5.0 → scientific_method_engine-0.7.0}/src/scientific_method_engine/x86/trace.py +17 -8
  10. scientific_method_engine-0.7.0/tests/differential.py +120 -0
  11. scientific_method_engine-0.7.0/tests/oracle.py +75 -0
  12. scientific_method_engine-0.7.0/tests/test_differential.py +71 -0
  13. {scientific_method_engine-0.5.0 → scientific_method_engine-0.7.0}/tests/test_dispatch.py +1 -1
  14. {scientific_method_engine-0.5.0 → scientific_method_engine-0.7.0}/tests/test_effect_order.py +3 -1
  15. scientific_method_engine-0.7.0/tests/test_nested_frame_request.py +61 -0
  16. scientific_method_engine-0.7.0/tests/test_oracle.py +222 -0
  17. {scientific_method_engine-0.5.0 → scientific_method_engine-0.7.0}/tests/test_pe.py +16 -1
  18. {scientific_method_engine-0.5.0 → scientific_method_engine-0.7.0}/tests/test_x86.py +55 -1
  19. scientific_method_engine-0.5.0/src/scientific_method_engine/x86/machine.py +0 -693
  20. {scientific_method_engine-0.5.0 → scientific_method_engine-0.7.0}/.gitignore +0 -0
  21. {scientific_method_engine-0.5.0 → scientific_method_engine-0.7.0}/LICENSE +0 -0
  22. {scientific_method_engine-0.5.0 → scientific_method_engine-0.7.0}/src/scientific_method_engine/__init__.py +0 -0
  23. {scientific_method_engine-0.5.0 → scientific_method_engine-0.7.0}/src/scientific_method_engine/__main__.py +0 -0
  24. {scientific_method_engine-0.5.0 → scientific_method_engine-0.7.0}/src/scientific_method_engine/cli.py +0 -0
  25. {scientific_method_engine-0.5.0 → scientific_method_engine-0.7.0}/src/scientific_method_engine/ghidra/ClearNoReturnFunctions.java +0 -0
  26. {scientific_method_engine-0.5.0 → scientific_method_engine-0.7.0}/src/scientific_method_engine/ghidra/CreateFunctions.java +0 -0
  27. {scientific_method_engine-0.5.0 → scientific_method_engine-0.7.0}/src/scientific_method_engine/ghidra/ExportBoundedFlow.java +0 -0
  28. {scientific_method_engine-0.5.0 → scientific_method_engine-0.7.0}/src/scientific_method_engine/ghidra/ExportFunctionFingerprints.java +0 -0
  29. {scientific_method_engine-0.5.0 → scientific_method_engine-0.7.0}/src/scientific_method_engine/ghidra/ExportFunctionInventory.java +0 -0
  30. {scientific_method_engine-0.5.0 → scientific_method_engine-0.7.0}/src/scientific_method_engine/ghidra/MergeFallThroughFragment.java +0 -0
  31. {scientific_method_engine-0.5.0 → scientific_method_engine-0.7.0}/src/scientific_method_engine/ghidra/RecoverCitedFunctions.java +0 -0
  32. {scientific_method_engine-0.5.0 → scientific_method_engine-0.7.0}/src/scientific_method_engine/ghidra/RepairReturningCallers.java +0 -0
  33. {scientific_method_engine-0.5.0 → scientific_method_engine-0.7.0}/src/scientific_method_engine/ghidra/ReportCallArguments.java +0 -0
  34. {scientific_method_engine-0.5.0 → scientific_method_engine-0.7.0}/src/scientific_method_engine/ghidra/ReportCallPaths.java +0 -0
  35. {scientific_method_engine-0.5.0 → scientific_method_engine-0.7.0}/src/scientific_method_engine/ghidra/ReportCallSitesWithScalars.java +0 -0
  36. {scientific_method_engine-0.5.0 → scientific_method_engine-0.7.0}/src/scientific_method_engine/ghidra/ReportCallsToRange.java +0 -0
  37. {scientific_method_engine-0.5.0 → scientific_method_engine-0.7.0}/src/scientific_method_engine/ghidra/ReportConstantFirstArgumentCalls.java +0 -0
  38. {scientific_method_engine-0.5.0 → scientific_method_engine-0.7.0}/src/scientific_method_engine/ghidra/ReportDataBytes.java +0 -0
  39. {scientific_method_engine-0.5.0 → scientific_method_engine-0.7.0}/src/scientific_method_engine/ghidra/ReportDecompileMatches.java +0 -0
  40. {scientific_method_engine-0.5.0 → scientific_method_engine-0.7.0}/src/scientific_method_engine/ghidra/ReportDecompileWindow.java +0 -0
  41. {scientific_method_engine-0.5.0 → scientific_method_engine-0.7.0}/src/scientific_method_engine/ghidra/ReportFilePatternInMemory.java +0 -0
  42. {scientific_method_engine-0.5.0 → scientific_method_engine-0.7.0}/src/scientific_method_engine/ghidra/ReportFirstArgumentCallSummary.java +0 -0
  43. {scientific_method_engine-0.5.0 → scientific_method_engine-0.7.0}/src/scientific_method_engine/ghidra/ReportFunctionScalarConstants.java +0 -0
  44. {scientific_method_engine-0.5.0 → scientific_method_engine-0.7.0}/src/scientific_method_engine/ghidra/ReportFunctionSummary.java +0 -0
  45. {scientific_method_engine-0.5.0 → scientific_method_engine-0.7.0}/src/scientific_method_engine/ghidra/ReportInstructionContext.java +0 -0
  46. {scientific_method_engine-0.5.0 → scientific_method_engine-0.7.0}/src/scientific_method_engine/ghidra/ReportInstructionWindow.java +0 -0
  47. {scientific_method_engine-0.5.0 → scientific_method_engine-0.7.0}/src/scientific_method_engine/ghidra/ReportMemoryBlockForFileOffset.java +0 -0
  48. {scientific_method_engine-0.5.0 → scientific_method_engine-0.7.0}/src/scientific_method_engine/ghidra/ReportMemoryBlocks.java +0 -0
  49. {scientific_method_engine-0.5.0 → scientific_method_engine-0.7.0}/src/scientific_method_engine/ghidra/ReportRandomnessCandidates.java +0 -0
  50. {scientific_method_engine-0.5.0 → scientific_method_engine-0.7.0}/src/scientific_method_engine/ghidra/ReportReferences.java +0 -0
  51. {scientific_method_engine-0.5.0 → scientific_method_engine-0.7.0}/src/scientific_method_engine/ghidra/ReportScalarConstants.java +0 -0
  52. {scientific_method_engine-0.5.0 → scientific_method_engine-0.7.0}/src/scientific_method_engine/ghidra/ReportStringReferences.java +0 -0
  53. {scientific_method_engine-0.5.0 → scientific_method_engine-0.7.0}/src/scientific_method_engine/ghidra/ReportSymbolReferences.java +0 -0
  54. {scientific_method_engine-0.5.0 → scientific_method_engine-0.7.0}/src/scientific_method_engine/x86/__init__.py +0 -0
  55. {scientific_method_engine-0.5.0 → scientific_method_engine-0.7.0}/src/scientific_method_engine/x86/dispatch.py +0 -0
  56. {scientific_method_engine-0.5.0 → scientific_method_engine-0.7.0}/src/scientific_method_engine/x86/effect_order.py +0 -0
  57. {scientific_method_engine-0.5.0 → scientific_method_engine-0.7.0}/src/scientific_method_engine/x86/image.py +0 -0
  58. {scientific_method_engine-0.5.0 → scientific_method_engine-0.7.0}/src/scientific_method_engine/x86/pe.py +0 -0
  59. {scientific_method_engine-0.5.0 → scientific_method_engine-0.7.0}/src/scientific_method_engine/x86/reports.py +0 -0
  60. {scientific_method_engine-0.5.0 → scientific_method_engine-0.7.0}/src/scientific_method_engine/x86/result_flow.py +0 -0
  61. {scientific_method_engine-0.5.0 → scientific_method_engine-0.7.0}/src/scientific_method_engine/x86/values.py +0 -0
  62. {scientific_method_engine-0.5.0 → scientific_method_engine-0.7.0}/tests/test_table_continuations.py +0 -0
@@ -1,20 +1,24 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: scientific-method-engine
3
- Version: 0.5.0
3
+ Version: 0.7.0
4
4
  Summary: Bounded instruction-derived x86 evidence reports for segmented MZ/FBOV and PE32/i386 code.
5
5
  Project-URL: Source, https://github.com/kibertoad/refurbished-dinosaurs-toolkit/tree/main/packages/scientific-method-engine
6
6
  Author: kibertoad
7
7
  License-Expression: MIT
8
8
  License-File: LICENSE
9
- Requires-Python: >=3.10
9
+ Requires-Python: >=3.12
10
10
  Requires-Dist: capstone==5.0.7
11
+ Requires-Dist: pypcode==4.0.0
12
+ Provides-Extra: test
13
+ Requires-Dist: unicorn==2.1.4; extra == 'test'
11
14
  Description-Content-Type: text/markdown
12
15
 
13
16
  # scientific-method-engine
14
17
 
15
18
  Bounded instruction-derived x86 evidence reports for segmented 16-bit MZ/FBOV code and
16
- PE32/i386 code. The engine decodes instructions with Capstone, follows bounded paths and emits
17
- `bounded-x86-v1` JSON. It never runs the original program.
19
+ PE32/i386 code. The engine decodes instructions with Capstone, takes their values, flags and
20
+ branch conditions from Ghidra's SLEIGH specification through pypcode, follows bounded paths and
21
+ emits `bounded-x86-v1` JSON. It never runs the original program. It needs Python 3.12 or later.
18
22
 
19
23
  ```sh
20
24
  uv add --group research scientific-method-engine # or: pip install scientific-method-engine
@@ -1,8 +1,9 @@
1
1
  # scientific-method-engine
2
2
 
3
3
  Bounded instruction-derived x86 evidence reports for segmented 16-bit MZ/FBOV code and
4
- PE32/i386 code. The engine decodes instructions with Capstone, follows bounded paths and emits
5
- `bounded-x86-v1` JSON. It never runs the original program.
4
+ PE32/i386 code. The engine decodes instructions with Capstone, takes their values, flags and
5
+ branch conditions from Ghidra's SLEIGH specification through pypcode, follows bounded paths and
6
+ emits `bounded-x86-v1` JSON. It never runs the original program. It needs Python 3.12 or later.
6
7
 
7
8
  ```sh
8
9
  uv add --group research scientific-method-engine # or: pip install scientific-method-engine
@@ -5,13 +5,18 @@ build-backend = "hatchling.build"
5
5
  [project]
6
6
  name = "scientific-method-engine"
7
7
  # The release workflow writes the published version from the package's release tag.
8
- version = "0.5.0"
8
+ version = "0.7.0"
9
9
  description = "Bounded instruction-derived x86 evidence reports for segmented MZ/FBOV and PE32/i386 code."
10
10
  readme = "README.md"
11
- requires-python = ">=3.10"
11
+ requires-python = ">=3.12"
12
12
  license = "MIT"
13
13
  authors = [{ name = "kibertoad" }]
14
- dependencies = ["capstone==5.0.7"]
14
+ dependencies = ["capstone==5.0.7", "pypcode==4.0.0"]
15
+
16
+ [project.optional-dependencies]
17
+ # Unicorn is the tests' concrete oracle (ADR 0003, decision 4). Its core is GPLv2, so it never
18
+ # becomes a runtime dependency.
19
+ test = ["unicorn==2.1.4"]
15
20
 
16
21
  [project.scripts]
17
22
  scientific-method-engine = "scientific_method_engine.cli:run"
@@ -0,0 +1,391 @@
1
+ """The handwritten semantics backend (ADR 0003): value computation, flag predicates and string bodies.
2
+
3
+ ADR 0003 decision 6 freezes this module. No mnemonic, flag rule or value computation is added here;
4
+ the pypcode backend replaces it group by group, and phase 5 deletes it.
5
+ """
6
+ from capstone.x86 import X86_OP_IMM, X86_OP_REG, X86_OP_MEM
7
+ from . import semantics
8
+ from .machine import ALIASES, StopPath
9
+ from .values import Value, const, unknown, op, extract, join, resize, sources
10
+
11
+ CARRY_BRANCHES = {"jb": True, "jc": True, "jnae": True, "jae": False, "jnb": False, "jnc": False}
12
+ # Branches taken when CF or OF is set; logic operations clear both whatever their operands.
13
+ CLEARED_BY_LOGIC = {**CARRY_BRANCHES, "jo": True, "jno": False}
14
+
15
+
16
+ def predicate(state, mnemonic):
17
+ flags = state.flags
18
+ if flags is None and state.carry is not None and mnemonic in CARRY_BRANCHES:
19
+ info = {"predicate": mnemonic, "flag": "CF", "carry": state.carry.report()}
20
+ if state.carry.number is None:
21
+ return None, {**info, "reason": "carry unresolved"}
22
+ return bool(state.carry.number) == CARRY_BRANCHES[mnemonic], info
23
+ if flags is None:
24
+ return None, {"predicate": mnemonic, "reason": "flag producer unresolved",
25
+ "flagProducer": state.unknown_flag_site, "flagGeneration": state.flag_epoch}
26
+ a, b, operation, site = flags
27
+ info = {"predicate": mnemonic, "flagProducer": site, "operation": operation,
28
+ "left": a.report(), "right": b.report()}
29
+ if a.number is not None and b.number is not None:
30
+ x, y = a.number, b.number
31
+ elif operation in ("cmp", "sub", "xor") and a.term == b.term:
32
+ # Any value compared with, subtracted from or XORed with itself yields zero.
33
+ x = y = 0
34
+ elif operation in ("test", "and", "or", "xor") and mnemonic in CLEARED_BY_LOGIC:
35
+ return not CLEARED_BY_LOGIC[mnemonic], info
36
+ else:
37
+ return None, info
38
+ bits = a.bits
39
+ if operation in ("cmp", "sub"):
40
+ raw = x - y
41
+ result = raw % (1 << bits)
42
+ cf = x < y
43
+ of = bool(((x ^ y) & (x ^ result)) & (1 << (bits - 1)))
44
+ elif operation == "add":
45
+ raw = x + y
46
+ result = raw % (1 << bits)
47
+ cf = raw >= 1 << bits
48
+ of = bool((~(x ^ y) & (x ^ result)) & (1 << (bits - 1)))
49
+ elif operation in ("test", "and", "or", "xor"):
50
+ result = {"test": x & y, "and": x & y, "or": x | y, "xor": x ^ y}[operation]
51
+ cf = of = False
52
+ else:
53
+ return None, info
54
+ zf, sf = result == 0, bool(result & (1 << (bits - 1)))
55
+ conditions = {"je": zf, "jz": zf, "jne": not zf, "jnz": not zf,
56
+ "jb": cf, "jc": cf, "jnae": cf, "jae": not cf, "jnb": not cf, "jnc": not cf,
57
+ "jbe": cf or zf, "jna": cf or zf, "ja": not cf and not zf, "jnbe": not cf and not zf,
58
+ "jl": sf != of, "jnge": sf != of, "jge": sf == of, "jnl": sf == of,
59
+ "jle": zf or sf != of, "jng": zf or sf != of, "jg": not zf and sf == of,
60
+ "jnle": not zf and sf == of, "js": sf, "jns": not sf, "jo": of, "jno": not of}
61
+ return conditions.get(mnemonic), info
62
+
63
+
64
+ def ordinary(state, ins, image):
65
+ m, operands = ins.mnemonic, ins.operands
66
+ if m in ("cld", "std", "cli", "sti"):
67
+ value = const(1 if m in ("std", "sti") else 0, 1, state.at)
68
+ flag = "DF" if m in ("cld", "std") else "IF"
69
+ if flag == "DF": state.direction_flag = value
70
+ else: state.interrupt_flag = value
71
+ state.event("flag-write", flag=flag, value=value.report(),
72
+ interpretation="local flag effect only; interrupts and timing are not simulated")
73
+ return
74
+ if m in ("pushf", "pushfd", "popf", "popfd"):
75
+ bits = 32 if (0x66 in ins.prefix) != state.flat else 16
76
+ if m.startswith("push"): state.save_flags(bits)
77
+ else: state.restore_flags(bits)
78
+ return
79
+ if m == "nop":
80
+ return
81
+ if m in ("mov", "movzx", "movsx"):
82
+ if state.flat and operands[0].type == X86_OP_REG and ins.reg_name(operands[0].reg) in state.segment_bases:
83
+ raise StopPath("Segment selector assignment requires a descriptor model")
84
+ value = state.get(ins, operands[1], image)
85
+ result = resize(value, operands[0].size * 8, signed=m == "movsx")
86
+ state.put(ins, operands[0], result)
87
+ if not state.value_transfers:
88
+ return
89
+
90
+ def location(operand):
91
+ return {"kind": "register", "register": ins.reg_name(operand.reg)} if operand.type == X86_OP_REG else {"kind": "memory"} if operand.type == X86_OP_MEM else {"kind": "immediate"}
92
+ destination_container = ALIASES[ins.reg_name(operands[0].reg)][0] if operands[0].type == X86_OP_REG else None
93
+ state.event("value-transfer", operation=m, source=location(operands[1]), destination=location(operands[0]),
94
+ destinationContainer=destination_container, destinationContainerValue=state.reg(destination_container).report() if destination_container else None,
95
+ sourceBits=value.bits, destinationBits=result.bits, sourceValue=value.report(), resultValue=result.report(),
96
+ conversion="truncate" if result.bits < value.bits else "signExtend" if m == "movsx" else "zeroExtend" if result.bits > value.bits else "sameWidth")
97
+ return
98
+ if m == "xchg":
99
+ values = [state.get(ins, operand, image) for operand in operands]
100
+ addresses = [state.address(ins, operand) if operand.type == X86_OP_MEM else None for operand in operands]
101
+ for index, operand in enumerate(operands):
102
+ value = values[1-index]
103
+ if addresses[index] is None:
104
+ state.put(ins, operand, value)
105
+ else:
106
+ segment, offset, register = addresses[index]
107
+ state.access(segment, offset, operand.size, resize(value, operand.size * 8), addressing_register=register)
108
+ return
109
+ if m == "imul" and len(operands) in (2, 3):
110
+ left, right = (state.get(ins, operand, image) for operand in (operands if len(operands) == 2 else operands[1:]))
111
+ left = resize(left, operands[0].size * 8)
112
+ right = resize(right, left.bits, signed=True)
113
+ result = op("mul", left, right, state.at)
114
+ state.put(ins, operands[0], result)
115
+ state.forget_flags() # CF/OF require the full signed product; other flags are undefined.
116
+ state.event("arithmetic", operation="imul", left=left.report(), right=right.report(),
117
+ result=result.report(), modulus=1 << left.bits, flags="unresolved signed-product overflow")
118
+ return
119
+ if m == "lea":
120
+ segment, offset, register = state.address(ins, operands[1])
121
+ state.put(ins, operands[0], offset)
122
+ state.event("address-formation", value=offset.report(), addressingSegment=segment.report(),
123
+ addressingSegmentRegister=register, destinationRegister=ins.reg_name(operands[0].reg),
124
+ note="LEA does not access memory; this addressing default does not bind a later dereference")
125
+ return
126
+ if m in ("lds", "les"):
127
+ if state.flat:
128
+ raise StopPath("Descriptor loads are outside the PE32 flat model")
129
+ segment, offset, register = state.address(ins, operands[1])
130
+ if operands[0].size != 2:
131
+ raise StopPath("Only 16:16 pointer loads are supported")
132
+ value = state.access(segment, offset, 4, role="far-pointer", addressing_register=register)
133
+ state.put(ins, operands[0], extract(value, 0, 16))
134
+ state.setreg("ds" if m == "lds" else "es", extract(value, 16, 16), state.at)
135
+ return
136
+ if m == "push":
137
+ if state.flat and operands[0].type == X86_OP_REG and ins.reg_name(operands[0].reg) in state.segment_bases:
138
+ raise StopPath("Segment stack operations require a descriptor model")
139
+ state.push(state.get(ins, operands[0], image))
140
+ return
141
+ if m == "pop":
142
+ if state.flat and operands[0].type == X86_OP_REG and ins.reg_name(operands[0].reg) in state.segment_bases:
143
+ raise StopPath("Segment selector assignment requires a descriptor model")
144
+ state.put(ins, operands[0], state.pop(operands[0].size))
145
+ return
146
+ if m == "leave":
147
+ if 0x66 in ins.prefix:
148
+ raise StopPath("Operand-size override on LEAVE is unsupported")
149
+ state.setreg(state.sp, state.reg(state.bp), state.at)
150
+ state.setreg(state.bp, state.pop(state.bits // 8), state.at)
151
+ return
152
+ if m in ("cmp", "test"):
153
+ a, b = (state.get(ins, o, image) for o in operands)
154
+ state.set_flags(a, resize(b, a.bits), m)
155
+ state.event("compare", operation=m, left=a.report(), right=b.report())
156
+ return
157
+ if m in ("add", "sub", "and", "or", "xor", "shl", "sal", "shr", "sar"):
158
+ a, b = (state.get(ins, o, image) for o in operands)
159
+ b = resize(b, a.bits)
160
+ result = op("shl" if m == "sal" else m, a, b, state.at)
161
+ state.put(ins, operands[0], result)
162
+ if m in ("add", "sub", "and", "or", "xor"):
163
+ state.set_flags(a, b, m)
164
+ else:
165
+ shift_carry(state, m, a, b)
166
+ state.event("arithmetic", operation=m, left=a.report(), right=b.report(), result=result.report(), modulus=1 << a.bits)
167
+ return
168
+ if m in ("inc", "dec"):
169
+ a = state.get(ins, operands[0], image)
170
+ result = op("add" if m == "inc" else "sub", a, const(1, a.bits), state.at)
171
+ state.put(ins, operands[0], result)
172
+ # Carry is preserved; the other flags are not modeled for INC/DEC.
173
+ state.forget_flags(keep_carry=True)
174
+ return
175
+ if m in ("cbw", "cwde"):
176
+ # Capstone 5 names these inconsistently in 16-bit mode. Use effective size.
177
+ # The prefix toggles the mode's default operand size (16-bit real mode, 32-bit flat).
178
+ wide = (0x66 in ins.prefix) != state.flat
179
+ source, destination = ("ax", "eax") if wide else ("al", "ax")
180
+ source_value = state.reg(source)
181
+ value = resize(source_value, 32 if wide else 16, True)
182
+ state.setreg(destination, value, state.at)
183
+ state.event("conversion", sourceRegister=source, destinationRegister=destination,
184
+ effectiveOperandBits=32 if wide else 16, decoderMnemonic=m,
185
+ mnemonicWidthMismatch=m != ("cwde" if wide else "cbw"), result=value.report(),
186
+ sourceValue=source_value.report(), sourceBits=source_value.bits, destinationBits=value.bits, conversion="signExtend")
187
+ return
188
+ if m in ("cwd", "cdq"):
189
+ wide = (0x66 in ins.prefix) != state.flat
190
+ source, destination = ("eax", "edx") if wide else ("ax", "dx")
191
+ bits = 32 if wide else 16
192
+ source_value = state.reg(source)
193
+ value = resize(extract(source_value, bits-1, 1), bits, signed=True)
194
+ state.setreg(destination, value, state.at)
195
+ state.event("conversion", sourceRegister=source, destinationRegister=destination,
196
+ effectiveOperandBits=bits, decoderMnemonic=m,
197
+ mnemonicWidthMismatch=m != ("cdq" if wide else "cwd"), result=value.report(),
198
+ sourceValue=source_value.report(), sourceBits=source_value.bits, destinationBits=value.bits, conversion="signFillHighHalf")
199
+ return
200
+ if m in ("clc", "stc", "cmc"):
201
+ if m == "cmc":
202
+ value = op("xor", state.carry_value(), const(1, 1), state.at)
203
+ else:
204
+ value = const(int(m == "stc"), 1, state.at)
205
+ state.forget_flags()
206
+ state.carry = value
207
+ state.event("flag-write", flag="CF", value=value.report(), interpretation="local carry effect")
208
+ return
209
+ if m in ("not", "neg"):
210
+ a = state.get(ins, operands[0], image)
211
+ if m == "not":
212
+ state.put(ins, operands[0], op("xor", a, const((1 << a.bits) - 1, a.bits), state.at))
213
+ return
214
+ result = op("sub", const(0, a.bits), a, state.at)
215
+ state.put(ins, operands[0], result)
216
+ state.set_flags(const(0, a.bits, state.at), a, "sub")
217
+ state.event("arithmetic", operation="neg", left=a.report(), result=result.report(), modulus=1 << a.bits)
218
+ return
219
+ if m in ("adc", "sbb"):
220
+ a, b = (state.get(ins, o, image) for o in operands)
221
+ b = resize(b, a.bits)
222
+ carry = state.carry_value()
223
+ name = "add" if m == "adc" else "sub"
224
+ result = op(name, op(name, a, b, state.at), resize(carry, a.bits), state.at)
225
+ state.put(ins, operands[0], result)
226
+ state.forget_flags()
227
+ if None not in (a.number, b.number, carry.number):
228
+ raw = a.number + b.number + carry.number if m == "adc" else a.number - b.number - carry.number
229
+ state.carry = const(int(raw < 0 or raw >= 1 << a.bits), 1, state.at)
230
+ else:
231
+ state.carry = unknown(f"carry:{state.at}:{state.flag_serial}", 1, state.at)
232
+ state.event("arithmetic", operation=m, left=a.report(), right=b.report(), carryIn=carry.report(),
233
+ result=result.report(), carryOut=state.carry.report(), modulus=1 << a.bits)
234
+ return
235
+ if m in ("rol", "ror", "rcl", "rcr"):
236
+ a = state.get(ins, operands[0], image)
237
+ count = state.get(ins, operands[1], image) if len(operands) > 1 else const(1, 8)
238
+ if len(operands) > 1 and operands[1].type == X86_OP_IMM:
239
+ # Capstone gives RCL's implicit count of 1 on a memory operand a size of 0.
240
+ count = const(operands[1].imm, 8, state.at)
241
+ if count.number is None:
242
+ raise StopPath("rotate count unresolved")
243
+ masked = count.number & 31
244
+ if masked == 0:
245
+ return # The value and flags are unchanged.
246
+ bits = a.bits
247
+ through = m in ("rcl", "rcr")
248
+ n = masked % (bits + 1 if through else bits)
249
+ if through and n == 0:
250
+ # A full rotation through CF restores the value and CF; OF is undefined.
251
+ state.forget_flags(keep_carry=True)
252
+ return
253
+ carry_in = resize(state.carry_value(), bits)
254
+ # Each form is an OR of shifted copies (positive shifts left), built once so the
255
+ # expression does not repeat the operand for every bit rotated.
256
+ parts = {"rol": [(a, n), (a, n - bits)],
257
+ "ror": [(a, -n), (a, bits - n)],
258
+ "rcl": [(a, n), (carry_in, n - 1), (a, n - bits - 1)],
259
+ "rcr": [(a, -n), (carry_in, bits - n), (a, bits + 1 - n)]}[m]
260
+ value = None
261
+ for part, shift in parts:
262
+ if abs(shift) >= bits:
263
+ continue # Every bit leaves the operand; op() would mask the count instead.
264
+ if shift:
265
+ part = op("shl" if shift > 0 else "shr", part, const(abs(shift), bits), state.at)
266
+ value = part if value is None else op("or", value, part, state.at)
267
+ # CF is the last bit rotated out: the result's low bit for ROL/RCL and its high bit for ROR/RCR.
268
+ out = {"rol": (bits - n) % bits, "ror": (n - 1) % bits, "rcl": bits - n, "rcr": n - 1}[m]
269
+ carry = extract(a, out, 1)
270
+ # A count read from CL is an input of the result and of CF.
271
+ value = Value(value.bits, value.term, sources(value, count, site=state.at))
272
+ state.put(ins, operands[0], value)
273
+ state.forget_flags()
274
+ state.carry = Value(1, carry.term, sources(carry, count, site=state.at))
275
+ state.event("arithmetic", operation=m, left=a.report(), count=n, result=value.report(),
276
+ carryOut=state.carry.report(), modulus=1 << bits)
277
+ return
278
+ if m in ("mul", "imul") and len(operands) == 1:
279
+ source = state.get(ins, operands[0], image)
280
+ bits = source.bits
281
+ low_reg, high_reg = {8: ("al", "ah"), 16: ("ax", "dx"), 32: ("eax", "edx")}[bits]
282
+ signed = m == "imul"
283
+ multiplicand = state.reg(low_reg)
284
+ product = op("mul", resize(multiplicand, 2 * bits, signed), resize(source, 2 * bits, signed), state.at)
285
+ low = extract(product, 0, bits)
286
+ if bits == 8:
287
+ state.setreg("ax", product, state.at)
288
+ else:
289
+ state.setreg(low_reg, low, state.at)
290
+ state.setreg(high_reg, extract(product, bits, bits), state.at)
291
+ state.forget_flags()
292
+ if product.number is None:
293
+ state.carry = unknown(f"carry:{state.at}:{state.flag_serial}", 1, state.at)
294
+ else:
295
+ # CF and OF say whether the high half carries information beyond the low half.
296
+ state.carry = const(int(resize(low, 2 * bits, signed).number != product.number), 1, state.at)
297
+ state.event("arithmetic", operation=m, left=multiplicand.report(), right=source.report(),
298
+ result=product.report(), resultBits=2 * bits, carryOut=state.carry.report())
299
+ return
300
+ if m in ("div", "idiv"):
301
+ divisor = state.get(ins, operands[0], image)
302
+ bits = divisor.bits
303
+ signed = m == "idiv"
304
+ if bits == 8:
305
+ dividend = state.reg("ax")
306
+ else:
307
+ dividend = join([state.reg({16: "ax", 32: "eax"}[bits]), state.reg({16: "dx", 32: "edx"}[bits])])
308
+ quotient_reg, remainder_reg = {8: ("al", "ah"), 16: ("ax", "dx"), 32: ("eax", "edx")}[bits]
309
+ fault = None
310
+ if divisor.number == 0:
311
+ raise StopPath("divide by zero raises interrupt 0; its handler is not modeled")
312
+ if None not in (dividend.number, divisor.number):
313
+ x, y = dividend.number, divisor.number
314
+ if signed:
315
+ x -= (x >> (2 * bits - 1)) << (2 * bits)
316
+ y -= (y >> (bits - 1)) << bits
317
+ q = abs(x) // abs(y) * (1 if (x < 0) == (y < 0) else -1)
318
+ r = x - q * y
319
+ if not ((-(1 << (bits - 1)) <= q < 1 << (bits - 1)) if signed else q < 1 << bits):
320
+ raise StopPath("divide overflow raises interrupt 0; its handler is not modeled")
321
+ # Both results are computed from the dividend and divisor, so they name their producers.
322
+ origin = sources(dividend, divisor, site=state.at)
323
+ quotient, remainder = Value(bits, const(q, bits).term, origin), Value(bits, const(r, bits).term, origin)
324
+ else:
325
+ wide = resize(divisor, 2 * bits, signed)
326
+ origin = sources(dividend, divisor, site=state.at)
327
+ quotient = extract(Value(2 * bits, ("sdiv" if signed else "udiv", dividend.term, wide.term), origin), 0, bits)
328
+ remainder = extract(Value(2 * bits, ("smod" if signed else "umod", dividend.term, wide.term), origin), 0, bits)
329
+ fault = "possible divide error (interrupt 0) unresolved; this path assumes none"
330
+ state.conditional.append({"site": state.at, "assumption": "no divide error"})
331
+ state.setreg(quotient_reg, quotient, state.at)
332
+ state.setreg(remainder_reg, remainder, state.at)
333
+ state.forget_flags()
334
+ state.event("arithmetic", operation=m, dividend=dividend.report(), divisor=divisor.report(),
335
+ quotient=quotient.report(), remainder=remainder.report(), fault=fault)
336
+ return
337
+ raise StopPath("Unsupported instruction semantics: " + m)
338
+
339
+
340
+ def shift_carry(state, m, a, count):
341
+ """CF after SHL/SHR/SAR: the last bit shifted out, when the count is known."""
342
+ n = None if count.number is None else count.number & 31
343
+ if n == 0:
344
+ return # A zero count leaves every flag unchanged.
345
+ state.forget_flags()
346
+ if n is None or n > a.bits:
347
+ state.carry = unknown(f"carry:{state.at}:{state.flag_serial}", 1, state.at)
348
+ elif m in ("shl", "sal"):
349
+ state.carry = Value(1, extract(a, a.bits - n, 1).term, sources(a, site=state.at))
350
+ else:
351
+ state.carry = Value(1, extract(a, n - 1, 1).term, sources(a, site=state.at))
352
+
353
+
354
+ def string_iteration(state, ins, operation, width, source_name, delta):
355
+ if operation in ("cmps", "scas"):
356
+ raise StopPath("Unsupported instruction semantics: " + ins.mnemonic)
357
+ si, di = ("esi", "edi") if state.flat else ("si", "di")
358
+ # DF chooses the step's sign, so the instruction that set DF is one of the step's producers.
359
+ step = Value(state.bits, const(delta, state.bits).term, state.direction_flag.sources)
360
+ if operation in ("movs", "lods"):
361
+ value = state.access(state.segment(source_name), state.reg(si), width, role="string-source", addressing_register=source_name)
362
+ state.setreg(si, op("add", state.reg(si), step, state.at), state.at)
363
+ else:
364
+ value = state.reg({1:"al",2:"ax",4:"eax"}[width])
365
+ if operation in ("movs", "stos"):
366
+ state.access(state.segment("es"), state.reg(di), width, value, role="string-destination", addressing_register="es")
367
+ state.setreg(di, op("add", state.reg(di), step, state.at), state.at)
368
+ else:
369
+ state.setreg({1:"al",2:"ax",4:"eax"}[width], value, state.at)
370
+
371
+
372
+ class Handwritten:
373
+ """The handwritten semantics backend: ``ordinary``, ``predicate`` and ``string_iteration``."""
374
+
375
+ name = "handwritten"
376
+
377
+ def ordinary(self, state, ins, image):
378
+ ordinary(state, ins, image)
379
+
380
+ def condition(self, state, mnemonic):
381
+ return predicate(state, mnemonic)
382
+
383
+ def string_iteration(self, state, ins, operation, width, source_segment, delta):
384
+ string_iteration(state, ins, operation, width, source_segment, delta)
385
+
386
+ def __deepcopy__(self, memo):
387
+ # Backends hold no path state, so every copied path shares one.
388
+ return self
389
+
390
+
391
+ semantics.register(Handwritten(), default=True)