scientific-method-engine 0.1.0__tar.gz → 0.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/PKG-INFO +1 -1
  2. {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/pyproject.toml +1 -1
  3. {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/cli.py +1 -1
  4. {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/x86/machine.py +43 -9
  5. {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/x86/reports.py +205 -2
  6. scientific_method_engine-0.3.0/src/scientific_method_engine/x86/result_flow.py +131 -0
  7. {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/x86/trace.py +11 -33
  8. {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/x86/values.py +20 -2
  9. {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/tests/test_x86.py +288 -0
  10. {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/.gitignore +0 -0
  11. {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/LICENSE +0 -0
  12. {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/README.md +0 -0
  13. {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/__init__.py +0 -0
  14. {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/__main__.py +0 -0
  15. {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/ghidra/ClearNoReturnFunctions.java +0 -0
  16. {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/ghidra/CreateFunctions.java +0 -0
  17. {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/ghidra/ExportBoundedFlow.java +0 -0
  18. {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/ghidra/ExportFunctionFingerprints.java +0 -0
  19. {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/ghidra/ExportFunctionInventory.java +0 -0
  20. {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/ghidra/MergeFallThroughFragment.java +0 -0
  21. {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/ghidra/RecoverCitedFunctions.java +0 -0
  22. {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/ghidra/RepairReturningCallers.java +0 -0
  23. {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/ghidra/ReportCallArguments.java +0 -0
  24. {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/ghidra/ReportCallPaths.java +0 -0
  25. {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/ghidra/ReportCallSitesWithScalars.java +0 -0
  26. {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/ghidra/ReportCallsToRange.java +0 -0
  27. {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/ghidra/ReportConstantFirstArgumentCalls.java +0 -0
  28. {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/ghidra/ReportDataBytes.java +0 -0
  29. {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/ghidra/ReportDecompileMatches.java +0 -0
  30. {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/ghidra/ReportDecompileWindow.java +0 -0
  31. {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/ghidra/ReportFilePatternInMemory.java +0 -0
  32. {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/ghidra/ReportFirstArgumentCallSummary.java +0 -0
  33. {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/ghidra/ReportFunctionScalarConstants.java +0 -0
  34. {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/ghidra/ReportFunctionSummary.java +0 -0
  35. {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/ghidra/ReportInstructionContext.java +0 -0
  36. {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/ghidra/ReportInstructionWindow.java +0 -0
  37. {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/ghidra/ReportMemoryBlockForFileOffset.java +0 -0
  38. {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/ghidra/ReportMemoryBlocks.java +0 -0
  39. {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/ghidra/ReportRandomnessCandidates.java +0 -0
  40. {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/ghidra/ReportReferences.java +0 -0
  41. {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/ghidra/ReportScalarConstants.java +0 -0
  42. {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/ghidra/ReportStringReferences.java +0 -0
  43. {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/ghidra/ReportSymbolReferences.java +0 -0
  44. {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/x86/__init__.py +0 -0
  45. {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/x86/dispatch.py +0 -0
  46. {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/x86/image.py +0 -0
  47. {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/src/scientific_method_engine/x86/pe.py +0 -0
  48. {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/tests/test_dispatch.py +0 -0
  49. {scientific_method_engine-0.1.0 → scientific_method_engine-0.3.0}/tests/test_pe.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: scientific-method-engine
3
- Version: 0.1.0
3
+ Version: 0.3.0
4
4
  Summary: Bounded instruction-derived x86 evidence reports for segmented MZ/FBOV and PE32/i386 code.
5
5
  Project-URL: Source, https://github.com/kibertoad/refurbished-dinosaurs-toolkit/tree/main/packages/scientific-method-engine
6
6
  Author: kibertoad
@@ -5,7 +5,7 @@ build-backend = "hatchling.build"
5
5
  [project]
6
6
  name = "scientific-method-engine"
7
7
  # The release workflow writes the published version from the package's release tag.
8
- version = "0.1.0"
8
+ version = "0.3.0"
9
9
  description = "Bounded instruction-derived x86 evidence reports for segmented MZ/FBOV and PE32/i386 code."
10
10
  readme = "README.md"
11
11
  requires-python = ">=3.10"
@@ -8,7 +8,7 @@ from . import PREPARED_PROTOCOL
8
8
  CONFIG_LIMIT = 1024 * 1024
9
9
  PREPARED_CONFIG_LIMIT = 16 * 1024 * 1024
10
10
  USAGE = ("Usage: scientific-method-engine <operand|operand-candidates|target|bounds|owner|callees|trace|uses|arguments|"
11
- "effects|returns|memory|incoming|guards|allocation|dispatch> <config.json|->\n"
11
+ "effects|returns|memory|incoming|call-order|guards|allocation|dispatch> <config.json|->\n"
12
12
  " scientific-method-engine ghidra-scripts")
13
13
 
14
14
 
@@ -1,7 +1,7 @@
1
1
  """Path-specific instruction effects. Unsupported semantics stop the path."""
2
2
  from copy import deepcopy
3
3
  from capstone.x86 import X86_OP_REG, X86_OP_IMM, X86_OP_MEM
4
- from .values import Value, const, unknown, op, extract, join, resize, sources, address_parts
4
+ from .values import Value, const, unknown, op, extract, join, resize, sources, address_parts, producers
5
5
 
6
6
  REGISTERS = ("eax", "ebx", "ecx", "edx", "esi", "edi", "ebp", "esp", "cs", "ds", "es", "ss", "fs", "gs")
7
7
  ALIASES = {}
@@ -41,6 +41,8 @@ class State:
41
41
  self.sp, self.bp = ("esp", "ebp") if self.flat else ("sp", "bp")
42
42
  self.at = entry
43
43
  self.regs = {r: unknown("initial:" + r, ALIASES[r][2]) for r in REGISTERS}
44
+ # Producers per register byte, so a partial write replaces only the bytes it stores.
45
+ self.reg_sources = {r: [()] * (ALIASES[r][2] // 8) for r in REGISTERS}
44
46
  self.regs["esp"] = resize(unknown("entry:sp", self.bits), 32)
45
47
  self.setreg(self.sp, unknown("entry:sp", self.bits), None)
46
48
  self.segment_bases = {r: const(0, 32) if r in ("cs", "ds", "es", "ss") else unknown("initial-base:" + r, 32)
@@ -55,6 +57,8 @@ class State:
55
57
  self.memory_groups = {}
56
58
  self.memory_epoch = 0
57
59
  self.events = []
60
+ # Value transfers are recorded only for queries that trace declared return results.
61
+ self.value_transfers = bool(config.get("returnContracts"))
58
62
  self.guards = []
59
63
  self.assumptions = {}
60
64
  self.flags = None
@@ -134,7 +138,8 @@ class State:
134
138
 
135
139
  def reg(self, name):
136
140
  root, low, bits = alias(name)
137
- return extract(self.regs[root], low, bits)
141
+ value = extract(self.regs[root], low, bits)
142
+ return Value(bits, value.term, tuple(sorted(set().union(*self.reg_sources[root][low // 8:(low + bits) // 8]))))
138
143
 
139
144
  def setreg(self, name, value, site):
140
145
  root, low, bits = alias(name)
@@ -147,6 +152,7 @@ class State:
147
152
  chunks = [extract(old, n, 8) for n in range(0, old.bits, 8)]
148
153
  chunks[low // 8:(low + bits) // 8] = [extract(value, n, 8) for n in range(0, bits, 8)]
149
154
  self.regs[root] = join(chunks)
155
+ self.reg_sources[root][low // 8:(low + bits) // 8] = [value.sources] * (bits // 8)
150
156
 
151
157
  def event(self, kind, **fields):
152
158
  event = {"kind": kind, "site": self.at, "entry": self.frames[-1]["entry"],
@@ -225,7 +231,7 @@ class State:
225
231
  event = self.event("write" if write is not None else "read", segment=segment.report(), offset=offset.report(),
226
232
  width=width, effectiveSegmentRegister=addressing_register, segmentInterpretation="base" if self.flat else "selector-paragraph", interval={"segment": seg, "base": base, "start": delta, "end": delta + width},
227
233
  value=value.report(), missingByteProducers=missing,
228
- byteProducers=[{"index": i, "producers": list(self.memory[key].sources) if key in self.memory else []} for i, key in enumerate(keys)],
234
+ byteProducers=[{"index": i, "producers": producers(self.memory[key]) if key in self.memory else []} for i, key in enumerate(keys)],
229
235
  guards=deepcopy(relevant), role=role,
230
236
  uncertainAliasesInvalidated=len(uncertain))
231
237
  f = self.frames[-1]
@@ -234,7 +240,7 @@ class State:
234
240
  relative = (mem_delta - stack_delta) % (1 << self.bits)
235
241
  if write is None and segment.term == self.segment("ss").term and mem_base == stack_base and f["returnBytes"] <= relative < 1 << (self.bits - 1):
236
242
  event["argument"] = {"offsetFromEntrySP": relative, "width": width, "returnFrameBytes": f["returnBytes"],
237
- "pushProducers": list(value.sources), "grouping": role or "consumed width only"}
243
+ "pushProducers": producers(value), "grouping": role or "consumed width only"}
238
244
  return value
239
245
 
240
246
  def address(self, ins, operand):
@@ -287,6 +293,19 @@ class State:
287
293
  return value
288
294
 
289
295
 
296
+ # Synonymous and complementary branches on one flag producer share a single assumption.
297
+ BRANCH_CONDITIONS = {}
298
+ for names, condition in ((("je", "jz"), "z"), (("jb", "jc", "jnae"), "c"), (("jbe", "jna"), "be"),
299
+ (("jl", "jnge"), "l"), (("jle", "jng"), "le"), (("js",), "s"),
300
+ (("jo",), "o"), (("jp", "jpe"), "p")):
301
+ for name in names:
302
+ BRANCH_CONDITIONS[name] = (condition, False)
303
+ for names, condition in ((("jne", "jnz"), "z"), (("jae", "jnb", "jnc"), "c"), (("ja", "jnbe"), "be"),
304
+ (("jge", "jnl"), "l"), (("jg", "jnle"), "le"), (("jns",), "s"),
305
+ (("jno",), "o"), (("jnp", "jpo"), "p")):
306
+ for name in names:
307
+ BRANCH_CONDITIONS[name] = (condition, True)
308
+
290
309
  CARRY_BRANCHES = {"jb": True, "jc": True, "jnae": True, "jae": False, "jnb": False, "jnc": False}
291
310
  # Branches taken when CF or OF is set; logic operations clear both whatever their operands.
292
311
  CLEARED_BY_LOGIC = {**CARRY_BRANCHES, "jo": True, "jno": False}
@@ -361,7 +380,18 @@ def ordinary(state, ins, image):
361
380
  if state.flat and operands[0].type == X86_OP_REG and ins.reg_name(operands[0].reg) in state.segment_bases:
362
381
  raise StopPath("Segment selector assignment requires a descriptor model")
363
382
  value = state.get(ins, operands[1], image)
364
- state.put(ins, operands[0], resize(value, operands[0].size * 8, signed=m == "movsx"))
383
+ result = resize(value, operands[0].size * 8, signed=m == "movsx")
384
+ state.put(ins, operands[0], result)
385
+ if not state.value_transfers:
386
+ return
387
+
388
+ def location(operand):
389
+ return {"kind": "register", "register": ins.reg_name(operand.reg)} if operand.type == X86_OP_REG else {"kind": "memory"} if operand.type == X86_OP_MEM else {"kind": "immediate"}
390
+ destination_container = ALIASES[ins.reg_name(operands[0].reg)][0] if operands[0].type == X86_OP_REG else None
391
+ state.event("value-transfer", operation=m, source=location(operands[1]), destination=location(operands[0]),
392
+ destinationContainer=destination_container, destinationContainerValue=state.reg(destination_container).report() if destination_container else None,
393
+ sourceBits=value.bits, destinationBits=result.bits, sourceValue=value.report(), resultValue=result.report(),
394
+ conversion="truncate" if result.bits < value.bits else "signExtend" if m == "movsx" else "zeroExtend" if result.bits > value.bits else "sameWidth")
365
395
  return
366
396
  if m == "xchg":
367
397
  values = [state.get(ins, operand, image) for operand in operands]
@@ -445,21 +475,25 @@ def ordinary(state, ins, image):
445
475
  # The prefix toggles the mode's default operand size (16-bit real mode, 32-bit flat).
446
476
  wide = (0x66 in ins.prefix) != state.flat
447
477
  source, destination = ("ax", "eax") if wide else ("al", "ax")
448
- value = resize(state.reg(source), 32 if wide else 16, True)
478
+ source_value = state.reg(source)
479
+ value = resize(source_value, 32 if wide else 16, True)
449
480
  state.setreg(destination, value, state.at)
450
481
  state.event("conversion", sourceRegister=source, destinationRegister=destination,
451
482
  effectiveOperandBits=32 if wide else 16, decoderMnemonic=m,
452
- mnemonicWidthMismatch=m != ("cwde" if wide else "cbw"), result=value.report())
483
+ mnemonicWidthMismatch=m != ("cwde" if wide else "cbw"), result=value.report(),
484
+ sourceValue=source_value.report(), sourceBits=source_value.bits, destinationBits=value.bits, conversion="signExtend")
453
485
  return
454
486
  if m in ("cwd", "cdq"):
455
487
  wide = (0x66 in ins.prefix) != state.flat
456
488
  source, destination = ("eax", "edx") if wide else ("ax", "dx")
457
489
  bits = 32 if wide else 16
458
- value = resize(extract(state.reg(source), bits-1, 1), bits, signed=True)
490
+ source_value = state.reg(source)
491
+ value = resize(extract(source_value, bits-1, 1), bits, signed=True)
459
492
  state.setreg(destination, value, state.at)
460
493
  state.event("conversion", sourceRegister=source, destinationRegister=destination,
461
494
  effectiveOperandBits=bits, decoderMnemonic=m,
462
- mnemonicWidthMismatch=m != ("cdq" if wide else "cwd"), result=value.report())
495
+ mnemonicWidthMismatch=m != ("cdq" if wide else "cwd"), result=value.report(),
496
+ sourceValue=source_value.report(), sourceBits=source_value.bits, destinationBits=value.bits, conversion="signFillHighHalf")
463
497
  return
464
498
  if m in ("clc", "stc", "cmc"):
465
499
  if m == "cmc":
@@ -5,6 +5,7 @@ from capstone import CS_AC_READ, CS_AC_WRITE
5
5
  from capstone.x86 import X86_OP_IMM, X86_OP_MEM, X86_OP_REG
6
6
  from .machine import State, StopPath, REGISTERS, ALIASES, segment_register
7
7
  from .values import unknown
8
+ from .result_flow import return_flows
8
9
  from .image import Image, integer
9
10
  from .trace import (trace, walk, call_target, unsupported_transfer, uncovered, base_mnemonic, OVERLAP_REASON, CONTESTED_REASON,
10
11
  RETURNS, INTERRUPTS, PORTS)
@@ -298,7 +299,9 @@ def uses(image, config):
298
299
  if state is None:
299
300
  # Registers are unknown here; name them for this operand site so no entry value is implied.
300
301
  state = State(at, image, {})
301
- state.regs.update({r: unknown(f"CFG-operand:{at}:{r}", ALIASES[r][2]) for r in REGISTERS if r != "cs"})
302
+ for r in REGISTERS:
303
+ if r != "cs":
304
+ state.setreg(r, unknown(f"CFG-operand:{at}:{r}", ALIASES[r][2]), None)
302
305
  try:
303
306
  segment_value, offset_value, segment_name = state.address(ins, operand)
304
307
  except StopPath as error:
@@ -994,6 +997,200 @@ def callees(image, config):
994
997
  "that does not. Neither proves runtime recursion."}
995
998
 
996
999
 
1000
+ # These branches test CX/ECX (LOOPE/LOOPNE also ZF), so an adjacent CMP/TEST never describes their predicate.
1001
+ COUNT_BRANCHES = frozenset(("jcxz", "jecxz", "jrcxz", "loop", "loope", "loopne", "loopz", "loopnz"))
1002
+
1003
+
1004
+ def _operand_view(ins, o):
1005
+ if o.type == X86_OP_REG:
1006
+ return {"kind": "register", "name": ins.reg_name(o.reg), "width": o.size}
1007
+ if o.type == X86_OP_IMM:
1008
+ return {"kind": "immediate", "value": o.imm, "width": o.size}
1009
+ if o.type == X86_OP_MEM:
1010
+ return {"kind": "memory", "width": memory_width(ins, o), "segmentRegister": segment_register(ins, o.mem),
1011
+ "baseRegister": ins.reg_name(o.mem.base) or None, "indexRegister": ins.reg_name(o.mem.index) or None,
1012
+ "displacement": o.mem.disp}
1013
+ return {"kind": "unresolved"}
1014
+
1015
+
1016
+ def _stack_cleanup(ins):
1017
+ """The bytes an immediate ADD SP/ESP releases, or None for any other instruction or a negative immediate."""
1018
+ if (ins is None or ins.mnemonic != "add" or len(ins.operands) != 2 or ins.operands[0].type != X86_OP_REG
1019
+ or ins.reg_name(ins.operands[0].reg) not in ("sp", "esp") or ins.operands[1].type != X86_OP_IMM):
1020
+ return None
1021
+ bits = 8 * ins.operands[0].size
1022
+ amount = ins.operands[1].imm & ((1 << bits) - 1)
1023
+ return amount if 0 < amount < 1 << (bits - 1) else None
1024
+
1025
+
1026
+ def call_order(image, config):
1027
+ """Group the confirmed incoming calls of each containing entry by necessary guards and CFG order."""
1028
+ flat = incoming(image, config)
1029
+ entry_limit = integer(config.get("entryLimit", 64), 1, 256, "entry limit")
1030
+ analysis_limit = integer(config.get("analysisLimit", 1000000), 1, 10000000, "call order analysis limit")
1031
+ established = entries(image)
1032
+ bodies = {e: body(image, e, config.get("instructionLimit", 10000)) for e in established[:entry_limit]}
1033
+ conflicts = _cross_entry_overlaps(bodies)
1034
+ unchecked = established[entry_limit:]
1035
+ sites = {r["site"] for r in flat["confirmed"]}
1036
+ owners = {at: [e for e, b in bodies.items() if at in b["instructions"]] for at in sites}
1037
+ reports = []
1038
+ for entry, b in bodies.items():
1039
+ selected = sorted(at for at in sites if entry in owners[at])
1040
+ if not selected:
1041
+ continue
1042
+ usable = b["complete"] and not unchecked and not conflicts.get(entry) and not flat["truncated"] and all(owners[at] == [entry] for at in selected)
1043
+ instructions, successors, branches, exits = b["instructions"], {}, [], set()
1044
+ ending = {}
1045
+ for at, ins in instructions.items():
1046
+ ending.setdefault(at + ins.size, []).append(at)
1047
+ for at, ins in instructions.items():
1048
+ m, following = base_mnemonic(ins), at + ins.size
1049
+ targets = []
1050
+ if m in RETURNS or m == "hlt" or unsupported_transfer(image, ins):
1051
+ pass
1052
+ elif m in ("jmp", "ljmp"):
1053
+ d = image.indirect_jumps.get(at)
1054
+ targets = [r["target"] for r in d["rows"]] if d else [call_target(image, at, ins)[0]]
1055
+ elif m.startswith("j") or m.startswith("loop"):
1056
+ target = call_target(image, at, ins)[0]
1057
+ targets = [target, following]
1058
+ if target != following:
1059
+ producer = ending.get(at, []) if m not in COUNT_BRANCHES and at != entry else []
1060
+ comparison = instructions[producer[0]] if len(producer) == 1 else None
1061
+ context = ({"site": producer[0], "mnemonic": comparison.mnemonic,
1062
+ "operands": [_operand_view(comparison, o) for o in comparison.operands]}
1063
+ if comparison and comparison.mnemonic in ("cmp", "test") else None)
1064
+ for taken, to in ((True, target), (False, following)):
1065
+ if to is not None:
1066
+ branches.append({"site": at, "to": to, "taken": taken, "predicate": m,
1067
+ "comparison": context, "interpretation": "necessary caller CFG edge, not a runtime value or preserved guard"})
1068
+ else:
1069
+ targets = [following]
1070
+ successors[at] = sorted(set(t for t in targets if t in instructions))
1071
+ # Returns, halts, unsupported frames and transfers out of the body leave the caller CFG.
1072
+ if not targets or any(t not in instructions for t in targets):
1073
+ exits.add(at)
1074
+ predecessors = {}
1075
+ for source, targets in successors.items():
1076
+ for to in targets:
1077
+ predecessors.setdefault(to, set()).add(source)
1078
+ for guard in branches:
1079
+ if guard["comparison"] and predecessors.get(guard["site"], set()) != {guard["comparison"]["site"]}:
1080
+ guard["comparison"] = None
1081
+ cache, spent, capped = {}, 0, False
1082
+ def reachable(start, blocked=None, stops=frozenset()):
1083
+ nonlocal spent, capped
1084
+ key = (start, blocked, stops)
1085
+ if key in cache:
1086
+ return cache[key]
1087
+ todo, seen = [start], set()
1088
+ while todo:
1089
+ at = todo.pop()
1090
+ if at in seen or at not in instructions or at in stops:
1091
+ continue
1092
+ if spent >= analysis_limit:
1093
+ capped = True
1094
+ return set()
1095
+ spent += 1
1096
+ seen.add(at)
1097
+ todo.extend(to for to in successors[at] if (at, to) != blocked)
1098
+ cache[key] = seen
1099
+ return seen
1100
+
1101
+ def dominates(first, second):
1102
+ """Whether every route from the entry to second passes first."""
1103
+ return second not in reachable(entry, stops=frozenset((first,)))
1104
+
1105
+ def must_follow(first, second, guards):
1106
+ """Whether every route from first's continuation reaches second before an exit or a guard revisit."""
1107
+ start = first + instructions[first].size
1108
+ if start == second:
1109
+ return True
1110
+ if start in guards:
1111
+ return False
1112
+ seen = reachable(start, stops=guards | {second})
1113
+ return not seen & exits and not any(to in guards for at in seen for to in successors[at])
1114
+ necessary = {at: [] for at in selected}
1115
+ for guard in branches:
1116
+ remaining = reachable(entry, (guard["site"], guard["to"]))
1117
+ for at in selected:
1118
+ if at not in remaining:
1119
+ necessary[at].append({k: v for k, v in guard.items() if k != "to"})
1120
+ later = {at: reachable(at + instructions[at].size) & sites for at in selected}
1121
+ rows = []
1122
+ for at in selected:
1123
+ following = at + instructions[at].size
1124
+ amount = _stack_cleanup(instructions.get(following))
1125
+ rows.append({"site": at, "necessaryGuards": necessary[at] if not capped else [],
1126
+ "cleanup": {"continuation": following, "site": following if amount is not None else None,
1127
+ "argumentBytes": amount, "status": "observed after assumed return" if amount is not None else "unread cleanup",
1128
+ "assumption": "callee returns to the next instruction"},
1129
+ "calleeEffects": {"status": "unresolved", "reason": "caller CFG never proves callee return success or restored/preserved state"}})
1130
+ partitions = {}
1131
+ for row in rows:
1132
+ key = tuple((g["site"], g["taken"]) for g in row["necessaryGuards"])
1133
+ partitions.setdefault(key, []).append(row["site"])
1134
+ groups = []
1135
+ for members in partitions.values():
1136
+ shared_guards = next(r["necessaryGuards"] for r in rows if r["site"] == members[0]) if not capped else []
1137
+ guard_sites = frozenset(g["site"] for g in shared_guards)
1138
+ local_later = {at: reachable(at + instructions[at].size, stops=guard_sites) & sites for at in members}
1139
+ pairs = [(a, z) for index, a in enumerate(members) for z in members[index + 1:]]
1140
+ sequence = all((z in local_later[a]) != (a in local_later[z]) for a, z in pairs) and all(a not in local_later[a] for a in members)
1141
+ alternatives = len(members) > 1 and all(z not in local_later[a] and a not in local_later[z] for a, z in pairs)
1142
+ kind = "unread" if not usable or capped else "sequence" if sequence else "branchAlternatives" if alternatives else "unread"
1143
+ order = sorted(members, key=lambda a: sum(a in local_later[z] for z in members if z != a)) if kind == "sequence" else None
1144
+ # Equal single-edge guards do not make every member run: a later call may be reached around an
1145
+ # earlier one (two branches to it, a jump table) or be skipped after it. Each call must dominate
1146
+ # the next and the next must follow it on every route within the visit.
1147
+ if order and not all(dominates(a, z) and must_follow(a, z, guard_sites) for a, z in zip(order, order[1:])):
1148
+ kind, order = "unread", None
1149
+ groups.append({"kind": kind, "sites": members, "order": order,
1150
+ "sharedGuards": shared_guards,
1151
+ "scope": "one visit past the shared guard edges" if guard_sites else "caller CFG",
1152
+ "mayRepeatAcrossGuardVisits": any(a in later[a] for a in members)})
1153
+ if capped or not usable:
1154
+ # A cycle found in a partial read still may repeat; finding none there proves nothing.
1155
+ for group in groups:
1156
+ group["mayRepeatAcrossGuardVisits"] = group["mayRepeatAcrossGuardVisits"] or None
1157
+ if capped:
1158
+ for group in groups:
1159
+ group.update(kind="unread", order=None, sharedGuards=[])
1160
+ for row in rows:
1161
+ row["necessaryGuards"] = []
1162
+ relations = []
1163
+ for index, first in enumerate(selected):
1164
+ for second in selected[index + 1:]:
1165
+ forward, backward = second in later[first], first in later[second]
1166
+ kind = ("unread" if not usable or capped or forward and backward else
1167
+ "sequence" if forward or backward else "branchAlternatives")
1168
+ relations.append({"sites": [first, second], "kind": kind,
1169
+ "order": [first, second] if kind == "sequence" and forward else [second, first] if kind == "sequence" else None})
1170
+ if relations and all(r["kind"] == "branchAlternatives" for r in relations):
1171
+ common = [g for g in rows[0]["necessaryGuards"] if all(any(g["site"] == h["site"] and g["taken"] == h["taken"] for h in row["necessaryGuards"]) for row in rows)]
1172
+ groups = [{"kind": "branchAlternatives", "sites": selected, "order": None, "sharedGuards": common,
1173
+ "scope": "caller CFG", "mayRepeatAcrossGuardVisits": any(a in later[a] for a in selected)}]
1174
+ reports.append({"entry": entry, "ranges": b["intervals"], "boundaryUsable": usable, "orderingUsable": usable and not capped,
1175
+ "gaps": b["gaps"], "conflicts": conflicts.get(entry, []), "analysisCapped": capped, "analysisSteps": spent,
1176
+ "calls": rows, "groups": groups, "relations": relations, "assumptions": b["assumedContinuations"]})
1177
+ controls = config.get("orderControls", [])
1178
+ if not isinstance(controls, list) or len(controls) > 256:
1179
+ raise ValueError("Invalid call order controls")
1180
+ for control in controls:
1181
+ if not isinstance(control, dict) or control.get("kind") not in ("sequence", "branchAlternatives") or not isinstance(control.get("sites"), list) or not control["sites"] or len(control["sites"]) > 256 or any(type(at) is not int for at in control["sites"]) or type(control.get("entry")) is not int:
1182
+ raise ValueError("Invalid call order control")
1183
+ report = next((r for r in reports if r["entry"] == control.get("entry")), None)
1184
+ # Sequence sites are compared in order; branch alternatives have none.
1185
+ expected = control["sites"] if control["kind"] == "sequence" else sorted(control["sites"])
1186
+ group = next((g for g in report["groups"] if g["kind"] == control["kind"] and (g["order"] if g["kind"] == "sequence" else sorted(g["sites"])) == expected), None) if report else None
1187
+ if group is None:
1188
+ raise ValueError("Call order positive control missed")
1189
+ return {"incoming": flat, "callers": reports, "uncheckedEntries": unchecked, "orderControls": controls,
1190
+ "unownedSites": sorted(at for at in sites if not owners[at]),
1191
+ "interpretation": "Order and guards describe a conditional caller CFG. Cleanup follows an assumed return; callees' effects, successful return and state restoration remain unresolved."}
1192
+
1193
+
997
1194
  def _body_report(b):
998
1195
  return {k: v for k, v in b.items() if k != "instructions"} | {"instructionCount": len(b["instructions"])}
999
1196
 
@@ -1158,6 +1355,8 @@ def _run_report(image, config, command):
1158
1355
  return owner(image, config)
1159
1356
  if command == "callees":
1160
1357
  return callees(image, config)
1358
+ if command == "call-order":
1359
+ return call_order(image, config)
1161
1360
  if command == "incoming":
1162
1361
  return incoming(image, config)
1163
1362
  if command == "operand-candidates":
@@ -1180,6 +1379,8 @@ def _run_report(image, config, command):
1180
1379
  report = near_pointer_provenance(report, config)
1181
1380
  if command == "allocation":
1182
1381
  return allocations(report, config)
1382
+ if command == "returns":
1383
+ report = return_flows(report, config)
1183
1384
  if command != "trace":
1184
1385
  kinds = {"arguments": ("address-formation", "read", "call", "call-return"), "effects": ("address-formation", "write", "call", "call-return", "return", "branch", "string-operation",
1185
1386
  "flag-assumption", "flag-write", "flags-save", "flags-restore", "local-iret"),
@@ -1187,7 +1388,9 @@ def _run_report(image, config, command):
1187
1388
  "guards": ("compare", "branch", "read", "write", "call", "call-return"),
1188
1389
  "memory": ("read", "write", "address-formation")}[command]
1189
1390
  for path in report["paths"]:
1190
- path["events"] = [e for e in path["events"] if e["kind"] in kinds or
1391
+ # Returns keep the transfers, conversions and reads that depend on a declared result.
1392
+ consumed = {c["order"] for f in path.get("returnFlows", {}).get("results", ()) for c in f["consumers"]}
1393
+ path["events"] = [e for e in path["events"] if e["kind"] in kinds or e["order"] in consumed or
1191
1394
  (command == "effects" and e["kind"] == "read" and (e.get("nearPointerAccessCandidates") or e.get("nearPointerArgumentCandidates")))]
1192
1395
  return report
1193
1396
 
@@ -0,0 +1,131 @@
1
+ """Declared result roles and bounded producer dependencies across caller continuations."""
2
+ from .image import integer
3
+ from .machine import ALIASES, BRANCH_CONDITIONS
4
+ from .values import result_marker
5
+
6
+
7
+ def validate_contracts(config, image):
8
+ contracts = config.get("returnContracts", [])
9
+ if not isinstance(contracts, list) or len(contracts) > 256:
10
+ raise ValueError("At most 256 return contracts")
11
+ keys = set()
12
+ for c in contracts:
13
+ if not isinstance(c, dict):
14
+ raise ValueError("Return contract must be an object")
15
+ entry = integer(c.get("entry"), 0, len(image.data) - 1, "return contract entry")
16
+ register = c.get("register")
17
+ if not isinstance(register, str) or register not in ALIASES or not isinstance(c.get("evidence"), str) or not c["evidence"].strip():
18
+ raise ValueError("Return contract requires register and evidence")
19
+ if (entry, register) in keys:
20
+ raise ValueError("Return contract entry/register must be unique")
21
+ keys.add((entry, register))
22
+ bits = ALIASES[register][2]
23
+ failures = c.get("failures", [])
24
+ if not isinstance(failures, list) or len(failures) > 256 or any(type(n) is not int or not 0 <= n < 1 << bits for n in failures):
25
+ raise ValueError("Failure encodings must fit the consumed return width")
26
+ encodings = c.get("encodings", [])
27
+ if not isinstance(encodings, list) or len(encodings) > 256:
28
+ raise ValueError("At most 256 result encodings")
29
+ for encoding in encodings:
30
+ if not isinstance(encoding, dict) or type(encoding.get("value")) is not int or not 0 <= encoding["value"] < 1 << bits:
31
+ raise ValueError("Result encoding must fit the declared return width")
32
+ if not all(isinstance(encoding.get(k), str) and encoding[k].strip() for k in ("role", "evidence")):
33
+ raise ValueError("Result encoding requires role and evidence")
34
+ return contracts
35
+
36
+
37
+ def result_contracts(state, contracts, entry):
38
+ rows = []
39
+ for c in contracts:
40
+ if c["entry"] != entry:
41
+ continue
42
+ register = c["register"]
43
+ # Tag the declared value with a marker unique to the return event about to be
44
+ # recorded. The instruction site would also tag the SP the return pops, every
45
+ # register a call model clobbers and every later execution of the same return.
46
+ # A producer dependency is still weaker than unchanged value identity.
47
+ state.setreg(register, state.reg(register), result_marker(len(state.events)))
48
+ value = state.reg(register)
49
+ rows.append({"entry": entry, "register": register, "value": value.report(),
50
+ "failureEncodings": c.get("failures", []), "encodings": c.get("encodings", []),
51
+ "matchesFailureEncoding": None if value.number is None else value.number in c.get("failures", []),
52
+ "matchingRoles": None if value.number is None else [e for e in c.get("encodings", []) if e["value"] == value.number],
53
+ "evidence": c["evidence"], "encodingsExhaustive": False, "successEstablished": False})
54
+ return rows
55
+
56
+
57
+ # Sign and overflow tests read the operand as a signed value; carry tests as unsigned.
58
+ DOMAINS = {"l": "signed", "le": "signed", "s": "signed", "o": "signed", "c": "unsigned", "be": "unsigned"}
59
+
60
+
61
+ VALUE_FIELDS = ("sourceValue", "resultValue", "destinationContainerValue", "result", "value", "left", "right", "carry", "count")
62
+ CONSUMER_KINDS = ("value-transfer", "conversion", "read", "write", "compare", "branch", "return")
63
+ ROW_FIELDS = ("kind", "site", "entry", "depth", "order", "operation", "source", "destination", "sourceBits", "destinationBits",
64
+ "destinationContainer", "conversion", "sourceRegister", "destinationRegister", "effectiveOperandBits", "decoderMnemonic",
65
+ "width", "segment", "offset", "effectiveSegmentRegister", "predicate", "taken", "flagProducer")
66
+
67
+
68
+ def _event_values(event):
69
+ values = {k: event[k] for k in VALUE_FIELDS if isinstance(event.get(k), dict)}
70
+ if event["kind"] == "return":
71
+ values.update({"return:" + c["register"]: c["value"] for c in event.get("resultContracts", [])})
72
+ return values
73
+
74
+
75
+ def return_flows(report, config):
76
+ limit = integer(config.get("returnFlowLimit", 128), 1, 1024, "return flow limit")
77
+ consumer_limit = integer(config.get("returnConsumerLimit", 256), 1, 10000, "return consumer limit")
78
+ analysis_limit = integer(config.get("returnFlowAnalysisLimit", 1000000), 1, 10000000, "return flow analysis limit")
79
+ steps, capped = 0, False
80
+ for path in report["paths"]:
81
+ events = path["events"]
82
+ event_values = {}
83
+ flows, omitted = [], 0
84
+ for index, event in enumerate(events):
85
+ if event["kind"] != "return" and not (event["kind"] == "call-return" and event.get("modeled")):
86
+ continue
87
+ for contract in event.get("resultContracts", []):
88
+ if len(flows) >= limit or capped:
89
+ omitted += 1
90
+ continue
91
+ consumers, dropped, scanned = [], 0, True
92
+ marker = event["order"]
93
+ for later_index in range(index + 1, len(events)):
94
+ if steps >= analysis_limit:
95
+ capped, scanned = True, False
96
+ break
97
+ steps += 1
98
+ later = events[later_index]
99
+ if later["kind"] not in CONSUMER_KINDS:
100
+ continue
101
+ if later_index not in event_values:
102
+ event_values[later_index] = _event_values(later)
103
+ values = event_values[later_index]
104
+ dependent = {k: v for k, v in values.items() if marker in v.get("resultOrigins", ())}
105
+ if not dependent:
106
+ continue
107
+ if len(consumers) >= consumer_limit:
108
+ dropped += 1
109
+ continue
110
+ row = {k: later[k] for k in ROW_FIELDS if k in later}
111
+ widths = {k: "narrower" if v["bits"] < contract["value"]["bits"] else
112
+ "wider" if v["bits"] > contract["value"]["bits"] else "sameWidth"
113
+ for k, v in dependent.items()}
114
+ row.update(values=values, dependentValueFields=list(dependent), returnWidthRelationships=widths,
115
+ relationship="producer dependency only; not value or storage identity")
116
+ if later["kind"] == "branch":
117
+ condition = BRANCH_CONDITIONS.get(later.get("predicate"), (None,))[0]
118
+ row["predicateDomain"] = DOMAINS.get(condition, "flags/equality")
119
+ consumers.append(row)
120
+ flows.append({"originOrder": event["order"], "originSite": event["site"], "calleeEntry": contract["entry"],
121
+ "callSite": event.get("callSite"), "callerEntry": event.get("callerEntry"),
122
+ "conditionalModel": event.get("modeled", False), "resultContract": contract,
123
+ "consumers": consumers, "consumersOmitted": dropped, "consumerScanComplete": scanned,
124
+ "successEstablished": False})
125
+ path["returnFlows"] = {"results": flows, "resultsOmitted": omitted, "resultLimit": limit,
126
+ "consumerLimit": consumer_limit,
127
+ "complete": not omitted and all(not f["consumersOmitted"] and f["consumerScanComplete"] for f in flows)
128
+ and path["returned"] and not report["gaps"],
129
+ "interpretation": "conditional static dependency paths; encodings are declared evidence, not live occurrence; a branch never establishes initialization, accepted contents or extent"}
130
+ report["returnFlowAnalysis"] = {"limit": analysis_limit, "steps": steps, "capped": capped}
131
+ return report
@@ -2,23 +2,10 @@
2
2
  from copy import deepcopy
3
3
  from capstone.x86 import X86_OP_IMM, X86_OP_REG, X86_OP_MEM
4
4
  from .image import integer
5
- from .machine import (State, StopPath, ordinary, predicate, REGISTERS, ALIASES, string_instruction, string_count,
6
- string_effect, check_string_form)
5
+ from .machine import (State, StopPath, ordinary, predicate, REGISTERS, ALIASES, BRANCH_CONDITIONS, string_instruction,
6
+ string_count, string_effect, check_string_form)
7
7
  from .values import const, unknown, sources, op, Value
8
-
9
- # Synonymous and complementary branches on one flag producer share a single assumption.
10
- BRANCH_CONDITIONS = {}
11
- for names, condition in ((("je", "jz"), "z"), (("jb", "jc", "jnae"), "c"), (("jbe", "jna"), "be"),
12
- (("jl", "jnge"), "l"), (("jle", "jng"), "le"), (("js",), "s"),
13
- (("jo",), "o"), (("jp", "jpe"), "p")):
14
- for name in names:
15
- BRANCH_CONDITIONS[name] = (condition, False)
16
- for names, condition in ((("jne", "jnz"), "z"), (("jae", "jnb", "jnc"), "c"), (("ja", "jnbe"), "be"),
17
- (("jge", "jnl"), "l"), (("jg", "jnle"), "le"), (("jns",), "s"),
18
- (("jno",), "o"), (("jnp", "jpo"), "p")):
19
- for name in names:
20
- BRANCH_CONDITIONS[name] = (condition, True)
21
-
8
+ from .result_flow import validate_contracts, result_contracts
22
9
 
23
10
  def call_target(image, site, ins):
24
11
  if ins.mnemonic in ("lcall", "ljmp"):
@@ -237,6 +224,7 @@ def trace(image, config):
237
224
  integer(config.get("returnBytes", image.bits // 8), 2, 4, "returnBytes")
238
225
  if config.get("returnBytes", image.bits // 8) not in ((4,) if image.flat else (2, 4)):
239
226
  raise ValueError("returnBytes must agree with the selected near/far frame model")
227
+ contracts = validate_contracts(config, image)
240
228
  models = config.get("callModels", [])
241
229
  if not isinstance(models, list) or len(models) > 64:
242
230
  raise ValueError("At most 64 explicit call models")
@@ -386,7 +374,7 @@ def trace(image, config):
386
374
  created += 1
387
375
  for r in REGISTERS:
388
376
  if r not in model.get("preserves", []) and r not in ("esp", "cs"):
389
- child.regs[r] = unknown(f"modeled-call:{at}:{r}", ALIASES[r][2], at)
377
+ child.setreg(r, unknown(f"modeled-call:{at}:{r}", ALIASES[r][2]), at)
390
378
  child.clear_memory()
391
379
  child.forget_flags()
392
380
  child.direction_flag = unknown(f"modeled-call:{at}:DF:{child.flag_serial}", 1, at)
@@ -395,7 +383,8 @@ def trace(image, config):
395
383
  child.setreg(r, const(n, ALIASES[r][2], at), at)
396
384
  child.conditional.append({"site": at, "evidence": model["evidence"],
397
385
  "assumption": "call returns with balanced stack; memory effects unresolved"})
398
- child.event("call-return", callSite=at, registers=snapshot(child), modeled=True,
386
+ child.event("call-return", callSite=at, callerEntry=state.frames[-1]["entry"],
387
+ resultContracts=result_contracts(child, contracts, target), registers=snapshot(child), modeled=True,
399
388
  unknownMemoryEffects=True)
400
389
  child.at = following
401
390
  pending.append(child)
@@ -451,21 +440,10 @@ def trace(image, config):
451
440
  if image.flat and m == "retf":
452
441
  raise StopPath("Far return is outside the PE32 flat model")
453
442
  frame = state.frames[-1]
454
- roles = []
455
- for contract in config.get("returnContracts", []):
456
- if contract.get("entry") != frame["entry"]:
457
- continue
458
- register = contract.get("register")
459
- if register not in ALIASES or not contract.get("evidence"):
460
- raise ValueError("Return contract requires register and evidence")
461
- value = state.reg(register)
462
- failures = contract.get("failures", [])
463
- if not isinstance(failures, list) or len(failures) > 256 or any(type(n) is not int or not 0 <= n < 1 << value.bits for n in failures):
464
- raise ValueError("Failure encodings must fit the consumed return width")
465
- roles.append({"register": register, "value": value.report(), "failureEncodings": failures,
466
- "matchesFailureEncoding": None if value.number is None else value.number in failures,
467
- "evidence": contract["evidence"]})
468
- state.event("return", registers=snapshot(state), cleanupBytes=ins.operands[0].imm if ins.operands else 0, resultContracts=roles)
443
+ roles = result_contracts(state, contracts, frame["entry"])
444
+ state.event("return", registers=snapshot(state), cleanupBytes=ins.operands[0].imm if ins.operands else 0,
445
+ resultContracts=roles, callSite=frame.get("callSite"),
446
+ callerEntry=state.frames[-2]["entry"] if len(state.frames) > 1 else None)
469
447
  # The entry frame gets the same width and balance checks as a traced call.
470
448
  expected = 4 if m == "retf" else image.bits // 8
471
449
  if expected != frame["returnBytes"] or state.reg(state.sp).term != frame["sp"].term:
@@ -22,8 +22,26 @@ class Value:
22
22
  return self.term[1] if self.term[0] == "constant" else None
23
23
 
24
24
  def report(self):
25
- return {"bits": self.bits, "expression": self.term, "value": self.number,
26
- "producers": list(self.sources)}
25
+ row = {"bits": self.bits, "expression": self.term, "value": self.number, "producers": producers(self)}
26
+ origins = result_origins(self)
27
+ if origins:
28
+ row["resultOrigins"] = origins
29
+ return row
30
+
31
+
32
+ def result_marker(order):
33
+ """A source tag naming the return event at ``order``; instruction sites are never negative."""
34
+ return -order - 1
35
+
36
+
37
+ def producers(v):
38
+ """The instruction sites among a value's sources, without return-result markers."""
39
+ return [s for s in v.sources if s >= 0]
40
+
41
+
42
+ def result_origins(v):
43
+ """The event orders of the declared return results a value depends on."""
44
+ return sorted(-s - 1 for s in v.sources if s < 0)
27
45
 
28
46
 
29
47
  def const(n, bits, site=None):
@@ -62,6 +62,163 @@ def events(result, kind):
62
62
  return [e for path in result["paths"] for e in path["events"] if e["kind"] == kind]
63
63
 
64
64
 
65
+ class ReturnFlowTests(unittest.TestCase):
66
+ def flow(self, code, contracts, **extra):
67
+ return report(code, "returns", returnContracts=contracts, **extra)
68
+
69
+ def contract(self, entry, register="ax", **extra):
70
+ return {"entry": entry, "register": register, "evidence": "synthetic result contract", **extra}
71
+
72
+ def test_nested_failure_word_byte_store_zero_extension_and_nonzero_gate(self):
73
+ c = Code().branch("e8", "wrapper").emit("0f b6 c0 85 c0").branch("74", "zero").emit("c3")
74
+ c.label("zero").emit("c3").label("wrapper").branch("e8", "callee").emit("a2 20 00 a0 20 00 c3")
75
+ c.label("callee").emit("b8 ff ff c3")
76
+ r = self.flow(c, [self.contract(c.labels["callee"], failures=[65535]), self.contract(c.labels["wrapper"], "al")], registers={"ds": 8192, "ss": 8192, "sp": 32768})
77
+ self.assertTrue(r["completeWithinModel"])
78
+ rows = r["paths"][0]["returnFlows"]["results"]
79
+ failure = next(f for f in rows if f["calleeEntry"] == c.labels["callee"])
80
+ self.assertTrue(failure["resultContract"]["matchesFailureEncoding"])
81
+ self.assertEqual(failure["callerEntry"], c.labels["wrapper"])
82
+ store = next(e for e in failure["consumers"] if e["kind"] == "write")
83
+ self.assertEqual(store["width"], 1)
84
+ self.assertEqual(store["values"]["value"]["value"], 255)
85
+ self.assertEqual(store["returnWidthRelationships"]["value"], "narrower")
86
+ extension = next(e for e in failure["consumers"] if e.get("conversion") == "zeroExtend")
87
+ self.assertEqual((extension["sourceBits"], extension["destinationBits"]), (8, 16))
88
+ branch = next(e for e in failure["consumers"] if e["kind"] == "branch")
89
+ self.assertEqual((branch["predicate"], branch["taken"], branch["values"]["left"]["value"]), ("je", False, 255))
90
+ self.assertFalse(failure["successEstablished"])
91
+
92
+ def test_full_word_failure_normalization(self):
93
+ c = Code().branch("e8", "callee").emit("83 f8 ff").branch("74", "bad").emit("b0 01 c3")
94
+ c.label("bad").emit("b0 00 c3").label("callee").emit("b8 ff ff c3")
95
+ r = self.flow(c, [self.contract(c.labels["callee"], failures=[65535]), self.contract(0, "al")])
96
+ f = r["paths"][0]["returnFlows"]["results"][0]
97
+ branch = next(e for e in f["consumers"] if e["kind"] == "branch")
98
+ self.assertEqual((branch["values"]["left"]["bits"], branch["taken"]), (16, True))
99
+ self.assertEqual(branch["values"]["right"]["value"], 65535)
100
+ self.assertNotIn("right", branch["dependentValueFields"])
101
+ self.assertEqual(r["paths"][0]["registers"]["al"]["value"], 0)
102
+
103
+ def test_signed_consumer_and_raw_field_role_remain_distinct(self):
104
+ c = Code().branch("e8", "callee").emit("83 f8 01").branch("7c", "reject").emit("c3")
105
+ c.label("reject").emit("c3").label("callee").emit("b8 00 80 c3")
106
+ encoding = {"value": 32768, "role": "raw field", "evidence": "synthetic field contract"}
107
+ r = self.flow(c, [self.contract(c.labels["callee"], failures=[65535], encodings=[encoding])])
108
+ f = r["paths"][0]["returnFlows"]["results"][0]
109
+ self.assertFalse(f["resultContract"]["matchesFailureEncoding"])
110
+ self.assertEqual(f["resultContract"]["matchingRoles"], [encoding])
111
+ branch = next(e for e in f["consumers"] if e["kind"] == "branch")
112
+ self.assertEqual((branch["predicateDomain"], branch["taken"]), ("signed", True))
113
+
114
+ def test_sign_extension_and_unrelated_equal_constant(self):
115
+ c = Code().branch("e8", "callee").emit("0f be d0 bb ff ff 83 fb ff c3")
116
+ c.label("callee").emit("b0 ff c3")
117
+ r = self.flow(c, [self.contract(c.labels["callee"], "al", failures=[255])])
118
+ consumers = r["paths"][0]["returnFlows"]["results"][0]["consumers"]
119
+ extension = next(e for e in consumers if e.get("conversion") == "signExtend")
120
+ self.assertEqual(extension["values"]["resultValue"]["value"], 65535)
121
+ self.assertFalse(any(e["kind"] == "compare" for e in consumers))
122
+
123
+ def test_implicit_sign_extension_retains_effective_widths(self):
124
+ c = Code().branch("e8", "callee").emit("98 99 c3").label("callee").emit("b0 ff c3")
125
+ r = self.flow(c, [self.contract(c.labels["callee"], "al", failures=[255])])
126
+ rows = r["paths"][0]["returnFlows"]["results"][0]["consumers"]
127
+ conversions = [e for e in rows if e["kind"] == "conversion"]
128
+ self.assertEqual([e["conversion"] for e in conversions], ["signExtend", "signFillHighHalf"])
129
+ self.assertEqual((conversions[0]["sourceBits"], conversions[0]["destinationBits"]), (8, 16))
130
+
131
+ def test_sibling_high_byte_zeroing_is_retained_before_word_zero_gate(self):
132
+ c = Code().branch("e8", "callee").emit("b4 00 09 c0").branch("75", "nonzero").emit("c3")
133
+ c.label("nonzero").emit("c3").label("callee").emit("b0 ff c3")
134
+ r = self.flow(c, [self.contract(c.labels["callee"], "al", failures=[255])])
135
+ rows = r["paths"][0]["returnFlows"]["results"][0]["consumers"]
136
+ write = next(e for e in rows if e["kind"] == "value-transfer")
137
+ self.assertEqual(write["destination"]["register"], "ah")
138
+ self.assertEqual(write["values"]["sourceValue"]["value"], 0)
139
+ self.assertIn("destinationContainerValue", write["dependentValueFields"])
140
+ branch = next(e for e in rows if e["kind"] == "branch")
141
+ self.assertEqual((branch["operation"], branch["predicate"], branch["taken"], branch["values"]["left"]["value"]), ("or", "jne", True, 255))
142
+
143
+ def test_models_stops_and_all_summary_caps_remain_explicit(self):
144
+ c = Code().branch("e8", "callee").emit("a2 20 00 85 c0").branch("74", "end").emit("90")
145
+ c.label("end").emit("c3").label("callee").emit("b8 ff ff c3")
146
+ contracts = [self.contract(c.labels["callee"], failures=[65535]), self.contract(0)]
147
+ for extra in ({"returnFlowLimit": 1}, {"returnConsumerLimit": 1}, {"returnFlowAnalysisLimit": 1}, {"maxSteps": 2}, {"maxDepth": 1}):
148
+ r = self.flow(c, contracts, **extra)
149
+ self.assertFalse(r["paths"][0]["returnFlows"]["complete"])
150
+ r = self.flow(c, contracts, callModels=[{"site": 0, "evidence": "synthetic conditional case", "cases": [{"registers": {"ax": 65535}}]}])
151
+ f = r["paths"][0]["returnFlows"]["results"][0]
152
+ self.assertTrue(f["conditionalModel"])
153
+ self.assertTrue(r["paths"][0]["conditionalModels"])
154
+
155
+ def test_values_sharing_only_the_return_or_call_site_are_not_dependent(self):
156
+ # The return pops SP at its own site; a call model clobbers BX at the call site.
157
+ c = Code().branch("e8", "callee").emit("89 e5 39 e9 89 d9 c3").label("callee").emit("b8 ff ff c3")
158
+ contracts = [self.contract(c.labels["callee"])]
159
+ r = self.flow(c, contracts, registers={"ss": 8192, "sp": 32768})
160
+ self.assertEqual(r["paths"][0]["returnFlows"]["results"][0]["consumers"], [])
161
+ r = self.flow(c, contracts, callModels=[{"site": 0, "evidence": "synthetic conditional case", "cases": [{"registers": {"ax": 1}}]}])
162
+ f = r["paths"][0]["returnFlows"]["results"][0]
163
+ self.assertTrue(f["conditionalModel"])
164
+ self.assertEqual(f["consumers"], [])
165
+ self.assertNotIn(0, r["paths"][0]["registers"]["sp"].get("resultOrigins", []))
166
+
167
+ def test_each_execution_of_one_return_is_a_separate_origin(self):
168
+ c = Code().branch("e8", "callee").emit("89 c3").branch("e8", "callee").emit("89 c1 c3")
169
+ c.label("callee").emit("66 b8 ff ff 00 00 c3")
170
+ rows = self.flow(c, [self.contract(c.labels["callee"])])["paths"][0]["returnFlows"]["results"]
171
+ self.assertEqual(len(rows), 2)
172
+ self.assertEqual([[e["site"] for e in f["consumers"]] for f in rows], [[3], [8]])
173
+ self.assertEqual(rows[0]["resultContract"]["value"]["resultOrigins"], [rows[0]["originOrder"]])
174
+ self.assertTrue(all(isinstance(p, int) and p >= 0 for p in rows[0]["resultContract"]["value"]["producers"]))
175
+
176
+ def test_sign_flag_gate_is_a_signed_predicate(self):
177
+ c = Code().branch("e8", "callee").emit("85 c0").branch("78", "negative").emit("c3")
178
+ c.label("negative").emit("c3").label("callee").emit("b8 ff ff c3")
179
+ rows = self.flow(c, [self.contract(c.labels["callee"], failures=[65535])])["paths"][0]["returnFlows"]["results"][0]["consumers"]
180
+ branch = next(e for e in rows if e["kind"] == "branch")
181
+ self.assertEqual((branch["predicate"], branch["predicateDomain"], branch["taken"]), ("js", "signed", True))
182
+
183
+ def test_analysis_cap_marks_the_truncated_flow(self):
184
+ c = Code().branch("e8", "callee").emit("a2 20 00 85 c0").branch("74", "end").emit("90")
185
+ c.label("end").emit("c3").label("callee").emit("b8 ff ff c3")
186
+ r = self.flow(c, [self.contract(c.labels["callee"], failures=[65535]), self.contract(0)], returnFlowAnalysisLimit=1)
187
+ flows = r["paths"][0]["returnFlows"]
188
+ self.assertTrue(r["returnFlowAnalysis"]["capped"])
189
+ self.assertFalse(flows["results"][0]["consumerScanComplete"])
190
+ self.assertEqual(flows["resultsOmitted"], 1)
191
+ self.assertFalse(flows["complete"])
192
+ r = self.flow(c, [self.contract(c.labels["callee"], failures=[65535]), self.contract(0)])
193
+ self.assertTrue(all(f["consumerScanComplete"] for f in r["paths"][0]["returnFlows"]["results"]))
194
+ self.assertTrue(r["paths"][0]["returnFlows"]["complete"])
195
+
196
+ def test_overwriting_the_result_register_ends_the_dependency_byte_by_byte(self):
197
+ # mov ax,5 replaces both result bytes; mov al,5 leaves AH, so the word compare still depends on it.
198
+ for overwrite, dependent in (("b8 05 00", False), ("b0 05", True)):
199
+ c = Code().branch("e8", "callee").emit(overwrite + " 83 f8 03 c3").label("callee").emit("b8 ff ff c3")
200
+ rows = self.flow(c, [self.contract(c.labels["callee"], failures=[65535])])["paths"][0]["returnFlows"]["results"][0]["consumers"]
201
+ self.assertEqual(any(e["kind"] == "compare" for e in rows), dependent)
202
+ r = report("b8 01 00 b0 02 c3")
203
+ self.assertEqual(r["paths"][0]["registers"]["ah"]["producers"], [0])
204
+ self.assertEqual(r["paths"][0]["registers"]["ax"]["producers"], [0, 3])
205
+
206
+ def test_value_transfers_are_recorded_only_for_declared_results(self):
207
+ c = Code().branch("e8", "callee").emit("89 c3 89 d1 8b 16 20 00 c3").label("callee").emit("b8 ff ff c3")
208
+ self.assertEqual(events(report(c), "value-transfer"), [])
209
+ r = self.flow(c, [self.contract(c.labels["callee"])], registers={"ds": 8192})
210
+ consumers = r["paths"][0]["returnFlows"]["results"][0]["consumers"]
211
+ self.assertEqual([e["site"] for e in consumers if e["kind"] == "value-transfer"], [3])
212
+ # Transfers and reads that do not depend on the result stay out of the returns events.
213
+ self.assertEqual([e["site"] for e in r["paths"][0]["events"] if e["kind"] in ("value-transfer", "read")], [3])
214
+
215
+ def test_invalid_unreachable_contracts_are_rejected(self):
216
+ for bad in (self.contract(0, failures=[65536]), self.contract(0, encodings=[{"value": 1, "role": "failure"}]),
217
+ self.contract(0, "fpu"), self.contract(0, ["ax"]), self.contract(0, evidence="")):
218
+ with self.assertRaises(ValueError):
219
+ self.flow(Code().emit("eb fe c3"), [bad], maxSteps=1)
220
+
221
+
65
222
  class CalleeGraphTests(unittest.TestCase):
66
223
  def graph(self, code, names, **extra):
67
224
  data = code.bytes()
@@ -349,6 +506,137 @@ class NearPointerSegmentTests(unittest.TestCase):
349
506
  self.assertTrue(all(p["dereferenceSegmentRegister"] == "es" and p["segmentRelationship"] == "sameWithinModel" for p in links))
350
507
 
351
508
 
509
+ class CallOrderTests(unittest.TestCase):
510
+ def run_order(self, code, **extra):
511
+ data = code.bytes()
512
+ cfg = configuration(data, target=code.labels["helper"], **extra)
513
+ cfg["regions"][0]["entries"] = [0, code.labels["helper"]]
514
+ return run_report(data, cfg, "call-order"), cfg
515
+
516
+ def test_guarded_sequence_retains_shared_cmp_cleanup_and_unread_effects(self):
517
+ c = Code().emit("83 7e fa 05").branch("7c", "end")
518
+ c.label("one").branch("e8", "helper").emit("83 c4 08")
519
+ c.label("two").branch("e8", "helper").emit("83 c4 08")
520
+ c.label("end").emit("c3").label("helper").emit("c3")
521
+ r, cfg = self.run_order(c, controls=[c.labels["one"], c.labels["two"]], orderControls=[{"entry": 0, "kind": "sequence", "sites": [c.labels["one"], c.labels["two"]]}])
522
+ caller = r["callers"][0]
523
+ self.assertEqual(caller["groups"][0]["kind"], "sequence")
524
+ self.assertEqual(caller["groups"][0]["sharedGuards"][0]["comparison"]["operands"][0]["segmentRegister"], "ss")
525
+ self.assertTrue(all(row["cleanup"]["argumentBytes"] == 8 and row["calleeEffects"]["status"] == "unresolved" for row in caller["calls"]))
526
+ self.assertEqual(len(r["incoming"]["confirmed"]), 2)
527
+
528
+ def test_adjacent_comparison_is_not_claimed_when_another_edge_enters_the_branch(self):
529
+ c = Code().emit("85 c0").branch("74", "compare").branch("e9", "condition")
530
+ c.label("compare").emit("83 fb 05").label("condition").branch("7c", "end").branch("e8", "helper")
531
+ c.label("end").emit("c3").label("helper").emit("c3")
532
+ r, _ = self.run_order(c)
533
+ guards = r["callers"][0]["groups"][0]["sharedGuards"]
534
+ conditional = next(g for g in guards if g["site"] == c.labels["condition"])
535
+ self.assertIsNone(conditional["comparison"])
536
+
537
+ def test_sequence_order_comes_from_flow_not_ascending_addresses(self):
538
+ c = Code().branch("e9", "first").label("second").branch("e8", "helper").emit("c3")
539
+ c.label("first").branch("e8", "helper").branch("e9", "second").label("helper").emit("c3")
540
+ r, _ = self.run_order(c)
541
+ self.assertEqual(r["callers"][0]["groups"][0]["order"], [c.labels["first"], c.labels["second"]])
542
+
543
+ def test_branch_alternatives_do_not_become_a_sequence(self):
544
+ c = Code().emit("85 c0").branch("74", "other").branch("e8", "helper").emit("c3")
545
+ c.label("other").branch("e8", "helper").emit("c3").label("helper").emit("c3")
546
+ r, cfg = self.run_order(c)
547
+ self.assertEqual(r["callers"][0]["groups"][0]["kind"], "branchAlternatives")
548
+ cfg["orderControls"] = [{"entry": 0, "kind": "sequence", "sites": r["callers"][0]["groups"][0]["sites"]}]
549
+ with self.assertRaisesRegex(ValueError, "positive control"):
550
+ run_report(c.bytes(), cfg, "call-order")
551
+
552
+ def test_outer_loop_keeps_sequence_within_a_guard_visit_and_reports_recurrence(self):
553
+ c = Code().label("guard").emit("83 f8 05").branch("7c", "end")
554
+ c.label("one").branch("e8", "helper").label("two").branch("e8", "helper").branch("eb", "guard")
555
+ c.label("end").emit("c3").label("helper").emit("c3")
556
+ r, _ = self.run_order(c)
557
+ g = r["callers"][0]["groups"][0]
558
+ self.assertEqual(g["kind"], "sequence")
559
+ self.assertEqual(g["order"], [c.labels["one"], c.labels["two"]])
560
+ self.assertTrue(g["mayRepeatAcrossGuardVisits"])
561
+ self.assertEqual(g["scope"], "one visit past the shared guard edges")
562
+
563
+ def test_shared_ownership_never_verifies_order(self):
564
+ c = Code().emit("90").branch("e8", "helper").emit("c3").label("helper").emit("c3")
565
+ cfg = configuration(c.bytes(), target=c.labels["helper"])
566
+ cfg["regions"][0]["entries"] = [0, 1, c.labels["helper"]]
567
+ r = run_report(c.bytes(), cfg, "call-order")
568
+ self.assertTrue(r["callers"])
569
+ self.assertTrue(all(not caller["orderingUsable"] for caller in r["callers"]))
570
+ self.assertTrue(all(g["kind"] == "unread" for caller in r["callers"] for g in caller["groups"]))
571
+
572
+ def test_overlapping_entries_are_rejected_by_the_flat_boundary_inventory(self):
573
+ c = Code().emit("66 90").branch("e8", "helper").emit("c3").label("helper").emit("c3")
574
+ cfg = configuration(c.bytes(), target=c.labels["helper"])
575
+ cfg["regions"][0]["entries"] = [0, 1, c.labels["helper"]]
576
+ r = run_report(c.bytes(), cfg, "call-order")
577
+ self.assertFalse(r["incoming"]["confirmed"])
578
+ self.assertGreater(sum(r["incoming"]["counts"].values()), 0)
579
+ self.assertFalse(r["callers"])
580
+
581
+ def test_caps_and_recurring_calls_leave_order_unread(self):
582
+ c = Code().branch("e8", "helper").branch("e8", "helper").emit("c3").label("helper").emit("c3")
583
+ for options in ({"entryLimit": 1}, {"limit": 1}, {"instructionLimit": 1}, {"analysisLimit": 1}):
584
+ r, _ = self.run_order(c, **options)
585
+ self.assertTrue(all(g["kind"] == "unread" for caller in r["callers"] for g in caller["groups"]))
586
+ loop = Code().label("again").branch("e8", "helper").branch("eb", "again").label("helper").emit("c3")
587
+ r, _ = self.run_order(loop)
588
+ self.assertTrue(all(g["kind"] == "unread" for caller in r["callers"] for g in caller["groups"]))
589
+
590
+ def test_capped_analysis_does_not_deny_recurrence(self):
591
+ c = Code().branch("e8", "helper").emit("90 c3").label("helper").emit("c3")
592
+ r, _ = self.run_order(c, analysisLimit=1)
593
+ self.assertTrue(r["callers"][0]["analysisCapped"])
594
+ self.assertIsNone(r["callers"][0]["groups"][0]["mayRepeatAcrossGuardVisits"])
595
+
596
+ def test_a_call_reached_around_another_is_not_sequenced_after_it(self):
597
+ # Two branches reach the first call, so neither edge alone is necessary, and the JMP skips it.
598
+ c = Code().emit("85 c0").branch("74", "first").branch("75", "first").branch("eb", "second")
599
+ c.label("first").branch("e8", "helper").label("second").branch("e8", "helper").emit("c3").label("helper").emit("c3")
600
+ r, cfg = self.run_order(c)
601
+ self.assertEqual(r["callers"][0]["groups"][0]["kind"], "unread")
602
+ cfg["orderControls"] = [{"entry": 0, "kind": "sequence", "sites": [c.labels["first"], c.labels["second"]]}]
603
+ with self.assertRaisesRegex(ValueError, "positive control"):
604
+ run_report(c.bytes(), cfg, "call-order")
605
+
606
+ def test_a_call_that_can_be_skipped_after_another_is_not_sequenced_after_it(self):
607
+ c = Code().label("first").branch("e8", "helper").emit("85 c0").branch("74", "second").branch("75", "second").emit("c3")
608
+ c.label("second").branch("e8", "helper").emit("c3").label("helper").emit("c3")
609
+ r, _ = self.run_order(c)
610
+ self.assertEqual(r["callers"][0]["groups"][0]["kind"], "unread")
611
+
612
+ def test_alternatives_have_the_full_group_shape_and_unordered_controls(self):
613
+ c = Code().emit("85 c0").branch("74", "other").label("one").branch("e8", "helper").emit("c3")
614
+ c.label("other").branch("e8", "helper").emit("c3").label("helper").emit("c3")
615
+ r, _ = self.run_order(c, orderControls=[{"entry": 0, "kind": "branchAlternatives", "sites": [c.labels["other"], c.labels["one"]]}])
616
+ g = r["callers"][0]["groups"][0]
617
+ self.assertEqual(g["kind"], "branchAlternatives")
618
+ self.assertEqual((g["scope"], g["mayRepeatAcrossGuardVisits"]), ("caller CFG", False))
619
+
620
+ def test_count_branches_and_entry_branches_take_no_adjacent_comparison(self):
621
+ c = Code().emit("83 f8 05").branch("e3", "end").branch("e8", "helper").label("end").emit("c3").label("helper").emit("c3")
622
+ r, _ = self.run_order(c)
623
+ self.assertIsNone(r["callers"][0]["calls"][0]["necessaryGuards"][0]["comparison"])
624
+ # The CMP before the entry is its only CFG predecessor, but the entry is also entered from outside.
625
+ c = Code().label("compare").emit("83 f8 05").label("entry").branch("7c", "end").branch("e8", "helper").branch("eb", "compare")
626
+ c.label("end").emit("c3").label("helper").emit("c3")
627
+ data = c.bytes()
628
+ cfg = configuration(data, target=c.labels["helper"])
629
+ cfg["regions"][0]["entries"] = [c.labels["entry"], c.labels["helper"]]
630
+ r = run_report(data, cfg, "call-order")
631
+ self.assertIsNone(r["callers"][0]["calls"][0]["necessaryGuards"][0]["comparison"])
632
+
633
+ def test_a_negative_stack_adjustment_is_not_cleanup(self):
634
+ c = Code().branch("e8", "helper").emit("81 c4 00 80 c3").label("helper").emit("c3")
635
+ r, _ = self.run_order(c)
636
+ self.assertEqual(r["callers"][0]["calls"][0]["cleanup"]["status"], "unread cleanup")
637
+ self.assertIsNone(r["callers"][0]["calls"][0]["cleanup"]["argumentBytes"])
638
+
639
+
352
640
  class ReporterTests(unittest.TestCase):
353
641
  def test_register_parts_preserve_neighbor(self):
354
642
  r = report("b8 34 12 b0 00 c3")