scientific-method-engine 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. scientific_method_engine/__init__.py +5 -0
  2. scientific_method_engine/__main__.py +3 -0
  3. scientific_method_engine/cli.py +75 -0
  4. scientific_method_engine/ghidra/ClearNoReturnFunctions.java +22 -0
  5. scientific_method_engine/ghidra/CreateFunctions.java +27 -0
  6. scientific_method_engine/ghidra/ExportBoundedFlow.java +59 -0
  7. scientific_method_engine/ghidra/ExportFunctionFingerprints.java +135 -0
  8. scientific_method_engine/ghidra/ExportFunctionInventory.java +50 -0
  9. scientific_method_engine/ghidra/MergeFallThroughFragment.java +63 -0
  10. scientific_method_engine/ghidra/RecoverCitedFunctions.java +93 -0
  11. scientific_method_engine/ghidra/RepairReturningCallers.java +76 -0
  12. scientific_method_engine/ghidra/ReportCallArguments.java +65 -0
  13. scientific_method_engine/ghidra/ReportCallPaths.java +106 -0
  14. scientific_method_engine/ghidra/ReportCallSitesWithScalars.java +95 -0
  15. scientific_method_engine/ghidra/ReportCallsToRange.java +67 -0
  16. scientific_method_engine/ghidra/ReportConstantFirstArgumentCalls.java +55 -0
  17. scientific_method_engine/ghidra/ReportDataBytes.java +36 -0
  18. scientific_method_engine/ghidra/ReportDecompileMatches.java +71 -0
  19. scientific_method_engine/ghidra/ReportDecompileWindow.java +61 -0
  20. scientific_method_engine/ghidra/ReportFilePatternInMemory.java +102 -0
  21. scientific_method_engine/ghidra/ReportFirstArgumentCallSummary.java +63 -0
  22. scientific_method_engine/ghidra/ReportFunctionScalarConstants.java +56 -0
  23. scientific_method_engine/ghidra/ReportFunctionSummary.java +64 -0
  24. scientific_method_engine/ghidra/ReportInstructionContext.java +64 -0
  25. scientific_method_engine/ghidra/ReportInstructionWindow.java +36 -0
  26. scientific_method_engine/ghidra/ReportMemoryBlockForFileOffset.java +105 -0
  27. scientific_method_engine/ghidra/ReportMemoryBlocks.java +52 -0
  28. scientific_method_engine/ghidra/ReportRandomnessCandidates.java +74 -0
  29. scientific_method_engine/ghidra/ReportReferences.java +42 -0
  30. scientific_method_engine/ghidra/ReportScalarConstants.java +55 -0
  31. scientific_method_engine/ghidra/ReportStringReferences.java +102 -0
  32. scientific_method_engine/ghidra/ReportSymbolReferences.java +72 -0
  33. scientific_method_engine/x86/__init__.py +0 -0
  34. scientific_method_engine/x86/dispatch.py +51 -0
  35. scientific_method_engine/x86/image.py +172 -0
  36. scientific_method_engine/x86/machine.py +659 -0
  37. scientific_method_engine/x86/pe.py +96 -0
  38. scientific_method_engine/x86/reports.py +1259 -0
  39. scientific_method_engine/x86/trace.py +547 -0
  40. scientific_method_engine/x86/values.py +123 -0
  41. scientific_method_engine-0.1.0.dist-info/METADATA +110 -0
  42. scientific_method_engine-0.1.0.dist-info/RECORD +45 -0
  43. scientific_method_engine-0.1.0.dist-info/WHEEL +4 -0
  44. scientific_method_engine-0.1.0.dist-info/entry_points.txt +2 -0
  45. scientific_method_engine-0.1.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,1259 @@
1
+ """Focused reports derived from instruction paths and explicit source bounds."""
2
+ from bisect import bisect_right
3
+ from collections import deque
4
+ from capstone import CS_AC_READ, CS_AC_WRITE
5
+ from capstone.x86 import X86_OP_IMM, X86_OP_MEM, X86_OP_REG
6
+ from .machine import State, StopPath, REGISTERS, ALIASES, segment_register
7
+ from .values import unknown
8
+ from .image import Image, integer
9
+ from .trace import (trace, walk, call_target, unsupported_transfer, uncovered, base_mnemonic, OVERLAP_REASON, CONTESTED_REASON,
10
+ RETURNS, INTERRUPTS, PORTS)
11
+
12
+
13
+ def entries(image):
14
+ return sorted(set(at for r in image.regions for at in r["entries"]))
15
+
16
+
17
+ def memory_width(ins, operand):
18
+ # Capstone reports the LDS/LES source as a word, but the load reads the full selector:offset pointer.
19
+ return 2 + ins.operands[0].size if ins.mnemonic in ("lds", "les") else operand.size
20
+
21
+
22
+ # Capstone reports several x87 stores (fst, fstp m32/m64, fist, fistp m16/m32, fnstcw) as reads and frstor
23
+ # as a write; for these the mnemonic, not Capstone's access flags, fixes the direction.
24
+ X87_MEMORY_STORES = {"fst", "fstp", "fist", "fistp", "fisttp", "fbstp", "fnstcw", "fstcw", "fnstsw", "fstsw",
25
+ "fnstenv", "fstenv", "fnsave", "fsave"}
26
+ X87_MEMORY_LOADS = {"fldcw", "fldenv", "frstor"}
27
+
28
+
29
+ def memory_access(ins, operand):
30
+ if ins.mnemonic in X87_MEMORY_STORES:
31
+ return ["write"]
32
+ if ins.mnemonic in X87_MEMORY_LOADS:
33
+ return ["read"]
34
+ return [name for flag, name in ((CS_AC_READ, "read"), (CS_AC_WRITE, "write")) if operand.access & flag]
35
+
36
+
37
+ def search_coverage(image, spans):
38
+ """How much of each complete segment, overlay or section the searched byte spans cover.
39
+
40
+ spans maps each searched region name to the (start, end) bytes the scan actually read."""
41
+ sections = (image.config.get("peMetadata") or {}).get("sections", [])
42
+ containers, alone = {}, []
43
+ for r in image.regions:
44
+ if r["name"] not in spans:
45
+ continue
46
+ holders = [r["container"]] if r.get("container") else [
47
+ {"view": "segment " + d["name"], "start": d["start"], "end": d["end"]}
48
+ for d in image.segments if d["start"] < r["end"] and r["start"] < d["end"]]
49
+ if not holders:
50
+ section = next((s for s in sections if s["index"] == r.get("sectionIndex")), None)
51
+ if section is not None:
52
+ holders = [{"view": "section " + section["name"], "start": section["rawStart"],
53
+ "end": section["rawStart"] + section["loadedRawSize"]}]
54
+ if not holders:
55
+ alone.append(r["name"])
56
+ for c in holders:
57
+ containers.setdefault((c["view"], c["start"], c["end"]), []).append(r["name"])
58
+ rows = []
59
+ for (view, start, end), names in containers.items():
60
+ missing = uncovered(start, end, [spans[name] for name in names])
61
+ rows.append({"container": {"view": view, "start": start, "end": end}, "regions": names, "unsearched": missing,
62
+ "partial": bool(missing),
63
+ "meaning": "partial search: callers in the unsearched ranges are not covered" if missing else
64
+ "the searched regions cover the complete container"})
65
+ if alone:
66
+ rows.append({"container": None, "regions": alone, "partial": False,
67
+ "meaning": "no overlay, section or declared segment contains these regions; the search covers them only"})
68
+ return rows
69
+
70
+
71
+ def incoming(image, config):
72
+ target = integer(config.get("target"), 0, len(image.data) - 1, "target")
73
+ if image.region(target) is None:
74
+ raise ValueError("Incoming target is outside declared code")
75
+ limit = integer(config.get("limit", 100), 1, 10000, "result limit")
76
+ seen, gaps, edges, undecoded, contested = walk(image, entries(image), config.get("instructionLimit", 10000))
77
+ hits, candidates, scanned, partial, disputed = [], [], {}, [], []
78
+
79
+ def classify(at):
80
+ if at in seen:
81
+ return "entry-path instruction"
82
+ return CONTESTED_REASON if at in contested else "raw byte candidate"
83
+ scans = config.get("searchRegions", [r["name"] for r in image.regions])
84
+ if not isinstance(scans, list) or not scans or len(set(scans)) != len(scans):
85
+ raise ValueError("searchRegions must be unique region names")
86
+ scan_limit = integer(config.get("scanLimit", 65536), 1, 1048576, "scanLimit")
87
+ scanned_bytes, read = 0, {}
88
+ for name in scans:
89
+ r = next((r for r in image.regions if r["name"] == name), None)
90
+ if r is None:
91
+ raise ValueError("Unknown search region")
92
+ read[name] = (r["start"], r["end"])
93
+ # Scan the entire declared region, including sites after the target returns.
94
+ for at in range(r["start"], r["end"]):
95
+ if scanned_bytes >= scan_limit:
96
+ gaps.append({"region": name, "unsearchedStart": at, "end": r["end"], "reason": "raw scan limit"})
97
+ read[name] = (r["start"], at)
98
+ break
99
+ scanned_bytes += 1
100
+ if image.data[at] not in (0xe8, 0x9a):
101
+ continue
102
+ ins = image.decode(at)
103
+ if ins is None or ins.mnemonic not in ("call", "lcall"):
104
+ continue
105
+ resolved, provenance = call_target(image, at, ins)
106
+ row = {"site": at, "target": resolved, "encoding": ins.mnemonic,
107
+ "classification": classify(at), "provenance": provenance, "region": name}
108
+ scanned[at] = row
109
+ if resolved == target:
110
+ (hits if at in seen else disputed if at in contested else candidates).append(row)
111
+ elif resolved is None:
112
+ partial.append(row)
113
+ for at, ins in sorted({**seen, **contested}.items()):
114
+ if at in scanned or ins.mnemonic not in ("call", "lcall"):
115
+ continue
116
+ region = image.region(at)
117
+ if region["name"] not in scans:
118
+ continue
119
+ # A reached call can start with a prefix, so the raw E8/9A scan above never sees it.
120
+ if unsupported_transfer(image, ins):
121
+ partial.append({"site": at, "target": None, "encoding": ins.mnemonic, "region": region["name"],
122
+ "classification": "unsupported control-transfer frame encoding"})
123
+ continue
124
+ resolved, provenance = call_target(image, at, ins)
125
+ row = {"site": at, "target": resolved, "encoding": ins.mnemonic,
126
+ "classification": classify(at), "provenance": provenance, "region": region["name"]}
127
+ scanned[at] = row
128
+ if resolved == target:
129
+ (hits if at in seen else disputed).append(row)
130
+ elif resolved is None:
131
+ partial.append(row)
132
+ for edge in edges:
133
+ if edge.get("overlappingTarget") and edge["site"] in scanned:
134
+ scanned[edge["site"]]["overlappingTarget"] = True
135
+ scanned[edge["site"]]["boundaryEvidence"] = edge["boundaryEvidence"]
136
+ hits.sort(key=lambda row: row["site"])
137
+ disputed.sort(key=lambda row: row["site"])
138
+ controls = config.get("controls", [])
139
+ if not isinstance(controls, list) or len(controls) > 256:
140
+ raise ValueError("Invalid positive controls")
141
+ for at in controls:
142
+ if type(at) is not int or at not in scanned or at not in seen or scanned[at]["target"] is None:
143
+ raise ValueError(f"Positive control {at} missed or not verified")
144
+ truncated = len(hits) + len(candidates) + len(disputed) + len(partial) > limit
145
+ budget = limit
146
+ def bounded(rows):
147
+ nonlocal budget
148
+ result = rows[:budget]
149
+ budget -= len(result)
150
+ return result
151
+ sections = {
152
+ "residentRelocatedFar": [h["site"] for h in hits if h["encoding"] == "lcall" and h["provenance"].get("relocation", {}).get("descriptor") is None],
153
+ "overlayFixupFar": [h["site"] for h in hits if h["encoding"] == "lcall" and h["provenance"].get("relocation", {}).get("descriptor") is not None],
154
+ "relative": [h["site"] for h in hits if h["encoding"] == "call"],
155
+ }
156
+ # Say where each unverified row sits, so a reader knows whether a route to it is still unread
157
+ # (undecoded bytes, perhaps behind a computed transfer) or whether it is bytes of another instruction.
158
+ # Undecoded ranges exclude only established instructions, so contested starts are checked first.
159
+ holes = sorted(undecoded, key=lambda u: u["start"])
160
+ hole_starts = [u["start"] for u in holes]
161
+ for row in candidates + disputed + partial:
162
+ site = row["site"]
163
+ if site in seen:
164
+ continue
165
+ if site in contested:
166
+ row["position"] = {"meaning": "start of a contested instruction"}
167
+ continue
168
+ # An x86 instruction is at most 15 bytes, so only the starts just before the site can hold it.
169
+ inside = next((at for at in range(max(site - 14, 0), site) if at in seen and site < at + seen[at].size), None)
170
+ if inside is not None:
171
+ row["position"] = {"insideInstruction": inside, "meaning": "bytes of a reached instruction; a call here needs an overlapping start"}
172
+ continue
173
+ i = bisect_right(hole_starts, site) - 1
174
+ if i >= 0 and site < holes[i]["end"]:
175
+ row["position"] = {"undecodedRange": holes[i], "meaning": "no established path reaches these bytes; an unread or computed route may"}
176
+ transfers = [{"site": e["site"], "kind": e["kind"], "reason": e["provenance"].get("reason", "target outside declared regions")}
177
+ for e in edges if e["target"] is None and image.region(e["site"])["name"] in scans]
178
+ coverage = search_coverage(image, read)
179
+ partial_scope = any(c["partial"] for c in coverage)
180
+ return {"target": target, "sections": {k: v[:limit] for k, v in sections.items()}, "confirmed": bounded(hits), "candidates": bounded(candidates),
181
+ "contested": bounded(disputed), "unresolved": bounded(partial),
182
+ "counts": {"confirmed": len(hits), "candidates": len(candidates), "contested": len(disputed), "unresolved": len(partial)},
183
+ "truncated": truncated, "controls": [scanned[at] for at in controls],
184
+ "searched": [r for r in image.regions if r["name"] in scans], "coverage": coverage, "partialSearch": partial_scope,
185
+ "unresolvedTransfers": sorted(transfers, key=lambda t: t["site"]), "undecodedRanges": undecoded, "gaps": gaps,
186
+ "negativeUsable": bool(controls) and not (hits or candidates or disputed or partial or gaps or truncated or undecoded or partial_scope),
187
+ "exclusions": ["computed call targets", "unrelocated far calls", "undeclared mappings", "prefix-started raw candidates off the entry path"],
188
+ "scope": "All bytes of declared search regions; verified calls are reachable from accepted starts. Never proves universal absence."}
189
+
190
+
191
+ def uses(image, config):
192
+ query = config.get("query", {})
193
+ offset = integer(query.get("offset"), 0, image.mask, "query offset")
194
+ width = integer(query.get("width", 1), 1, 32, "query width")
195
+ if offset + width > 1 << image.bits:
196
+ raise ValueError("Query crosses address boundary")
197
+ segment = query.get("segment")
198
+ if segment is not None:
199
+ integer(segment, 0, 65535, "query segment")
200
+ if image.flat:
201
+ raise ValueError("PE32 variable queries use flat VA offsets, not segment selectors")
202
+ mode = query.get("access", "both")
203
+ if mode not in ("read", "write", "both"):
204
+ raise ValueError("query access must be read, write or both")
205
+ controls = config.get("controls", [])
206
+ if not isinstance(controls, list) or len(controls) > 256:
207
+ raise ValueError("Invalid positive controls")
208
+ result_limit = integer(config.get("limit", 100), 1, 10000, "result limit")
209
+ seen, gaps, _, undecoded, contested = walk(image, entries(image), config.get("instructionLimit", 10000))
210
+ # A site reached only through a rejected start is as unverified as the start itself.
211
+ unverified = {g["site"] for g in gaps if g.get("reason") == OVERLAP_REASON} | set(contested)
212
+ matches, unresolved, unique = [], [], set()
213
+ # Trace each established entry independently; never decode a whole segment as one stream.
214
+ remaining = integer(config.get("totalSteps", 20000), 1, 100000, "totalSteps")
215
+ # One string iteration budget spans every traced entry, as totalSteps does.
216
+ string_remaining = integer(config.get("stringIterations", 4096), 0, 65536, "string iteration budget")
217
+ entry_limit = integer(config.get("entryLimit", 64), 1, 256, "entryLimit")
218
+ # CFG points where value propagation stopped (or never started), with why; operands after them are inventoried below.
219
+ stops = {}
220
+ established = entries(image)
221
+ for index, at in enumerate(established):
222
+ if remaining <= 0 or index >= entry_limit:
223
+ gaps.append({"entry": at, "reason": "entry or total instruction budget exhausted"})
224
+ for root in established[index:]:
225
+ stops.setdefault(root, "entry not traced: entry or total instruction budget exhausted")
226
+ break
227
+ report = trace(image, {**config, "entry": at, "totalSteps": remaining, "stringIterations": string_remaining})
228
+ remaining -= report["stepsUsed"]
229
+ string_remaining -= report["stringIterationsUsed"]
230
+ if not report["completeWithinModel"]:
231
+ gaps.append({"entry": at, "reason": "incomplete path effects", "stops": list({p["stop"] for p in report["paths"] if p["stop"]})})
232
+ for p in report["paths"]:
233
+ if p["stop"] and p["stopSite"] is not None:
234
+ stops.setdefault(p["stopSite"], p["stop"])
235
+ for g in report["gaps"]:
236
+ if "site" in g:
237
+ stops.setdefault(g["site"], g["reason"])
238
+ for path in report["paths"]:
239
+ for e in path["events"]:
240
+ if e["kind"] not in ("read", "write") or mode not in ("both", e["kind"]):
241
+ continue
242
+ if e["site"] not in seen:
243
+ key = (e["site"], e["kind"], "unverified-boundary")
244
+ if key not in unique:
245
+ classification = ("unverified overlapping instruction path" if e["site"] in unverified
246
+ else "outside the bounded entry walk")
247
+ unresolved.append({**e, "classification": classification}); unique.add(key)
248
+ continue
249
+ off, seg = e["offset"]["value"], e["segment"]["value"]
250
+ if off is None or ((segment is not None or image.flat) and seg is None):
251
+ key = (e["site"], e["kind"], repr(e["offset"]["expression"]), repr(e["segment"]["expression"]))
252
+ if key not in unique:
253
+ unresolved.append(e); unique.add(key)
254
+ continue
255
+ if image.flat:
256
+ off += seg
257
+ if segment is None:
258
+ overlap = max(offset, off) < min(offset + width, off + e["width"])
259
+ else:
260
+ a, b = segment * 16 + offset, seg * 16 + off
261
+ overlap = max(a, b) < min(a + width, b + e["width"])
262
+ if overlap:
263
+ key = (e["site"], e["kind"], repr(e["value"]["expression"]))
264
+ if key not in unique:
265
+ matches.append(e); unique.add(key)
266
+ # Operand discovery is distinct from value propagation. An unread call stops
267
+ # trace effects, but it must not erase a later instruction reached by the CFG.
268
+ # Only the CFG reachable from a stop is inventoried; fully traced accesses keep their values.
269
+ # Those operands stay out of matches: each names the stops whose CFG reaches it, so
270
+ # reading one callee later shows exactly which accesses depended on it.
271
+ reported = {(e["site"], e["kind"]) for e in matches + unresolved}
272
+ instruction_limit = config.get("instructionLimit", 10000)
273
+ after_stop, stop_gaps, _, _, _ = walk(image, list(stops), instruction_limit) if stops else ({}, [], None, None, None)
274
+ gaps.extend(g for g in stop_gaps if g["reason"] == "instruction limit")
275
+ # A call past a stop was never traced either, so code after it also depends on it returning.
276
+ starts = [(root, root, reason) for root, reason in stops.items()]
277
+ starts += [(at, at + ins.size, "call past a stop; assumed to return")
278
+ for at, ins in after_stop.items() if ins.mnemonic in ("call", "lcall") and at not in stops]
279
+ depends = {}
280
+ for site, start, reason in sorted(starts):
281
+ reached, _, _, _, _ = walk(image, [start], instruction_limit)
282
+ for at in reached:
283
+ depends.setdefault(at, []).append({"site": site, "reason": reason})
284
+ conditional = []
285
+ for at, ins in sorted(after_stop.items()):
286
+ if ins.mnemonic == "lea":
287
+ continue # Address formation is not a memory use.
288
+ if ins.mnemonic == "xlatb":
289
+ gaps.append({"site": at, "reason": "implicit DS:[(E)BX+AL] operand is not inventoried"}); continue
290
+ state = None
291
+ for operand in ins.operands:
292
+ if operand.type != X86_OP_MEM:
293
+ continue
294
+ kinds = [kind for flag, kind in ((CS_AC_READ, "read"), (CS_AC_WRITE, "write"))
295
+ if operand.access & flag and mode in ("both", kind) and (at, kind) not in reported]
296
+ if not kinds:
297
+ continue
298
+ if state is None:
299
+ # Registers are unknown here; name them for this operand site so no entry value is implied.
300
+ state = State(at, image, {})
301
+ state.regs.update({r: unknown(f"CFG-operand:{at}:{r}", ALIASES[r][2]) for r in REGISTERS if r != "cs"})
302
+ try:
303
+ segment_value, offset_value, segment_name = state.address(ins, operand)
304
+ except StopPath as error:
305
+ gaps.append({"site": at, "reason": str(error)}); continue
306
+ size = memory_width(ins, operand)
307
+ off = offset_value.number
308
+ overlaps = off is not None and max(offset, off) < min(offset + width, off + size)
309
+ if off is not None and not overlaps:
310
+ continue
311
+ for kind in kinds:
312
+ # A concrete segment query cannot bind an unpropagated DS/SS.
313
+ conditional.append({"site": at, "kind": kind, "width": size,
314
+ "segment": segment_value.report(), "offset": offset_value.report(),
315
+ "value": unknown(f"CFG-operand:{at}", size * 8).report(),
316
+ "effectiveSegmentRegister": segment_name,
317
+ "address": "overlaps query" if overlaps and segment is None else "possible alias",
318
+ "classification": ("unverified overlapping instruction path" if at in unverified else
319
+ "entry-CFG operand past a stop; values and callee effects unresolved"),
320
+ "dependsOn": depends.get(at, []),
321
+ "reachability": "conditional on encoded branch outcomes and on execution continuing past every named stop"})
322
+ # A control proves the search reaches a known use, which an operand found past a stop still shows.
323
+ found = matches + [e for e in conditional if e["address"] == "overlaps query" and e["site"] not in unverified]
324
+ for at in controls:
325
+ if type(at) is not int or not any(e["site"] == at for e in found):
326
+ raise ValueError(f"Positive variable-use control {at} missed; negative result rejected")
327
+ raw = []
328
+ scanned_bytes = 0
329
+ scan_limit = integer(config.get("scanLimit", 65536), 1, 1048576, "scanLimit")
330
+ for r in image.regions:
331
+ for at in range(r["start"], r["end"]):
332
+ if scanned_bytes >= scan_limit:
333
+ gaps.append({"region": r["name"], "unsearchedStart": at, "end": r["end"], "reason": "raw scan limit"})
334
+ break
335
+ scanned_bytes += 1
336
+ ins = image.decode(at)
337
+ if ins and at not in seen and any(o.type == X86_OP_MEM and max(offset, o.mem.disp & image.mask) < min(offset + width, (o.mem.disp & image.mask) + max(o.size, 1))
338
+ for o in ins.operands):
339
+ if len(raw) < result_limit:
340
+ raw.append({"site": at, "size": ins.size, "classification": "unverified operand candidate"})
341
+ else:
342
+ gaps.append({"reason": "raw candidate limit"}); break
343
+ truncated = len(matches) + len(unresolved) + len(conditional) > result_limit
344
+ kept_matches = matches[:result_limit]
345
+ kept_unresolved = unresolved[:result_limit - len(kept_matches)]
346
+ return {"query": query, "matches": kept_matches, "unresolvedAccesses": kept_unresolved,
347
+ "conditionalAccesses": conditional[:result_limit - len(kept_matches) - len(kept_unresolved)],
348
+ "rawCandidates": raw, "controls": controls, "truncated": truncated, "gaps": gaps, "undecodedRanges": undecoded,
349
+ "negativeUsable": bool(controls) and not (matches or unresolved or conditional or raw or gaps or truncated or undecoded),
350
+ "interpretation": "Unknown segments or addresses remain possible aliases; raw candidates are never counted as uses. "
351
+ "Matches were traced; conditionalAccesses were reached only past the stops each one names."}
352
+
353
+
354
+ # Legacy prefix bytes in encoded order (segment overrides, operand/address size, LOCK/REP).
355
+ PREFIX_BYTES = frozenset((0x26, 0x2e, 0x36, 0x3e, 0x64, 0x65, 0x66, 0x67, 0xf0, 0xf2, 0xf3))
356
+
357
+
358
+ def prefixes(ins):
359
+ count = 0
360
+ while count < ins.size and ins.bytes[count] in PREFIX_BYTES:
361
+ count += 1
362
+ return list(ins.bytes[:count])
363
+
364
+
365
+ def _relative_transfer(ins):
366
+ m = base_mnemonic(ins)
367
+ return m in ("call", "jmp") or m.startswith(("j", "loop"))
368
+
369
+
370
+ def operand_candidates(image, config):
371
+ query = config.get("query", {})
372
+ if not isinstance(query, dict):
373
+ raise ValueError("Candidate query must be an object")
374
+ value = integer(query.get("offset"), 0, image.mask, "candidate literal")
375
+ limit = integer(config.get("limit", 100), 1, 10000, "candidate result limit")
376
+ scan_limit = integer(config.get("scanLimit", 100000), 1, 1048576, "candidate scan limit")
377
+ seen, gaps, _, _, contested = walk(image, entries(image), config.get("instructionLimit", 10000))
378
+ ambiguous = {g["site"] for g in gaps if g.get("reason") == OVERLAP_REASON} | set(contested)
379
+ intervals = sorted((at, at + ins.size) for at, ins in seen.items())
380
+ starts = [a for a, _ in intervals]
381
+ rows, controls_found, scanned, read = [], set(), 0, {}
382
+ low = value & 0xff
383
+ counts = {"verifiedMemoryUses": 0, "verifiedOtherOperands": 0, "rejectedOverlap": 0, "unresolvedBoundary": 0}
384
+ for region in image.regions:
385
+ end = region["start"]
386
+ for at in range(region["start"], region["end"]):
387
+ if scanned >= scan_limit:
388
+ break
389
+ scanned += 1
390
+ end = at + 1
391
+ # Every encoded width of the literal (including sign-extended 8-bit forms) holds its low byte.
392
+ if image.data.find(low, at, min(at + 15, region["end"])) < 0:
393
+ continue
394
+ ins = image.decode(at)
395
+ if ins is None:
396
+ continue
397
+ for index, operand in enumerate(ins.operands):
398
+ # Only encoded literals: implicit forms (SHL r,1; [BX]) and relative branch targets are excluded.
399
+ if operand.type == X86_OP_MEM:
400
+ if not ins.disp_size:
401
+ continue
402
+ elif operand.type != X86_OP_IMM or not ins.imm_size or _relative_transfer(ins):
403
+ continue
404
+ literal = operand.mem.disp if operand.type == X86_OP_MEM else operand.imm
405
+ if literal & image.mask != value:
406
+ continue
407
+ # At most 15 bytes precede a partly overlapping x86 instruction.
408
+ lo, hi = bisect_right(starts, at - 15), bisect_right(starts, at + ins.size - 1)
409
+ overlaps = [{"site": a, "end": z} for a, z in intervals[lo:hi] if a != at and a < at + ins.size and z > at]
410
+ memory = operand.type == X86_OP_MEM and ins.mnemonic != "lea" and bool(operand.access & (CS_AC_READ | CS_AC_WRITE))
411
+ classification = ("unresolvedBoundary" if at in ambiguous else
412
+ "verifiedMemoryUses" if at in seen and memory else
413
+ "verifiedOtherOperands" if at in seen else
414
+ "rejectedOverlap" if overlaps else "unresolvedBoundary")
415
+ counts[classification] += 1
416
+ if classification == "verifiedMemoryUses":
417
+ controls_found.add(at)
418
+ if len(rows) >= limit:
419
+ continue
420
+ rows.append({"site": at, "end": at + ins.size, "operandIndex": index,
421
+ "operandKind": "memory" if operand.type == X86_OP_MEM else "immediate",
422
+ "width": 2 + ins.operands[0].size if ins.mnemonic in ("lds", "les", "lss", "lfs", "lgs") and operand.type == X86_OP_MEM else operand.size,
423
+ "prefixes": prefixes(ins), "mnemonic": ins.mnemonic,
424
+ "access": [name for flag, name in ((CS_AC_READ, "read"), (CS_AC_WRITE, "write")) if operand.access & flag],
425
+ "effectiveSegmentRegister": segment_register(ins, operand.mem) if operand.type == X86_OP_MEM else None,
426
+ "classification": classification, "countedAsUse": classification == "verifiedMemoryUses",
427
+ "overlapsVerified": overlaps, "region": region["name"]})
428
+ read[region["name"]] = (region["start"], end)
429
+ controls = config.get("controls", [])
430
+ if not isinstance(controls, list) or len(controls) > 256 or any(type(at) is not int or at not in controls_found for at in controls):
431
+ raise ValueError("Candidate positive control missed or is not a verified memory use")
432
+ total = sum(counts.values())
433
+ groups = []
434
+ for row in sorted(rows, key=lambda r: (r["site"], r["end"])):
435
+ if groups and row["site"] < groups[-1]["end"]:
436
+ groups[-1]["end"] = max(groups[-1]["end"], row["end"])
437
+ groups[-1]["members"].append({"site": row["site"], "operandIndex": row["operandIndex"]})
438
+ else:
439
+ groups.append({"start": row["site"], "end": row["end"], "members": [{"site": row["site"], "operandIndex": row["operandIndex"]}]})
440
+ coverage = search_coverage(image, read)
441
+ region_coverage = [{"region": r["name"], "declared": {"start": r["start"], "end": r["end"]},
442
+ "unsearched": uncovered(r["start"], r["end"], [read[r["name"]]])} for r in image.regions]
443
+ return {"query": query, "regionCoverage": region_coverage, "candidates": rows, "counts": counts, "truncated": total > limit,
444
+ "scannedStarts": scanned, "coverage": coverage, "partialSearch": any(r["partial"] for r in coverage) or any(r["unsearched"] for r in region_coverage),
445
+ "overlapGroups": [g | {"completeWithinSearch": total <= limit} for g in groups if len({m["site"] for m in g["members"]}) > 1],
446
+ "controls": controls, "gaps": gaps,
447
+ "exclusions": ["computed displacements", "implicit operands", "relative branch targets", "segment-value alias proof", "runtime reachability"],
448
+ "interpretation": "Only entry-path memory operand starts count as uses of this literal representation; local decodability never establishes a boundary. Groups cover returned candidates only when truncated."}
449
+
450
+
451
+ def dispatch(image, config):
452
+ d = config.get("dispatch", {})
453
+ site = integer(d.get("site"), 0, len(image.data) - 1, "dispatch site")
454
+ table = d.get("table", {})
455
+ start = integer(table.get("start"), 0, len(image.data), "table start")
456
+ count = integer(table.get("count"), 1, 4096, "table count")
457
+ stride = integer(table.get("stride"), 1, 64, "table stride")
458
+ width = integer(table.get("width", 2), 1, 4, "table width")
459
+ if width > stride or start + count * stride > len(image.data):
460
+ raise ValueError("Table layout exceeds declared source bounds")
461
+ if not table.get("countEvidence") or not d.get("indexEvidence"):
462
+ raise ValueError("Dispatch requires count and index provenance")
463
+ values = d.get("inputs")
464
+ if not isinstance(values, list) or not 1 <= len(values) <= 256:
465
+ raise ValueError("Dispatch requires 1..256 explicit input cases")
466
+ from .machine import ALIASES
467
+ input_reg, index_reg = d.get("inputRegister"), d.get("indexRegister")
468
+ if input_reg not in ALIASES or index_reg not in ALIASES:
469
+ raise ValueError("Dispatch needs valid input/index registers")
470
+ rows = [int.from_bytes(image.data[start+i*stride:start+i*stride+width], "little") for i in range(count)]
471
+ results = []
472
+ divisor = integer(d.get("indexDivisor", stride), 1, 64, "index divisor")
473
+ ins = image.decode(site)
474
+ if ins is None or ins.mnemonic != "jmp" or len(ins.operands) != 1 or ins.operands[0].type != X86_OP_MEM:
475
+ raise ValueError("Dispatch site must be an indirect near memory jump")
476
+ if ins.addr_size != image.bits // 8:
477
+ raise ValueError("Dispatch address-size override is unsupported")
478
+ mem = ins.operands[0].mem
479
+ if image.flat and mem.segment and ins.reg_name(mem.segment) in ("fs", "gs"):
480
+ raise ValueError("Dispatch table has an unknown segment base")
481
+ address_reg = ins.reg_name(mem.base) if mem.base else None
482
+ scale = 1
483
+ if image.flat and mem.index and not mem.base:
484
+ address_reg, scale = ins.reg_name(mem.index), mem.scale
485
+ elif mem.index:
486
+ raise ValueError("Dispatch needs one address register")
487
+ if address_reg != index_reg or ins.operands[0].size != width or divisor != stride:
488
+ raise ValueError("Dispatch index register/stride/width differs from the encoded access")
489
+ if integer(table.get("offset"), 0, image.mask, "table memory offset") != (mem.disp & image.mask) or not table.get("mappingEvidence"):
490
+ raise ValueError("Dispatch requires the encoded table displacement and mapping evidence")
491
+ if image.config.get("peMetadata") and image.file_offset(table["offset"], count * stride) != start:
492
+ raise ValueError("Dispatch table mapping differs from PE source sections")
493
+ for value in values:
494
+ integer(value, 0, (1 << ALIASES[input_reg][2]) - 1, "input value")
495
+ report = trace(image, {**config, "registers": {**config.get("registers", {}), input_reg: value}})
496
+ outcomes = []
497
+ for path in report["paths"]:
498
+ reached = bool(path["instructionPath"]) and path["instructionPath"][-1] == site
499
+ index_value = path["registers"].get(index_reg, {}).get("value")
500
+ if reached and index_value is not None:
501
+ index_value *= scale
502
+ if index_value % divisor or index_value // divisor >= count:
503
+ outcomes.append({"status": "out-of-layout index", "encodedIndex": index_value})
504
+ else:
505
+ index = index_value // divisor
506
+ outcomes.append({"status": "selected", "position": index, "rawTarget": rows[index], "encodedIndex": index_value})
507
+ else:
508
+ outcomes.append({"status": "returned-before-dispatch" if path["returned"] else "unresolved", "stop": path["stop"]})
509
+ outcomes[-1]["transformations"] = [e for e in path["events"] if e["kind"] == "arithmetic"]
510
+ outcomes[-1]["guards"] = path["guards"]
511
+ results.append({"input": value, "outcomes": outcomes, "gaps": report["gaps"]})
512
+ return {"cases": results, "table": table, "indexEvidence": d["indexEvidence"],
513
+ "interpretation": "Raw table targets; indexDivisor is an evidenced layout mapping. Cases do not establish native input coverage."}
514
+
515
+
516
+ def allocations(report, config):
517
+ requests = config.get("allocations", [])
518
+ if not isinstance(requests, list) or not 1 <= len(requests) <= 64:
519
+ raise ValueError("Declare 1..64 allocation call contracts")
520
+ results = []
521
+ flat = config.get("addressModel") == "flat32"
522
+ paragraph = 1 if flat else 16
523
+ for a in requests:
524
+ site = integer(a.get("site"), 0, 0x7fffffff, "allocation site")
525
+ unit = integer(a.get("unitBytes"), 1, 65536, "allocator unit bytes")
526
+ if not a.get("unitEvidence") or not a.get("requestRegister"):
527
+ raise ValueError("Allocation unit and request register require provenance")
528
+ for path_index, path in enumerate(report["paths"]):
529
+ for event in path["events"]:
530
+ if event["kind"] != "call" or event["site"] != site:
531
+ continue
532
+ request = event["registers"].get(a["requestRegister"])
533
+ if request is None:
534
+ raise ValueError("Unsupported allocation request register")
535
+ returns = next((e for e in path["events"][event["order"]+1:] if e["kind"] == "call-return" and e["callSite"] == site), None)
536
+ header = a.get("headerBytes")
537
+ if header is not None:
538
+ integer(header, 0, 65535, "header bytes")
539
+ if not a.get("headerEvidence"):
540
+ raise ValueError("Header extent needs evidence")
541
+ extent, pointer, capacity, comparisons = None, None, None, []
542
+ for name in ("extent", "pointer"):
543
+ observation = a.get(name)
544
+ if observation is None:
545
+ continue
546
+ if not isinstance(observation, dict) or not observation.get("evidence"):
547
+ raise ValueError("Allocation observations need evidence")
548
+ integer(observation.get("site"), 0, 0x7fffffff, "observation site")
549
+ checkpoint = next((e for e in path["events"] if e["kind"] == "checkpoint" and e["site"] == observation["site"] and e["order"] > event["order"]), None)
550
+ if checkpoint is None:
551
+ continue
552
+ if name == "extent":
553
+ unit_bytes = integer(observation.get("unitBytes"), 1, 65536, "extent units")
554
+ value = checkpoint["registers"].get(observation.get("register"))
555
+ if value is None:
556
+ raise ValueError("Invalid extent register")
557
+ extent = {"value": value, "unitBytes": unit_bytes, "evidence": observation["evidence"], "site": checkpoint["site"]}
558
+ if value["value"] is not None:
559
+ capacity = value["value"] * unit_bytes
560
+ else:
561
+ if flat and observation.get("segmentRegister") is not None:
562
+ raise ValueError("Flat allocation pointer observations use only offsetRegister")
563
+ segment = ({"bits": 32, "expression": ("constant", 0), "value": 0, "producers": [],
564
+ "provenance": "PE32 flat base assumption"} if flat
565
+ else checkpoint["registers"].get(observation.get("segmentRegister")))
566
+ offset = checkpoint["registers"].get(observation.get("offsetRegister"))
567
+ if segment is None or offset is None:
568
+ raise ValueError("Invalid returned pointer registers")
569
+ pointer = {"segment": segment, "offset": offset, "evidence": observation["evidence"], "site": checkpoint["site"], "order": checkpoint["order"]}
570
+ if pointer and capacity is not None:
571
+ seg, off = pointer["segment"]["value"], pointer["offset"]["value"]
572
+ for write in path["events"][pointer["order"] + 1:]:
573
+ if write["kind"] != "write":
574
+ continue
575
+ ws, wo = write["segment"]["value"], write["offset"]["value"]
576
+ relative = None if None in (seg, off, ws, wo) else (ws - seg) * paragraph + wo - off
577
+ comparisons.append({"site": write["site"], "relativeStart": relative, "width": write["width"],
578
+ "withinObservedExtent": None if relative is None else 0 <= relative and relative + write["width"] <= capacity,
579
+ "association": "address comparison only; write ownership remains a reading"})
580
+ results.append({"path": path_index, "site": site, "request": request,
581
+ "requestModulus": 1 << request["bits"], "unitBytes": unit, "unitEvidence": a["unitEvidence"],
582
+ "requestedBytes": None if request["value"] is None else request["value"] * unit,
583
+ "headerBytes": header, "headerEvidence": a.get("headerEvidence"),
584
+ "returnedRegisters": returns["registers"] if returns else None,
585
+ "orderedWrites": [e for e in path["events"][event["order"]+1:] if e["kind"] == "write"],
586
+ "arithmetic": [e for e in path["events"] if e["kind"] == "arithmetic"],
587
+ "guards": path["guards"], "pathStop": path["stop"],
588
+ "allocatorEffects": "conditional model; memory unresolved" if returns and returns.get("modeled") else "see path writes and unresolved exits",
589
+ "extentObservation": extent, "pointerObservation": pointer, "writeComparisons": comparisons,
590
+ "observedExtentBytes": capacity,
591
+ "capacity": "conditional on evidenced extent units and pointer identity" if extent else "unresolved: request units and bounded writes do not establish allocated extent",
592
+ "rollback": "unproven; failure returns do not undo earlier writes"})
593
+ return {"allocations": results, "paths": report["paths"], "gaps": report["gaps"], "completeWithinModel": report["completeWithinModel"]}
594
+
595
+
596
+ def operand_provenance(image, config):
597
+ query = config.get("query", {})
598
+ if not isinstance(query, dict):
599
+ raise ValueError("Operand query must be an object")
600
+ site = integer(query.get("site"), 0, len(image.data)-1, "instruction site")
601
+ word_site = integer(query.get("operandSite"), 0, len(image.data)-2, "segment operand site")
602
+ offset = integer(query.get("targetOffset", 0), 0, 65535, "target offset")
603
+ if image.flat:
604
+ raise ValueError("Segment relocation operands require the segmented16 model")
605
+ seen, gaps, _, _, _ = walk(image, entries(image), config.get("instructionLimit", 10000))
606
+ ins = seen.get(site)
607
+ if ins is None:
608
+ raise ValueError("Operand instruction is not a verified entry-path boundary")
609
+ if ins.mnemonic not in ("mov", "push") or ins.imm_size != 2 or site + ins.imm_offset != word_site:
610
+ raise ValueError("Selected word is not the complete 16-bit immediate of a supported MOV/PUSH")
611
+ raw = int.from_bytes(image.data[word_site:word_site+2], "little")
612
+ fixup = image.fixups.get(word_site)
613
+ destination = ins.operands[0]
614
+ kind = "pushed word" if ins.mnemonic == "push" else ("stored word" if destination.type == X86_OP_MEM else "register immediate")
615
+ result = {"instructionSite":site, "operandSite":word_site, "mnemonic":ins.mnemonic,
616
+ "instructionSize":ins.size, "immediateWidth":2, "representation":kind,
617
+ "destinationRegister":ins.reg_name(destination.reg) if destination.type == X86_OP_REG else None,
618
+ "raw":raw, "rawToken":f"{raw:04X}", "relocated":fixup is not None,
619
+ "boundaryEvidence":"decoded from established entries", "gaps":gaps,
620
+ "nativeReachability":"unconfirmed"}
621
+ if fixup:
622
+ if "raw" not in fixup:
623
+ raise ValueError("Relocation lacks its source raw word; use the hash-guarded source loader")
624
+ if fixup["raw"] != raw:
625
+ raise ValueError("Relocation raw word disagrees with the selected operand")
626
+ result.update({"relocation":fixup, "descriptor":fixup.get("descriptor"),
627
+ "canonicalMappedSegment":fixup["segment"],
628
+ "loadedAddress":f"{fixup['segment']:04X}:{offset:04X}"})
629
+ else:
630
+ result["reason"] = "No declared relocation or fixup; raw operand does not establish a segment"
631
+ return result
632
+
633
+
634
+ def _segmented(segment, offset):
635
+ return f"{segment:04X}:{offset:04X}"
636
+
637
+
638
+ def _citation(image, target):
639
+ """How the standard cites a canonical file offset: resident code by mapped address, overlay code by file offset."""
640
+ region = image.region(target)
641
+ if region is None:
642
+ return {"fileOffset": target, "region": None, "citation": None,
643
+ "reason": "canonical target is outside the declared regions"}
644
+ ip = (region["ip"] + target - region["start"]) & image.mask
645
+ if image.flat:
646
+ return {"fileOffset": target, "region": region["name"], "citation": f"{ip:08X}", "form": "preferred-base virtual address"}
647
+ if region.get("resident", False):
648
+ return {"fileOffset": target, "region": region["name"], "citation": _segmented(region["segment"], ip),
649
+ "form": "resident load-image address"}
650
+ return {"fileOffset": target, "region": region["name"], "citation": f"+0x{target:08X}",
651
+ "form": "file offset; prefix the path the build entry gives",
652
+ "analysisView": _segmented(region["segment"], ip)}
653
+
654
+
655
+ def call_target_report(image, config):
656
+ """One direct transfer: the raw operand, its relocation or fixup chain and the address it may be cited by."""
657
+ query = config.get("query", {})
658
+ if not isinstance(query, dict):
659
+ raise ValueError("Target query must be an object")
660
+ site = integer(query.get("site"), 0, len(image.data) - 1, "call site")
661
+ ins = image.decode(site)
662
+ if ins is None or ins.mnemonic not in ("call", "lcall", "jmp", "ljmp") or not ins.operands or ins.operands[0].type != X86_OP_IMM:
663
+ raise ValueError("Target site must decode as a direct call or jump inside a declared region")
664
+ if unsupported_transfer(image, ins):
665
+ raise ValueError("Operand-size or far control transfer is outside the selected frame model")
666
+ far = ins.mnemonic in ("lcall", "ljmp")
667
+ if far:
668
+ target, provenance = image.far_target(site, ins)
669
+ if "rawSegment" not in provenance:
670
+ raise ValueError("Only the ptr16:16 far transfer encoding is supported")
671
+ seen, gaps, _, _, contested = walk(image, entries(image), config.get("instructionLimit", 10000))
672
+ # A walk stopped by its instruction limit leaves later boundaries unverified, not disproved.
673
+ truncated = any(g.get("reason") == "instruction limit" for g in gaps)
674
+ boundary = ("entry-path instruction" if site in seen else CONTESTED_REASON if site in contested
675
+ else "raw byte candidate; instruction boundary unverified"
676
+ + ("; the entry walk stopped at its instruction limit" if truncated else ""))
677
+ result = {"site": site, "mnemonic": ins.mnemonic, "size": ins.size, "boundary": boundary,
678
+ "nativeReachability": "unconfirmed"}
679
+ region = image.region(site)
680
+ if not far:
681
+ loaded = ins.operands[0].imm & image.mask
682
+ target = image.near_target(site, loaded)
683
+ result.update({"encoding": "relative", "loadedTarget": loaded,
684
+ "loadedAddress": f"{loaded:08X}" if image.flat else _segmented(region["segment"], loaded),
685
+ "relocated": None, "relocation": "relative transfers carry no relocation",
686
+ "mapping": "source PE section table" if image.config.get("peMetadata") else f"declared mapping of region {region['name']}",
687
+ "canonicalTarget": target, "target": None if target is None else _citation(image, target)})
688
+ else:
689
+ raw_offset, raw_segment = provenance["offset"], provenance["rawSegment"]
690
+ result.update({"encoding": "ptr16:16", "operandSite": site + 3, "rawOffset": raw_offset, "rawSegment": raw_segment,
691
+ "rawOperand": _segmented(raw_segment, raw_offset)})
692
+ fixup = provenance.get("relocation")
693
+ if fixup is None:
694
+ result.update({"relocated": False, "canonicalTarget": None, "target": None,
695
+ "reason": "no relocation or fixup covers the segment word; the raw operand is not a loaded address and no target is assigned"})
696
+ else:
697
+ if "raw" in fixup and fixup["raw"] != raw_segment:
698
+ raise ValueError("Relocation raw word disagrees with the encoded segment operand")
699
+ overlay = fixup.get("descriptor") is not None
700
+ result.update({"relocated": True, "kind": "FBOV fixup" if overlay else "MZ relocation", "evidence": fixup["evidence"],
701
+ "loadSegment": fixup.get("loadSegment"), "resolvedSegment": fixup["segment"],
702
+ "loadedAddress": _segmented(fixup["segment"], raw_offset)})
703
+ if overlay:
704
+ result.update({"storedWord": raw_segment, "descriptor": fixup["descriptor"], "storedLowBits": raw_segment & 7,
705
+ "descriptorSegment": fixup.get("descriptorSegment"), "descriptorFlags": fixup.get("descriptorFlags"),
706
+ "note": "the stored word is the descriptor index shifted left by three; it is neither the index nor a segment"})
707
+ else:
708
+ result["note"] = "the raw word is relative to the load image; the loader adds the load segment"
709
+ for name in ("loadedTarget", "trampoline", "targetError"):
710
+ if fixup.get(name) is not None:
711
+ result[name] = fixup[name]
712
+ if "raw" not in fixup:
713
+ result["mappingProvenance"] = "relocation metadata supplied by the caller, not read from the source"
714
+ if fixup.get("targetError") is not None:
715
+ # The source loader could not resolve the loaded address; a declared analysis view must not stand in for it.
716
+ target = None
717
+ result["reason"] = "the source loader could not resolve the loaded address; no target is assigned"
718
+ result["canonicalTarget"] = target
719
+ result["target"] = None if target is None else _citation(image, target)
720
+ analyzer = query.get("analyzerAddress")
721
+ if analyzer is not None:
722
+ if not isinstance(analyzer, dict) or not analyzer.get("evidence"):
723
+ raise ValueError("analyzerAddress needs segment, offset and evidence")
724
+ if image.flat:
725
+ raise ValueError("analyzerAddress compares segment:offset identities and needs the segmented16 model")
726
+ shown = _segmented(integer(analyzer.get("segment"), 0, 65535, "analyzer segment"),
727
+ integer(analyzer.get("offset"), 0, 65535, "analyzer offset"))
728
+ cited = result.get("target") or {}
729
+ identities = {"raw operand": result.get("rawOperand"), "loaded address": result.get("loadedAddress"),
730
+ "canonical target": cited.get("analysisView") or cited.get("citation")}
731
+ matched = [name for name, value in identities.items() if value == shown]
732
+ result["analyzer"] = {"address": shown, "evidence": analyzer["evidence"], "matches": matched, "disagrees": not matched,
733
+ "interpretation": ("equal only to the raw operand, which names unrelocated bytes"
734
+ if matched == ["raw operand"] else
735
+ "kept beside the derived chain; it never replaces the relocation, descriptor or trampoline identities")}
736
+ result["gaps"] = [g for g in gaps if g.get("site") == site or g.get("reason") == "instruction limit"]
737
+ result["walkComplete"] = not truncated
738
+ return result
739
+
740
+
741
+ def body(image, entry, limit=10000):
742
+ """Every instruction one entry reaches without entering a callee, and every way out of it.
743
+
744
+ Calls, interrupts and port accesses are followed to the next instruction, and each such
745
+ continuation is listed as an assumption. A direct jump or conditional branch to another
746
+ established entry or another region, and every far jump, is a tail transfer.
747
+ """
748
+ integer(limit, 1, 100000, "instruction limit")
749
+ established = set(entries(image))
750
+ pending, seen, exits, calls, gaps, assumed, shared = [entry], {}, [], [], [], [], set()
751
+
752
+ def leaves(at, target):
753
+ return (target in established and target != entry) or image.region(target) is not image.region(at)
754
+ while pending:
755
+ at = pending.pop()
756
+ if at in seen:
757
+ continue
758
+ if len(seen) >= limit:
759
+ gaps.append({"site": at, "reason": "instruction limit"})
760
+ break
761
+ ins = image.decode(at)
762
+ if ins is None:
763
+ gaps.append({"site": at, "reason": "undecoded or unmapped edge"})
764
+ continue
765
+ seen[at] = ins
766
+ if at != entry and at in established:
767
+ shared.add(at)
768
+ m, following = base_mnemonic(ins), at + ins.size
769
+ if unsupported_transfer(image, ins):
770
+ gaps.append({"site": at, "reason": "unsupported control-transfer frame encoding"})
771
+ continue
772
+ if m in RETURNS:
773
+ exits.append({"site": at, "kind": RETURNS[m], "cleanupBytes": ins.operands[0].imm if ins.operands else 0})
774
+ continue
775
+ if m == "hlt":
776
+ exits.append({"site": at, "kind": "halt"})
777
+ continue
778
+ if m in INTERRUPTS or m in PORTS:
779
+ assumed.append({"site": at, "assumption": ("the interrupt returns to the next instruction" if m in INTERRUPTS
780
+ else "the port access continues to the next instruction")})
781
+ pending.append(following)
782
+ continue
783
+ if m in ("jmp", "ljmp"):
784
+ declaration = image.indirect_jumps.get(at)
785
+ if declaration is not None:
786
+ targets = sorted(set(row["target"] for row in declaration["rows"]))
787
+ # The full declaration is reported once, in indirectJumpDeclarations.
788
+ assumed.append({"site": at, "assumption": "indirect jump consumes the declared source table",
789
+ "targets": targets, "exhaustive": declaration["exhaustive"]})
790
+ for target in targets:
791
+ if leaves(at, target):
792
+ exits.append({"site": at, "kind": "tail transfer", "target": target,
793
+ "mapping": "declared indirect jump table"})
794
+ else:
795
+ pending.append(target)
796
+ if not declaration["exhaustive"]:
797
+ # Undeclared routes remain a way out of the body, as for any unresolved computed jump.
798
+ exits.append({"site": at, "kind": "unresolved jump", "reason": "indirect jump table is not declared exhaustive"})
799
+ gaps.append({"site": at, "reason": "indirect jump table is not declared exhaustive"})
800
+ continue
801
+ target, provenance = call_target(image, at, ins)
802
+ if target is None:
803
+ exits.append({"site": at, "kind": "unresolved jump", "reason": provenance.get("reason")})
804
+ gaps.append({"site": at, "reason": "jump target unresolved; the body may continue elsewhere"})
805
+ elif m == "ljmp" or leaves(at, target):
806
+ exits.append({"site": at, "kind": "tail transfer", "target": target})
807
+ else:
808
+ pending.append(target)
809
+ continue
810
+ if m in ("call", "lcall"):
811
+ target, provenance = call_target(image, at, ins)
812
+ calls.append({"site": at, "target": target, "encoding": m,
813
+ **({} if target is not None else {"reason": provenance.get("reason")})})
814
+ assumed.append({"site": at, "assumption": "the callee returns to the next instruction"})
815
+ pending.append(following)
816
+ continue
817
+ if m.startswith("j") or m.startswith("loop"):
818
+ target, provenance = call_target(image, at, ins)
819
+ if target is None:
820
+ gaps.append({"site": at, "reason": provenance.get("reason", "branch target outside declared regions")})
821
+ elif leaves(at, target):
822
+ exits.append({"site": at, "kind": "tail transfer", "target": target, "conditional": True})
823
+ else:
824
+ pending.append(target)
825
+ pending.append(following)
826
+ intervals = sorted((at, at + ins.size) for at, ins in seen.items())
827
+ runs, overlaps = [], []
828
+ for start, end in intervals:
829
+ if runs and start < runs[-1][1]:
830
+ overlaps.append(start)
831
+ if runs and start <= runs[-1][1]:
832
+ runs[-1][1] = max(runs[-1][1], end)
833
+ else:
834
+ runs.append([start, end])
835
+ for at in overlaps:
836
+ gaps.append({"site": at, "reason": OVERLAP_REASON})
837
+ holes = [{"start": a[1], "end": b[0]} for a, b in zip(runs, runs[1:])]
838
+ covered = sum(end - start for start, end in runs)
839
+ return {"entry": entry, "instructions": seen, "intervals": [{"start": a, "end": b} for a, b in runs], "holes": holes,
840
+ "span": {"start": runs[0][0], "end": runs[-1][1]} if runs else None, "coveredBytes": covered,
841
+ "exits": sorted(exits, key=lambda e: e["site"]), "calls": sorted(calls, key=lambda c: c["site"]),
842
+ "assumedContinuations": sorted(assumed, key=lambda a: a["site"]), "sharedEntries": sorted(shared),
843
+ "gaps": gaps, "complete": bool(exits) and not gaps}
844
+
845
+
846
+ def callees(image, config):
847
+ root = integer(config.get("entry"), 0, len(image.data) - 1, "callee root")
848
+ established = set(entries(image))
849
+ if root not in established:
850
+ raise ValueError("Callee root must be an established entry")
851
+ node_limit = integer(config.get("nodeLimit", 64), 1, 128, "callee node limit")
852
+ edge_limit = integer(config.get("edgeLimit", 512), 1, 2048, "callee edge limit")
853
+ depth_limit = integer(config.get("depthLimit", 16), 1, 128, "callee depth limit")
854
+ instruction_limit = integer(config.get("instructionLimit", 10000), 1, 100000, "instruction limit")
855
+ controls = config.get("controls", {})
856
+ if not isinstance(controls, dict) or set(controls) - {"sharedSites", "recursiveSites", "writeSites"}:
857
+ raise ValueError("Invalid callee controls")
858
+ nodes, edges, omitted = {}, [], []
859
+
860
+ def read(entry):
861
+ b = body(image, entry, instruction_limit)
862
+ observations = []
863
+ for site, ins in sorted(b["instructions"].items()):
864
+ for index, operand in enumerate(ins.operands):
865
+ access = memory_access(ins, operand) if operand.type == X86_OP_MEM and ins.mnemonic != "lea" else []
866
+ if not access:
867
+ continue
868
+ observations.append({"entry": entry, "site": site, "operandIndex": index,
869
+ "width": memory_width(ins, operand),
870
+ "access": access,
871
+ "segmentRegister": segment_register(ins, operand.mem),
872
+ "displacement": operand.mem.disp,
873
+ "baseRegister": ins.reg_name(operand.mem.base) or None,
874
+ "indexRegister": ins.reg_name(operand.mem.index) or None,
875
+ "interpretation": "explicit operand reached in conditional entry CFG; effective address and runtime execution unresolved"})
876
+ nodes[entry] = {"entry": entry, "body": b, "memoryObservations": observations, "contestedBy": []}
877
+ return b
878
+
879
+ # Breadth-first reading gives every node its shortest depth, so depth/node/edge limits do not depend on
880
+ # which caller happened to be read first. paths holds each admitted node's tree path from the root.
881
+ paths, queue = {root: [root]}, [root]
882
+ for entry in queue:
883
+ b, path = read(entry), paths[entry]
884
+ routes = b["calls"] + [e | {"encoding": "tail transfer"} for e in b["exits"] if e["kind"] == "tail transfer"]
885
+ for route in sorted(routes, key=lambda r: (r["site"], -1 if r.get("target") is None else r["target"])):
886
+ if len(edges) >= edge_limit:
887
+ omitted.append({"id": len(omitted), "entry": entry, "site": route["site"], "reason": "edge limit; route not traversed"})
888
+ continue
889
+ target = route.get("target")
890
+ edge = {"id": len(edges), "caller": entry, "site": route["site"], "target": target, "kind": route["encoding"],
891
+ "path": path, "classification": "unresolved", "dependencies": []}
892
+ edges.append(edge)
893
+ if target is None or target not in established:
894
+ edge["dependencies"].append({"reason": route.get("reason") or "target is not an established entry"})
895
+ elif target in paths:
896
+ edge["classification"] = None # recursivePath or sharedNodeReuse, once the read graph is known
897
+ elif len(path) >= depth_limit or len(paths) >= node_limit:
898
+ edge["dependencies"].append({"reason": "depth limit" if len(path) >= depth_limit else "node limit"})
899
+ else:
900
+ edge["classification"] = "newNode"
901
+ paths[target] = path + [target]
902
+ queue.append(target)
903
+ outgoing = {}
904
+ for edge in edges:
905
+ outgoing.setdefault(edge["caller"], []).append(edge)
906
+
907
+ def walk(start):
908
+ """Shortest-route predecessors of every read node reachable from start."""
909
+ previous, todo = {start: None}, [start]
910
+ for at in todo:
911
+ for e in outgoing.get(at, []):
912
+ if e["target"] in nodes and e["target"] not in previous:
913
+ previous[e["target"]] = at
914
+ todo.append(e["target"])
915
+ return previous
916
+ reach = {entry: walk(entry) for entry in nodes}
917
+ # A non-tree edge closes a cycle exactly when its target reaches its caller; this is independent of read order.
918
+ for edge in edges:
919
+ if edge["classification"] is None:
920
+ previous = reach[edge["target"]]
921
+ if edge["caller"] in previous:
922
+ cycle, at = [], edge["caller"]
923
+ while at is not None:
924
+ cycle.append(at)
925
+ at = previous[at]
926
+ edge.update(classification="recursivePath", cyclePath=cycle[::-1] + [edge["target"]])
927
+ else:
928
+ edge["classification"] = "sharedNodeReuse"
929
+ # A decoded instruction partly overlapping another entry's cannot verify ownership/effects;
930
+ # an identical instruction both bodies reach (a shared tail) is not a conflict.
931
+ for entry, rows in _cross_entry_overlaps({entry: n["body"] for entry, n in nodes.items()}).items():
932
+ nodes[entry]["contestedBy"] = sorted(set(r["entry"] for r in rows))
933
+ unchecked = sorted(established - nodes.keys())
934
+ for entry, n in nodes.items():
935
+ n["boundaryUsable"] = n["body"]["complete"] and not n["contestedBy"] and not unchecked
936
+ n["dependencies"] = list(n["body"]["gaps"])
937
+ if unchecked:
938
+ n["dependencies"].append({"reason": "declared entries not checked for boundary conflicts", "entries": unchecked})
939
+ if n["contestedBy"]:
940
+ n["dependencies"].append({"reason": "cross-entry instruction overlap", "entries": n["contestedBy"]})
941
+ for observation in n["memoryObservations"]:
942
+ observation["boundaryUsable"] = n["boundaryUsable"]
943
+ # Every edge's own dependencies are final before any summary refers to them.
944
+ for edge in edges:
945
+ edge["boundaryUsable"] = nodes[edge["caller"]]["boundaryUsable"] and edge["target"] in nodes and nodes[edge["target"]]["boundaryUsable"]
946
+ if edge["classification"] == "recursivePath" and any(not nodes[e]["boundaryUsable"] for e in edge["cyclePath"]):
947
+ edge["classification"] = "unresolvedBackEdge"
948
+ edge["dependencies"].append({"reason": "cycle path has an incomplete or contested body"})
949
+ # One summary per read node, shared by reference by every edge into it, keeps output linear in the graph
950
+ # while each caller still retains the callee's observations, assumptions and dependencies through it.
951
+ summaries = {}
952
+ for entry in sorted(nodes):
953
+ reached = sorted(reach[entry])
954
+ observations = [o for at in reached for o in nodes[at]["memoryObservations"]]
955
+ summaries[entry] = {
956
+ "entry": entry, "entries": reached,
957
+ "dependencyEntries": [at for at in reached if nodes[at]["dependencies"]],
958
+ "dependencyEdges": [e["id"] for at in reached for e in outgoing.get(at, []) if e["dependencies"]],
959
+ "omittedRoutes": [o["id"] for o in omitted if o["entry"] in reach[entry]],
960
+ "counts": {"memoryObservations": len(observations),
961
+ "writeObservations": sum("write" in o["access"] for o in observations),
962
+ "assumptions": sum(len(nodes[at]["body"]["assumedContinuations"]) for at in reached)},
963
+ "effectComplete": False,
964
+ "interpretation": "references to the explicit memory observations, continuation assumptions and unread dependencies "
965
+ "of every reached node; never a read-only or callee-effect guarantee"}
966
+ for edge in edges:
967
+ edge["calleeSummary"] = edge["target"] if edge["target"] in summaries else None
968
+ # A reused node that reaches the active path, or whose reached bodies were capped or are unusable, may lead
969
+ # back into the active path, so such reuse is no shared-node control.
970
+ capped = {"depth limit", "node limit", "instruction limit"}
971
+
972
+ def shared_control(e):
973
+ if e["classification"] != "sharedNodeReuse":
974
+ return False
975
+ s = summaries[e["target"]]
976
+ return (e["boundaryUsable"]
977
+ and not set(e["path"]) & set(s["entries"]) and all(nodes[at]["boundaryUsable"] for at in s["entries"])
978
+ and not s["omittedRoutes"]
979
+ and not any(d.get("reason") in capped for at in s["entries"] for d in nodes[at]["dependencies"])
980
+ and not any(d.get("reason") in capped for i in s["dependencyEdges"] for d in edges[i]["dependencies"]))
981
+ known = {"sharedSites": {e["site"] for e in edges if shared_control(e)},
982
+ "recursiveSites": {e["site"] for e in edges if e["classification"] == "recursivePath"},
983
+ "writeSites": {o["site"] for n in nodes.values() for o in n["memoryObservations"] if o["boundaryUsable"] and "write" in o["access"]}}
984
+ for kind, sites in controls.items():
985
+ if not isinstance(sites, list) or len(sites) > 256 or any(type(at) is not int or at not in known[kind] for at in sites):
986
+ raise ValueError("Callee positive control missed: " + kind)
987
+ return {"root": root, "nodes": [{k: v for k, v in n.items() if k != "body"} | {"body": _body_report(n["body"])} for n in nodes.values()],
988
+ "edges": edges, "calleeSummaries": list(summaries.values()), "omittedRoutes": omitted, "uncheckedEntries": unchecked,
989
+ "controls": controls,
990
+ "completeWithinDeclaredGraph": not omitted and all(n["boundaryUsable"] for n in nodes.values()) and not any(e["dependencies"] for e in edges),
991
+ "exclusions": ["implicit memory effects", "computed/unestablished targets", "argument-sensitive effects", "runtime reachability"],
992
+ "interpretation": "Nodes are read breadth-first; path is the shortest read route to the caller. A recursivePath is a "
993
+ "non-tree entry-CFG edge whose target reaches its caller; sharedNodeReuse is a previously read node "
994
+ "that does not. Neither proves runtime recursion."}
995
+
996
+
997
+ def _body_report(b):
998
+ return {k: v for k, v in b.items() if k != "instructions"} | {"instructionCount": len(b["instructions"])}
999
+
1000
+
1001
+ def _analyzer_function(config):
1002
+ claim = config.get("analyzerFunction")
1003
+ if claim is None:
1004
+ return None
1005
+ if not isinstance(claim, dict) or not claim.get("evidence"):
1006
+ raise ValueError("analyzerFunction needs start and evidence")
1007
+ return claim
1008
+
1009
+
1010
+ def bounds(image, config):
1011
+ entry = integer(config.get("entry"), 0, len(image.data) - 1, "entry")
1012
+ if entry not in entries(image):
1013
+ raise ValueError("Bounds entry must be an established region entry")
1014
+ b = body(image, entry, config.get("instructionLimit", 10000))
1015
+ result = _body_report(b)
1016
+ claim = _analyzer_function(config)
1017
+ if claim is not None:
1018
+ start = integer(claim.get("start"), 0, len(image.data) - 1, "analyzer start")
1019
+ size = integer(claim.get("bodyBytes"), 1, len(image.data), "analyzer body bytes")
1020
+ end = start + size
1021
+ result["analyzer"] = {
1022
+ "start": start, "bodyBytes": size, "evidence": claim["evidence"], "startMatches": start == entry,
1023
+ "bodyBytesMatch": size == b["coveredBytes"], "startPlusBodyBytes": end,
1024
+ "exitsAtOrBeyond": [e for e in b["exits"] if e["site"] >= end],
1025
+ "instructionsAtOrBeyond": sorted(at for at in b["instructions"] if at >= end),
1026
+ "interpretation": "an analyzer size counts body bytes; start plus size is not an end address unless the body is one contiguous run"}
1027
+ result["interpretation"] = ("complete means every reached path ends in a listed exit within the declared regions and the "
1028
+ "listed continuation assumptions; it is not a complete reading under the standard")
1029
+ return result
1030
+
1031
+
1032
+ def _cross_entry_overlaps(bodies):
1033
+ """For each entry, the instructions of other entries' bodies that partly overlap one of its own.
1034
+
1035
+ An overlap inside one body is already a gap of that body, so only pairs from different entries count.
1036
+ """
1037
+ starts = {}
1038
+ for entry, b in bodies.items():
1039
+ for at, ins in b["instructions"].items():
1040
+ starts.setdefault(at, (at + ins.size, set()))[1].add(entry)
1041
+ conflicts, active = {}, []
1042
+ for start in sorted(starts):
1043
+ end, holders = starts[start]
1044
+ active = [a for a in active if starts[a][0] > start]
1045
+ for a in active:
1046
+ for first in starts[a][1]:
1047
+ for second in holders:
1048
+ if first != second:
1049
+ conflicts.setdefault(first, []).append({"entry": second, "site": a, "otherSite": start})
1050
+ conflicts.setdefault(second, []).append({"entry": first, "site": start, "otherSite": a})
1051
+ active.append(start)
1052
+ return {entry: sorted(rows, key=lambda r: (r["site"], r["entry"], r["otherSite"])) for entry, rows in conflicts.items()}
1053
+
1054
+
1055
+ def _overlay_exports(config):
1056
+ """Source-derived overlay export rows (from the Node MZ/FBOV loader) grouped by entry offset."""
1057
+ rows = config.get("overlayExports", [])
1058
+ if not isinstance(rows, list):
1059
+ raise ValueError("overlayExports must be a list")
1060
+ grouped = {}
1061
+ for e in rows:
1062
+ if not isinstance(e, dict) or type(e.get("entry")) is not int:
1063
+ raise ValueError("Each overlay export needs an integer entry")
1064
+ grouped.setdefault(e["entry"], []).append(e)
1065
+ return grouped
1066
+
1067
+
1068
+ def owner(image, config):
1069
+ query = config.get("query", {})
1070
+ if not isinstance(query, dict):
1071
+ raise ValueError("Owner query must be an object")
1072
+ site = integer(query.get("site"), 0, len(image.data) - 1, "owner site")
1073
+ if image.region(site) is None:
1074
+ raise ValueError("Owner site is outside declared code")
1075
+ entry_limit = integer(config.get("entryLimit", 64), 1, 256, "entryLimit")
1076
+ limit = config.get("instructionLimit", 10000)
1077
+ claim = _analyzer_function(config)
1078
+ start = None if claim is None else integer(claim.get("start"), 0, len(image.data) - 1, "analyzer start")
1079
+ established = entries(image)
1080
+ bodies, owners, inside, incomplete, gaps = {}, [], [], [], []
1081
+ for index, entry in enumerate(established):
1082
+ if index >= entry_limit:
1083
+ gaps.append({"entries": established[index:], "reason": "entry limit; these entries were not checked"})
1084
+ break
1085
+ b = bodies[entry] = body(image, entry, limit)
1086
+ if site in b["instructions"]:
1087
+ owners.append({"entry": entry, "complete": b["complete"], "span": b["span"],
1088
+ "exitsBeforeSiteByAddress": [e for e in b["exits"] if entry <= e["site"] < site]})
1089
+ continue
1090
+ if any(at < site < at + ins.size for at, ins in b["instructions"].items()):
1091
+ inside.append({"entry": entry, "reason": "the site is inside an instruction this entry reaches, not at its start"})
1092
+ if b["gaps"]:
1093
+ # A body that stopped at a gap may still reach the site beyond it.
1094
+ incomplete.append({"entry": entry, "gaps": b["gaps"]})
1095
+ # Two checked bodies that decode overlapping instructions cannot both be right. Which one
1096
+ # is misdecoded is not decided here; an owner on either side only leaves the site unresolved.
1097
+ conflicts = _cross_entry_overlaps(bodies)
1098
+ for o in owners:
1099
+ o["contestedBy"] = conflicts.get(o["entry"], [])
1100
+ exports = _overlay_exports(config)
1101
+ checked = {}
1102
+ for entry, b in bodies.items():
1103
+ region = image.region(entry)
1104
+ checked[entry] = {"entry": entry, "span": b["span"], "ranges": b["intervals"],
1105
+ "complete": b["complete"], "gaps": b["gaps"],
1106
+ "assumedContinuations": b["assumedContinuations"],
1107
+ "entryEvidence": region["evidence"], "container": region.get("container"),
1108
+ "overlayExports": exports.get(entry, [])}
1109
+ for o in owners:
1110
+ row = checked[o["entry"]]
1111
+ o.update({k: row[k] for k in ("ranges", "assumedContinuations", "container", "overlayExports")})
1112
+ # Entries past the entry limit were never decoded, so an overlap with them is unknown, not absent.
1113
+ o["boundaryCheck"] = {"performed": True, "instructionStartReached": True,
1114
+ "joinableWithinModel": row["complete"] and not o["contestedBy"] and not gaps,
1115
+ "meaning": "entry-path instruction under complete bounded traversal and no cross-entry overlap "
1116
+ "among all established entries; not player reachability"}
1117
+ contested = [o["entry"] for o in owners if o["contestedBy"]]
1118
+ if contested:
1119
+ verdict = "unresolved: an owner's body overlaps instructions another checked entry decodes"
1120
+ elif owners:
1121
+ verdict = "shared by several entries" if len(owners) > 1 else "one established entry reaches this site"
1122
+ elif gaps or incomplete:
1123
+ verdict = "unresolved: no checked body reaches this site, but some entries were unchecked or their bodies stopped at a gap"
1124
+ else:
1125
+ verdict = "unowned: no established entry reaches this site"
1126
+ result = {"site": site, "checkedEntries": list(checked.values()), "owners": owners, "insideOtherInstructions": inside, "incompleteEntries": incomplete, "gaps": gaps,
1127
+ "contestedOwners": contested, "shared": len(owners) > 1, "verdict": verdict,
1128
+ "interpretation": "ownership is reachability from established entries without entering callees; a return or "
1129
+ "prologue between an entry and the site by address is a warning, never a boundary"}
1130
+ if claim is not None:
1131
+ hypothesis = bodies.get(start) or (body(image, start, limit) if image.region(start) else None)
1132
+ reaches = None if hypothesis is None else site in hypothesis["instructions"]
1133
+ result["analyzer"] = {
1134
+ "start": start, "evidence": claim["evidence"], "established": start in established,
1135
+ "agrees": start in established and bool(reaches), "contested": start in contested,
1136
+ "reachesSite": reaches,
1137
+ "boundaryCheck": {"performed": hypothesis is not None, "instructionStartReached": reaches,
1138
+ "joinableWithinModel": bool(hypothesis and hypothesis["complete"] and reaches and start in bodies
1139
+ and not gaps and not conflicts.get(start))},
1140
+ "span": None if hypothesis is None else hypothesis["span"],
1141
+ "ranges": [] if hypothesis is None else hypothesis["intervals"],
1142
+ "complete": False if hypothesis is None else hypothesis["complete"],
1143
+ "gaps": [] if hypothesis is None else hypothesis["gaps"],
1144
+ "assumedContinuations": [] if hypothesis is None else hypothesis["assumedContinuations"],
1145
+ "exitsBeforeSiteByAddress": [] if hypothesis is None else [e for e in hypothesis["exits"] if start <= e["site"] < site],
1146
+ "interpretation": "disagreement means the analyzer's function and the established entries assign this site differently"}
1147
+ return result
1148
+
1149
+
1150
+ def _run_report(image, config, command):
1151
+ if command == "operand":
1152
+ return operand_provenance(image, config)
1153
+ if command == "target":
1154
+ return call_target_report(image, config)
1155
+ if command == "bounds":
1156
+ return bounds(image, config)
1157
+ if command == "owner":
1158
+ return owner(image, config)
1159
+ if command == "callees":
1160
+ return callees(image, config)
1161
+ if command == "incoming":
1162
+ return incoming(image, config)
1163
+ if command == "operand-candidates":
1164
+ return operand_candidates(image, config)
1165
+ if command == "uses":
1166
+ return uses(image, config)
1167
+ if command == "dispatch":
1168
+ return dispatch(image, config)
1169
+ if command not in ("trace", "arguments", "effects", "returns", "guards", "memory", "allocation"):
1170
+ raise ValueError("Unknown x86 report command")
1171
+ if command == "allocation":
1172
+ checkpoints = set(config.get("checkpoints", []))
1173
+ for a in config.get("allocations", []):
1174
+ for field in ("extent", "pointer"):
1175
+ if a.get(field):
1176
+ checkpoints.add(a[field]["site"])
1177
+ config = {**config, "checkpoints": sorted(checkpoints)}
1178
+ report = trace(image, config)
1179
+ if command in ("arguments", "effects"):
1180
+ report = near_pointer_provenance(report, config)
1181
+ if command == "allocation":
1182
+ return allocations(report, config)
1183
+ if command != "trace":
1184
+ kinds = {"arguments": ("address-formation", "read", "call", "call-return"), "effects": ("address-formation", "write", "call", "call-return", "return", "branch", "string-operation",
1185
+ "flag-assumption", "flag-write", "flags-save", "flags-restore", "local-iret"),
1186
+ "returns": ("return", "call-return", "compare", "branch", "write"),
1187
+ "guards": ("compare", "branch", "read", "write", "call", "call-return"),
1188
+ "memory": ("read", "write", "address-formation")}[command]
1189
+ for path in report["paths"]:
1190
+ path["events"] = [e for e in path["events"] if e["kind"] in kinds or
1191
+ (command == "effects" and e["kind"] == "read" and (e.get("nearPointerAccessCandidates") or e.get("nearPointerArgumentCandidates")))]
1192
+ return report
1193
+
1194
+
1195
+ def _formation_links(formations, value):
1196
+ """Relate one offset/value to the retained LEA formations that produced it."""
1197
+ matched = sorted((f for site in value["producers"] for f in formations.get(site, ())), key=lambda f: f["order"])
1198
+ results = []
1199
+ for formed in matched:
1200
+ original = formed["value"]
1201
+ if original["bits"] != value["bits"]:
1202
+ continue
1203
+ a, b = original["expression"], value["expression"]
1204
+ ab, ad = (a[1], a[2]) if a[0] == "offset" else (a, 0)
1205
+ bb, bd = (b[1], b[2]) if b[0] == "offset" else (b, 0)
1206
+ relation = "sameOffset" if a == b else "affineFieldOffset" if ab == bb and ab[0] != "constant" else "producerOnly"
1207
+ delta = (bd - ad) % (1 << value["bits"]) if relation != "producerOnly" else None
1208
+ results.append({"formationSite": formed["site"], "formationOrder": formed["order"],
1209
+ "formationValue": original, "formationAddressingSegment": formed["addressingSegment"],
1210
+ "formationSegmentRegister": formed.get("addressingSegmentRegister"),
1211
+ "offsetRelation": relation, "offsetDeltaModulo": delta,
1212
+ "note": "LEA addressing default is not a segment binding"})
1213
+ return results
1214
+
1215
+
1216
+ def near_pointer_provenance(report, config):
1217
+ limit = integer(config.get("pointerFormationLimit", 128), 1, 1024, "pointer formation limit")
1218
+ for path in report["paths"]:
1219
+ formations, retained, omitted = {}, deque(), 0
1220
+ for event in path["events"]:
1221
+ if event["kind"] == "address-formation":
1222
+ if len(retained) >= limit:
1223
+ # Evict the oldest formation so the most recent ones stay linkable.
1224
+ oldest = retained.popleft()
1225
+ formations[oldest["site"]].pop(0)
1226
+ omitted += 1
1227
+ formations.setdefault(event["site"], []).append(event)
1228
+ retained.append(event)
1229
+ continue
1230
+ if event["kind"] not in ("read", "write"):
1231
+ continue
1232
+ if event.get("argument"):
1233
+ arguments = _formation_links(formations, event["value"])
1234
+ if arguments:
1235
+ event["nearPointerArgumentCandidates"] = arguments
1236
+ accesses = _formation_links(formations, event["offset"])
1237
+ for candidate in accesses:
1238
+ formed_segment, accessed_segment = candidate["formationAddressingSegment"], event["segment"]
1239
+ relation = ("sameWithinModel" if formed_segment["bits"] == accessed_segment["bits"] and formed_segment["expression"] == accessed_segment["expression"] else
1240
+ "differentWithinModel" if formed_segment["value"] is not None and accessed_segment["value"] is not None else "unresolved")
1241
+ candidate.update(dereferenceSegment=accessed_segment, dereferenceSegmentRegister=event.get("effectiveSegmentRegister"),
1242
+ segmentRelationship=relation,
1243
+ mayMergeStorage=relation == "sameWithinModel" and candidate["offsetRelation"] != "producerOnly" and not omitted,
1244
+ interpretation="offset and segment comparison within the propagated model only; unknown segment relationships remain possible aliases")
1245
+ if accesses:
1246
+ event["nearPointerAccessCandidates"] = accesses
1247
+ event["pointerFormationsOmittedBeforeEvent"] = omitted
1248
+ path["nearPointerProvenance"] = {"formationLimit": limit, "formationsOmitted": omitted,
1249
+ "interpretation": "bounded provenance candidates; no absence or runtime alias proof"}
1250
+ return report
1251
+
1252
+
1253
+ def run_report(data, config, command):
1254
+ image = Image(data, config)
1255
+ result = _run_report(image, image.config, command)
1256
+ return {"instructionModel": {"bits": image.bits, "addressModel": "flat32" if image.flat else "segmented16",
1257
+ "flatAssumption": "CS/DS/ES/SS bases zero; FS/GS bases unknown" if image.flat else None},
1258
+ "sourceMapping": image.config.get("peMetadata"), "formatTables": image.config.get("formatTables"),
1259
+ "declaredRegions": image.regions, "indirectJumpDeclarations": list(image.indirect_jumps.values()), **result}