cseq 0.0.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
cseq/__init__.py ADDED
@@ -0,0 +1,9 @@
1
+ """cseq: C source sequence analysis tool."""
2
+
3
+ __version__ = "0.0.1"
4
+
5
+ from .api import Analysis, Explanation, MarkerSuggestion, QueryResult, RuntimeSession, analyze
6
+
7
+ __all__ = [
8
+ "Analysis", "Explanation", "MarkerSuggestion", "QueryResult", "RuntimeSession", "analyze", "__version__"
9
+ ]
cseq/acceptance.py ADDED
@@ -0,0 +1,162 @@
1
+ from __future__ import annotations
2
+
3
+ from dataclasses import asdict, dataclass
4
+ from pathlib import Path
5
+ import shutil
6
+
7
+ from .explain import explain_call
8
+ from .html import build_html_payload
9
+ from .marker import apply_marker, apply_marker_overlay, undo_marker
10
+ from .model import CallTargetKind
11
+ from .project import Project
12
+ from .runtime import RuntimePattern, build_trace_session
13
+ from .sequence import build_sequence, render_plantuml
14
+ from .trace_analysis import ReconstructionState, reconstruct_static_paths
15
+
16
+
17
+ @dataclass(frozen=True, slots=True)
18
+ class AcceptanceCriterion:
19
+ criterion_id: str
20
+ name: str
21
+ status: str
22
+ evidence: str
23
+ detail: str = ""
24
+
25
+
26
+ @dataclass(frozen=True, slots=True)
27
+ class AcceptanceReport:
28
+ criteria: tuple[AcceptanceCriterion, ...]
29
+
30
+ @property
31
+ def functional_pass(self) -> bool:
32
+ return all(c.status == "PASS" for c in self.criteria if c.status != "DEFERRED_TO_LATER_GATE")
33
+
34
+ def to_dict(self) -> dict:
35
+ return {
36
+ "functional_pass": self.functional_pass,
37
+ "criteria": [asdict(c) for c in self.criteria],
38
+ }
39
+
40
+
41
+ def run_poc_functional_acceptance(fixtures_root: str | Path, work_root: str | Path) -> AcceptanceReport:
42
+ fixtures = Path(fixtures_root)
43
+ work = Path(work_root)
44
+ work.mkdir(parents=True, exist_ok=True)
45
+ criteria: list[AcceptanceCriterion] = []
46
+
47
+ # A/B: direct call + external separation.
48
+ basic = Project(fixtures / "basic", use_cache=False).index()
49
+ calls = {(c.caller_name, c.callee_name): c for c in basic.callsites}
50
+ direct_ok = (
51
+ calls[("main", "helper")].target_kind == CallTargetKind.INTERNAL_EXACT
52
+ and calls[("main", "printf")].target_kind == CallTargetKind.EXTERNAL_DECLARED
53
+ )
54
+ criteria.append(AcceptanceCriterion(
55
+ "POC-01", "Direct Call基本精度とExternal分離",
56
+ "PASS" if direct_ok else "FAIL", "fixtures/basic",
57
+ ))
58
+
59
+ # C/D/E: indirect target explanation.
60
+ fptr = Project(fixtures / "fptr_local", use_cache=False).index()
61
+ explanation = explain_call(fptr, caller="run")
62
+ indirect_calls = [c for c in explanation.get("calls", []) if c.get("resolution", "").startswith("INDIRECT")]
63
+ indirect_ok = bool(indirect_calls) and all(c.get("targets") for c in indirect_calls) and any(c.get("value_flow") for c in indirect_calls)
64
+ criteria.append(AcceptanceCriterion(
65
+ "POC-02", "Indirect Callbackを理由付きで説明可能",
66
+ "PASS" if indirect_ok else "FAIL", "fixtures/fptr_local + explain_call",
67
+ ))
68
+
69
+ # L/M: unknown syntax does not drop surrounding functions.
70
+ unknown = Project(fixtures / "unknown_compiler", use_cache=False).index()
71
+ names = {f.name for f in unknown.functions()}
72
+ opaque_count = sum(len(a.opaque_regions) for a in unknown.parse_artifacts)
73
+ unknown_ok = {"helper", "main"} <= names and opaque_count >= 1
74
+ criteria.append(AcceptanceCriterion(
75
+ "POC-03", "未知構文でも解析可能部分を保持",
76
+ "PASS" if unknown_ok else "FAIL", "fixtures/unknown_compiler",
77
+ f"functions={sorted(names)}, opaque_regions={opaque_count}",
78
+ ))
79
+
80
+ # H: finite representation of infinite loop.
81
+ rtos = Project(fixtures / "rtos_infinite", use_cache=False).index()
82
+ rtos_model = build_sequence(rtos, "main")
83
+ puml = render_plantuml(rtos_model)
84
+ rtos_ok = "loop [infinite]" in puml and puml.count("dispatch()") == 1
85
+ criteria.append(AcceptanceCriterion(
86
+ "POC-04", "RTOS無限loopを有限表示",
87
+ "PASS" if rtos_ok else "FAIL", "fixtures/rtos_infinite + PlantUML",
88
+ ))
89
+
90
+ # I: runtime observation + static reconstruction.
91
+ pattern = RuntimePattern("[{cpu}] {time} {message}", "%H:%M:%S.%f")
92
+ session = build_trace_session("acceptance", [fixtures / "runtime" / "a.log", fixtures / "runtime" / "b.log"], pattern=pattern)
93
+ recon_index = Project(fixtures / "reconstruction" / "unique", use_cache=False).index()
94
+ reconstruction = reconstruct_static_paths(recon_index, "a", "f")
95
+ runtime_ok = len(session.events) == 3 and reconstruction.state == ReconstructionState.EXACT_PATH
96
+ criteria.append(AcceptanceCriterion(
97
+ "POC-05", "Runtime Marker観測と静的経路復元",
98
+ "PASS" if runtime_ok else "FAIL", "fixtures/runtime + reconstruction/unique",
99
+ ))
100
+
101
+ # G/J: generic CPU rule without company-specific core naming.
102
+ cpu_root = fixtures / "cpu_rules"
103
+ cpu = Project(cpu_root, config=cpu_root / "cseq.toml", configuration="GROUP_A_BUILD", use_cache=False).index()
104
+ cpu_values = {e.value for e in cpu.cpu_evidence}
105
+ cpu_ok = "CPU_GROUP_A" in cpu_values and "CPU_A2" in cpu_values
106
+ criteria.append(AcceptanceCriterion(
107
+ "POC-06", "一般化CPU Ruleで割当可能",
108
+ "PASS" if cpu_ok else "FAIL", "fixtures/cpu_rules",
109
+ ))
110
+
111
+ # K: marker rewrite minimal and reversible.
112
+ marker_src = fixtures / "marker" / "main.c"
113
+ marker_dst = work / "marker_acceptance.c"
114
+ shutil.copyfile(marker_src, marker_dst)
115
+ before = marker_dst.read_bytes()
116
+ manifest = apply_marker(marker_dst, literal_contains="rx=%d", marker_id="ACCEPT01")
117
+ after = marker_dst.read_bytes()
118
+ marker_token = b"[[CSEQ:ACCEPT01]] "
119
+ minimal_ok = after.replace(marker_token, b"", 1) == before and after.count(marker_token) == 1
120
+ undo_marker(marker_dst, manifest)
121
+ marker_ok = minimal_ok and marker_dst.read_bytes() == before
122
+ criteria.append(AcceptanceCriterion(
123
+ "POC-07", "Marker直接編集が最小差分・可逆",
124
+ "PASS" if marker_ok else "FAIL", "fixtures/marker + apply/undo",
125
+ ))
126
+
127
+ # PlantUML and HTML consume the exact same SequenceModel instance.
128
+ basic_model = build_sequence(basic, "main")
129
+ html_payload = build_html_payload(basic_model)
130
+ puml2 = render_plantuml(basic_model)
131
+ shared_ir_ok = len(html_payload["messages"]) == len(basic_model.messages) and "@startuml" in puml2
132
+ criteria.append(AcceptanceCriterion(
133
+ "POC-08", "PlantUML / HTMLを同一Sequence IRから生成",
134
+ "PASS" if shared_ir_ok else "FAIL", "SequenceModel -> render_plantuml/build_html_payload",
135
+ ))
136
+
137
+ # K2: overlay marker mode leaves the original source untouched.
138
+ overlay_src = fixtures / "marker" / "main.c"
139
+ overlay_dst = work / "overlay" / "marker_acceptance.c"
140
+ overlay_before = overlay_src.read_bytes()
141
+ overlay_manifest = apply_marker_overlay(
142
+ overlay_src, overlay_dst, literal_contains="rx=%d", marker_id="OVERLAY01"
143
+ )
144
+ overlay_ok = (
145
+ overlay_src.read_bytes() == overlay_before
146
+ and b"[[CSEQ:OVERLAY01]] " in overlay_dst.read_bytes()
147
+ )
148
+ undo_marker(overlay_dst, overlay_manifest)
149
+ overlay_ok = overlay_ok and overlay_dst.read_bytes() == overlay_before
150
+ criteria.append(AcceptanceCriterion(
151
+ "POC-09", "Marker overlay mode preserves the original source",
152
+ "PASS" if overlay_ok else "FAIL", "apply_marker_overlay + undo",
153
+ ))
154
+
155
+ # Performance-dependent criterion is intentionally delegated to G5.
156
+ criteria.append(AcceptanceCriterion(
157
+ "POC-10", "1M Event級UIが構造的に破綻しない",
158
+ "DEFERRED_TO_LATER_GATE", "G5 Runtime/UI受入",
159
+ "Functional E2Eでは判定せず、固定済み性能Gateで正式受入する",
160
+ ))
161
+
162
+ return AcceptanceReport(tuple(criteria))
cseq/api.py ADDED
@@ -0,0 +1,328 @@
1
+ from __future__ import annotations
2
+
3
+ from dataclasses import dataclass
4
+ from pathlib import Path
5
+ from typing import Iterable, Mapping
6
+
7
+ from .explain import explain_call
8
+ from .html import render_html
9
+ from .model import ProjectIndex
10
+ from .project import Project
11
+ from .query import query_calls, query_paths, query_unresolved
12
+ from .runtime import RuntimePattern, TraceSession as _TraceSession, build_trace_session
13
+ from .runtime_cpu import apply_runtime_cpu_evidence
14
+ from .runtime_address import discover_runtime_address_evidence
15
+ from .sequence import SequenceModel, build_sequence, render_plantuml
16
+ from .trace_analysis import ReconstructionState, reconstruct_static_paths, score_marker_suggestions, discover_correlations, fit_clock_alignments
17
+
18
+
19
+ @dataclass(frozen=True, slots=True)
20
+ class Explanation:
21
+ """Stable public explanation facade; internal Evidence/IR objects stay hidden."""
22
+ _data: Mapping[str, object]
23
+
24
+ @property
25
+ def found(self) -> bool:
26
+ return bool(self._data.get("found"))
27
+
28
+ def to_dict(self) -> dict[str, object]:
29
+ return dict(self._data)
30
+
31
+
32
+ @dataclass(frozen=True, slots=True)
33
+ class QueryResult:
34
+ kind: str
35
+ _data: Mapping[str, object]
36
+
37
+ def to_dict(self) -> dict[str, object]:
38
+ return dict(self._data)
39
+
40
+
41
+ @dataclass(frozen=True, slots=True)
42
+ class RuntimeSession:
43
+ """Read-only runtime-session facade for Python callers."""
44
+ _session: _TraceSession
45
+
46
+ @property
47
+ def name(self) -> str:
48
+ return self._session.name
49
+
50
+ @property
51
+ def event_count(self) -> int:
52
+ return len(self._session.events)
53
+
54
+ @property
55
+ def cpus(self) -> tuple[str, ...]:
56
+ return tuple(sorted({e.cpu_hint for e in self._session.events if e.cpu_hint}))
57
+
58
+ @property
59
+ def tasks(self) -> tuple[str, ...]:
60
+ return tuple(sorted({e.task_hint for e in self._session.events if e.task_hint}))
61
+
62
+ def rows(self) -> tuple[dict[str, object], ...]:
63
+ return tuple({
64
+ "event_index": e.event_index,
65
+ "marker_id": e.marker_id,
66
+ "timestamp": e.timestamp_raw,
67
+ "cpu": e.cpu_hint,
68
+ "task": e.task_hint,
69
+ "message": e.message,
70
+ "payload": e.payload_dict(),
71
+ } for e in self._session.events)
72
+
73
+ def correlations(self) -> tuple[dict[str, object], ...]:
74
+ return tuple({
75
+ "left_event_index": r.left_event_index,
76
+ "right_event_index": r.right_event_index,
77
+ "relation_type": r.relation_type.value,
78
+ "keys": r.keys,
79
+ "values": r.values,
80
+ "score": r.score,
81
+ } for r in discover_correlations(self._session.events))
82
+
83
+ def clock_alignments(self) -> tuple[dict[str, object], ...]:
84
+ relations = discover_correlations(self._session.events)
85
+ return tuple({
86
+ "source_domain": a.source_domain,
87
+ "target_domain": a.target_domain,
88
+ "scale": a.scale,
89
+ "offset_seconds": a.offset_seconds,
90
+ "anchor_count": a.anchor_count,
91
+ "max_abs_residual_seconds": a.max_abs_residual_seconds,
92
+ "confidence": a.confidence,
93
+ } for a in fit_clock_alignments(self._session.events, relations))
94
+
95
+
96
+ @dataclass(frozen=True, slots=True)
97
+ class MarkerSuggestion:
98
+ function_name: str
99
+ score: float
100
+ distinguished_path_pairs: int
101
+ total_path_pairs: int
102
+ reason: str
103
+
104
+
105
+ @dataclass(slots=True)
106
+ class Analysis:
107
+ """Stable public facade over the internal ProjectIndex/IR implementation."""
108
+ project: Project
109
+ index: ProjectIndex
110
+
111
+ @property
112
+ def cache_hits(self) -> int:
113
+ return self.project.cache_hits
114
+
115
+ @property
116
+ def cache_misses(self) -> int:
117
+ return self.project.cache_misses
118
+
119
+ @property
120
+ def analysis_cache_hits(self) -> int:
121
+ return self.project.analysis_cache_hits
122
+
123
+ @property
124
+ def analysis_cache_misses(self) -> int:
125
+ return self.project.analysis_cache_misses
126
+
127
+ @property
128
+ def plugin_evidence(self) -> tuple[dict[str, object], ...]:
129
+ return tuple({
130
+ "plugin": e.plugin_name,
131
+ "kind": e.kind.value,
132
+ "target": e.target,
133
+ "type": e.evidence_type,
134
+ "value": e.value,
135
+ "confidence": e.confidence,
136
+ } for e in self.project.plugin_result.evidence)
137
+
138
+ @property
139
+ def linker_evidence(self) -> tuple[dict[str, object], ...]:
140
+ return tuple({
141
+ "symbol": e.symbol_name, "address": e.address,
142
+ "object_file": e.object_file, "section": e.section,
143
+ "matched_function_ids": e.matched_function_ids,
144
+ "provenance": e.provenance,
145
+ } for e in self.index.linker_evidence)
146
+
147
+ @property
148
+ def binary_evidence(self) -> tuple[dict[str, object], ...]:
149
+ return tuple({
150
+ "symbol": e.symbol_name, "address": e.address, "size": e.size,
151
+ "kind": e.symbol_kind, "matched_function_ids": e.matched_function_ids,
152
+ "binary_hash": e.binary_hash, "provenance": e.provenance,
153
+ } for e in self.index.binary_evidence)
154
+
155
+ def resolve_binary_address(self, address: int, *, load_bias: int = 0) -> dict[str, object] | None:
156
+ image = self.project.binary_image
157
+ if image is None:
158
+ return None
159
+ sym = image.resolve_address(address, load_bias=load_bias)
160
+ if sym is None:
161
+ return None
162
+ return {"name": sym.name, "address": sym.address, "size": sym.size, "kind": sym.kind}
163
+
164
+ @property
165
+ def dwarf_evidence(self) -> tuple[dict[str, object], ...]:
166
+ return tuple({
167
+ "function_id": e.function_id, "source": e.source_path,
168
+ "line": e.line, "address": e.address, "dwarf_file": e.dwarf_file,
169
+ "provenance": e.provenance,
170
+ } for e in self.index.dwarf_evidence)
171
+
172
+ def resolve_source_address(self, address: int, *, load_bias: int = 0) -> dict[str, object] | None:
173
+ dwarf = self.project.dwarf_index
174
+ if dwarf is None:
175
+ return None
176
+ row = dwarf.resolve_address(address, load_bias=load_bias)
177
+ if row is None:
178
+ return None
179
+ return {"file": row.file, "line": row.line, "address": row.address}
180
+
181
+ @property
182
+ def plugin_diagnostics(self) -> tuple[dict[str, str], ...]:
183
+ return tuple({
184
+ "plugin": d.plugin_name,
185
+ "severity": d.severity,
186
+ "code": d.code,
187
+ "message": d.message,
188
+ } for d in self.project.plugin_result.diagnostics)
189
+
190
+ def sequence(self, entry: str = "main") -> SequenceModel:
191
+ return build_sequence(self.index, entry)
192
+
193
+ def plantuml(self, entry: str = "main") -> str:
194
+ return render_plantuml(self.sequence(entry))
195
+
196
+ def html(self, entry: str = "main", **kwargs) -> str:
197
+ return render_html(self.sequence(entry), **kwargs)
198
+
199
+ # Backward-compatible raw result methods.
200
+ def explain(self, caller: str, callee: str | None = None) -> dict[str, object]:
201
+ return explain_call(self.index, caller=caller, callee_text=callee)
202
+
203
+ def calls(self, function: str) -> dict[str, object]:
204
+ return query_calls(self.index, function)
205
+
206
+ def unresolved(self) -> dict[str, object]:
207
+ return query_unresolved(self.index)
208
+
209
+ def paths(self, start: str, end: str) -> dict[str, object]:
210
+ return query_paths(self.index, start, end)
211
+
212
+ # Stable typed facades.
213
+ def explanation(self, caller: str, callee: str | None = None) -> Explanation:
214
+ return Explanation(self.explain(caller, callee))
215
+
216
+ def query(self, kind: str, **kwargs) -> QueryResult:
217
+ if kind == "calls":
218
+ function = kwargs.get("function")
219
+ if not isinstance(function, str) or not function:
220
+ raise ValueError("query('calls') requires function=<name>")
221
+ payload = self.calls(function)
222
+ elif kind == "unresolved":
223
+ payload = self.unresolved()
224
+ elif kind == "paths":
225
+ start, end = kwargs.get("start"), kwargs.get("end")
226
+ if not isinstance(start, str) or not isinstance(end, str) or not start or not end:
227
+ raise ValueError("query('paths') requires start=<name>, end=<name>")
228
+ payload = self.paths(start, end)
229
+ else:
230
+ raise ValueError(f"unknown query kind: {kind}")
231
+ return QueryResult(kind, payload)
232
+
233
+ def runtime_session(
234
+ self,
235
+ paths: Iterable[str | Path],
236
+ *,
237
+ name: str = "default",
238
+ pattern: str | None = None,
239
+ time_format: str | None = None,
240
+ ) -> RuntimeSession:
241
+ runtime_pattern = RuntimePattern(pattern, time_format) if pattern else None
242
+ return RuntimeSession(build_trace_session(name, paths, pattern=runtime_pattern))
243
+
244
+ def observe_runtime(
245
+ self,
246
+ paths: Iterable[str | Path],
247
+ *,
248
+ name: str = "default",
249
+ pattern: str | None = None,
250
+ time_format: str | None = None,
251
+ ) -> RuntimeSession:
252
+ """Import a runtime session and add unambiguous observed CPU evidence.
253
+
254
+ StaticAffinity and BuildDomain evidence are preserved; observations are
255
+ appended as the independent OBSERVED_RUNTIME_DOMAIN layer.
256
+ """
257
+ runtime_pattern = RuntimePattern(pattern, time_format) if pattern else None
258
+ session = build_trace_session(name, paths, pattern=runtime_pattern)
259
+ apply_runtime_cpu_evidence(self.index, session)
260
+ return RuntimeSession(session)
261
+
262
+ def runtime_address_evidence(
263
+ self,
264
+ session: RuntimeSession,
265
+ *,
266
+ load_bias: int = 0,
267
+ trusted_keys: Iterable[str] = (),
268
+ ) -> tuple[dict[str, object], ...]:
269
+ image = self.project.binary_image
270
+ if image is None:
271
+ return ()
272
+ rows = discover_runtime_address_evidence(
273
+ session._session.events, image, dwarf=self.project.dwarf_index,
274
+ load_bias=load_bias, trusted_keys=trusted_keys,
275
+ )
276
+ return tuple({
277
+ "event_index": e.event_index, "marker_id": e.marker_id,
278
+ "key": e.payload_key, "value": e.payload_value,
279
+ "runtime_address": e.runtime_address, "linked_address": e.linked_address,
280
+ "symbol": e.symbol_name, "symbol_size": e.symbol_size,
281
+ "source_file": e.source_file, "source_line": e.source_line,
282
+ "confidence": e.confidence, "provenance": e.provenance,
283
+ } for e in rows)
284
+
285
+ def marker_suggestions(self, start: str, end: str, *, count: int = 5) -> tuple[MarkerSuggestion, ...]:
286
+ reconstruction = reconstruct_static_paths(self.index, start, end)
287
+ if reconstruction.state != ReconstructionState.AMBIGUOUS_PATH:
288
+ return ()
289
+ suggestions = score_marker_suggestions(reconstruction.paths, count=count)
290
+ return tuple(MarkerSuggestion(
291
+ s.function_name,
292
+ s.information_gain_score,
293
+ s.distinguished_path_pairs,
294
+ s.total_path_pairs,
295
+ s.reason,
296
+ ) for s in suggestions)
297
+
298
+
299
+ def analyze(
300
+ path: str | Path,
301
+ *,
302
+ config: str | Path | None = None,
303
+ configuration: str | None = None,
304
+ defines: Iterable[str] = (),
305
+ compile_db: str | Path | None = None,
306
+ linker_map: str | Path | None = None,
307
+ binary: str | Path | None = None,
308
+ dwarf: str | Path | None = None,
309
+ use_cache: bool = True,
310
+ cache_dir: str | Path | None = None,
311
+ jobs: int = 1,
312
+ static_store: str | Path | None = None,
313
+ ) -> Analysis:
314
+ project = Project(
315
+ path,
316
+ config=config,
317
+ configuration=configuration,
318
+ defines=tuple(defines),
319
+ compile_db=compile_db,
320
+ linker_map=linker_map,
321
+ binary=binary,
322
+ dwarf=dwarf,
323
+ use_cache=use_cache,
324
+ cache_dir=cache_dir,
325
+ jobs=jobs,
326
+ static_store=static_store,
327
+ )
328
+ return Analysis(project=project, index=project.index())
cseq/binary.py ADDED
@@ -0,0 +1,122 @@
1
+ from __future__ import annotations
2
+
3
+ from dataclasses import dataclass
4
+ from hashlib import sha256
5
+ from pathlib import Path
6
+ import bisect
7
+ import subprocess
8
+ import shutil
9
+ import os
10
+
11
+ from .model import BinaryEvidenceRecord, ProjectIndex
12
+
13
+
14
+ @dataclass(frozen=True, slots=True)
15
+ class BinarySymbol:
16
+ name: str
17
+ address: int
18
+ size: int
19
+ kind: str
20
+
21
+
22
+ @dataclass(frozen=True, slots=True)
23
+ class BinaryImage:
24
+ path: str
25
+ content_hash: str
26
+ symbols: tuple[BinarySymbol, ...]
27
+
28
+ def resolve_address(self, runtime_address: int, *, load_bias: int = 0) -> BinarySymbol | None:
29
+ """Resolve a runtime PC/function-pointer address against linked symbols.
30
+
31
+ ``load_bias`` is subtracted explicitly for PIE/ASLR images. Zero-size
32
+ symbols are bounded by the next symbol address when possible.
33
+ """
34
+ address = int(runtime_address) - int(load_bias)
35
+ symbols = self.symbols
36
+ starts = [s.address for s in symbols]
37
+ pos = bisect.bisect_right(starts, address) - 1
38
+ if pos < 0:
39
+ return None
40
+ sym = symbols[pos]
41
+ if sym.size > 0:
42
+ return sym if sym.address <= address < sym.address + sym.size else None
43
+ next_start = symbols[pos + 1].address if pos + 1 < len(symbols) else None
44
+ if next_start is None:
45
+ return sym if address == sym.address else None
46
+ return sym if sym.address <= address < next_start else None
47
+
48
+
49
+ def _existing_executable_path(path: str | Path) -> Path:
50
+ p = Path(path)
51
+ if p.exists():
52
+ return p
53
+ if os.name == "nt":
54
+ candidate = Path(str(p) + ".exe")
55
+ if candidate.exists():
56
+ return candidate
57
+ return p
58
+
59
+
60
+ def load_binary_image(path: str | Path, *, nm: str = "nm") -> BinaryImage:
61
+ p = _existing_executable_path(path)
62
+ data = p.read_bytes()
63
+ nm_exe = shutil.which(nm) or (shutil.which("llvm-nm") if nm == "nm" else None) or nm
64
+ proc = subprocess.run(
65
+ [nm_exe, "-n", "-S", "--defined-only", str(p)],
66
+ stdout=subprocess.PIPE,
67
+ stderr=subprocess.PIPE,
68
+ text=True,
69
+ check=False,
70
+ )
71
+ if proc.returncode != 0:
72
+ raise RuntimeError(f"nm failed for {p}: {proc.stderr.strip()}")
73
+ symbols: list[BinarySymbol] = []
74
+ for line in proc.stdout.splitlines():
75
+ parts = line.split()
76
+ if len(parts) < 3:
77
+ continue
78
+ try:
79
+ address = int(parts[0], 16)
80
+ if len(parts) >= 4:
81
+ size = int(parts[1], 16)
82
+ kind = parts[2]
83
+ name = " ".join(parts[3:])
84
+ else:
85
+ # PE/COFF nm commonly omits symbol sizes even with -S.
86
+ # Keep zero-size symbols; resolve_address bounds them by the
87
+ # next symbol just as it already does for ELF zero-size rows.
88
+ size = 0
89
+ kind = parts[1]
90
+ name = " ".join(parts[2:])
91
+ except ValueError:
92
+ continue
93
+ # Text/weak text symbols are executable function candidates. Keep no
94
+ # object/data symbols in the function-address index.
95
+ if kind not in {"T", "t", "W", "w"}:
96
+ continue
97
+ symbols.append(BinarySymbol(name, address, size, kind))
98
+ symbols.sort(key=lambda x: (x.address, x.name))
99
+ return BinaryImage(str(p), sha256(data).hexdigest(), tuple(symbols))
100
+
101
+
102
+ def apply_binary_evidence(index: ProjectIndex, image: BinaryImage) -> list[BinaryEvidenceRecord]:
103
+ by_name: dict[str, list] = {}
104
+ for fn in index.functions():
105
+ by_name.setdefault(fn.name, []).append(fn)
106
+ out: list[BinaryEvidenceRecord] = []
107
+ for sym in image.symbols:
108
+ matches = by_name.get(sym.name, [])
109
+ # A binary symbol alone does not identify which duplicate source
110
+ # definition supplied it. Bind only unique source-name matches.
111
+ ids = (matches[0].qualified_id,) if len(matches) == 1 else ()
112
+ out.append(BinaryEvidenceRecord(
113
+ symbol_name=sym.name,
114
+ address=sym.address,
115
+ size=sym.size,
116
+ symbol_kind=sym.kind,
117
+ matched_function_ids=ids,
118
+ binary_hash=image.content_hash,
119
+ provenance=f"binary:{image.path}",
120
+ ))
121
+ index.binary_evidence.extend(out)
122
+ return out