diffgenome 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- diffgenome/__init__.py +7 -0
- diffgenome/__main__.py +240 -0
- diffgenome/_collectors/go/dg/dg.go +623 -0
- diffgenome/_collectors/go/go.mod +3 -0
- diffgenome/_collectors/go/instrument/facts.go +346 -0
- diffgenome/_collectors/go/instrument/main.go +484 -0
- diffgenome/_collectors/node/instrument.js +289 -0
- diffgenome/_collectors/node/jest-setup.js +40 -0
- diffgenome/_collectors/node/package-lock.json +35 -0
- diffgenome/_collectors/node/package.json +11 -0
- diffgenome/_collectors/node/runtime.js +426 -0
- diffgenome/ambiguity.py +122 -0
- diffgenome/api.py +67 -0
- diffgenome/change.py +86 -0
- diffgenome/change_artifact.py +310 -0
- diffgenome/collect/__init__.py +2 -0
- diffgenome/collect/go_test.py +271 -0
- diffgenome/collect/node_jest.py +319 -0
- diffgenome/collect/py_monitoring.py +985 -0
- diffgenome/collect/py_runtime.py +116 -0
- diffgenome/collect/py_symbols.py +238 -0
- diffgenome/collect/pytest_plugin.py +130 -0
- diffgenome/compose.py +469 -0
- diffgenome/dependence.py +264 -0
- diffgenome/evaluate.py +669 -0
- diffgenome/frontends/__init__.py +0 -0
- diffgenome/frontends/python_ir.py +335 -0
- diffgenome/genome.py +1016 -0
- diffgenome/genome_pipeline.py +674 -0
- diffgenome/genome_prompt.py +33 -0
- diffgenome/genome_state.py +2118 -0
- diffgenome/graph.py +426 -0
- diffgenome/llm.py +189 -0
- diffgenome/model.py +364 -0
- diffgenome/mvp.py +398 -0
- diffgenome/probe.py +509 -0
- diffgenome/projection.py +308 -0
- diffgenome/py.typed +0 -0
- diffgenome/render.py +118 -0
- diffgenome/report.py +363 -0
- diffgenome/resolve.py +37 -0
- diffgenome/runtime.py +74 -0
- diffgenome/runtime_evidence.py +261 -0
- diffgenome/sandbox.py +166 -0
- diffgenome/serialize.py +96 -0
- diffgenome/sites.py +19 -0
- diffgenome/static_types.py +69 -0
- diffgenome/structure.py +462 -0
- diffgenome-0.1.0.dist-info/METADATA +139 -0
- diffgenome-0.1.0.dist-info/RECORD +53 -0
- diffgenome-0.1.0.dist-info/WHEEL +4 -0
- diffgenome-0.1.0.dist-info/entry_points.txt +2 -0
- diffgenome-0.1.0.dist-info/licenses/LICENSE +202 -0
diffgenome/change.py
ADDED
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
"""Change-centred analysis: from a diff (or an explicit symbol) to changed symbols.
|
|
2
|
+
|
|
3
|
+
Diff parsing is language-neutral (paths and line ranges on the new side). Mapping lines
|
|
4
|
+
to symbols goes through a `SymbolIndex`, which a language adapter implements.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import re
|
|
10
|
+
import subprocess
|
|
11
|
+
from collections.abc import Sequence
|
|
12
|
+
from dataclasses import dataclass
|
|
13
|
+
from pathlib import Path
|
|
14
|
+
from typing import Protocol
|
|
15
|
+
|
|
16
|
+
from diffgenome.model import SymbolId
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class Located(Protocol):
|
|
20
|
+
@property
|
|
21
|
+
def symbol(self) -> SymbolId: ...
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class SymbolIndex(Protocol):
|
|
25
|
+
def symbols_at(self, rel_path: str, lines: set[int]) -> Sequence[Located]: ...
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
@dataclass(frozen=True)
|
|
29
|
+
class ChangedRange:
|
|
30
|
+
path: str
|
|
31
|
+
lines: frozenset[int] # new-side line numbers; empty for pure deletions
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
@dataclass
|
|
35
|
+
class ChangeSet:
|
|
36
|
+
description: str
|
|
37
|
+
ranges: list[ChangedRange]
|
|
38
|
+
symbols: list[SymbolId]
|
|
39
|
+
unmapped_paths: list[str] # changed files no symbol could be mapped for (non-code, deleted)
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
_HUNK = re.compile(r"^@@ -\d+(?:,\d+)? \+(\d+)(?:,(\d+))? @@")
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def parse_unified_diff(text: str) -> list[ChangedRange]:
|
|
46
|
+
ranges: dict[str, set[int]] = {}
|
|
47
|
+
path: str | None = None
|
|
48
|
+
for line in text.splitlines():
|
|
49
|
+
if line.startswith("+++ "):
|
|
50
|
+
target = line[4:].strip()
|
|
51
|
+
path = None if target == "/dev/null" else target.removeprefix("b/")
|
|
52
|
+
if path is not None:
|
|
53
|
+
ranges.setdefault(path, set())
|
|
54
|
+
elif line.startswith("@@") and path is not None:
|
|
55
|
+
m = _HUNK.match(line)
|
|
56
|
+
if m:
|
|
57
|
+
start, count = int(m.group(1)), int(m.group(2) or "1")
|
|
58
|
+
ranges[path].update(range(start, start + max(count, 1)))
|
|
59
|
+
return [ChangedRange(p, frozenset(ls)) for p, ls in sorted(ranges.items())]
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def git_diff(repo: Path, revspec: str) -> str:
|
|
63
|
+
"""`revspec` like ``base..head`` or a single commit (diffed against its parent)."""
|
|
64
|
+
spec = revspec if ".." in revspec else f"{revspec}^..{revspec}"
|
|
65
|
+
return subprocess.run(
|
|
66
|
+
["git", "diff", "--unified=0", "--no-color", spec],
|
|
67
|
+
cwd=repo, capture_output=True, text=True, check=True,
|
|
68
|
+
).stdout # fmt: skip
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def changes_from_diff(text: str, index: SymbolIndex, description: str) -> ChangeSet:
|
|
72
|
+
ranges = parse_unified_diff(text)
|
|
73
|
+
symbols: list[SymbolId] = []
|
|
74
|
+
unmapped: list[str] = []
|
|
75
|
+
for r in ranges:
|
|
76
|
+
defs = index.symbols_at(r.path, set(r.lines)) if r.lines else []
|
|
77
|
+
found = [d.symbol for d in defs]
|
|
78
|
+
if found:
|
|
79
|
+
symbols.extend(s for s in found if s not in symbols)
|
|
80
|
+
else:
|
|
81
|
+
unmapped.append(r.path)
|
|
82
|
+
return ChangeSet(description, ranges, symbols, unmapped)
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def changes_from_symbols(symbols: list[SymbolId]) -> ChangeSet:
|
|
86
|
+
return ChangeSet("symbol-centred: " + ", ".join(symbols), [], list(symbols), [])
|
|
@@ -0,0 +1,310 @@
|
|
|
1
|
+
"""The integration artifact: ``diffgenome-change/1``.
|
|
2
|
+
|
|
3
|
+
A bounded, deterministic, change-centred projection of the behavioral graph for consumers
|
|
4
|
+
such as Sydes. It is derived from the same graph as ``map.md`` and ``slice.json`` but
|
|
5
|
+
exposes no internal objects: symbols become nodes with stable ids and locations, evidence
|
|
6
|
+
becomes one edge per (caller, callee) with its evidence class, join grade, STATE status,
|
|
7
|
+
exit verdict and provenance, stand-in leaves become boundaries, and every join the composer
|
|
8
|
+
rejected on a seam in the neighborhood is listed with its reason. It is consumable without
|
|
9
|
+
the trace corpus. Nothing in it is a confidence score: the grades are the evidence.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import json
|
|
15
|
+
from collections import Counter, defaultdict
|
|
16
|
+
from datetime import UTC, datetime
|
|
17
|
+
from typing import Any
|
|
18
|
+
|
|
19
|
+
from diffgenome import __version__
|
|
20
|
+
from diffgenome.graph import BehavioralGraph, GraphEdge
|
|
21
|
+
from diffgenome.model import (
|
|
22
|
+
CallNode,
|
|
23
|
+
EvidenceKind,
|
|
24
|
+
Execution,
|
|
25
|
+
JoinStrength,
|
|
26
|
+
Origin,
|
|
27
|
+
Stimulus,
|
|
28
|
+
SymbolId,
|
|
29
|
+
)
|
|
30
|
+
from diffgenome.projection import case_metrics, project, render_behavior_map
|
|
31
|
+
|
|
32
|
+
FORMAT = "diffgenome-change/1"
|
|
33
|
+
|
|
34
|
+
# Evidence classes a consumer merges on. Observed and composed are runtime evidence of
|
|
35
|
+
# different standing; the rest are places where DiffGenome's knowledge stops.
|
|
36
|
+
EVIDENCE_OBSERVED = "observed"
|
|
37
|
+
EVIDENCE_COMPOSED = "composed"
|
|
38
|
+
EVIDENCE_EXTERNAL = "external"
|
|
39
|
+
EVIDENCE_UNRESOLVED = "unresolved"
|
|
40
|
+
EVIDENCE_GAP = "gap"
|
|
41
|
+
EVIDENCE_DECLARATION = "declaration"
|
|
42
|
+
EVIDENCE_OS = "os"
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def _name(symbol: SymbolId) -> str:
|
|
46
|
+
return symbol.split(":", 1)[1] if ":" in symbol else symbol
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def _execution_label(execution_id: str) -> str:
|
|
50
|
+
return execution_id.split("@")[0]
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def _test_file(ex: Execution) -> str | None:
|
|
54
|
+
"""The test source file of an execution: the location of the first test-origin call
|
|
55
|
+
the collector attributed a location to (the stimulus root itself may carry none, as a
|
|
56
|
+
Go subtest's does)."""
|
|
57
|
+
symbols = {sym.id: sym for sym in ex.symbols}
|
|
58
|
+
for node in ex.nodes:
|
|
59
|
+
if isinstance(node, CallNode):
|
|
60
|
+
sym = symbols.get(node.symbol)
|
|
61
|
+
if sym is not None and sym.origin is Origin.TEST and sym.location is not None:
|
|
62
|
+
return sym.location.path
|
|
63
|
+
return None
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def _state_status(graph: BehavioralGraph, edge: GraphEdge) -> str:
|
|
67
|
+
"""STATE evidence status of a composed edge, from the join attempts behind it:
|
|
68
|
+
``matched`` (STATE grade reached), ``conflict-filtered`` (candidates were rejected on
|
|
69
|
+
state; the accepted ones matched), ``unavailable`` (VALUE held but no state facts on a
|
|
70
|
+
side), ``not_consulted`` (below VALUE, where STATE is never evaluated)."""
|
|
71
|
+
if edge.kind is not EvidenceKind.COMPOSED:
|
|
72
|
+
return "n/a"
|
|
73
|
+
best = edge.best_join
|
|
74
|
+
if best is JoinStrength.STATE:
|
|
75
|
+
return "matched"
|
|
76
|
+
sites = {ev.site for ev in edge.evidence}
|
|
77
|
+
notes = [a.note for a in graph.attempts if a.site in sites and a.target == edge.callee]
|
|
78
|
+
if best is JoinStrength.VALUE:
|
|
79
|
+
if any("state unavailable" in n or "no common state" in n for n in notes):
|
|
80
|
+
return "unavailable"
|
|
81
|
+
return "unavailable"
|
|
82
|
+
return "not_consulted"
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def _exit_status(edge: GraphEdge) -> str | None:
|
|
86
|
+
if edge.kind is not EvidenceKind.COMPOSED:
|
|
87
|
+
return None
|
|
88
|
+
verdicts = {ev.exit for ev in edge.evidence if ev.exit}
|
|
89
|
+
if not verdicts:
|
|
90
|
+
return None
|
|
91
|
+
for v in ("same", "kind", "unknown"):
|
|
92
|
+
if verdicts == {v}:
|
|
93
|
+
return v
|
|
94
|
+
return "mixed:" + "/".join(sorted(verdicts))
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def build_change_artifact(
|
|
98
|
+
graph: BehavioralGraph,
|
|
99
|
+
seeds: list[SymbolId],
|
|
100
|
+
*,
|
|
101
|
+
up: int,
|
|
102
|
+
down: int,
|
|
103
|
+
change_spec: str,
|
|
104
|
+
runtime: str,
|
|
105
|
+
repo: str,
|
|
106
|
+
revision: str | None,
|
|
107
|
+
budget: dict[str, Any],
|
|
108
|
+
probes: list[dict[str, Any]],
|
|
109
|
+
notes: list[str],
|
|
110
|
+
graph_before: BehavioralGraph | None = None,
|
|
111
|
+
) -> dict[str, Any]:
|
|
112
|
+
"""Build the artifact from a graph and the changed symbols. Deterministic for a
|
|
113
|
+
given graph: every list is sorted, every set rendered as a sorted list."""
|
|
114
|
+
nb = graph.neighborhood(seeds, up=up, down=down)
|
|
115
|
+
projected = project(graph)
|
|
116
|
+
keys = {(e.caller, e.callee) for _, e in nb.edges.values()}
|
|
117
|
+
distance: dict[tuple[SymbolId, SymbolId], int] = {}
|
|
118
|
+
for d, e in nb.edges.values():
|
|
119
|
+
key = (e.caller, e.callee)
|
|
120
|
+
distance[key] = min(d, distance.get(key, d))
|
|
121
|
+
edges_by_key: dict[tuple[SymbolId, SymbolId], list[GraphEdge]] = defaultdict(list)
|
|
122
|
+
for _, e in nb.edges.values():
|
|
123
|
+
edges_by_key[(e.caller, e.callee)].append(e)
|
|
124
|
+
|
|
125
|
+
# ---- nodes: every production symbol touched by the neighborhood, plus the seeds
|
|
126
|
+
node_ids: set[SymbolId] = set(seeds)
|
|
127
|
+
for (c, k), be in projected.items():
|
|
128
|
+
if (c, k) in keys and graph.origin(c) is not Origin.TEST:
|
|
129
|
+
node_ids.add(c)
|
|
130
|
+
if be.boundary is None:
|
|
131
|
+
node_ids.add(k)
|
|
132
|
+
nodes = []
|
|
133
|
+
for sid in sorted(node_ids):
|
|
134
|
+
sym = graph.symbols.get(sid)
|
|
135
|
+
nodes.append(
|
|
136
|
+
{
|
|
137
|
+
"id": sid,
|
|
138
|
+
"name": _name(sid),
|
|
139
|
+
"file": sym.location.path if sym and sym.location else None,
|
|
140
|
+
"line": sym.location.line if sym and sym.location else None,
|
|
141
|
+
"kind": sym.kind if sym else "callable",
|
|
142
|
+
"origin": (sym.origin.value if sym else Origin.UNKNOWN.value),
|
|
143
|
+
"changed": sid in seeds,
|
|
144
|
+
"executed_by": len(graph.tests_by_symbol.get(sid, ())),
|
|
145
|
+
}
|
|
146
|
+
)
|
|
147
|
+
|
|
148
|
+
# ---- edges and boundaries
|
|
149
|
+
edges: list[dict[str, Any]] = []
|
|
150
|
+
boundaries: list[dict[str, Any]] = []
|
|
151
|
+
for (c, k), be in sorted(projected.items()):
|
|
152
|
+
if (c, k) not in keys or graph.origin(c) is Origin.TEST:
|
|
153
|
+
continue
|
|
154
|
+
raw = edges_by_key[(c, k)]
|
|
155
|
+
if be.boundary is not None:
|
|
156
|
+
rules = sorted({r for e in raw for r in e.rules})
|
|
157
|
+
boundaries.append(
|
|
158
|
+
{
|
|
159
|
+
"caller": c,
|
|
160
|
+
"target": k,
|
|
161
|
+
"target_name": _name(k),
|
|
162
|
+
"kind": be.boundary,
|
|
163
|
+
"rules": rules,
|
|
164
|
+
"distance": distance[(c, k)],
|
|
165
|
+
"executions": sorted(_execution_label(x) for x in be.tests | be.probes),
|
|
166
|
+
}
|
|
167
|
+
)
|
|
168
|
+
continue
|
|
169
|
+
composed = [e for e in raw if e.kind is EvidenceKind.COMPOSED]
|
|
170
|
+
observed = be.observed_executions
|
|
171
|
+
evidence = EVIDENCE_OBSERVED if observed else EVIDENCE_COMPOSED
|
|
172
|
+
best = be.best_grade
|
|
173
|
+
state = "n/a"
|
|
174
|
+
exit_ = None
|
|
175
|
+
if composed:
|
|
176
|
+
state = _state_status(graph, composed[0])
|
|
177
|
+
exit_ = _exit_status(composed[0])
|
|
178
|
+
edges.append(
|
|
179
|
+
{
|
|
180
|
+
"caller": c,
|
|
181
|
+
"callee": k,
|
|
182
|
+
"evidence": evidence,
|
|
183
|
+
"distance": distance[(c, k)],
|
|
184
|
+
"join": best if evidence == EVIDENCE_COMPOSED else None,
|
|
185
|
+
"also_composed_at": best if evidence == EVIDENCE_OBSERVED and best else None,
|
|
186
|
+
"state": state if evidence == EVIDENCE_COMPOSED else "n/a",
|
|
187
|
+
"exit": exit_,
|
|
188
|
+
"probe_derived": bool(be.probes) and not be.tests,
|
|
189
|
+
"executions": sorted(_execution_label(x) for x in be.tests),
|
|
190
|
+
"probes": sorted(_execution_label(x) for x in be.probes),
|
|
191
|
+
"composed_shapes_by_grade": dict(sorted(be.composed.items())),
|
|
192
|
+
"same_shape_alternates": be.alternates,
|
|
193
|
+
"ambiguous": be.ambiguous,
|
|
194
|
+
"rules": sorted(be.rules),
|
|
195
|
+
}
|
|
196
|
+
)
|
|
197
|
+
|
|
198
|
+
# ---- executions contributing
|
|
199
|
+
exec_ids = {
|
|
200
|
+
x
|
|
201
|
+
for be in projected.values()
|
|
202
|
+
for x in (be.tests | be.probes)
|
|
203
|
+
if (be.caller, be.callee) in keys
|
|
204
|
+
}
|
|
205
|
+
executions = []
|
|
206
|
+
for x in sorted(exec_ids):
|
|
207
|
+
stim = "generated_probe"
|
|
208
|
+
test_file: str | None = None
|
|
209
|
+
outcome: str | None = None
|
|
210
|
+
if graph.corpus is not None and x in graph.corpus.executions:
|
|
211
|
+
ex = graph.corpus.executions[x]
|
|
212
|
+
stim = ex.stimulus.value
|
|
213
|
+
test_file = _test_file(ex)
|
|
214
|
+
outcome = ex.outcome
|
|
215
|
+
elif "diffgenome_probe" not in x:
|
|
216
|
+
stim = Stimulus.EXISTING_TEST.value
|
|
217
|
+
executions.append(
|
|
218
|
+
{"id": _execution_label(x), "stimulus": stim, "file": test_file, "outcome": outcome}
|
|
219
|
+
)
|
|
220
|
+
|
|
221
|
+
# ---- ambiguous seams and rejected candidates on neighborhood seams
|
|
222
|
+
sites = {ev.site for _, e in nb.edges.values() for ev in e.evidence}
|
|
223
|
+
site_caller: dict[Any, SymbolId] = {}
|
|
224
|
+
for _, e in nb.edges.values():
|
|
225
|
+
for ev in e.evidence:
|
|
226
|
+
site_caller.setdefault(ev.site, e.caller)
|
|
227
|
+
rejected: list[dict[str, Any]] = []
|
|
228
|
+
by_seam: dict[tuple[SymbolId, SymbolId], Counter[str]] = defaultdict(Counter)
|
|
229
|
+
accepted_by_seam: dict[tuple[SymbolId, SymbolId], set[str]] = defaultdict(set)
|
|
230
|
+
for a in graph.attempts:
|
|
231
|
+
if a.site not in sites:
|
|
232
|
+
continue
|
|
233
|
+
seam = (site_caller.get(a.site, "?"), a.target)
|
|
234
|
+
if a.accepted:
|
|
235
|
+
accepted_by_seam[seam].add(a.fragment.execution)
|
|
236
|
+
continue
|
|
237
|
+
reason = a.note.split(";")[0].strip()
|
|
238
|
+
by_seam[seam][reason] += 1
|
|
239
|
+
rejected.append(
|
|
240
|
+
{
|
|
241
|
+
"caller": seam[0],
|
|
242
|
+
"target": a.target,
|
|
243
|
+
"candidate": _execution_label(a.fragment.execution),
|
|
244
|
+
"reason": reason,
|
|
245
|
+
}
|
|
246
|
+
)
|
|
247
|
+
rejected.sort(key=lambda r: (r["caller"], r["target"], r["candidate"], r["reason"]))
|
|
248
|
+
ambiguous_seams = [
|
|
249
|
+
{
|
|
250
|
+
"caller": c,
|
|
251
|
+
"target": k,
|
|
252
|
+
"accepted_candidates": len(accepted_by_seam.get((c, k), ())),
|
|
253
|
+
"rejected_candidates": sum(by_seam.get((c, k), Counter()).values()),
|
|
254
|
+
"rejection_reasons": dict(sorted(by_seam.get((c, k), Counter()).items())),
|
|
255
|
+
}
|
|
256
|
+
for (c, k) in sorted(set(by_seam) | {s for s, v in accepted_by_seam.items() if len(v) > 1})
|
|
257
|
+
]
|
|
258
|
+
|
|
259
|
+
metrics = case_metrics(graph, seeds, up, down, [], graph_before)
|
|
260
|
+
facts = {
|
|
261
|
+
k: v
|
|
262
|
+
for k, v in metrics.items()
|
|
263
|
+
if k
|
|
264
|
+
in (
|
|
265
|
+
"observed_edges", "composed_edges", "joins", "rejected_joins", "ambiguous_joins",
|
|
266
|
+
"internal_gaps", "declarations_no_in_repo_body", "unresolved_boundaries",
|
|
267
|
+
"external_boundaries", "existing_tests_contributing", "probe_derived_edges",
|
|
268
|
+
"structural_coverage_ratio_NOT_correctness",
|
|
269
|
+
)
|
|
270
|
+
} # fmt: skip
|
|
271
|
+
return {
|
|
272
|
+
"format": FORMAT,
|
|
273
|
+
"generated_by": f"diffgenome {__version__}",
|
|
274
|
+
"generated_at": datetime.now(tz=UTC).isoformat(timespec="seconds"),
|
|
275
|
+
"repository": {"root": repo, "runtime": runtime, "revision": revision or "checkout"},
|
|
276
|
+
"change": {
|
|
277
|
+
"spec": change_spec,
|
|
278
|
+
"symbols": sorted(seeds),
|
|
279
|
+
"symbols_never_executed": sorted(s for s in seeds if s not in graph.tests_by_symbol),
|
|
280
|
+
},
|
|
281
|
+
"neighborhood": {"up": up, "down": down, "symbols": len(node_ids)},
|
|
282
|
+
"budget": budget,
|
|
283
|
+
"nodes": nodes,
|
|
284
|
+
"edges": edges,
|
|
285
|
+
"boundaries": boundaries,
|
|
286
|
+
"executions": executions,
|
|
287
|
+
"ambiguous_seams": ambiguous_seams,
|
|
288
|
+
"rejected_candidates": rejected[:200],
|
|
289
|
+
"probes": probes,
|
|
290
|
+
"facts": facts,
|
|
291
|
+
"invariants": [
|
|
292
|
+
"observed edges were seen in one execution; composed edges are reconstructed "
|
|
293
|
+
"across a stand-in seam and are never presented as observed",
|
|
294
|
+
"join grades are evidence classes, not a probability: "
|
|
295
|
+
"SYMBOL < ARG_SHAPE < VALUE < STATE",
|
|
296
|
+
"structural_coverage_ratio_NOT_correctness counts symbol pairs with evidence; "
|
|
297
|
+
"it says nothing about whether the behavior is right",
|
|
298
|
+
"boundaries are where DiffGenome's knowledge stops: external stays substituted, "
|
|
299
|
+
"unresolved was not attributed, gap has no executing fragment",
|
|
300
|
+
],
|
|
301
|
+
"notes": list(notes),
|
|
302
|
+
}
|
|
303
|
+
|
|
304
|
+
|
|
305
|
+
def render_behavioral_map(graph: BehavioralGraph, seeds: list[SymbolId], up: int, down: int) -> str:
|
|
306
|
+
return render_behavior_map(graph, seeds, up=up, down=down)
|
|
307
|
+
|
|
308
|
+
|
|
309
|
+
def dumps(artifact: dict[str, Any]) -> str:
|
|
310
|
+
return json.dumps(artifact, indent=1, sort_keys=False) + "\n"
|
|
@@ -0,0 +1,271 @@
|
|
|
1
|
+
"""Go runtime adapter: source instrumentation + `go test`.
|
|
2
|
+
|
|
3
|
+
Everything Go-specific lives here and in diffgenome/_collectors/go. The workspace copy gets the
|
|
4
|
+
`dg` runtime package copied to `<module>/internal/diffgenome/dg`, its sources rewritten in
|
|
5
|
+
place by the instrumenter (built once with the host Go), and `go test` runs inside the
|
|
6
|
+
sandbox with the module cache read-only and GOPROXY=off (no network). Generated mock
|
|
7
|
+
packages are test origin; their methods claim the sole in-repo definer of the mocked
|
|
8
|
+
interface method (a static fact from the MockGen header and the package's method sets).
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import json
|
|
14
|
+
import os
|
|
15
|
+
import shutil
|
|
16
|
+
import subprocess
|
|
17
|
+
from dataclasses import dataclass
|
|
18
|
+
from pathlib import Path
|
|
19
|
+
|
|
20
|
+
from diffgenome.model import Execution, Stimulus, SymbolId
|
|
21
|
+
from diffgenome.probe import copy_probe_traces
|
|
22
|
+
from diffgenome.runtime import SymbolIndex
|
|
23
|
+
from diffgenome.sandbox import Workspace
|
|
24
|
+
from diffgenome.serialize import execution_from_json
|
|
25
|
+
|
|
26
|
+
TOOLS = Path(__file__).resolve().parent.parent / "_collectors" / "go"
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
@dataclass(frozen=True)
|
|
30
|
+
class Definition:
|
|
31
|
+
symbol: SymbolId
|
|
32
|
+
path: str
|
|
33
|
+
start: int
|
|
34
|
+
end: int
|
|
35
|
+
kind: str
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
class GoSymbolIndex:
|
|
39
|
+
def __init__(
|
|
40
|
+
self,
|
|
41
|
+
repo_root: Path,
|
|
42
|
+
source_roots: list[Path],
|
|
43
|
+
test_roots: list[Path],
|
|
44
|
+
index_file: Path,
|
|
45
|
+
mock_roots: list[Path] | None = None,
|
|
46
|
+
) -> None:
|
|
47
|
+
self.repo_root = repo_root.resolve()
|
|
48
|
+
self.source_roots = [p.resolve() for p in source_roots]
|
|
49
|
+
self.test_roots = [p.resolve() for p in test_roots]
|
|
50
|
+
self.mock_roots = [p.resolve() for p in (mock_roots or [])]
|
|
51
|
+
doc = json.loads(index_file.read_text())
|
|
52
|
+
self._defs = [
|
|
53
|
+
Definition(d["symbol"], d["path"], d["start"], d["end"], d["kind"])
|
|
54
|
+
for d in doc["definitions"]
|
|
55
|
+
]
|
|
56
|
+
self._by_symbol = {d.symbol: d for d in self._defs}
|
|
57
|
+
self._by_path: dict[str, list[Definition]] = {}
|
|
58
|
+
for d in self._defs:
|
|
59
|
+
self._by_path.setdefault(d.path, []).append(d)
|
|
60
|
+
|
|
61
|
+
def definitions(self, rel_path: str) -> list[Definition]:
|
|
62
|
+
return self._by_path.get(rel_path, [])
|
|
63
|
+
|
|
64
|
+
def symbols_at(self, rel_path: str, lines: set[int]) -> list[Definition]:
|
|
65
|
+
hits: dict[SymbolId, Definition] = {}
|
|
66
|
+
for line in lines:
|
|
67
|
+
covering = [d for d in self.definitions(rel_path) if d.start <= line <= d.end]
|
|
68
|
+
funcs = [d for d in covering if d.kind == "function"]
|
|
69
|
+
pick = (
|
|
70
|
+
max(funcs, key=lambda d: d.start)
|
|
71
|
+
if funcs
|
|
72
|
+
else (max(covering, key=lambda d: d.start) if covering else None)
|
|
73
|
+
)
|
|
74
|
+
if pick:
|
|
75
|
+
hits[pick.symbol] = pick
|
|
76
|
+
return sorted(hits.values(), key=lambda d: (d.path, d.start))
|
|
77
|
+
|
|
78
|
+
def find(self, symbol: SymbolId) -> Definition | None:
|
|
79
|
+
return self._by_symbol.get(symbol)
|
|
80
|
+
|
|
81
|
+
def source(self, symbol: SymbolId, context: int = 0) -> str | None:
|
|
82
|
+
d = self.find(symbol)
|
|
83
|
+
if d is None:
|
|
84
|
+
return None
|
|
85
|
+
lines = (self.repo_root / d.path).read_text(encoding="utf-8").splitlines()
|
|
86
|
+
lo, hi = max(0, d.start - 1 - context), min(len(lines), d.end + context)
|
|
87
|
+
return "\n".join(f"{i + 1:5d} {lines[i]}" for i in range(lo, hi))
|
|
88
|
+
|
|
89
|
+
def enclosing_class(self, symbol: SymbolId) -> Definition | None:
|
|
90
|
+
# go:pkg.Type.Method -> go:pkg.Type
|
|
91
|
+
name = symbol.split(":", 1)[1]
|
|
92
|
+
parts = name.split(".")
|
|
93
|
+
if len(parts) < 3:
|
|
94
|
+
return None
|
|
95
|
+
return self.find("go:" + ".".join(parts[:-1]))
|
|
96
|
+
|
|
97
|
+
def is_test(self, symbol: SymbolId) -> bool:
|
|
98
|
+
# Go tests live beside the code: test-ness is the _test.go suffix or a mock package.
|
|
99
|
+
d = self.find(symbol)
|
|
100
|
+
if d is None:
|
|
101
|
+
return False
|
|
102
|
+
if d.path.endswith("_test.go"):
|
|
103
|
+
return True
|
|
104
|
+
path = self.repo_root / d.path
|
|
105
|
+
return any(path.is_relative_to(r) for r in self.mock_roots)
|
|
106
|
+
|
|
107
|
+
def return_type_of(self, symbol: SymbolId) -> SymbolId | None:
|
|
108
|
+
return None # not extracted for this runtime yet
|
|
109
|
+
|
|
110
|
+
def module_of(self, symbol: SymbolId) -> str | None:
|
|
111
|
+
d = self.find(symbol)
|
|
112
|
+
return d.path.rsplit("/", 1)[0] if d and "/" in d.path else ("." if d else None)
|
|
113
|
+
|
|
114
|
+
def tests_importing(self, module: str, limit: int = 3) -> list[str]:
|
|
115
|
+
# Go tests live in the package directory itself.
|
|
116
|
+
pkg = self.repo_root / module
|
|
117
|
+
hits = (
|
|
118
|
+
sorted(str(f.relative_to(self.repo_root)) for f in pkg.glob("*_test.go"))
|
|
119
|
+
if pkg.is_dir()
|
|
120
|
+
else []
|
|
121
|
+
)
|
|
122
|
+
return hits[:limit]
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
class GoTestRuntime:
|
|
126
|
+
name = "go/test"
|
|
127
|
+
|
|
128
|
+
def __init__(
|
|
129
|
+
self,
|
|
130
|
+
repo: Path,
|
|
131
|
+
source_root: str,
|
|
132
|
+
test_root: str,
|
|
133
|
+
tests: str,
|
|
134
|
+
mock_dirs: list[str] | None = None,
|
|
135
|
+
) -> None:
|
|
136
|
+
self.repo = repo.resolve()
|
|
137
|
+
self.source_root = source_root
|
|
138
|
+
self.test_root = test_root # package dir whose tests are the stimuli, e.g. "api"
|
|
139
|
+
self.tests = tests # go test pattern, e.g. "./api/"
|
|
140
|
+
self.mock_dirs = mock_dirs or []
|
|
141
|
+
which = shutil.which("go")
|
|
142
|
+
if not which:
|
|
143
|
+
raise RuntimeError("go not found on PATH")
|
|
144
|
+
self.go = Path(which).resolve()
|
|
145
|
+
self.module = self._module_path()
|
|
146
|
+
self.gopath = Path(
|
|
147
|
+
subprocess.run(
|
|
148
|
+
[str(self.go), "env", "GOPATH"], capture_output=True, text=True
|
|
149
|
+
).stdout.strip()
|
|
150
|
+
)
|
|
151
|
+
self.gocache = Path(
|
|
152
|
+
subprocess.run(
|
|
153
|
+
[str(self.go), "env", "GOCACHE"], capture_output=True, text=True
|
|
154
|
+
).stdout.strip()
|
|
155
|
+
)
|
|
156
|
+
|
|
157
|
+
def _module_path(self) -> str:
|
|
158
|
+
for line in (self.repo / "go.mod").read_text().splitlines():
|
|
159
|
+
if line.startswith("module "):
|
|
160
|
+
return line.split()[1]
|
|
161
|
+
raise RuntimeError("go.mod has no module line")
|
|
162
|
+
|
|
163
|
+
def prepare(self, ws: Workspace) -> SymbolIndex:
|
|
164
|
+
# runtime package inside the module so no network/module edits are needed
|
|
165
|
+
dst = ws.repo / "internal" / "diffgenome" / "dg"
|
|
166
|
+
dst.mkdir(parents=True, exist_ok=True)
|
|
167
|
+
shutil.copy(TOOLS / "dg" / "dg.go", dst / "dg.go")
|
|
168
|
+
tool = ws.root / "dg-instrument"
|
|
169
|
+
# build from a copy inside the workspace: the installed package may be read-only
|
|
170
|
+
build_dir = ws.root / "go-collector"
|
|
171
|
+
if not build_dir.exists():
|
|
172
|
+
shutil.copytree(TOOLS, build_dir)
|
|
173
|
+
subprocess.run(
|
|
174
|
+
[str(self.go), "build", "-o", str(tool), "./instrument"],
|
|
175
|
+
cwd=build_dir, check=True, capture_output=True, text=True,
|
|
176
|
+
env={**os.environ, "GOFLAGS": "-mod=mod", "GOPROXY": "off"},
|
|
177
|
+
) # fmt: skip
|
|
178
|
+
index_file = ws.root / "symbol-index.json"
|
|
179
|
+
result = subprocess.run(
|
|
180
|
+
[
|
|
181
|
+
str(tool), "-root", str(ws.repo), "-module", self.module, "-src", self.source_root,
|
|
182
|
+
"-tests", ",".join(self.mock_dirs), "-index", str(index_file),
|
|
183
|
+
"-facts", str(ws.root / "mechanics-ir.json"),
|
|
184
|
+
],
|
|
185
|
+
capture_output=True, text=True, check=True,
|
|
186
|
+
) # fmt: skip
|
|
187
|
+
(ws.root / "instrument.log").write_text(result.stdout + result.stderr)
|
|
188
|
+
mock_roots = [self.repo / d for d in self.mock_dirs]
|
|
189
|
+
return GoSymbolIndex(
|
|
190
|
+
self.repo,
|
|
191
|
+
[self.repo / self.source_root],
|
|
192
|
+
[self.repo / self.test_root, *mock_roots],
|
|
193
|
+
index_file,
|
|
194
|
+
mock_roots,
|
|
195
|
+
)
|
|
196
|
+
|
|
197
|
+
def trace(
|
|
198
|
+
self, ws: Workspace, out_dir: Path, stimulus: Stimulus, only: list[str] | None
|
|
199
|
+
) -> tuple[list[Execution], str, str]:
|
|
200
|
+
tag = "existing" if stimulus is Stimulus.EXISTING_TEST else "probe"
|
|
201
|
+
traces_ws = ws.root / f"traces-{tag}-{len(list(ws.root.glob('traces-*')))}"
|
|
202
|
+
traces_ws.mkdir(parents=True, exist_ok=True)
|
|
203
|
+
env = {
|
|
204
|
+
"DIFFGENOME_OUT": str(traces_ws),
|
|
205
|
+
"DIFFGENOME_STIMULUS": stimulus.value,
|
|
206
|
+
"GOPATH": str(self.gopath),
|
|
207
|
+
"GOCACHE": str(ws.root / "gocache"),
|
|
208
|
+
"GOMODCACHE": str(self.gopath / "pkg" / "mod"),
|
|
209
|
+
"GOFLAGS": "-mod=mod",
|
|
210
|
+
"GOPROXY": "off",
|
|
211
|
+
"GOTOOLCHAIN": "local",
|
|
212
|
+
"CGO_ENABLED": "0",
|
|
213
|
+
"PATH": f"{self.go.parent}:/usr/bin:/bin",
|
|
214
|
+
}
|
|
215
|
+
argv = [str(self.go), "test", "-count=1", "-p", "1"]
|
|
216
|
+
if only:
|
|
217
|
+
# A probe file added after preparation must be instrumented too (Begin/End
|
|
218
|
+
# come from instrumentation); already-instrumented files are skipped.
|
|
219
|
+
tool = ws.root / "dg-instrument"
|
|
220
|
+
subprocess.run(
|
|
221
|
+
[
|
|
222
|
+
str(tool), "-root", str(ws.repo), "-module", self.module,
|
|
223
|
+
"-src", os.path.dirname(only[0]) or ".", "-tests", ",".join(self.mock_dirs),
|
|
224
|
+
],
|
|
225
|
+
capture_output=True, text=True, check=True,
|
|
226
|
+
) # fmt: skip
|
|
227
|
+
# a probe file lives in the package dir; run that package, only the probe tests
|
|
228
|
+
pkg = "./" + os.path.dirname(only[0]) + "/"
|
|
229
|
+
argv += ["-run", "DiffgenomeProbe", pkg]
|
|
230
|
+
else:
|
|
231
|
+
argv += [self.tests]
|
|
232
|
+
result = ws.run(argv, env=env, timeout=1800, allow_loopback=True)
|
|
233
|
+
out_dir.mkdir(parents=True, exist_ok=True)
|
|
234
|
+
copy_probe_traces(traces_ws, out_dir)
|
|
235
|
+
executions = [execution_from_json(f.read_text()) for f in sorted(out_dir.glob("*.json"))]
|
|
236
|
+
return executions, result.stdout, result.stderr
|
|
237
|
+
|
|
238
|
+
def probe_relpath(self, tag: str) -> str:
|
|
239
|
+
return f"{self.test_root}/diffgenome_probe_{tag}_test.go"
|
|
240
|
+
|
|
241
|
+
def conventions(self, index: SymbolIndex) -> str:
|
|
242
|
+
return (
|
|
243
|
+
f"Runtime: Go ({self._go_version()}), the standard `go test` runner with testify "
|
|
244
|
+
"(`require`) and gomock available. The probe is a file named "
|
|
245
|
+
f"diffgenome_probe_*_test.go placed in the package directory `{self.test_root}/` "
|
|
246
|
+
f"(package `{self._package_name()}`), so it can use unexported identifiers of that "
|
|
247
|
+
"package directly; test functions MUST be named TestDiffgenomeProbe... so the "
|
|
248
|
+
f"harness can select them. Module path: {self.module}. No network and no database "
|
|
249
|
+
"are available; substitute the Store with the generated gomock "
|
|
250
|
+
"(`mockdb.NewMockStore(ctrl)`) as the existing tests do. Only modules already in "
|
|
251
|
+
"go.mod can be imported (GOPROXY=off): the standard library, this module, testify, "
|
|
252
|
+
"gomock, pgx/v5 and gin; do not add new dependencies. Sqlc `Queries` take a `DBTX` "
|
|
253
|
+
"interface (pgx), so a hand-written fake DBTX returning a fake `pgx.Row` can drive "
|
|
254
|
+
"real query code without a database."
|
|
255
|
+
)
|
|
256
|
+
|
|
257
|
+
def _package_name(self) -> str:
|
|
258
|
+
pkg = self.repo / self.test_root
|
|
259
|
+
for f in sorted(pkg.glob("*.go")):
|
|
260
|
+
for line in f.read_text(encoding="utf-8").splitlines():
|
|
261
|
+
if line.startswith("package "):
|
|
262
|
+
return line.split()[1]
|
|
263
|
+
return os.path.basename(self.test_root)
|
|
264
|
+
|
|
265
|
+
def _go_version(self) -> str:
|
|
266
|
+
try:
|
|
267
|
+
return subprocess.run(
|
|
268
|
+
[str(self.go), "version"], capture_output=True, text=True, timeout=10
|
|
269
|
+
).stdout.strip()
|
|
270
|
+
except (OSError, subprocess.TimeoutExpired):
|
|
271
|
+
return "?"
|