diffgenome 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- diffgenome/__init__.py +7 -0
- diffgenome/__main__.py +240 -0
- diffgenome/_collectors/go/dg/dg.go +623 -0
- diffgenome/_collectors/go/go.mod +3 -0
- diffgenome/_collectors/go/instrument/facts.go +346 -0
- diffgenome/_collectors/go/instrument/main.go +484 -0
- diffgenome/_collectors/node/instrument.js +289 -0
- diffgenome/_collectors/node/jest-setup.js +40 -0
- diffgenome/_collectors/node/package-lock.json +35 -0
- diffgenome/_collectors/node/package.json +11 -0
- diffgenome/_collectors/node/runtime.js +426 -0
- diffgenome/ambiguity.py +122 -0
- diffgenome/api.py +67 -0
- diffgenome/change.py +86 -0
- diffgenome/change_artifact.py +310 -0
- diffgenome/collect/__init__.py +2 -0
- diffgenome/collect/go_test.py +271 -0
- diffgenome/collect/node_jest.py +319 -0
- diffgenome/collect/py_monitoring.py +985 -0
- diffgenome/collect/py_runtime.py +116 -0
- diffgenome/collect/py_symbols.py +238 -0
- diffgenome/collect/pytest_plugin.py +130 -0
- diffgenome/compose.py +469 -0
- diffgenome/dependence.py +264 -0
- diffgenome/evaluate.py +669 -0
- diffgenome/frontends/__init__.py +0 -0
- diffgenome/frontends/python_ir.py +335 -0
- diffgenome/genome.py +1016 -0
- diffgenome/genome_pipeline.py +674 -0
- diffgenome/genome_prompt.py +33 -0
- diffgenome/genome_state.py +2118 -0
- diffgenome/graph.py +426 -0
- diffgenome/llm.py +189 -0
- diffgenome/model.py +364 -0
- diffgenome/mvp.py +398 -0
- diffgenome/probe.py +509 -0
- diffgenome/projection.py +308 -0
- diffgenome/py.typed +0 -0
- diffgenome/render.py +118 -0
- diffgenome/report.py +363 -0
- diffgenome/resolve.py +37 -0
- diffgenome/runtime.py +74 -0
- diffgenome/runtime_evidence.py +261 -0
- diffgenome/sandbox.py +166 -0
- diffgenome/serialize.py +96 -0
- diffgenome/sites.py +19 -0
- diffgenome/static_types.py +69 -0
- diffgenome/structure.py +462 -0
- diffgenome-0.1.0.dist-info/METADATA +139 -0
- diffgenome-0.1.0.dist-info/RECORD +53 -0
- diffgenome-0.1.0.dist-info/WHEEL +4 -0
- diffgenome-0.1.0.dist-info/entry_points.txt +2 -0
- diffgenome-0.1.0.dist-info/licenses/LICENSE +202 -0
diffgenome/evaluate.py
ADDED
|
@@ -0,0 +1,669 @@
|
|
|
1
|
+
"""Ground-truth evaluation: how right is a reconstructed graph when a whole execution exists?
|
|
2
|
+
|
|
3
|
+
Ground truth is a set of complete executions (e.g. in-process end-to-end tests) that the
|
|
4
|
+
reconstruction was NOT allowed to consume. Both sides are reduced to in-repo caller→callee
|
|
5
|
+
edges with callee outcomes; the reconstruction is the graph's downstream neighborhood of
|
|
6
|
+
the entry symbols. Precision asks whether claimed edges really happen; recall asks how much
|
|
7
|
+
of what happens was claimed. Provenance explains claims; this module judges them.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
from collections import Counter, defaultdict
|
|
13
|
+
from dataclasses import dataclass, field
|
|
14
|
+
from itertools import pairwise
|
|
15
|
+
|
|
16
|
+
from diffgenome.graph import BehavioralGraph, GraphEdge
|
|
17
|
+
from diffgenome.model import (
|
|
18
|
+
CallNode,
|
|
19
|
+
EvidenceKind,
|
|
20
|
+
Execution,
|
|
21
|
+
JoinStrength,
|
|
22
|
+
NodeRef,
|
|
23
|
+
Origin,
|
|
24
|
+
SymbolId,
|
|
25
|
+
)
|
|
26
|
+
|
|
27
|
+
Edge = tuple[SymbolId, SymbolId]
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
@dataclass
|
|
31
|
+
class GroundTruth:
|
|
32
|
+
edges: set[Edge]
|
|
33
|
+
outcomes: dict[Edge, set[str]] # callee outcome kinds ("returned"/"raised") seen per edge
|
|
34
|
+
symbols: set[SymbolId]
|
|
35
|
+
executions: int
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def ground_truth_from(
|
|
39
|
+
executions: list[Execution], entries: list[SymbolId], origins: dict[SymbolId, Origin]
|
|
40
|
+
) -> GroundTruth:
|
|
41
|
+
"""In-repo edges reachable from the entry symbols inside the given executions."""
|
|
42
|
+
edges: set[Edge] = set()
|
|
43
|
+
outcomes: dict[Edge, set[str]] = defaultdict(set)
|
|
44
|
+
symbols: set[SymbolId] = set()
|
|
45
|
+
|
|
46
|
+
def under_entry(n: CallNode, by_id: dict[int, object]) -> bool:
|
|
47
|
+
cur = n
|
|
48
|
+
while cur.parent is not None:
|
|
49
|
+
p = by_id[cur.parent]
|
|
50
|
+
if not isinstance(p, CallNode):
|
|
51
|
+
return False
|
|
52
|
+
if p.symbol in entries:
|
|
53
|
+
return True
|
|
54
|
+
cur = p
|
|
55
|
+
return False
|
|
56
|
+
|
|
57
|
+
for ex in executions:
|
|
58
|
+
by_id: dict[int, object] = {n.id: n for n in ex.nodes}
|
|
59
|
+
for n in ex.nodes:
|
|
60
|
+
if not isinstance(n, CallNode) or n.parent is None:
|
|
61
|
+
continue
|
|
62
|
+
p = by_id[n.parent]
|
|
63
|
+
if not isinstance(p, CallNode):
|
|
64
|
+
continue
|
|
65
|
+
if origins.get(p.symbol) is not Origin.REPO or origins.get(n.symbol) is not Origin.REPO:
|
|
66
|
+
continue
|
|
67
|
+
if p.symbol in entries or under_entry(p, by_id):
|
|
68
|
+
e = (p.symbol, n.symbol)
|
|
69
|
+
edges.add(e)
|
|
70
|
+
outcomes[e].add(n.outcome.split(":")[0])
|
|
71
|
+
symbols.update(e)
|
|
72
|
+
return GroundTruth(edges, dict(outcomes), symbols, len(executions))
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
@dataclass
|
|
76
|
+
class Evaluation:
|
|
77
|
+
entries: list[SymbolId]
|
|
78
|
+
gt_edges: int
|
|
79
|
+
claimed_edges: int
|
|
80
|
+
true_positive: int
|
|
81
|
+
false_positive: int # claimed, not in ground truth
|
|
82
|
+
false_negative: int # in ground truth, not claimed
|
|
83
|
+
precision: float
|
|
84
|
+
recall: float
|
|
85
|
+
by_kind: dict[str, dict[str, int]] # kind -> {claimed, true}
|
|
86
|
+
by_join: dict[str, dict[str, int]] # join -> {claimed, true}
|
|
87
|
+
probe_derived: dict[str, int] # {claimed, true}
|
|
88
|
+
false_composed: list[Edge]
|
|
89
|
+
missing: list[Edge]
|
|
90
|
+
missing_explained: dict[str, list[Edge]] # "gap"/"unresolved"/"absent" -> edges
|
|
91
|
+
outcome_conflicts: list[tuple[Edge, str, str]] # (edge, reconstructed outcomes, gt outcomes)
|
|
92
|
+
alternates_total: int
|
|
93
|
+
notes: list[str] = field(default_factory=list)
|
|
94
|
+
|
|
95
|
+
def as_dict(self) -> dict[str, object]:
|
|
96
|
+
return {
|
|
97
|
+
"entries": self.entries,
|
|
98
|
+
"ground_truth_edges": self.gt_edges,
|
|
99
|
+
"claimed_edges": self.claimed_edges,
|
|
100
|
+
"true_positive": self.true_positive,
|
|
101
|
+
"false_positive": self.false_positive,
|
|
102
|
+
"false_negative": self.false_negative,
|
|
103
|
+
"precision": round(self.precision, 3),
|
|
104
|
+
"recall": round(self.recall, 3),
|
|
105
|
+
"by_evidence_kind": self.by_kind,
|
|
106
|
+
"by_join_strength": self.by_join,
|
|
107
|
+
"probe_derived": self.probe_derived,
|
|
108
|
+
"false_composed_continuations": [list(e) for e in self.false_composed],
|
|
109
|
+
"missing_continuations": [list(e) for e in self.missing],
|
|
110
|
+
"missing_explained": {
|
|
111
|
+
k: [list(e) for e in v] for k, v in self.missing_explained.items()
|
|
112
|
+
},
|
|
113
|
+
"outcome_conflicts": [[list(e), r, g] for e, r, g in self.outcome_conflicts],
|
|
114
|
+
"alternate_fragments_total": self.alternates_total,
|
|
115
|
+
"notes": self.notes,
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def _reconstructed(
|
|
120
|
+
graph: BehavioralGraph, entries: list[SymbolId], depth: int
|
|
121
|
+
) -> dict[Edge, GraphEdge]:
|
|
122
|
+
"""Downstream neighborhood restricted to in-repo symbol pairs, best edge per pair
|
|
123
|
+
(OBSERVED wins over COMPOSED)."""
|
|
124
|
+
nb = graph.neighborhood(entries, up=0, down=depth)
|
|
125
|
+
best: dict[Edge, GraphEdge] = {}
|
|
126
|
+
for _, e in nb.edges.values():
|
|
127
|
+
if e.kind not in (EvidenceKind.OBSERVED, EvidenceKind.COMPOSED):
|
|
128
|
+
continue
|
|
129
|
+
if graph.origin(e.caller) is not Origin.REPO or graph.origin(e.callee) is not Origin.REPO:
|
|
130
|
+
continue
|
|
131
|
+
key = (e.caller, e.callee)
|
|
132
|
+
if key not in best or (
|
|
133
|
+
e.kind is EvidenceKind.OBSERVED and best[key].kind is not EvidenceKind.OBSERVED
|
|
134
|
+
):
|
|
135
|
+
best[key] = e
|
|
136
|
+
return best
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def evaluate(
|
|
140
|
+
graph: BehavioralGraph, gt: GroundTruth, entries: list[SymbolId], depth: int = 6
|
|
141
|
+
) -> Evaluation:
|
|
142
|
+
claimed = _reconstructed(graph, entries, depth)
|
|
143
|
+
tp = {e for e in claimed if e in gt.edges}
|
|
144
|
+
fp = {e for e in claimed if e not in gt.edges}
|
|
145
|
+
fn = {e for e in gt.edges if e not in claimed}
|
|
146
|
+
by_kind: dict[str, Counter[str]] = defaultdict(Counter)
|
|
147
|
+
by_join: dict[str, Counter[str]] = defaultdict(Counter)
|
|
148
|
+
probe: Counter[str] = Counter()
|
|
149
|
+
for e, ge in claimed.items():
|
|
150
|
+
by_kind[ge.kind.value]["claimed"] += 1
|
|
151
|
+
by_kind[ge.kind.value]["true"] += e in gt.edges
|
|
152
|
+
if ge.kind is EvidenceKind.COMPOSED:
|
|
153
|
+
j = ge.best_join.name if ge.best_join else "NONE"
|
|
154
|
+
by_join[j]["claimed"] += 1
|
|
155
|
+
by_join[j]["true"] += e in gt.edges
|
|
156
|
+
if ge.probe_derived:
|
|
157
|
+
probe["claimed"] += 1
|
|
158
|
+
probe["true"] += e in gt.edges
|
|
159
|
+
# where did the missing edges stop in the reconstruction?
|
|
160
|
+
explained: dict[str, list[Edge]] = {"gap": [], "unresolved": [], "absent": []}
|
|
161
|
+
for e in sorted(fn):
|
|
162
|
+
kinds = {x.kind for x in graph.out.get(e[0], [])}
|
|
163
|
+
if any(
|
|
164
|
+
x.kind is EvidenceKind.INTERNAL_GAP and x.callee == e[1]
|
|
165
|
+
for x in graph.out.get(e[0], [])
|
|
166
|
+
):
|
|
167
|
+
explained["gap"].append(e)
|
|
168
|
+
elif EvidenceKind.UNRESOLVED_BOUNDARY in kinds:
|
|
169
|
+
explained["unresolved"].append(e)
|
|
170
|
+
else:
|
|
171
|
+
explained["absent"].append(e)
|
|
172
|
+
conflicts: list[tuple[Edge, str, str]] = []
|
|
173
|
+
for e in sorted(tp):
|
|
174
|
+
recon = set(graph.outcomes_by_symbol.get(e[1], Counter()).keys())
|
|
175
|
+
truth = gt.outcomes.get(e, set())
|
|
176
|
+
if recon and truth and not (recon & truth):
|
|
177
|
+
conflicts.append((e, ",".join(sorted(recon)), ",".join(sorted(truth))))
|
|
178
|
+
alternates = sum(len(ev.alternates) for ge in claimed.values() for ev in ge.evidence)
|
|
179
|
+
precision = len(tp) / len(claimed) if claimed else 0.0
|
|
180
|
+
recall = len(tp) / len(gt.edges) if gt.edges else 0.0
|
|
181
|
+
return Evaluation(
|
|
182
|
+
entries,
|
|
183
|
+
len(gt.edges),
|
|
184
|
+
len(claimed),
|
|
185
|
+
len(tp),
|
|
186
|
+
len(fp),
|
|
187
|
+
len(fn),
|
|
188
|
+
precision,
|
|
189
|
+
recall,
|
|
190
|
+
{k: dict(v) for k, v in by_kind.items()},
|
|
191
|
+
{k: dict(v) for k, v in by_join.items()},
|
|
192
|
+
dict(probe),
|
|
193
|
+
sorted(e for e in fp if claimed[e].kind is EvidenceKind.COMPOSED),
|
|
194
|
+
sorted(fn),
|
|
195
|
+
explained,
|
|
196
|
+
conflicts,
|
|
197
|
+
alternates,
|
|
198
|
+
)
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
def _row(label: str, v: dict[str, int]) -> str:
|
|
202
|
+
c, t = v.get("claimed", 0), v.get("true", 0)
|
|
203
|
+
return f"| {label} | {c} | {t} | {t / c:.3f} |" if c else f"| {label} | 0 | 0 | - |"
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
@dataclass
|
|
207
|
+
class SeamGrade:
|
|
208
|
+
grade: str
|
|
209
|
+
seams: int
|
|
210
|
+
consistent: int # continuation claims no edge the whole execution did not take
|
|
211
|
+
matching: int # continuation is exactly the whole execution's continuation
|
|
212
|
+
outcome_ok: int
|
|
213
|
+
|
|
214
|
+
|
|
215
|
+
def seam_precision(
|
|
216
|
+
graph: BehavioralGraph, corpus_graph_edges: dict[str, set[Edge]], gt: GroundTruth
|
|
217
|
+
) -> list[SeamGrade]:
|
|
218
|
+
"""Join-lattice test at the seam level. For every accepted join whose target is on a
|
|
219
|
+
ground-truth path: the borrowed fragment's first-level continuation (target → callees
|
|
220
|
+
inside the fragment execution) is compared with the ground truth's continuation from
|
|
221
|
+
that target. `corpus_graph_edges` maps execution id -> in-repo (caller, callee) edges
|
|
222
|
+
observed in that execution."""
|
|
223
|
+
per: dict[str, list[tuple[bool, bool, bool]]] = defaultdict(list)
|
|
224
|
+
gt_out: dict[SymbolId, set[SymbolId]] = defaultdict(set)
|
|
225
|
+
for a, b in gt.edges:
|
|
226
|
+
gt_out[a].add(b)
|
|
227
|
+
gt_targets = {b for _, b in gt.edges} | {a for a, _ in gt.edges}
|
|
228
|
+
for att in graph.attempts:
|
|
229
|
+
if not att.accepted or att.grade is None or att.target not in gt_targets:
|
|
230
|
+
continue
|
|
231
|
+
frag_edges = {
|
|
232
|
+
e for e in corpus_graph_edges.get(att.fragment.execution, set()) if e[0] == att.target
|
|
233
|
+
}
|
|
234
|
+
claimed_next = {b for _, b in frag_edges}
|
|
235
|
+
# Continuations the fragment reaches through its own seams count too: composed
|
|
236
|
+
# edges out of the target whose evidence site lies in this fragment's execution.
|
|
237
|
+
for ge in graph.out.get(att.target, []):
|
|
238
|
+
if ge.kind is EvidenceKind.COMPOSED and any(
|
|
239
|
+
ev.site.execution == att.fragment.execution for ev in ge.evidence
|
|
240
|
+
):
|
|
241
|
+
claimed_next.add(ge.callee)
|
|
242
|
+
truth_next = gt_out.get(att.target, set())
|
|
243
|
+
consistent = claimed_next <= truth_next
|
|
244
|
+
matching = claimed_next == truth_next
|
|
245
|
+
frag_outcome = graph.outcomes_by_symbol.get(att.target, Counter())
|
|
246
|
+
truth_outcomes = (
|
|
247
|
+
set().union(
|
|
248
|
+
*(
|
|
249
|
+
gt.outcomes.get((a, att.target), set())
|
|
250
|
+
for a in gt_out
|
|
251
|
+
if att.target in gt_out[a]
|
|
252
|
+
)
|
|
253
|
+
)
|
|
254
|
+
if any(att.target in v for v in gt_out.values())
|
|
255
|
+
else set()
|
|
256
|
+
)
|
|
257
|
+
outcome_ok = (not truth_outcomes) or bool(set(frag_outcome) & truth_outcomes)
|
|
258
|
+
per[att.grade.name].append((consistent, matching, outcome_ok))
|
|
259
|
+
out = []
|
|
260
|
+
for grade in ("SYMBOL", "ARG_SHAPE", "VALUE", "STATE"):
|
|
261
|
+
rows = per.get(grade, [])
|
|
262
|
+
if rows:
|
|
263
|
+
out.append(
|
|
264
|
+
SeamGrade(
|
|
265
|
+
grade,
|
|
266
|
+
len(rows),
|
|
267
|
+
sum(r[0] for r in rows),
|
|
268
|
+
sum(r[1] for r in rows),
|
|
269
|
+
sum(r[2] for r in rows),
|
|
270
|
+
)
|
|
271
|
+
)
|
|
272
|
+
return out
|
|
273
|
+
|
|
274
|
+
|
|
275
|
+
def render_seams(grades: list[SeamGrade]) -> str:
|
|
276
|
+
lines = ["| join grade | seams | consistent | matching | outcome ok |", "|---|---|---|---|---|"]
|
|
277
|
+
for g in grades:
|
|
278
|
+
lines.append(
|
|
279
|
+
f"| {g.grade} | {g.seams} | {g.consistent} ({g.consistent / g.seams:.2f}) "
|
|
280
|
+
f"| {g.matching} ({g.matching / g.seams:.2f}) | {g.outcome_ok} |"
|
|
281
|
+
)
|
|
282
|
+
return "\n".join(lines) + "\n"
|
|
283
|
+
|
|
284
|
+
|
|
285
|
+
def render_evaluation(ev: Evaluation) -> str:
|
|
286
|
+
lines = [
|
|
287
|
+
"# Ground-truth evaluation",
|
|
288
|
+
f"entries: {', '.join(s.split(':', 1)[1] for s in ev.entries)}",
|
|
289
|
+
f"ground-truth in-repo edges: {ev.gt_edges} claimed: {ev.claimed_edges} "
|
|
290
|
+
f"true: {ev.true_positive} false: {ev.false_positive} missing: {ev.false_negative}",
|
|
291
|
+
f"edge precision: {ev.precision:.3f} edge recall: {ev.recall:.3f}",
|
|
292
|
+
"",
|
|
293
|
+
"| evidence | claimed | true | precision |",
|
|
294
|
+
"|---|---|---|---|",
|
|
295
|
+
]
|
|
296
|
+
for k, v in sorted(ev.by_kind.items()):
|
|
297
|
+
lines.append(_row(k, v))
|
|
298
|
+
if ev.by_join:
|
|
299
|
+
lines += [
|
|
300
|
+
"",
|
|
301
|
+
"| join strength (composed) | claimed | true | precision |",
|
|
302
|
+
"|---|---|---|---|",
|
|
303
|
+
]
|
|
304
|
+
for k, v in sorted(
|
|
305
|
+
ev.by_join.items(),
|
|
306
|
+
key=lambda kv: JoinStrength[kv[0]].value if kv[0] in JoinStrength.__members__ else 0,
|
|
307
|
+
):
|
|
308
|
+
lines.append(_row(k, v))
|
|
309
|
+
if ev.probe_derived:
|
|
310
|
+
c, t = ev.probe_derived.get("claimed", 0), ev.probe_derived.get("true", 0)
|
|
311
|
+
lines += [
|
|
312
|
+
"",
|
|
313
|
+
f"probe-derived edges: claimed {c}, true {t}, precision {t / c:.3f}" if c else "",
|
|
314
|
+
]
|
|
315
|
+
if ev.false_composed:
|
|
316
|
+
lines += ["", "false composed continuations:"] + [
|
|
317
|
+
f" {a.split(':', 1)[1]} ⇢ {b.split(':', 1)[1]}" for a, b in ev.false_composed
|
|
318
|
+
]
|
|
319
|
+
if ev.missing:
|
|
320
|
+
lines += ["", "missing continuations (ground truth not claimed):"]
|
|
321
|
+
for k, es in ev.missing_explained.items():
|
|
322
|
+
for a, b in es:
|
|
323
|
+
lines.append(f" [{k}] {a.split(':', 1)[1]} → {b.split(':', 1)[1]}")
|
|
324
|
+
if ev.outcome_conflicts:
|
|
325
|
+
lines += ["", "outcome conflicts (reconstructed vs ground truth):"] + [
|
|
326
|
+
f" {a.split(':', 1)[1]} → {b.split(':', 1)[1]}: {r} vs {g}"
|
|
327
|
+
for (a, b), r, g in ev.outcome_conflicts
|
|
328
|
+
]
|
|
329
|
+
lines += ["", f"alternate fragments carried on claimed edges: {ev.alternates_total}"]
|
|
330
|
+
lines += [f"note: {n}" for n in ev.notes]
|
|
331
|
+
return "\n".join(lines) + "\n"
|
|
332
|
+
|
|
333
|
+
|
|
334
|
+
# --------------------------------------------------------------------------- path level
|
|
335
|
+
|
|
336
|
+
|
|
337
|
+
def execution_paths(
|
|
338
|
+
executions: list[Execution], entry: SymbolId, origins: dict[SymbolId, Origin]
|
|
339
|
+
) -> set[tuple[str, ...]]:
|
|
340
|
+
"""Pre-order sequences of in-repo (symbol, outcome-kind) under `entry` in whole executions.
|
|
341
|
+
One sequence per occurrence of the entry."""
|
|
342
|
+
paths: set[tuple[str, ...]] = set()
|
|
343
|
+
for ex in executions:
|
|
344
|
+
kids: dict[int, list[CallNode]] = defaultdict(list)
|
|
345
|
+
for n in ex.nodes:
|
|
346
|
+
if isinstance(n, CallNode) and n.parent is not None:
|
|
347
|
+
kids[n.parent].append(n)
|
|
348
|
+
for n in ex.nodes:
|
|
349
|
+
if isinstance(n, CallNode) and n.symbol == entry:
|
|
350
|
+
seq: list[str] = []
|
|
351
|
+
_walk_preorder(n, kids, origins, seq)
|
|
352
|
+
paths.add(tuple(seq))
|
|
353
|
+
return paths
|
|
354
|
+
|
|
355
|
+
|
|
356
|
+
def _walk_preorder(
|
|
357
|
+
node: CallNode, kids: dict[int, list[CallNode]], origins: dict[SymbolId, Origin], seq: list[str]
|
|
358
|
+
) -> None:
|
|
359
|
+
if origins.get(node.symbol) is Origin.REPO:
|
|
360
|
+
seq.append(f"{node.symbol}:{node.outcome.split(':')[0]}")
|
|
361
|
+
for k in kids.get(node.id, []):
|
|
362
|
+
_walk_preorder(k, kids, origins, seq)
|
|
363
|
+
|
|
364
|
+
|
|
365
|
+
def composed_paths(graph: BehavioralGraph, entry: SymbolId, depth: int = 8) -> set[tuple[str, ...]]:
|
|
366
|
+
"""Pre-order sequences the reconstruction implies from `entry`: one per combination of
|
|
367
|
+
composed alternatives, following the composition trees of every execution that ran the
|
|
368
|
+
entry (observed children in order, composed seams expanded to each accepted fragment)."""
|
|
369
|
+
corpus = graph.corpus
|
|
370
|
+
if corpus is None:
|
|
371
|
+
return set()
|
|
372
|
+
from diffgenome.compose import Branch, compose
|
|
373
|
+
from diffgenome.model import EvidenceKind as EK
|
|
374
|
+
|
|
375
|
+
out: set[tuple[str, ...]] = set()
|
|
376
|
+
for ex_id in graph.tests_by_symbol.get(entry, ()):
|
|
377
|
+
comp = compose(corpus, ex_id, use_state=graph.use_state)
|
|
378
|
+
|
|
379
|
+
def expand(branches: list[Branch], d: int) -> list[list[str]]:
|
|
380
|
+
# siblings concatenate; alternatives for one caller→callee pair multiply
|
|
381
|
+
seqs: list[list[str]] = [[]]
|
|
382
|
+
groups: dict[tuple[str, str, int], list[Branch]] = defaultdict(list)
|
|
383
|
+
order: list[tuple[str, str, int]] = []
|
|
384
|
+
for i, b in enumerate(branches):
|
|
385
|
+
# composed alternatives for one seam share a site; observed siblings are
|
|
386
|
+
# distinct calls even when they have the same callee
|
|
387
|
+
ev = b.edge.evidence
|
|
388
|
+
key = (b.edge.caller, b.edge.callee, ev.site.node if ev.kind is EK.COMPOSED else -i)
|
|
389
|
+
if key not in groups:
|
|
390
|
+
order.append(key)
|
|
391
|
+
groups[key].append(b)
|
|
392
|
+
for key in order:
|
|
393
|
+
alts = groups[key]
|
|
394
|
+
alt_seqs: list[list[str]] = []
|
|
395
|
+
for b in alts:
|
|
396
|
+
kind = b.edge.evidence.kind
|
|
397
|
+
if (
|
|
398
|
+
kind in (EK.OBSERVED, EK.COMPOSED)
|
|
399
|
+
and graph.origin(b.edge.callee) is Origin.REPO
|
|
400
|
+
):
|
|
401
|
+
outcome = "?"
|
|
402
|
+
ref = b.edge.evidence.fragment or b.edge.evidence.site
|
|
403
|
+
node = corpus.executions[ref.execution].nodes[ref.node]
|
|
404
|
+
if isinstance(node, CallNode):
|
|
405
|
+
outcome = node.outcome.split(":")[0]
|
|
406
|
+
head = [f"{b.edge.callee}:{outcome}"]
|
|
407
|
+
tails = expand(b.children, d + 1) if d < depth else [[]]
|
|
408
|
+
alt_seqs.extend(head + t for t in tails)
|
|
409
|
+
else:
|
|
410
|
+
alt_seqs.append([])
|
|
411
|
+
# dedupe alternatives
|
|
412
|
+
uniq = []
|
|
413
|
+
for a in alt_seqs:
|
|
414
|
+
if a not in uniq:
|
|
415
|
+
uniq.append(a)
|
|
416
|
+
seqs = [s_ + a for s_ in seqs for a in uniq]
|
|
417
|
+
return seqs
|
|
418
|
+
|
|
419
|
+
# find the entry branch(es) in the seed tree
|
|
420
|
+
stack = list(comp.branches)
|
|
421
|
+
while stack:
|
|
422
|
+
b = stack.pop()
|
|
423
|
+
if b.edge.callee == entry:
|
|
424
|
+
site = b.edge.evidence.site
|
|
425
|
+
node = corpus.executions[site.execution].nodes[site.node]
|
|
426
|
+
outcome = node.outcome.split(":")[0] if isinstance(node, CallNode) else "?"
|
|
427
|
+
for tail in expand(b.children, 1):
|
|
428
|
+
out.add(tuple([f"{entry}:{outcome}", *tail]))
|
|
429
|
+
stack.extend(b.children)
|
|
430
|
+
return out
|
|
431
|
+
|
|
432
|
+
|
|
433
|
+
@dataclass
|
|
434
|
+
class PathEvaluation:
|
|
435
|
+
entry: SymbolId
|
|
436
|
+
truth_paths: int
|
|
437
|
+
claimed_paths: int
|
|
438
|
+
matched: int # claimed sequences that equal a ground-truth sequence
|
|
439
|
+
extra: list[tuple[str, ...]] # claimed, not in truth (wrong composition or untaken alternative)
|
|
440
|
+
missed: list[tuple[str, ...]] # truth, not claimed
|
|
441
|
+
outcome_mismatches: (
|
|
442
|
+
int # extra sequences whose symbol order matches a truth sequence but outcomes differ
|
|
443
|
+
)
|
|
444
|
+
|
|
445
|
+
|
|
446
|
+
def evaluate_paths(
|
|
447
|
+
graph: BehavioralGraph,
|
|
448
|
+
gt_runs: list[Execution],
|
|
449
|
+
entry: SymbolId,
|
|
450
|
+
origins: dict[SymbolId, Origin],
|
|
451
|
+
) -> PathEvaluation:
|
|
452
|
+
truth = execution_paths(gt_runs, entry, origins)
|
|
453
|
+
claimed = composed_paths(graph, entry)
|
|
454
|
+
matched = {p for p in claimed if p in truth}
|
|
455
|
+
extra = sorted(p for p in claimed if p not in truth)
|
|
456
|
+
missed = sorted(p for p in truth if p not in claimed)
|
|
457
|
+
order_only = {tuple(x.rsplit(":", 1)[0] for x in p) for p in truth}
|
|
458
|
+
outcome_mm = sum(1 for p in extra if tuple(x.rsplit(":", 1)[0] for x in p) in order_only)
|
|
459
|
+
return PathEvaluation(entry, len(truth), len(claimed), len(matched), extra, missed, outcome_mm)
|
|
460
|
+
|
|
461
|
+
|
|
462
|
+
def path_scores(pe: PathEvaluation) -> tuple[float, float]:
|
|
463
|
+
precision = pe.matched / pe.claimed_paths if pe.claimed_paths else 0.0
|
|
464
|
+
recall = pe.matched / pe.truth_paths if pe.truth_paths else 0.0
|
|
465
|
+
return precision, recall
|
|
466
|
+
|
|
467
|
+
|
|
468
|
+
def render_paths(pe: PathEvaluation) -> str:
|
|
469
|
+
def fmt(p: tuple[str, ...]) -> str:
|
|
470
|
+
return " → ".join(
|
|
471
|
+
x.split(":", 1)[1].split(":")[0].split(".")[-1] + ("!" if x.endswith(":raised") else "")
|
|
472
|
+
for x in p
|
|
473
|
+
)
|
|
474
|
+
|
|
475
|
+
pp, pr = path_scores(pe)
|
|
476
|
+
lines = [
|
|
477
|
+
f"### path-level: {pe.entry.split(':', 1)[1]}",
|
|
478
|
+
f"ground-truth sequences {pe.truth_paths}, claimed {pe.claimed_paths}, "
|
|
479
|
+
f"matched {pe.matched}, extra {len(pe.extra)} (of which outcome-only mismatches "
|
|
480
|
+
f"{pe.outcome_mismatches}), missed {len(pe.missed)}",
|
|
481
|
+
f"path precision {pp:.3f} path recall {pr:.3f}",
|
|
482
|
+
]
|
|
483
|
+
for p in pe.extra[:8]:
|
|
484
|
+
lines.append(f" extra: {fmt(p)}")
|
|
485
|
+
for p in pe.missed[:8]:
|
|
486
|
+
lines.append(f" missed: {fmt(p)}")
|
|
487
|
+
return "\n".join(lines) + "\n"
|
|
488
|
+
|
|
489
|
+
|
|
490
|
+
# --------------------------------------------------------------------------- join matrix
|
|
491
|
+
|
|
492
|
+
|
|
493
|
+
@dataclass
|
|
494
|
+
class MatrixRow:
|
|
495
|
+
caller: SymbolId
|
|
496
|
+
target: SymbolId
|
|
497
|
+
fragment: str
|
|
498
|
+
symbol: str # "✓" reached, "✗" failed at this rung, "·" not evaluated
|
|
499
|
+
arg_shape: str
|
|
500
|
+
value: str
|
|
501
|
+
state: str # ✓ | ✗ (conflict) | n/a (no facts to compare) | · (not evaluated / disabled)
|
|
502
|
+
exit: str # same | kind | unknown | conflict
|
|
503
|
+
truth: bool | None # the fragment's continuation equals a ground-truth execution's
|
|
504
|
+
accepted: bool
|
|
505
|
+
note: str
|
|
506
|
+
|
|
507
|
+
|
|
508
|
+
def join_matrix(
|
|
509
|
+
graph: BehavioralGraph,
|
|
510
|
+
corpus_graph_edges: dict[str, set[Edge]],
|
|
511
|
+
gt: GroundTruth,
|
|
512
|
+
truth_per_exec: list[set[Edge]] | None = None,
|
|
513
|
+
) -> list[MatrixRow]:
|
|
514
|
+
"""One row per (seam, candidate fragment) for seams whose target is on a ground-truth
|
|
515
|
+
path: which rungs the candidate reached, its exit verdict, whether its first-level
|
|
516
|
+
continuation is one some ground-truth execution actually took from that target, and
|
|
517
|
+
whether the composer accepted it. Rung marks are derived from the grade and the
|
|
518
|
+
rejection note, so a rejected candidate still shows how far it got."""
|
|
519
|
+
gt_out: dict[SymbolId, set[SymbolId]] = defaultdict(set)
|
|
520
|
+
for a, b in gt.edges:
|
|
521
|
+
gt_out[a].add(b)
|
|
522
|
+
truth_sets: dict[SymbolId, list[set[SymbolId]]] = defaultdict(list)
|
|
523
|
+
for edges in truth_per_exec or []:
|
|
524
|
+
outs: dict[SymbolId, set[SymbolId]] = defaultdict(set)
|
|
525
|
+
for a, b in edges:
|
|
526
|
+
outs[a].add(b)
|
|
527
|
+
for a, nxt in outs.items():
|
|
528
|
+
truth_sets[a].append(nxt)
|
|
529
|
+
targets = {b for _, b in gt.edges} | {a for a, _ in gt.edges}
|
|
530
|
+
callers: dict[tuple[NodeRef, str], str] = {}
|
|
531
|
+
for e in graph.edges.values():
|
|
532
|
+
for ev in e.evidence:
|
|
533
|
+
callers[(ev.site, e.callee)] = e.caller
|
|
534
|
+
rows: list[MatrixRow] = []
|
|
535
|
+
for att in graph.attempts:
|
|
536
|
+
if att.target not in targets:
|
|
537
|
+
continue
|
|
538
|
+
frag_next = {
|
|
539
|
+
b for a, b in corpus_graph_edges.get(att.fragment.execution, set()) if a == att.target
|
|
540
|
+
}
|
|
541
|
+
truth: bool | None
|
|
542
|
+
if truth_sets.get(att.target):
|
|
543
|
+
truth = frag_next in truth_sets[att.target]
|
|
544
|
+
elif att.target in gt_out:
|
|
545
|
+
truth = frag_next == gt_out[att.target]
|
|
546
|
+
else:
|
|
547
|
+
truth = None
|
|
548
|
+
g = att.grade.value if att.grade else 0
|
|
549
|
+
note = att.note
|
|
550
|
+
exit_ = att.exit if att.exit else "conflict"
|
|
551
|
+
sym = arg = val = state = "·"
|
|
552
|
+
if att.grade is not None:
|
|
553
|
+
sym = "✓"
|
|
554
|
+
arg = "✓" if g >= 2 else "✗"
|
|
555
|
+
val = "✓" if g >= 3 else ("·" if g < 2 else "✗")
|
|
556
|
+
if g >= 4:
|
|
557
|
+
state = "✓"
|
|
558
|
+
elif g == 3:
|
|
559
|
+
state = "n/a" if ("state unavailable" in note or "no common" in note) else "·"
|
|
560
|
+
elif "state conflict" in note:
|
|
561
|
+
sym = arg = val = "✓"
|
|
562
|
+
state = "✗"
|
|
563
|
+
elif "type conflict" in note:
|
|
564
|
+
sym, arg = "✓", "✗"
|
|
565
|
+
elif "exit conflict" in note:
|
|
566
|
+
sym = "✓"
|
|
567
|
+
rows.append(
|
|
568
|
+
MatrixRow(
|
|
569
|
+
callers.get((att.site, att.target), "?"),
|
|
570
|
+
att.target,
|
|
571
|
+
att.fragment.execution.split("::")[-1].split("@")[0],
|
|
572
|
+
sym,
|
|
573
|
+
arg,
|
|
574
|
+
val,
|
|
575
|
+
state,
|
|
576
|
+
exit_,
|
|
577
|
+
truth,
|
|
578
|
+
att.accepted,
|
|
579
|
+
note,
|
|
580
|
+
)
|
|
581
|
+
)
|
|
582
|
+
rows.sort(key=lambda r: (r.caller, r.target, r.fragment))
|
|
583
|
+
return rows
|
|
584
|
+
|
|
585
|
+
|
|
586
|
+
def render_matrix(rows: list[MatrixRow]) -> str:
|
|
587
|
+
def yn(b: bool) -> str:
|
|
588
|
+
return "✓" if b else "✗"
|
|
589
|
+
|
|
590
|
+
lines = [
|
|
591
|
+
"| seam | candidate fragment | SYMBOL | ARG_SHAPE | VALUE | STATE | EXIT "
|
|
592
|
+
"| truth | accepted |",
|
|
593
|
+
"|---|---|---|---|---|---|---|---|---|",
|
|
594
|
+
]
|
|
595
|
+
for r in rows:
|
|
596
|
+
seam = f"{r.caller.split('.')[-1]} ⇢ {r.target.split('.')[-1]}"
|
|
597
|
+
truth = "?" if r.truth is None else yn(r.truth)
|
|
598
|
+
lines.append(
|
|
599
|
+
f"| {seam} | {r.fragment} | {r.symbol} | {r.arg_shape} | {r.value} "
|
|
600
|
+
f"| {r.state} | {r.exit} | {truth} | {yn(r.accepted)} |"
|
|
601
|
+
)
|
|
602
|
+
return "\n".join(lines) + "\n"
|
|
603
|
+
|
|
604
|
+
|
|
605
|
+
def explain_paths(pe: PathEvaluation, graph: BehavioralGraph) -> list[str]:
|
|
606
|
+
"""For each extra path, which seam and join grade admitted it; for each missed path,
|
|
607
|
+
what evidence was absent."""
|
|
608
|
+
out: list[str] = []
|
|
609
|
+
for p in pe.extra:
|
|
610
|
+
# the first symbol whose edge from its predecessor is composed is the admitting seam
|
|
611
|
+
found = False
|
|
612
|
+
for a, b in pairwise(p):
|
|
613
|
+
ca, cb = a.rsplit(":", 1)[0], b.rsplit(":", 1)[0]
|
|
614
|
+
e = graph.edges.get((ca, cb, EvidenceKind.COMPOSED))
|
|
615
|
+
if e is not None:
|
|
616
|
+
grades = sorted({ev.join.name for ev in e.evidence if ev.join})
|
|
617
|
+
out.append(
|
|
618
|
+
f"extra {_fmt_path(p)}: admitted at seam {ca.split('.')[-1]} ⇢ "
|
|
619
|
+
f"{cb.split('.')[-1]} with join {'/'.join(grades)}"
|
|
620
|
+
)
|
|
621
|
+
found = True
|
|
622
|
+
break
|
|
623
|
+
if not found:
|
|
624
|
+
out.append(
|
|
625
|
+
f"extra {_fmt_path(p)}: all edges observed in some execution "
|
|
626
|
+
"(a real path the ground truth did not take)"
|
|
627
|
+
)
|
|
628
|
+
for p in pe.missed:
|
|
629
|
+
syms = {x.rsplit(":", 1)[0] for x in p}
|
|
630
|
+
reasons: list[str] = []
|
|
631
|
+
# a rejected candidate whose target lies on the path: cite the rejection
|
|
632
|
+
seen: set[str] = set()
|
|
633
|
+
for att in graph.attempts:
|
|
634
|
+
if att.accepted or att.target not in syms:
|
|
635
|
+
continue
|
|
636
|
+
if graph.corpus is not None:
|
|
637
|
+
# only candidates that actually contain the missed continuation
|
|
638
|
+
frag_syms = {
|
|
639
|
+
n.symbol
|
|
640
|
+
for n in graph.corpus.executions[att.fragment.execution].nodes
|
|
641
|
+
if isinstance(n, CallNode)
|
|
642
|
+
}
|
|
643
|
+
after = list(p)[[x.rsplit(":", 1)[0] for x in p].index(att.target) + 1 :]
|
|
644
|
+
if not all(x.rsplit(":", 1)[0] in frag_syms for x in after):
|
|
645
|
+
continue
|
|
646
|
+
head = att.note.split(";")[0]
|
|
647
|
+
key = f"{att.target}|{head}"
|
|
648
|
+
if key in seen:
|
|
649
|
+
continue
|
|
650
|
+
seen.add(key)
|
|
651
|
+
reasons.append(
|
|
652
|
+
f"candidate {att.fragment.execution.split('::')[-1].split('@')[0]} at "
|
|
653
|
+
f"⇢ {att.target.split('.')[-1]} rejected: {head}"
|
|
654
|
+
)
|
|
655
|
+
unreached = [
|
|
656
|
+
x.split(".")[-1] for x in syms if x not in graph.inc and x != p[0].rsplit(":", 1)[0]
|
|
657
|
+
]
|
|
658
|
+
if unreached:
|
|
659
|
+
reasons.append("no evidence reaches " + ", ".join(sorted(unreached)))
|
|
660
|
+
if not reasons:
|
|
661
|
+
reasons.append("the seed corpus never observed the entry under this exit path")
|
|
662
|
+
out.append(f"missed {_fmt_path(p)}: " + "; ".join(reasons))
|
|
663
|
+
return out
|
|
664
|
+
|
|
665
|
+
|
|
666
|
+
def _fmt_path(p: tuple[str, ...]) -> str:
|
|
667
|
+
return " → ".join(
|
|
668
|
+
x.rsplit(":", 1)[0].split(".")[-1] + ("!" if x.endswith(":raised") else "") for x in p
|
|
669
|
+
)
|
|
File without changes
|