diffgenome 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (53) hide show
  1. diffgenome/__init__.py +7 -0
  2. diffgenome/__main__.py +240 -0
  3. diffgenome/_collectors/go/dg/dg.go +623 -0
  4. diffgenome/_collectors/go/go.mod +3 -0
  5. diffgenome/_collectors/go/instrument/facts.go +346 -0
  6. diffgenome/_collectors/go/instrument/main.go +484 -0
  7. diffgenome/_collectors/node/instrument.js +289 -0
  8. diffgenome/_collectors/node/jest-setup.js +40 -0
  9. diffgenome/_collectors/node/package-lock.json +35 -0
  10. diffgenome/_collectors/node/package.json +11 -0
  11. diffgenome/_collectors/node/runtime.js +426 -0
  12. diffgenome/ambiguity.py +122 -0
  13. diffgenome/api.py +67 -0
  14. diffgenome/change.py +86 -0
  15. diffgenome/change_artifact.py +310 -0
  16. diffgenome/collect/__init__.py +2 -0
  17. diffgenome/collect/go_test.py +271 -0
  18. diffgenome/collect/node_jest.py +319 -0
  19. diffgenome/collect/py_monitoring.py +985 -0
  20. diffgenome/collect/py_runtime.py +116 -0
  21. diffgenome/collect/py_symbols.py +238 -0
  22. diffgenome/collect/pytest_plugin.py +130 -0
  23. diffgenome/compose.py +469 -0
  24. diffgenome/dependence.py +264 -0
  25. diffgenome/evaluate.py +669 -0
  26. diffgenome/frontends/__init__.py +0 -0
  27. diffgenome/frontends/python_ir.py +335 -0
  28. diffgenome/genome.py +1016 -0
  29. diffgenome/genome_pipeline.py +674 -0
  30. diffgenome/genome_prompt.py +33 -0
  31. diffgenome/genome_state.py +2118 -0
  32. diffgenome/graph.py +426 -0
  33. diffgenome/llm.py +189 -0
  34. diffgenome/model.py +364 -0
  35. diffgenome/mvp.py +398 -0
  36. diffgenome/probe.py +509 -0
  37. diffgenome/projection.py +308 -0
  38. diffgenome/py.typed +0 -0
  39. diffgenome/render.py +118 -0
  40. diffgenome/report.py +363 -0
  41. diffgenome/resolve.py +37 -0
  42. diffgenome/runtime.py +74 -0
  43. diffgenome/runtime_evidence.py +261 -0
  44. diffgenome/sandbox.py +166 -0
  45. diffgenome/serialize.py +96 -0
  46. diffgenome/sites.py +19 -0
  47. diffgenome/static_types.py +69 -0
  48. diffgenome/structure.py +462 -0
  49. diffgenome-0.1.0.dist-info/METADATA +139 -0
  50. diffgenome-0.1.0.dist-info/RECORD +53 -0
  51. diffgenome-0.1.0.dist-info/WHEEL +4 -0
  52. diffgenome-0.1.0.dist-info/entry_points.txt +2 -0
  53. diffgenome-0.1.0.dist-info/licenses/LICENSE +202 -0
diffgenome/evaluate.py ADDED
@@ -0,0 +1,669 @@
1
+ """Ground-truth evaluation: how right is a reconstructed graph when a whole execution exists?
2
+
3
+ Ground truth is a set of complete executions (e.g. in-process end-to-end tests) that the
4
+ reconstruction was NOT allowed to consume. Both sides are reduced to in-repo caller→callee
5
+ edges with callee outcomes; the reconstruction is the graph's downstream neighborhood of
6
+ the entry symbols. Precision asks whether claimed edges really happen; recall asks how much
7
+ of what happens was claimed. Provenance explains claims; this module judges them.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ from collections import Counter, defaultdict
13
+ from dataclasses import dataclass, field
14
+ from itertools import pairwise
15
+
16
+ from diffgenome.graph import BehavioralGraph, GraphEdge
17
+ from diffgenome.model import (
18
+ CallNode,
19
+ EvidenceKind,
20
+ Execution,
21
+ JoinStrength,
22
+ NodeRef,
23
+ Origin,
24
+ SymbolId,
25
+ )
26
+
27
+ Edge = tuple[SymbolId, SymbolId]
28
+
29
+
30
+ @dataclass
31
+ class GroundTruth:
32
+ edges: set[Edge]
33
+ outcomes: dict[Edge, set[str]] # callee outcome kinds ("returned"/"raised") seen per edge
34
+ symbols: set[SymbolId]
35
+ executions: int
36
+
37
+
38
+ def ground_truth_from(
39
+ executions: list[Execution], entries: list[SymbolId], origins: dict[SymbolId, Origin]
40
+ ) -> GroundTruth:
41
+ """In-repo edges reachable from the entry symbols inside the given executions."""
42
+ edges: set[Edge] = set()
43
+ outcomes: dict[Edge, set[str]] = defaultdict(set)
44
+ symbols: set[SymbolId] = set()
45
+
46
+ def under_entry(n: CallNode, by_id: dict[int, object]) -> bool:
47
+ cur = n
48
+ while cur.parent is not None:
49
+ p = by_id[cur.parent]
50
+ if not isinstance(p, CallNode):
51
+ return False
52
+ if p.symbol in entries:
53
+ return True
54
+ cur = p
55
+ return False
56
+
57
+ for ex in executions:
58
+ by_id: dict[int, object] = {n.id: n for n in ex.nodes}
59
+ for n in ex.nodes:
60
+ if not isinstance(n, CallNode) or n.parent is None:
61
+ continue
62
+ p = by_id[n.parent]
63
+ if not isinstance(p, CallNode):
64
+ continue
65
+ if origins.get(p.symbol) is not Origin.REPO or origins.get(n.symbol) is not Origin.REPO:
66
+ continue
67
+ if p.symbol in entries or under_entry(p, by_id):
68
+ e = (p.symbol, n.symbol)
69
+ edges.add(e)
70
+ outcomes[e].add(n.outcome.split(":")[0])
71
+ symbols.update(e)
72
+ return GroundTruth(edges, dict(outcomes), symbols, len(executions))
73
+
74
+
75
+ @dataclass
76
+ class Evaluation:
77
+ entries: list[SymbolId]
78
+ gt_edges: int
79
+ claimed_edges: int
80
+ true_positive: int
81
+ false_positive: int # claimed, not in ground truth
82
+ false_negative: int # in ground truth, not claimed
83
+ precision: float
84
+ recall: float
85
+ by_kind: dict[str, dict[str, int]] # kind -> {claimed, true}
86
+ by_join: dict[str, dict[str, int]] # join -> {claimed, true}
87
+ probe_derived: dict[str, int] # {claimed, true}
88
+ false_composed: list[Edge]
89
+ missing: list[Edge]
90
+ missing_explained: dict[str, list[Edge]] # "gap"/"unresolved"/"absent" -> edges
91
+ outcome_conflicts: list[tuple[Edge, str, str]] # (edge, reconstructed outcomes, gt outcomes)
92
+ alternates_total: int
93
+ notes: list[str] = field(default_factory=list)
94
+
95
+ def as_dict(self) -> dict[str, object]:
96
+ return {
97
+ "entries": self.entries,
98
+ "ground_truth_edges": self.gt_edges,
99
+ "claimed_edges": self.claimed_edges,
100
+ "true_positive": self.true_positive,
101
+ "false_positive": self.false_positive,
102
+ "false_negative": self.false_negative,
103
+ "precision": round(self.precision, 3),
104
+ "recall": round(self.recall, 3),
105
+ "by_evidence_kind": self.by_kind,
106
+ "by_join_strength": self.by_join,
107
+ "probe_derived": self.probe_derived,
108
+ "false_composed_continuations": [list(e) for e in self.false_composed],
109
+ "missing_continuations": [list(e) for e in self.missing],
110
+ "missing_explained": {
111
+ k: [list(e) for e in v] for k, v in self.missing_explained.items()
112
+ },
113
+ "outcome_conflicts": [[list(e), r, g] for e, r, g in self.outcome_conflicts],
114
+ "alternate_fragments_total": self.alternates_total,
115
+ "notes": self.notes,
116
+ }
117
+
118
+
119
+ def _reconstructed(
120
+ graph: BehavioralGraph, entries: list[SymbolId], depth: int
121
+ ) -> dict[Edge, GraphEdge]:
122
+ """Downstream neighborhood restricted to in-repo symbol pairs, best edge per pair
123
+ (OBSERVED wins over COMPOSED)."""
124
+ nb = graph.neighborhood(entries, up=0, down=depth)
125
+ best: dict[Edge, GraphEdge] = {}
126
+ for _, e in nb.edges.values():
127
+ if e.kind not in (EvidenceKind.OBSERVED, EvidenceKind.COMPOSED):
128
+ continue
129
+ if graph.origin(e.caller) is not Origin.REPO or graph.origin(e.callee) is not Origin.REPO:
130
+ continue
131
+ key = (e.caller, e.callee)
132
+ if key not in best or (
133
+ e.kind is EvidenceKind.OBSERVED and best[key].kind is not EvidenceKind.OBSERVED
134
+ ):
135
+ best[key] = e
136
+ return best
137
+
138
+
139
+ def evaluate(
140
+ graph: BehavioralGraph, gt: GroundTruth, entries: list[SymbolId], depth: int = 6
141
+ ) -> Evaluation:
142
+ claimed = _reconstructed(graph, entries, depth)
143
+ tp = {e for e in claimed if e in gt.edges}
144
+ fp = {e for e in claimed if e not in gt.edges}
145
+ fn = {e for e in gt.edges if e not in claimed}
146
+ by_kind: dict[str, Counter[str]] = defaultdict(Counter)
147
+ by_join: dict[str, Counter[str]] = defaultdict(Counter)
148
+ probe: Counter[str] = Counter()
149
+ for e, ge in claimed.items():
150
+ by_kind[ge.kind.value]["claimed"] += 1
151
+ by_kind[ge.kind.value]["true"] += e in gt.edges
152
+ if ge.kind is EvidenceKind.COMPOSED:
153
+ j = ge.best_join.name if ge.best_join else "NONE"
154
+ by_join[j]["claimed"] += 1
155
+ by_join[j]["true"] += e in gt.edges
156
+ if ge.probe_derived:
157
+ probe["claimed"] += 1
158
+ probe["true"] += e in gt.edges
159
+ # where did the missing edges stop in the reconstruction?
160
+ explained: dict[str, list[Edge]] = {"gap": [], "unresolved": [], "absent": []}
161
+ for e in sorted(fn):
162
+ kinds = {x.kind for x in graph.out.get(e[0], [])}
163
+ if any(
164
+ x.kind is EvidenceKind.INTERNAL_GAP and x.callee == e[1]
165
+ for x in graph.out.get(e[0], [])
166
+ ):
167
+ explained["gap"].append(e)
168
+ elif EvidenceKind.UNRESOLVED_BOUNDARY in kinds:
169
+ explained["unresolved"].append(e)
170
+ else:
171
+ explained["absent"].append(e)
172
+ conflicts: list[tuple[Edge, str, str]] = []
173
+ for e in sorted(tp):
174
+ recon = set(graph.outcomes_by_symbol.get(e[1], Counter()).keys())
175
+ truth = gt.outcomes.get(e, set())
176
+ if recon and truth and not (recon & truth):
177
+ conflicts.append((e, ",".join(sorted(recon)), ",".join(sorted(truth))))
178
+ alternates = sum(len(ev.alternates) for ge in claimed.values() for ev in ge.evidence)
179
+ precision = len(tp) / len(claimed) if claimed else 0.0
180
+ recall = len(tp) / len(gt.edges) if gt.edges else 0.0
181
+ return Evaluation(
182
+ entries,
183
+ len(gt.edges),
184
+ len(claimed),
185
+ len(tp),
186
+ len(fp),
187
+ len(fn),
188
+ precision,
189
+ recall,
190
+ {k: dict(v) for k, v in by_kind.items()},
191
+ {k: dict(v) for k, v in by_join.items()},
192
+ dict(probe),
193
+ sorted(e for e in fp if claimed[e].kind is EvidenceKind.COMPOSED),
194
+ sorted(fn),
195
+ explained,
196
+ conflicts,
197
+ alternates,
198
+ )
199
+
200
+
201
+ def _row(label: str, v: dict[str, int]) -> str:
202
+ c, t = v.get("claimed", 0), v.get("true", 0)
203
+ return f"| {label} | {c} | {t} | {t / c:.3f} |" if c else f"| {label} | 0 | 0 | - |"
204
+
205
+
206
+ @dataclass
207
+ class SeamGrade:
208
+ grade: str
209
+ seams: int
210
+ consistent: int # continuation claims no edge the whole execution did not take
211
+ matching: int # continuation is exactly the whole execution's continuation
212
+ outcome_ok: int
213
+
214
+
215
+ def seam_precision(
216
+ graph: BehavioralGraph, corpus_graph_edges: dict[str, set[Edge]], gt: GroundTruth
217
+ ) -> list[SeamGrade]:
218
+ """Join-lattice test at the seam level. For every accepted join whose target is on a
219
+ ground-truth path: the borrowed fragment's first-level continuation (target → callees
220
+ inside the fragment execution) is compared with the ground truth's continuation from
221
+ that target. `corpus_graph_edges` maps execution id -> in-repo (caller, callee) edges
222
+ observed in that execution."""
223
+ per: dict[str, list[tuple[bool, bool, bool]]] = defaultdict(list)
224
+ gt_out: dict[SymbolId, set[SymbolId]] = defaultdict(set)
225
+ for a, b in gt.edges:
226
+ gt_out[a].add(b)
227
+ gt_targets = {b for _, b in gt.edges} | {a for a, _ in gt.edges}
228
+ for att in graph.attempts:
229
+ if not att.accepted or att.grade is None or att.target not in gt_targets:
230
+ continue
231
+ frag_edges = {
232
+ e for e in corpus_graph_edges.get(att.fragment.execution, set()) if e[0] == att.target
233
+ }
234
+ claimed_next = {b for _, b in frag_edges}
235
+ # Continuations the fragment reaches through its own seams count too: composed
236
+ # edges out of the target whose evidence site lies in this fragment's execution.
237
+ for ge in graph.out.get(att.target, []):
238
+ if ge.kind is EvidenceKind.COMPOSED and any(
239
+ ev.site.execution == att.fragment.execution for ev in ge.evidence
240
+ ):
241
+ claimed_next.add(ge.callee)
242
+ truth_next = gt_out.get(att.target, set())
243
+ consistent = claimed_next <= truth_next
244
+ matching = claimed_next == truth_next
245
+ frag_outcome = graph.outcomes_by_symbol.get(att.target, Counter())
246
+ truth_outcomes = (
247
+ set().union(
248
+ *(
249
+ gt.outcomes.get((a, att.target), set())
250
+ for a in gt_out
251
+ if att.target in gt_out[a]
252
+ )
253
+ )
254
+ if any(att.target in v for v in gt_out.values())
255
+ else set()
256
+ )
257
+ outcome_ok = (not truth_outcomes) or bool(set(frag_outcome) & truth_outcomes)
258
+ per[att.grade.name].append((consistent, matching, outcome_ok))
259
+ out = []
260
+ for grade in ("SYMBOL", "ARG_SHAPE", "VALUE", "STATE"):
261
+ rows = per.get(grade, [])
262
+ if rows:
263
+ out.append(
264
+ SeamGrade(
265
+ grade,
266
+ len(rows),
267
+ sum(r[0] for r in rows),
268
+ sum(r[1] for r in rows),
269
+ sum(r[2] for r in rows),
270
+ )
271
+ )
272
+ return out
273
+
274
+
275
+ def render_seams(grades: list[SeamGrade]) -> str:
276
+ lines = ["| join grade | seams | consistent | matching | outcome ok |", "|---|---|---|---|---|"]
277
+ for g in grades:
278
+ lines.append(
279
+ f"| {g.grade} | {g.seams} | {g.consistent} ({g.consistent / g.seams:.2f}) "
280
+ f"| {g.matching} ({g.matching / g.seams:.2f}) | {g.outcome_ok} |"
281
+ )
282
+ return "\n".join(lines) + "\n"
283
+
284
+
285
+ def render_evaluation(ev: Evaluation) -> str:
286
+ lines = [
287
+ "# Ground-truth evaluation",
288
+ f"entries: {', '.join(s.split(':', 1)[1] for s in ev.entries)}",
289
+ f"ground-truth in-repo edges: {ev.gt_edges} claimed: {ev.claimed_edges} "
290
+ f"true: {ev.true_positive} false: {ev.false_positive} missing: {ev.false_negative}",
291
+ f"edge precision: {ev.precision:.3f} edge recall: {ev.recall:.3f}",
292
+ "",
293
+ "| evidence | claimed | true | precision |",
294
+ "|---|---|---|---|",
295
+ ]
296
+ for k, v in sorted(ev.by_kind.items()):
297
+ lines.append(_row(k, v))
298
+ if ev.by_join:
299
+ lines += [
300
+ "",
301
+ "| join strength (composed) | claimed | true | precision |",
302
+ "|---|---|---|---|",
303
+ ]
304
+ for k, v in sorted(
305
+ ev.by_join.items(),
306
+ key=lambda kv: JoinStrength[kv[0]].value if kv[0] in JoinStrength.__members__ else 0,
307
+ ):
308
+ lines.append(_row(k, v))
309
+ if ev.probe_derived:
310
+ c, t = ev.probe_derived.get("claimed", 0), ev.probe_derived.get("true", 0)
311
+ lines += [
312
+ "",
313
+ f"probe-derived edges: claimed {c}, true {t}, precision {t / c:.3f}" if c else "",
314
+ ]
315
+ if ev.false_composed:
316
+ lines += ["", "false composed continuations:"] + [
317
+ f" {a.split(':', 1)[1]} ⇢ {b.split(':', 1)[1]}" for a, b in ev.false_composed
318
+ ]
319
+ if ev.missing:
320
+ lines += ["", "missing continuations (ground truth not claimed):"]
321
+ for k, es in ev.missing_explained.items():
322
+ for a, b in es:
323
+ lines.append(f" [{k}] {a.split(':', 1)[1]} → {b.split(':', 1)[1]}")
324
+ if ev.outcome_conflicts:
325
+ lines += ["", "outcome conflicts (reconstructed vs ground truth):"] + [
326
+ f" {a.split(':', 1)[1]} → {b.split(':', 1)[1]}: {r} vs {g}"
327
+ for (a, b), r, g in ev.outcome_conflicts
328
+ ]
329
+ lines += ["", f"alternate fragments carried on claimed edges: {ev.alternates_total}"]
330
+ lines += [f"note: {n}" for n in ev.notes]
331
+ return "\n".join(lines) + "\n"
332
+
333
+
334
+ # --------------------------------------------------------------------------- path level
335
+
336
+
337
+ def execution_paths(
338
+ executions: list[Execution], entry: SymbolId, origins: dict[SymbolId, Origin]
339
+ ) -> set[tuple[str, ...]]:
340
+ """Pre-order sequences of in-repo (symbol, outcome-kind) under `entry` in whole executions.
341
+ One sequence per occurrence of the entry."""
342
+ paths: set[tuple[str, ...]] = set()
343
+ for ex in executions:
344
+ kids: dict[int, list[CallNode]] = defaultdict(list)
345
+ for n in ex.nodes:
346
+ if isinstance(n, CallNode) and n.parent is not None:
347
+ kids[n.parent].append(n)
348
+ for n in ex.nodes:
349
+ if isinstance(n, CallNode) and n.symbol == entry:
350
+ seq: list[str] = []
351
+ _walk_preorder(n, kids, origins, seq)
352
+ paths.add(tuple(seq))
353
+ return paths
354
+
355
+
356
+ def _walk_preorder(
357
+ node: CallNode, kids: dict[int, list[CallNode]], origins: dict[SymbolId, Origin], seq: list[str]
358
+ ) -> None:
359
+ if origins.get(node.symbol) is Origin.REPO:
360
+ seq.append(f"{node.symbol}:{node.outcome.split(':')[0]}")
361
+ for k in kids.get(node.id, []):
362
+ _walk_preorder(k, kids, origins, seq)
363
+
364
+
365
+ def composed_paths(graph: BehavioralGraph, entry: SymbolId, depth: int = 8) -> set[tuple[str, ...]]:
366
+ """Pre-order sequences the reconstruction implies from `entry`: one per combination of
367
+ composed alternatives, following the composition trees of every execution that ran the
368
+ entry (observed children in order, composed seams expanded to each accepted fragment)."""
369
+ corpus = graph.corpus
370
+ if corpus is None:
371
+ return set()
372
+ from diffgenome.compose import Branch, compose
373
+ from diffgenome.model import EvidenceKind as EK
374
+
375
+ out: set[tuple[str, ...]] = set()
376
+ for ex_id in graph.tests_by_symbol.get(entry, ()):
377
+ comp = compose(corpus, ex_id, use_state=graph.use_state)
378
+
379
+ def expand(branches: list[Branch], d: int) -> list[list[str]]:
380
+ # siblings concatenate; alternatives for one caller→callee pair multiply
381
+ seqs: list[list[str]] = [[]]
382
+ groups: dict[tuple[str, str, int], list[Branch]] = defaultdict(list)
383
+ order: list[tuple[str, str, int]] = []
384
+ for i, b in enumerate(branches):
385
+ # composed alternatives for one seam share a site; observed siblings are
386
+ # distinct calls even when they have the same callee
387
+ ev = b.edge.evidence
388
+ key = (b.edge.caller, b.edge.callee, ev.site.node if ev.kind is EK.COMPOSED else -i)
389
+ if key not in groups:
390
+ order.append(key)
391
+ groups[key].append(b)
392
+ for key in order:
393
+ alts = groups[key]
394
+ alt_seqs: list[list[str]] = []
395
+ for b in alts:
396
+ kind = b.edge.evidence.kind
397
+ if (
398
+ kind in (EK.OBSERVED, EK.COMPOSED)
399
+ and graph.origin(b.edge.callee) is Origin.REPO
400
+ ):
401
+ outcome = "?"
402
+ ref = b.edge.evidence.fragment or b.edge.evidence.site
403
+ node = corpus.executions[ref.execution].nodes[ref.node]
404
+ if isinstance(node, CallNode):
405
+ outcome = node.outcome.split(":")[0]
406
+ head = [f"{b.edge.callee}:{outcome}"]
407
+ tails = expand(b.children, d + 1) if d < depth else [[]]
408
+ alt_seqs.extend(head + t for t in tails)
409
+ else:
410
+ alt_seqs.append([])
411
+ # dedupe alternatives
412
+ uniq = []
413
+ for a in alt_seqs:
414
+ if a not in uniq:
415
+ uniq.append(a)
416
+ seqs = [s_ + a for s_ in seqs for a in uniq]
417
+ return seqs
418
+
419
+ # find the entry branch(es) in the seed tree
420
+ stack = list(comp.branches)
421
+ while stack:
422
+ b = stack.pop()
423
+ if b.edge.callee == entry:
424
+ site = b.edge.evidence.site
425
+ node = corpus.executions[site.execution].nodes[site.node]
426
+ outcome = node.outcome.split(":")[0] if isinstance(node, CallNode) else "?"
427
+ for tail in expand(b.children, 1):
428
+ out.add(tuple([f"{entry}:{outcome}", *tail]))
429
+ stack.extend(b.children)
430
+ return out
431
+
432
+
433
+ @dataclass
434
+ class PathEvaluation:
435
+ entry: SymbolId
436
+ truth_paths: int
437
+ claimed_paths: int
438
+ matched: int # claimed sequences that equal a ground-truth sequence
439
+ extra: list[tuple[str, ...]] # claimed, not in truth (wrong composition or untaken alternative)
440
+ missed: list[tuple[str, ...]] # truth, not claimed
441
+ outcome_mismatches: (
442
+ int # extra sequences whose symbol order matches a truth sequence but outcomes differ
443
+ )
444
+
445
+
446
+ def evaluate_paths(
447
+ graph: BehavioralGraph,
448
+ gt_runs: list[Execution],
449
+ entry: SymbolId,
450
+ origins: dict[SymbolId, Origin],
451
+ ) -> PathEvaluation:
452
+ truth = execution_paths(gt_runs, entry, origins)
453
+ claimed = composed_paths(graph, entry)
454
+ matched = {p for p in claimed if p in truth}
455
+ extra = sorted(p for p in claimed if p not in truth)
456
+ missed = sorted(p for p in truth if p not in claimed)
457
+ order_only = {tuple(x.rsplit(":", 1)[0] for x in p) for p in truth}
458
+ outcome_mm = sum(1 for p in extra if tuple(x.rsplit(":", 1)[0] for x in p) in order_only)
459
+ return PathEvaluation(entry, len(truth), len(claimed), len(matched), extra, missed, outcome_mm)
460
+
461
+
462
+ def path_scores(pe: PathEvaluation) -> tuple[float, float]:
463
+ precision = pe.matched / pe.claimed_paths if pe.claimed_paths else 0.0
464
+ recall = pe.matched / pe.truth_paths if pe.truth_paths else 0.0
465
+ return precision, recall
466
+
467
+
468
+ def render_paths(pe: PathEvaluation) -> str:
469
+ def fmt(p: tuple[str, ...]) -> str:
470
+ return " → ".join(
471
+ x.split(":", 1)[1].split(":")[0].split(".")[-1] + ("!" if x.endswith(":raised") else "")
472
+ for x in p
473
+ )
474
+
475
+ pp, pr = path_scores(pe)
476
+ lines = [
477
+ f"### path-level: {pe.entry.split(':', 1)[1]}",
478
+ f"ground-truth sequences {pe.truth_paths}, claimed {pe.claimed_paths}, "
479
+ f"matched {pe.matched}, extra {len(pe.extra)} (of which outcome-only mismatches "
480
+ f"{pe.outcome_mismatches}), missed {len(pe.missed)}",
481
+ f"path precision {pp:.3f} path recall {pr:.3f}",
482
+ ]
483
+ for p in pe.extra[:8]:
484
+ lines.append(f" extra: {fmt(p)}")
485
+ for p in pe.missed[:8]:
486
+ lines.append(f" missed: {fmt(p)}")
487
+ return "\n".join(lines) + "\n"
488
+
489
+
490
+ # --------------------------------------------------------------------------- join matrix
491
+
492
+
493
+ @dataclass
494
+ class MatrixRow:
495
+ caller: SymbolId
496
+ target: SymbolId
497
+ fragment: str
498
+ symbol: str # "✓" reached, "✗" failed at this rung, "·" not evaluated
499
+ arg_shape: str
500
+ value: str
501
+ state: str # ✓ | ✗ (conflict) | n/a (no facts to compare) | · (not evaluated / disabled)
502
+ exit: str # same | kind | unknown | conflict
503
+ truth: bool | None # the fragment's continuation equals a ground-truth execution's
504
+ accepted: bool
505
+ note: str
506
+
507
+
508
+ def join_matrix(
509
+ graph: BehavioralGraph,
510
+ corpus_graph_edges: dict[str, set[Edge]],
511
+ gt: GroundTruth,
512
+ truth_per_exec: list[set[Edge]] | None = None,
513
+ ) -> list[MatrixRow]:
514
+ """One row per (seam, candidate fragment) for seams whose target is on a ground-truth
515
+ path: which rungs the candidate reached, its exit verdict, whether its first-level
516
+ continuation is one some ground-truth execution actually took from that target, and
517
+ whether the composer accepted it. Rung marks are derived from the grade and the
518
+ rejection note, so a rejected candidate still shows how far it got."""
519
+ gt_out: dict[SymbolId, set[SymbolId]] = defaultdict(set)
520
+ for a, b in gt.edges:
521
+ gt_out[a].add(b)
522
+ truth_sets: dict[SymbolId, list[set[SymbolId]]] = defaultdict(list)
523
+ for edges in truth_per_exec or []:
524
+ outs: dict[SymbolId, set[SymbolId]] = defaultdict(set)
525
+ for a, b in edges:
526
+ outs[a].add(b)
527
+ for a, nxt in outs.items():
528
+ truth_sets[a].append(nxt)
529
+ targets = {b for _, b in gt.edges} | {a for a, _ in gt.edges}
530
+ callers: dict[tuple[NodeRef, str], str] = {}
531
+ for e in graph.edges.values():
532
+ for ev in e.evidence:
533
+ callers[(ev.site, e.callee)] = e.caller
534
+ rows: list[MatrixRow] = []
535
+ for att in graph.attempts:
536
+ if att.target not in targets:
537
+ continue
538
+ frag_next = {
539
+ b for a, b in corpus_graph_edges.get(att.fragment.execution, set()) if a == att.target
540
+ }
541
+ truth: bool | None
542
+ if truth_sets.get(att.target):
543
+ truth = frag_next in truth_sets[att.target]
544
+ elif att.target in gt_out:
545
+ truth = frag_next == gt_out[att.target]
546
+ else:
547
+ truth = None
548
+ g = att.grade.value if att.grade else 0
549
+ note = att.note
550
+ exit_ = att.exit if att.exit else "conflict"
551
+ sym = arg = val = state = "·"
552
+ if att.grade is not None:
553
+ sym = "✓"
554
+ arg = "✓" if g >= 2 else "✗"
555
+ val = "✓" if g >= 3 else ("·" if g < 2 else "✗")
556
+ if g >= 4:
557
+ state = "✓"
558
+ elif g == 3:
559
+ state = "n/a" if ("state unavailable" in note or "no common" in note) else "·"
560
+ elif "state conflict" in note:
561
+ sym = arg = val = "✓"
562
+ state = "✗"
563
+ elif "type conflict" in note:
564
+ sym, arg = "✓", "✗"
565
+ elif "exit conflict" in note:
566
+ sym = "✓"
567
+ rows.append(
568
+ MatrixRow(
569
+ callers.get((att.site, att.target), "?"),
570
+ att.target,
571
+ att.fragment.execution.split("::")[-1].split("@")[0],
572
+ sym,
573
+ arg,
574
+ val,
575
+ state,
576
+ exit_,
577
+ truth,
578
+ att.accepted,
579
+ note,
580
+ )
581
+ )
582
+ rows.sort(key=lambda r: (r.caller, r.target, r.fragment))
583
+ return rows
584
+
585
+
586
+ def render_matrix(rows: list[MatrixRow]) -> str:
587
+ def yn(b: bool) -> str:
588
+ return "✓" if b else "✗"
589
+
590
+ lines = [
591
+ "| seam | candidate fragment | SYMBOL | ARG_SHAPE | VALUE | STATE | EXIT "
592
+ "| truth | accepted |",
593
+ "|---|---|---|---|---|---|---|---|---|",
594
+ ]
595
+ for r in rows:
596
+ seam = f"{r.caller.split('.')[-1]} ⇢ {r.target.split('.')[-1]}"
597
+ truth = "?" if r.truth is None else yn(r.truth)
598
+ lines.append(
599
+ f"| {seam} | {r.fragment} | {r.symbol} | {r.arg_shape} | {r.value} "
600
+ f"| {r.state} | {r.exit} | {truth} | {yn(r.accepted)} |"
601
+ )
602
+ return "\n".join(lines) + "\n"
603
+
604
+
605
+ def explain_paths(pe: PathEvaluation, graph: BehavioralGraph) -> list[str]:
606
+ """For each extra path, which seam and join grade admitted it; for each missed path,
607
+ what evidence was absent."""
608
+ out: list[str] = []
609
+ for p in pe.extra:
610
+ # the first symbol whose edge from its predecessor is composed is the admitting seam
611
+ found = False
612
+ for a, b in pairwise(p):
613
+ ca, cb = a.rsplit(":", 1)[0], b.rsplit(":", 1)[0]
614
+ e = graph.edges.get((ca, cb, EvidenceKind.COMPOSED))
615
+ if e is not None:
616
+ grades = sorted({ev.join.name for ev in e.evidence if ev.join})
617
+ out.append(
618
+ f"extra {_fmt_path(p)}: admitted at seam {ca.split('.')[-1]} ⇢ "
619
+ f"{cb.split('.')[-1]} with join {'/'.join(grades)}"
620
+ )
621
+ found = True
622
+ break
623
+ if not found:
624
+ out.append(
625
+ f"extra {_fmt_path(p)}: all edges observed in some execution "
626
+ "(a real path the ground truth did not take)"
627
+ )
628
+ for p in pe.missed:
629
+ syms = {x.rsplit(":", 1)[0] for x in p}
630
+ reasons: list[str] = []
631
+ # a rejected candidate whose target lies on the path: cite the rejection
632
+ seen: set[str] = set()
633
+ for att in graph.attempts:
634
+ if att.accepted or att.target not in syms:
635
+ continue
636
+ if graph.corpus is not None:
637
+ # only candidates that actually contain the missed continuation
638
+ frag_syms = {
639
+ n.symbol
640
+ for n in graph.corpus.executions[att.fragment.execution].nodes
641
+ if isinstance(n, CallNode)
642
+ }
643
+ after = list(p)[[x.rsplit(":", 1)[0] for x in p].index(att.target) + 1 :]
644
+ if not all(x.rsplit(":", 1)[0] in frag_syms for x in after):
645
+ continue
646
+ head = att.note.split(";")[0]
647
+ key = f"{att.target}|{head}"
648
+ if key in seen:
649
+ continue
650
+ seen.add(key)
651
+ reasons.append(
652
+ f"candidate {att.fragment.execution.split('::')[-1].split('@')[0]} at "
653
+ f"⇢ {att.target.split('.')[-1]} rejected: {head}"
654
+ )
655
+ unreached = [
656
+ x.split(".")[-1] for x in syms if x not in graph.inc and x != p[0].rsplit(":", 1)[0]
657
+ ]
658
+ if unreached:
659
+ reasons.append("no evidence reaches " + ", ".join(sorted(unreached)))
660
+ if not reasons:
661
+ reasons.append("the seed corpus never observed the entry under this exit path")
662
+ out.append(f"missed {_fmt_path(p)}: " + "; ".join(reasons))
663
+ return out
664
+
665
+
666
+ def _fmt_path(p: tuple[str, ...]) -> str:
667
+ return " → ".join(
668
+ x.rsplit(":", 1)[0].split(".")[-1] + ("!" if x.endswith(":raised") else "") for x in p
669
+ )
File without changes