diffgenome 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (53) hide show
  1. diffgenome/__init__.py +7 -0
  2. diffgenome/__main__.py +240 -0
  3. diffgenome/_collectors/go/dg/dg.go +623 -0
  4. diffgenome/_collectors/go/go.mod +3 -0
  5. diffgenome/_collectors/go/instrument/facts.go +346 -0
  6. diffgenome/_collectors/go/instrument/main.go +484 -0
  7. diffgenome/_collectors/node/instrument.js +289 -0
  8. diffgenome/_collectors/node/jest-setup.js +40 -0
  9. diffgenome/_collectors/node/package-lock.json +35 -0
  10. diffgenome/_collectors/node/package.json +11 -0
  11. diffgenome/_collectors/node/runtime.js +426 -0
  12. diffgenome/ambiguity.py +122 -0
  13. diffgenome/api.py +67 -0
  14. diffgenome/change.py +86 -0
  15. diffgenome/change_artifact.py +310 -0
  16. diffgenome/collect/__init__.py +2 -0
  17. diffgenome/collect/go_test.py +271 -0
  18. diffgenome/collect/node_jest.py +319 -0
  19. diffgenome/collect/py_monitoring.py +985 -0
  20. diffgenome/collect/py_runtime.py +116 -0
  21. diffgenome/collect/py_symbols.py +238 -0
  22. diffgenome/collect/pytest_plugin.py +130 -0
  23. diffgenome/compose.py +469 -0
  24. diffgenome/dependence.py +264 -0
  25. diffgenome/evaluate.py +669 -0
  26. diffgenome/frontends/__init__.py +0 -0
  27. diffgenome/frontends/python_ir.py +335 -0
  28. diffgenome/genome.py +1016 -0
  29. diffgenome/genome_pipeline.py +674 -0
  30. diffgenome/genome_prompt.py +33 -0
  31. diffgenome/genome_state.py +2118 -0
  32. diffgenome/graph.py +426 -0
  33. diffgenome/llm.py +189 -0
  34. diffgenome/model.py +364 -0
  35. diffgenome/mvp.py +398 -0
  36. diffgenome/probe.py +509 -0
  37. diffgenome/projection.py +308 -0
  38. diffgenome/py.typed +0 -0
  39. diffgenome/render.py +118 -0
  40. diffgenome/report.py +363 -0
  41. diffgenome/resolve.py +37 -0
  42. diffgenome/runtime.py +74 -0
  43. diffgenome/runtime_evidence.py +261 -0
  44. diffgenome/sandbox.py +166 -0
  45. diffgenome/serialize.py +96 -0
  46. diffgenome/sites.py +19 -0
  47. diffgenome/static_types.py +69 -0
  48. diffgenome/structure.py +462 -0
  49. diffgenome-0.1.0.dist-info/METADATA +139 -0
  50. diffgenome-0.1.0.dist-info/RECORD +53 -0
  51. diffgenome-0.1.0.dist-info/WHEEL +4 -0
  52. diffgenome-0.1.0.dist-info/entry_points.txt +2 -0
  53. diffgenome-0.1.0.dist-info/licenses/LICENSE +202 -0
@@ -0,0 +1,674 @@
1
+ # ruff: noqa: E501
2
+ """`diffgenome genome`: the Behavioral Genome of a change, as a product step.
3
+
4
+ Input is the output directory of `diffgenome change` (its artifact, mechanics and existing-test
5
+ traces). Four steps, each generic (no framework knowledge):
6
+
7
+ 1. bundle: the proposer's instructions (the v5 schema), required sites, the observed execution
8
+ structure, mechanics of the vocabulary's functions, the tests' event logs, the change's
9
+ diff and the sources;
10
+ 2. propose: a model writes the semantic layer (recorded proposals, or OpenAI live);
11
+ 3. establish: the checker assigns statuses from ALL the traces, and every scenario is predicted
12
+ and compared with what executed (consistency, not a held-out score);
13
+ 4. summarize: only checked claims (verified, or supported and never contradicted) go to the
14
+ artifact's `genome` section. Hypotheses and rejected items are counted, never exported.
15
+ """
16
+
17
+ from __future__ import annotations
18
+
19
+ import hashlib
20
+ import json
21
+ import os
22
+ import subprocess
23
+ import time
24
+ from pathlib import Path
25
+ from typing import Any
26
+
27
+ from diffgenome.genome import (
28
+ SUPPORTED,
29
+ VERIFIED,
30
+ Genome,
31
+ Item,
32
+ Substrate,
33
+ bindings_of,
34
+ dumps,
35
+ genome_from_proposals,
36
+ render_markdown,
37
+ )
38
+ from diffgenome.genome_prompt import FORMAT, HEAD
39
+ from diffgenome.genome_state import (
40
+ Mechanics,
41
+ Observations,
42
+ StateSubstrate,
43
+ change_sites,
44
+ compare_sequence,
45
+ establish_state,
46
+ genome_vocabulary,
47
+ predict_sequence,
48
+ )
49
+ from diffgenome.model import CallNode, Execution, SubstitutionNode
50
+ from diffgenome.serialize import execution_from_json
51
+ from diffgenome.structure import build_skeleton, canonical, render
52
+
53
+ SUMMARY_FORMAT = "diffgenome-genome-summary/1"
54
+ MAX_VOCAB = 24
55
+ MAX_TESTS = 40
56
+ MAX_LOG_LINES = 120
57
+ MAX_EXTRA_FILES = 8
58
+ MAX_FILE_LINES = 400
59
+ WINDOW = 12
60
+ _SOURCE_EXT = (".go", ".py", ".js", ".ts", ".rs")
61
+
62
+
63
+ def _is_test_path(f: str) -> bool:
64
+ name = f.rsplit("/", 1)[-1]
65
+ return (
66
+ name.endswith(("_test.go", ".test.ts", ".test.js"))
67
+ or name.startswith("test_")
68
+ or "/tests/" in f
69
+ )
70
+
71
+
72
+ def _short(sym: str) -> str:
73
+ return sym.split(":", 1)[-1]
74
+
75
+
76
+ class Run:
77
+ """A `diffgenome change` output directory."""
78
+
79
+ def __init__(self, run: Path) -> None:
80
+ self.dir = run
81
+ self.artifact: dict[str, Any] = json.loads((run / "diffgenome-change.json").read_text())
82
+ self.mech_fns: list[dict[str, Any]] = json.loads((run / "mechanics.json").read_text())[
83
+ "functions"
84
+ ]
85
+ self.mech = Mechanics(self.mech_fns)
86
+ self.executions: dict[str, Execution] = {
87
+ e.stimulus_ref: e
88
+ for e in (
89
+ execution_from_json(f.read_text())
90
+ for f in sorted((run / "traces-existing").glob("*.json"))
91
+ )
92
+ }
93
+ ch = self.artifact["change"]
94
+ self.changed: list[str] = list(ch["symbols"])
95
+ # executed = seen in a trace (the artifact's list may come from another run)
96
+ seen = {
97
+ n.symbol for e in self.executions.values() for n in e.nodes if isinstance(n, CallNode)
98
+ }
99
+ self.executed_changed = [s for s in self.changed if s in seen]
100
+ repo = self.artifact["repository"]
101
+ self.root = Path(repo["root"])
102
+ self.revision: str = repo.get("revision") or "HEAD"
103
+ self.runtime: str = repo.get("runtime", "")
104
+ self.spec: str = ch.get("spec", "")
105
+
106
+ def git(self, *args: str) -> str:
107
+ return subprocess.run(
108
+ ["git", *args], cwd=self.root, capture_output=True, text=True, check=False
109
+ ).stdout
110
+
111
+
112
+ def vocabulary(run: Run) -> list[str]:
113
+ """Changed functions that executed, functions nested in them, and their direct callers
114
+ and callees among the repository's functions (by observed call nesting)."""
115
+ mech_syms = {f["symbol"] for f in run.mech_fns}
116
+ changed = set(run.executed_changed)
117
+ vocab = [s for s in run.executed_changed if s in mech_syms]
118
+ nested: set[str] = set()
119
+ near: dict[str, int] = {}
120
+ for ex in run.executions.values():
121
+ by_id = {n.id: n for n in ex.nodes}
122
+ for n in ex.nodes:
123
+ if not isinstance(n, CallNode):
124
+ continue
125
+ base = n.symbol.split(".<anon>")[0]
126
+ if base in changed and n.symbol not in changed:
127
+ nested.add(n.symbol)
128
+ parent = by_id.get(n.parent) if n.parent is not None else None
129
+ psym: str = getattr(parent, "symbol", "") or ""
130
+ if n.symbol in changed and psym in mech_syms and psym not in changed:
131
+ near[psym] = near.get(psym, 0) + 1
132
+ if psym in changed and n.symbol in mech_syms and n.symbol not in changed:
133
+ near[n.symbol] = near.get(n.symbol, 0) + 1
134
+ vocab += sorted(nested - set(vocab))
135
+ vocab += [s for s, _ in sorted(near.items(), key=lambda kv: (-kv[1], kv[0])) if s not in vocab]
136
+ return vocab[:MAX_VOCAB]
137
+
138
+
139
+ def relevant_tests(run: Run) -> dict[str, Execution]:
140
+ """Executions that call a changed function (or a function nested in one)."""
141
+ changed = set(run.executed_changed)
142
+ return {
143
+ ref: ex
144
+ for ref, ex in sorted(run.executions.items())
145
+ if any(
146
+ isinstance(n, CallNode)
147
+ and (n.symbol in changed or n.symbol.split(".<anon>")[0] in changed)
148
+ for n in ex.nodes
149
+ )
150
+ }
151
+
152
+
153
+ def shown_tests(tests: dict[str, Execution], required: set[str]) -> dict[str, Execution]:
154
+ """At most MAX_TESTS for the bundle: one test per distinct observed behavior (the
155
+ outcome vector of the required sites and the changed calls' exits) first, then the rest."""
156
+
157
+ def signature(ex: Execution) -> tuple[Any, ...]:
158
+ return (
159
+ tuple((b.site, b.outcome) for b in ex.branches if b.site in required),
160
+ tuple(sorted({n.outcome for n in ex.nodes if isinstance(n, CallNode)})),
161
+ )
162
+
163
+ first: dict[tuple[Any, ...], str] = {}
164
+ for ref, ex in tests.items():
165
+ first.setdefault(signature(ex), ref)
166
+ picked = list(first.values())
167
+ picked += [r for r in tests if r not in picked]
168
+ keep = set(picked[:MAX_TESTS])
169
+ return {r: ex for r, ex in tests.items() if r in keep}
170
+
171
+
172
+ def event_log(ex: Execution, vocab: list[str], preds: dict[str, str]) -> list[str]:
173
+ nodes = ex.nodes
174
+ name = {
175
+ n.id: canonical(vocab, getattr(n, "symbol", None) or getattr(n, "claimed_target", "") or "")
176
+ for n in nodes
177
+ }
178
+
179
+ def depth(i: int) -> int:
180
+ d, p = 0, nodes[i].parent
181
+ while p is not None:
182
+ d += name.get(p) is not None
183
+ p = nodes[p].parent
184
+ return d
185
+
186
+ events: list[tuple[int, int, Any]] = [(n.id, 1, n) for n in nodes if name.get(n.id)]
187
+ events += [(b.seq, 0, b) for b in ex.branches if name.get(b.node)]
188
+ out = []
189
+ for _, kind, o in sorted(events, key=lambda e: (e[0], e[1])):
190
+ if kind == 0:
191
+ out.append(
192
+ f"{' ' * (depth(o.node) + 1)}branch {o.site} `{preds.get(o.site, '?')[:70]}` = {'T' if o.outcome else 'F'}"
193
+ )
194
+ else:
195
+ args = ", ".join(
196
+ f"{a}={t}" + (f"#{d[:6]}" if d else "") for a, t, d in getattr(o, "args", ())
197
+ )
198
+ res = f" -> #{o.result[:6]}" if getattr(o, "result", "") else ""
199
+ tag = " (stand-in, mocked)" if isinstance(o, SubstitutionNode) else ""
200
+ out.append(
201
+ f"{' ' * depth(o.id)}call {_short(name[o.id] or '')}({args}) [{getattr(o, 'outcome', '?')}]{res}{tag}"
202
+ )
203
+ if len(out) > MAX_LOG_LINES:
204
+ out = [*out[:MAX_LOG_LINES], f"... ({len(out) - MAX_LOG_LINES} more events)"]
205
+ return out
206
+
207
+
208
+ def build_bundle(run: Run) -> tuple[str, list[str], dict[str, Execution]]:
209
+ vocab = vocabulary(run)
210
+ everything = relevant_tests(run)
211
+ required = change_sites(run.mech, Observations(list(everything.values())), run.changed)
212
+ tests = shown_tests(everything, required)
213
+ sk = build_skeleton(list(tests.values()), vocab)
214
+ preds = {s: rec["pred"] for s, rec in run.mech.sites.items()}
215
+ go = run.runtime.startswith("go")
216
+ null_note = "The digest `#5da3a4` is Go's `nil`." if go else ""
217
+ parts = [
218
+ HEAD.format(
219
+ spec=run.spec or "(unspecified)",
220
+ repo=run.root.name,
221
+ runtime=run.runtime,
222
+ symbols=", ".join(_short(s) for s in run.changed),
223
+ null_note=null_note,
224
+ )
225
+ + FORMAT.replace('"lang": "go"', f'"lang": "{"go" if go else "python"}"')
226
+ ]
227
+ seen = {b.site for ex in tests.values() for b in ex.branches}
228
+ parts.append(
229
+ "## Required sites\n\nEvery OBSERVED evaluation of these sites is scored, in every test.\n\n"
230
+ + "\n".join(
231
+ f"- {s} {run.mech.sites[s]['symbol']} line {run.mech.sites[s]['line']}: `{run.mech.sites[s]['pred']}`"
232
+ + ("" if s in seen else " (not evaluated by any test)")
233
+ for s in sorted(
234
+ (s for s in required if s in run.mech.sites),
235
+ key=lambda s: (run.mech.sites[s]["symbol"], run.mech.sites[s]["line"]),
236
+ )
237
+ )
238
+ + "\n"
239
+ )
240
+ runtime_only: dict[str, set[str]] = {}
241
+ for ex in tests.values():
242
+ for b in ex.branches:
243
+ if b.site in required and b.site not in run.mech.sites:
244
+ runtime_only.setdefault(b.site, set()).add(getattr(ex.nodes[b.node], "symbol", "?"))
245
+ if runtime_only:
246
+ parts.append(
247
+ "Runtime-only required sites (observed; no static facts, e.g. in closures): "
248
+ + "; ".join(f"{s} in {', '.join(sorted(v))}" for s, v in sorted(runtime_only.items()))
249
+ + "\n"
250
+ )
251
+ parts.append(
252
+ "## Observed execution structure (deterministic)\n\n```\n" + render(sk) + "\n```\n"
253
+ )
254
+ lines = []
255
+ files: list[str] = []
256
+ for f in run.mech_fns:
257
+ if canonical(vocab, f["symbol"]) is None:
258
+ continue
259
+ if f["file"] not in files:
260
+ files.append(f["file"])
261
+ lines.append(f"### {_short(f['symbol'])} ({f['file']})")
262
+ for s in f["sites"]:
263
+ ops = "; ".join(f"{o['path']} ← {', '.join(o['origins'])}" for o in s["operands"])
264
+ req = " & ".join(
265
+ f"{r[0]}={'T' if r[1] else 'F'}" if r[1] is not None else r[0]
266
+ for r in s["requires"]
267
+ )
268
+ lines.append(
269
+ f"- site {s['site']} line {s['line']}: `{s['pred']}` | operands: {ops} | requires: {req or '—'} | then exits: {s['then_exits']}, else exits: {s['else_exits']}"
270
+ )
271
+ for c in f["calls"]:
272
+ req = " & ".join(
273
+ f"{r[0]}={'T' if r[1] else 'F'}" if r[1] is not None else r[0]
274
+ for r in c["requires"]
275
+ )
276
+ lines.append(f"- call `{c['callee'][:80]}` line {c['line']} requires: {req or '—'}")
277
+ lines.append("")
278
+ parts.append("## Mechanics (deterministic, intra-procedural)\n\n" + "\n".join(lines))
279
+ logs = []
280
+ entries = set(run.executed_changed)
281
+ for ref, ex in tests.items():
282
+ exits = [
283
+ f"{_short(n.symbol)} {n.outcome}"
284
+ for n in ex.nodes
285
+ if isinstance(n, CallNode) and n.symbol in entries
286
+ ][:6]
287
+ logs.append(
288
+ f"### {ref} test: {ex.outcome} exits: {'; '.join(exits)}\n"
289
+ + "\n".join(event_log(ex, vocab, preds))
290
+ )
291
+ parts.append(
292
+ "## Observed event logs\n\nCalls of the listed functions in chronological order, indented by depth among them, "
293
+ "with argument shapes and value digests, results, exits, and the observed outcome of every decision site "
294
+ "evaluated in those calls.\n\n```\n" + "\n\n".join(logs) + "\n```\n"
295
+ )
296
+ rng = run.spec.split()[-1] if run.spec.startswith("git diff") else ""
297
+ if rng:
298
+ # the change's non-test files beside the vocabulary's (constants and literals live there)
299
+ touched = [
300
+ f
301
+ for f in run.git("diff", "--name-only", rng).split()
302
+ if f not in files and not _is_test_path(f) and f.endswith(_SOURCE_EXT)
303
+ ]
304
+ # smallest first (constants files are small), generated code skipped
305
+ heads = {f: run.git("show", f"{run.revision}:{f}") for f in touched}
306
+ touched = sorted(
307
+ (f for f in touched if heads[f] and "Code generated" not in heads[f][:400]),
308
+ key=lambda f: len(heads[f]),
309
+ )
310
+ files += touched[:MAX_EXTRA_FILES]
311
+ diff = run.git("diff", "-U25", rng, "--", *files)
312
+ parts.append(f"## The change ({run.spec}, the files shown below)\n\n```diff\n{diff}```\n")
313
+ test_files = sorted({f for f in (_test_file(run, r) for r in tests) if f})
314
+ marks: dict[str, set[int]] = {}
315
+ for f in run.mech_fns:
316
+ if canonical(vocab, f["symbol"]) is not None:
317
+ marks.setdefault(f["file"], set()).update(
318
+ x["line"] for x in (*f["sites"], *f["calls"]) if x.get("line")
319
+ )
320
+ test_names = {r.split("::")[-1].split("/")[0] for r in tests}
321
+ for src in files + test_files:
322
+ text = run.git("show", f"{run.revision}:{src}").splitlines()
323
+ if not text:
324
+ continue
325
+ keep = _windows(text, marks.get(src, set()), test_names)
326
+ body, prev = [], 0
327
+ for i in keep:
328
+ if i != prev + 1:
329
+ body.append(" ...")
330
+ body.append(f"{i:4d} {text[i - 1]}")
331
+ prev = i
332
+ parts.append(f"## Source: {src}\n\n```\n" + "\n".join(body) + "\n```\n")
333
+ if len(tests) < len(everything):
334
+ parts.append(
335
+ f"({len(everything) - len(tests)} further tests executed the change and are checked too; not listed.)\n"
336
+ )
337
+ return "\n".join(parts), vocab, everything
338
+
339
+
340
+ def _windows(text: list[str], marks: set[int], test_names: set[str]) -> list[int]:
341
+ """All lines of a small file; of a large one, windows around the marked lines and the
342
+ bodies of the listed tests."""
343
+ n = len(text)
344
+ if n <= MAX_FILE_LINES:
345
+ return list(range(1, n + 1))
346
+ keep: set[int] = set()
347
+ for m in marks:
348
+ keep.update(range(max(1, m - WINDOW), min(n, m + WINDOW) + 1))
349
+ for i, ln in enumerate(text, 1):
350
+ s = ln.lstrip()
351
+ if s.startswith(("def ", "func ", "async def ")) and any(f"{t}(" in s for t in test_names):
352
+ ind = len(ln) - len(s)
353
+ j = i + 1
354
+ while j <= n and (
355
+ not text[j - 1].strip()
356
+ or len(text[j - 1]) - len(text[j - 1].lstrip()) > ind
357
+ or text[j - 1].lstrip().startswith((")", "}"))
358
+ ):
359
+ j += 1
360
+ keep.update(range(i, min(n, j) + 1))
361
+ return sorted(keep)
362
+
363
+
364
+ def _test_file(run: Run, ref: str) -> str | None:
365
+ for e in run.artifact.get("executions") or []:
366
+ if e.get("id") == ref:
367
+ return str(e["file"]) if e.get("file") else None
368
+ return None
369
+
370
+
371
+ def propose(bundle: str, writer: str) -> tuple[dict[str, Any], str, dict[str, Any]]:
372
+ """`recorded:<proposals.json>` replays a proposal; `openai[:<model>]` asks OpenAI."""
373
+ t0 = time.monotonic()
374
+ if writer.startswith("recorded:"):
375
+ path = Path(writer.split(":", 1)[1])
376
+ meta = (
377
+ json.loads(path.with_suffix(".meta.json").read_text())
378
+ if path.with_suffix(".meta.json").exists()
379
+ else {}
380
+ )
381
+ return (
382
+ json.loads(path.read_text()),
383
+ str(meta.get("model", f"recorded:{path.name}")),
384
+ {"seconds": 0.0},
385
+ )
386
+ if writer.startswith("openai"):
387
+ from diffgenome.llm import OpenAIProbeWriter
388
+
389
+ model = writer.split(":", 1)[1] if ":" in writer else None
390
+ w = OpenAIProbeWriter(model=model, timeout=3600)
391
+ effort = os.environ.get("DIFFGENOME_OPENAI_REASONING_EFFORT", "").strip()
392
+ data = w._post(
393
+ "/chat/completions",
394
+ {
395
+ **({"reasoning_effort": effort} if effort else {}),
396
+ "model": w.model,
397
+ "messages": [
398
+ {
399
+ "role": "system",
400
+ "content": "You write Behavioral Genome proposals. Output one JSON object only.",
401
+ },
402
+ {"role": "user", "content": bundle},
403
+ ],
404
+ "response_format": {"type": "json_object"},
405
+ },
406
+ )
407
+ choices = data.get("choices")
408
+ assert isinstance(choices, list) and choices
409
+ usage = data.get("usage") or {}
410
+ return (
411
+ json.loads(choices[0]["message"]["content"]),
412
+ w.model,
413
+ {"seconds": round(time.monotonic() - t0, 1), "usage": usage},
414
+ )
415
+ raise ValueError(f"unknown writer {writer!r}")
416
+
417
+
418
+ def _resolve(key: str, refs: list[str]) -> str | None:
419
+ """A scenario key names a test by its full id, or by a unique suffix after '/'."""
420
+ if key in refs:
421
+ return key
422
+ hits = [r for r in refs if r.endswith(("/" + key, "::" + key))]
423
+ return hits[0] if len(hits) == 1 else None
424
+
425
+
426
+ def changed_lines(run: Any) -> dict[str, set[int]]:
427
+ """Head-side line numbers the change added or modified, per file."""
428
+ rng = run.spec.split()[-1] if getattr(run, "spec", "").startswith("git diff") else ""
429
+ out: dict[str, set[int]] = {}
430
+ if not rng:
431
+ return out
432
+ cur = ""
433
+ for ln in run.git("diff", "-U0", rng).splitlines():
434
+ if ln.startswith("+++ "):
435
+ cur = ln[6:] if ln.startswith("+++ b/") else ""
436
+ elif ln.startswith("@@") and cur:
437
+ new = ln.split("+", 1)[1].split(" ", 1)[0]
438
+ start, _, count = new.partition(",")
439
+ n = int(count) if count else 1
440
+ out.setdefault(cur, set()).update(range(int(start), int(start) + n))
441
+ return out
442
+
443
+
444
+ def _site_changed(run: Any, site: str | None, touched: dict[str, set[int]]) -> bool:
445
+ s = run.mech.sites.get(site or "")
446
+ return bool(s) and s["line"] in touched.get(s.get("file", ""), set())
447
+
448
+
449
+ def eligible(it: Item) -> bool:
450
+ return it.status == VERIFIED or (it.status == SUPPORTED and not it.contradicted)
451
+
452
+
453
+ def establish(
454
+ run: Run, proposals: dict[str, Any], model: str, tests: dict[str, Execution], tree: Path
455
+ ) -> tuple[Genome, dict[str, Any]]:
456
+ refs = list(tests)
457
+ scen_in = proposals.get("scenarios") or {}
458
+ scenarios = {r: s for k, s in scen_in.items() if (r := _resolve(k, refs))}
459
+ obs = Observations(list(tests.values()))
460
+ g = genome_from_proposals(proposals, {"repo": run.root.name, "change": run.spec}, model)
461
+ agree: dict[str, list[tuple[str, bool]]] = {}
462
+ for ref, sc in scenarios.items():
463
+ ob = obs.get(ref)
464
+ assert ob is not None
465
+ pred = predict_sequence(g, sc, min_status="hypothesis")
466
+ for d in g.decisions:
467
+ if not d.site:
468
+ continue
469
+ p = [e[2] for e in pred.events if e[0] == "branch" and e[1] == d.site]
470
+ o = [b.outcome for b in ob.branches if b.site == d.site]
471
+ if (not p and not o) or (pred.indeterminate and len(p) < len(o)):
472
+ continue
473
+ agree.setdefault(d.id, []).append((ref, p == o))
474
+ sub = StateSubstrate(Substrate([], set(), tree, set()), run.mech, obs)
475
+ establish_state(g, sub, agreement_seq=agree, scenarios=scenarios)
476
+ sk = build_skeleton(list(tests.values()), genome_vocabulary(g))
477
+ required = change_sites(run.mech, obs, run.changed)
478
+ sites = {d.site for d in g.decisions if d.site}
479
+ per_test: list[dict[str, Any]] = []
480
+ for ref, sc in sorted(scenarios.items()):
481
+ ob = obs.get(ref)
482
+ assert ob is not None
483
+ cmp = compare_sequence(g, predict_sequence(g, sc), ob, sites, required, sk, sc)
484
+ per_test.append(
485
+ {"test": ref, "match": bool(cmp["match"]), "indeterminate": cmp["indeterminate"] or ""}
486
+ )
487
+ ev = {
488
+ "tests": len(tests),
489
+ "scenarios": len(scenarios),
490
+ "unresolved_scenarios": sorted(k for k in scen_in if not _resolve(k, refs)),
491
+ "consistent": sum(r["match"] for r in per_test),
492
+ "indeterminate": sum(1 for r in per_test if r["indeterminate"]),
493
+ "contradicted": sum(1 for r in per_test if not r["match"] and not r["indeterminate"]),
494
+ "per_test": per_test,
495
+ "site_agreement": {k: [ok for _, ok in v] for k, v in agree.items()},
496
+ }
497
+ return g, ev
498
+
499
+
500
+ def summary(
501
+ run: Run,
502
+ g: Genome,
503
+ ev: dict[str, Any],
504
+ model: str,
505
+ proposals: dict[str, Any],
506
+ bundle_digest: str,
507
+ cost: dict[str, Any],
508
+ ) -> dict[str, Any]:
509
+ """The artifact's `genome` section: checked claims only."""
510
+
511
+ def site_ref(site: str | None) -> dict[str, Any]:
512
+ s = run.mech.sites.get(site or "")
513
+ return {"file": s.get("file"), "line": s["line"], "source": s["pred"]} if s else {}
514
+
515
+ def outcome(b: Any) -> str | None:
516
+ o = b.outcome or {}
517
+ return f"{_short(o.get('entity', ''))} {o['is']}" if o.get("is") else None
518
+
519
+ agreement = ev.get("site_agreement") or {}
520
+ touched = changed_lines(run)
521
+ rules = []
522
+ for d in g.decisions:
523
+ # a supported rule is shown only when its predicted outcomes matched the observed
524
+ # ones in every test that exercised it (a checked meaning, not only a cited source)
525
+ seen = agreement.get(d.id) or []
526
+ if not eligible(d) or (d.status != VERIFIED and not (seen and all(seen))):
527
+ continue
528
+ rules.append(
529
+ {
530
+ "id": d.id,
531
+ "status": d.status,
532
+ "entity": _short(d.entity),
533
+ "site": d.site,
534
+ **site_ref(d.site),
535
+ "meaning": d.predicate,
536
+ "inputs": d.inputs,
537
+ "when_true": d.true_branch.effect or None,
538
+ "when_false": d.false_branch.effect or None,
539
+ "outcome_true": outcome(d.true_branch),
540
+ "outcome_false": outcome(d.false_branch),
541
+ "basis": d.status_reason[-200:],
542
+ "agreeing_tests": sum(seen),
543
+ "at_changed_line": _site_changed(run, d.site, touched),
544
+ }
545
+ )
546
+ rules.sort(key=lambda r: (not r["at_changed_line"], r["status"] != VERIFIED))
547
+ identities = []
548
+ literals = []
549
+ for v in g.variables:
550
+ bs = bindings_of(v)
551
+ ids = [b for b in bs if b.get("kind") == "identity"]
552
+ if len(ids) >= 2 and eligible(v):
553
+ identities.append(
554
+ {
555
+ "name": v.name,
556
+ "status": v.status,
557
+ "same_value_at": [
558
+ f"{_short(b['at']['entity'])} {b['at']['point']}" for b in ids
559
+ ],
560
+ "basis": v.status_reason[-200:],
561
+ }
562
+ )
563
+ for b in bs:
564
+ if b.get("kind") == "equals_literal" and b.get("_literal_ok"):
565
+ lit = b.get("literal") or {}
566
+ src = lit.get("source") or {}
567
+ literals.append(
568
+ {
569
+ "name": v.name,
570
+ "at": f"{_short(b['at']['entity'])} {b['at']['point']}",
571
+ "equals": lit.get("value"),
572
+ "written_at": f"{src.get('file')}:{src.get('line')}",
573
+ }
574
+ )
575
+ transitions = [
576
+ {"id": t.id, "entity": _short(t.entity), "when": t.when, "sets": t.sets, "status": t.status}
577
+ for t in g.transitions
578
+ if t.status == VERIFIED
579
+ ]
580
+ kinds = ("variables", "decisions", "transitions", "procedures", "regions", "rules", "regimes")
581
+ counts = {
582
+ st: sum(1 for k in kinds for i in getattr(g, k) if i.status == st)
583
+ for st in ("verified", "supported", "hypothesis", "rejected")
584
+ }
585
+ counts["contradicted_not_exported"] = sum(
586
+ 1 for k in kinds for i in getattr(g, k) if i.contradicted and i.status != VERIFIED
587
+ )
588
+ return {
589
+ "format": SUMMARY_FORMAT,
590
+ "proposed_by": model,
591
+ "bundle_digest": bundle_digest,
592
+ "established_from": f"{ev['tests']} executions of existing tests",
593
+ # three different counts; a consumer must not report one as another
594
+ "accounting": {
595
+ "relevant_executions_checked": ev["tests"],
596
+ "scenario_predictions_checked": ev["scenarios"],
597
+ "executions_listed_in_artifact": len(
598
+ getattr(run, "artifact", {}).get("executions") or []
599
+ ),
600
+ },
601
+ "consistency": {
602
+ k: ev[k] for k in ("scenarios", "consistent", "indeterminate", "contradicted")
603
+ },
604
+ "decision_rules": rules,
605
+ "identities": identities,
606
+ "literals": literals,
607
+ "transitions": transitions,
608
+ "statuses": counts,
609
+ "unknowns": [u for u in (proposals.get("unknowns") or []) if isinstance(u, dict)][:12],
610
+ "cost": cost,
611
+ "note": "Only verified claims, and supported claims no execution contradicted, are listed. Meanings are the model's words; statuses come from the traces.",
612
+ }
613
+
614
+
615
+ def main(argv: list[str] | None = None) -> int:
616
+ import argparse
617
+
618
+ ap = argparse.ArgumentParser(prog="diffgenome genome")
619
+ ap.add_argument(
620
+ "--run", required=True, type=Path, help="output directory of `diffgenome change`"
621
+ )
622
+ ap.add_argument(
623
+ "--tree",
624
+ type=Path,
625
+ default=None,
626
+ help="source tree at head (default: the artifact's repository root)",
627
+ )
628
+ ap.add_argument(
629
+ "--writer",
630
+ default="bundle-only",
631
+ help="bundle-only | recorded:<proposals.json> | openai[:<model>]",
632
+ )
633
+ ap.add_argument(
634
+ "--out",
635
+ type=Path,
636
+ default=None,
637
+ help="directory for the bundle, proposals, genome and evaluation (default: <run>/genome)",
638
+ )
639
+ ap.add_argument(
640
+ "--attach",
641
+ action="store_true",
642
+ help="write the summary into <run>/diffgenome-change.json as `genome`",
643
+ )
644
+ args = ap.parse_args(argv)
645
+ run = Run(args.run)
646
+ out = args.out or args.run / "genome"
647
+ out.mkdir(parents=True, exist_ok=True)
648
+ bundle, vocab, tests = build_bundle(run)
649
+ digest = hashlib.sha256(bundle.encode()).hexdigest()[:16]
650
+ (out / "context-bundle.md").write_text(bundle)
651
+ print(
652
+ f"bundle {digest}: {len(vocab)} functions, {len(tests)} tests, {len(bundle) // 1024} KiB -> {out / 'context-bundle.md'}"
653
+ )
654
+ if args.writer == "bundle-only":
655
+ return 0
656
+ proposals, model, cost = propose(bundle, args.writer)
657
+ (out / "proposals.json").write_text(json.dumps(proposals, indent=1) + "\n")
658
+ g, ev = establish(run, proposals, model, tests, args.tree or run.root)
659
+ s = summary(run, g, ev, model, proposals, digest, cost)
660
+ (out / "genome.json").write_text(dumps(g))
661
+ (out / "genome.md").write_text(render_markdown(g))
662
+ (out / "evaluation.json").write_text(json.dumps(ev, indent=1) + "\n")
663
+ (out / "genome-summary.json").write_text(json.dumps(s, indent=1) + "\n")
664
+ print(json.dumps({k: s[k] for k in ("consistency", "statuses")}))
665
+ print(
666
+ f"exported: {len(s['decision_rules'])} decision rules, {len(s['identities'])} identities, {len(s['literals'])} literals, {len(s['transitions'])} transitions"
667
+ )
668
+ if args.attach:
669
+ art_path = args.run / "diffgenome-change.json"
670
+ art = json.loads(art_path.read_text())
671
+ art["genome"] = s
672
+ art_path.write_text(json.dumps(art, indent=1) + "\n")
673
+ print(f"attached to {art_path}")
674
+ return 0