diffgenome 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- diffgenome/__init__.py +7 -0
- diffgenome/__main__.py +240 -0
- diffgenome/_collectors/go/dg/dg.go +623 -0
- diffgenome/_collectors/go/go.mod +3 -0
- diffgenome/_collectors/go/instrument/facts.go +346 -0
- diffgenome/_collectors/go/instrument/main.go +484 -0
- diffgenome/_collectors/node/instrument.js +289 -0
- diffgenome/_collectors/node/jest-setup.js +40 -0
- diffgenome/_collectors/node/package-lock.json +35 -0
- diffgenome/_collectors/node/package.json +11 -0
- diffgenome/_collectors/node/runtime.js +426 -0
- diffgenome/ambiguity.py +122 -0
- diffgenome/api.py +67 -0
- diffgenome/change.py +86 -0
- diffgenome/change_artifact.py +310 -0
- diffgenome/collect/__init__.py +2 -0
- diffgenome/collect/go_test.py +271 -0
- diffgenome/collect/node_jest.py +319 -0
- diffgenome/collect/py_monitoring.py +985 -0
- diffgenome/collect/py_runtime.py +116 -0
- diffgenome/collect/py_symbols.py +238 -0
- diffgenome/collect/pytest_plugin.py +130 -0
- diffgenome/compose.py +469 -0
- diffgenome/dependence.py +264 -0
- diffgenome/evaluate.py +669 -0
- diffgenome/frontends/__init__.py +0 -0
- diffgenome/frontends/python_ir.py +335 -0
- diffgenome/genome.py +1016 -0
- diffgenome/genome_pipeline.py +674 -0
- diffgenome/genome_prompt.py +33 -0
- diffgenome/genome_state.py +2118 -0
- diffgenome/graph.py +426 -0
- diffgenome/llm.py +189 -0
- diffgenome/model.py +364 -0
- diffgenome/mvp.py +398 -0
- diffgenome/probe.py +509 -0
- diffgenome/projection.py +308 -0
- diffgenome/py.typed +0 -0
- diffgenome/render.py +118 -0
- diffgenome/report.py +363 -0
- diffgenome/resolve.py +37 -0
- diffgenome/runtime.py +74 -0
- diffgenome/runtime_evidence.py +261 -0
- diffgenome/sandbox.py +166 -0
- diffgenome/serialize.py +96 -0
- diffgenome/sites.py +19 -0
- diffgenome/static_types.py +69 -0
- diffgenome/structure.py +462 -0
- diffgenome-0.1.0.dist-info/METADATA +139 -0
- diffgenome-0.1.0.dist-info/RECORD +53 -0
- diffgenome-0.1.0.dist-info/WHEEL +4 -0
- diffgenome-0.1.0.dist-info/entry_points.txt +2 -0
- diffgenome-0.1.0.dist-info/licenses/LICENSE +202 -0
|
@@ -0,0 +1,674 @@
|
|
|
1
|
+
# ruff: noqa: E501
|
|
2
|
+
"""`diffgenome genome`: the Behavioral Genome of a change, as a product step.
|
|
3
|
+
|
|
4
|
+
Input is the output directory of `diffgenome change` (its artifact, mechanics and existing-test
|
|
5
|
+
traces). Four steps, each generic (no framework knowledge):
|
|
6
|
+
|
|
7
|
+
1. bundle: the proposer's instructions (the v5 schema), required sites, the observed execution
|
|
8
|
+
structure, mechanics of the vocabulary's functions, the tests' event logs, the change's
|
|
9
|
+
diff and the sources;
|
|
10
|
+
2. propose: a model writes the semantic layer (recorded proposals, or OpenAI live);
|
|
11
|
+
3. establish: the checker assigns statuses from ALL the traces, and every scenario is predicted
|
|
12
|
+
and compared with what executed (consistency, not a held-out score);
|
|
13
|
+
4. summarize: only checked claims (verified, or supported and never contradicted) go to the
|
|
14
|
+
artifact's `genome` section. Hypotheses and rejected items are counted, never exported.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
import hashlib
|
|
20
|
+
import json
|
|
21
|
+
import os
|
|
22
|
+
import subprocess
|
|
23
|
+
import time
|
|
24
|
+
from pathlib import Path
|
|
25
|
+
from typing import Any
|
|
26
|
+
|
|
27
|
+
from diffgenome.genome import (
|
|
28
|
+
SUPPORTED,
|
|
29
|
+
VERIFIED,
|
|
30
|
+
Genome,
|
|
31
|
+
Item,
|
|
32
|
+
Substrate,
|
|
33
|
+
bindings_of,
|
|
34
|
+
dumps,
|
|
35
|
+
genome_from_proposals,
|
|
36
|
+
render_markdown,
|
|
37
|
+
)
|
|
38
|
+
from diffgenome.genome_prompt import FORMAT, HEAD
|
|
39
|
+
from diffgenome.genome_state import (
|
|
40
|
+
Mechanics,
|
|
41
|
+
Observations,
|
|
42
|
+
StateSubstrate,
|
|
43
|
+
change_sites,
|
|
44
|
+
compare_sequence,
|
|
45
|
+
establish_state,
|
|
46
|
+
genome_vocabulary,
|
|
47
|
+
predict_sequence,
|
|
48
|
+
)
|
|
49
|
+
from diffgenome.model import CallNode, Execution, SubstitutionNode
|
|
50
|
+
from diffgenome.serialize import execution_from_json
|
|
51
|
+
from diffgenome.structure import build_skeleton, canonical, render
|
|
52
|
+
|
|
53
|
+
SUMMARY_FORMAT = "diffgenome-genome-summary/1"
|
|
54
|
+
MAX_VOCAB = 24
|
|
55
|
+
MAX_TESTS = 40
|
|
56
|
+
MAX_LOG_LINES = 120
|
|
57
|
+
MAX_EXTRA_FILES = 8
|
|
58
|
+
MAX_FILE_LINES = 400
|
|
59
|
+
WINDOW = 12
|
|
60
|
+
_SOURCE_EXT = (".go", ".py", ".js", ".ts", ".rs")
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def _is_test_path(f: str) -> bool:
|
|
64
|
+
name = f.rsplit("/", 1)[-1]
|
|
65
|
+
return (
|
|
66
|
+
name.endswith(("_test.go", ".test.ts", ".test.js"))
|
|
67
|
+
or name.startswith("test_")
|
|
68
|
+
or "/tests/" in f
|
|
69
|
+
)
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def _short(sym: str) -> str:
|
|
73
|
+
return sym.split(":", 1)[-1]
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
class Run:
|
|
77
|
+
"""A `diffgenome change` output directory."""
|
|
78
|
+
|
|
79
|
+
def __init__(self, run: Path) -> None:
|
|
80
|
+
self.dir = run
|
|
81
|
+
self.artifact: dict[str, Any] = json.loads((run / "diffgenome-change.json").read_text())
|
|
82
|
+
self.mech_fns: list[dict[str, Any]] = json.loads((run / "mechanics.json").read_text())[
|
|
83
|
+
"functions"
|
|
84
|
+
]
|
|
85
|
+
self.mech = Mechanics(self.mech_fns)
|
|
86
|
+
self.executions: dict[str, Execution] = {
|
|
87
|
+
e.stimulus_ref: e
|
|
88
|
+
for e in (
|
|
89
|
+
execution_from_json(f.read_text())
|
|
90
|
+
for f in sorted((run / "traces-existing").glob("*.json"))
|
|
91
|
+
)
|
|
92
|
+
}
|
|
93
|
+
ch = self.artifact["change"]
|
|
94
|
+
self.changed: list[str] = list(ch["symbols"])
|
|
95
|
+
# executed = seen in a trace (the artifact's list may come from another run)
|
|
96
|
+
seen = {
|
|
97
|
+
n.symbol for e in self.executions.values() for n in e.nodes if isinstance(n, CallNode)
|
|
98
|
+
}
|
|
99
|
+
self.executed_changed = [s for s in self.changed if s in seen]
|
|
100
|
+
repo = self.artifact["repository"]
|
|
101
|
+
self.root = Path(repo["root"])
|
|
102
|
+
self.revision: str = repo.get("revision") or "HEAD"
|
|
103
|
+
self.runtime: str = repo.get("runtime", "")
|
|
104
|
+
self.spec: str = ch.get("spec", "")
|
|
105
|
+
|
|
106
|
+
def git(self, *args: str) -> str:
|
|
107
|
+
return subprocess.run(
|
|
108
|
+
["git", *args], cwd=self.root, capture_output=True, text=True, check=False
|
|
109
|
+
).stdout
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def vocabulary(run: Run) -> list[str]:
|
|
113
|
+
"""Changed functions that executed, functions nested in them, and their direct callers
|
|
114
|
+
and callees among the repository's functions (by observed call nesting)."""
|
|
115
|
+
mech_syms = {f["symbol"] for f in run.mech_fns}
|
|
116
|
+
changed = set(run.executed_changed)
|
|
117
|
+
vocab = [s for s in run.executed_changed if s in mech_syms]
|
|
118
|
+
nested: set[str] = set()
|
|
119
|
+
near: dict[str, int] = {}
|
|
120
|
+
for ex in run.executions.values():
|
|
121
|
+
by_id = {n.id: n for n in ex.nodes}
|
|
122
|
+
for n in ex.nodes:
|
|
123
|
+
if not isinstance(n, CallNode):
|
|
124
|
+
continue
|
|
125
|
+
base = n.symbol.split(".<anon>")[0]
|
|
126
|
+
if base in changed and n.symbol not in changed:
|
|
127
|
+
nested.add(n.symbol)
|
|
128
|
+
parent = by_id.get(n.parent) if n.parent is not None else None
|
|
129
|
+
psym: str = getattr(parent, "symbol", "") or ""
|
|
130
|
+
if n.symbol in changed and psym in mech_syms and psym not in changed:
|
|
131
|
+
near[psym] = near.get(psym, 0) + 1
|
|
132
|
+
if psym in changed and n.symbol in mech_syms and n.symbol not in changed:
|
|
133
|
+
near[n.symbol] = near.get(n.symbol, 0) + 1
|
|
134
|
+
vocab += sorted(nested - set(vocab))
|
|
135
|
+
vocab += [s for s, _ in sorted(near.items(), key=lambda kv: (-kv[1], kv[0])) if s not in vocab]
|
|
136
|
+
return vocab[:MAX_VOCAB]
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def relevant_tests(run: Run) -> dict[str, Execution]:
|
|
140
|
+
"""Executions that call a changed function (or a function nested in one)."""
|
|
141
|
+
changed = set(run.executed_changed)
|
|
142
|
+
return {
|
|
143
|
+
ref: ex
|
|
144
|
+
for ref, ex in sorted(run.executions.items())
|
|
145
|
+
if any(
|
|
146
|
+
isinstance(n, CallNode)
|
|
147
|
+
and (n.symbol in changed or n.symbol.split(".<anon>")[0] in changed)
|
|
148
|
+
for n in ex.nodes
|
|
149
|
+
)
|
|
150
|
+
}
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def shown_tests(tests: dict[str, Execution], required: set[str]) -> dict[str, Execution]:
|
|
154
|
+
"""At most MAX_TESTS for the bundle: one test per distinct observed behavior (the
|
|
155
|
+
outcome vector of the required sites and the changed calls' exits) first, then the rest."""
|
|
156
|
+
|
|
157
|
+
def signature(ex: Execution) -> tuple[Any, ...]:
|
|
158
|
+
return (
|
|
159
|
+
tuple((b.site, b.outcome) for b in ex.branches if b.site in required),
|
|
160
|
+
tuple(sorted({n.outcome for n in ex.nodes if isinstance(n, CallNode)})),
|
|
161
|
+
)
|
|
162
|
+
|
|
163
|
+
first: dict[tuple[Any, ...], str] = {}
|
|
164
|
+
for ref, ex in tests.items():
|
|
165
|
+
first.setdefault(signature(ex), ref)
|
|
166
|
+
picked = list(first.values())
|
|
167
|
+
picked += [r for r in tests if r not in picked]
|
|
168
|
+
keep = set(picked[:MAX_TESTS])
|
|
169
|
+
return {r: ex for r, ex in tests.items() if r in keep}
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
def event_log(ex: Execution, vocab: list[str], preds: dict[str, str]) -> list[str]:
|
|
173
|
+
nodes = ex.nodes
|
|
174
|
+
name = {
|
|
175
|
+
n.id: canonical(vocab, getattr(n, "symbol", None) or getattr(n, "claimed_target", "") or "")
|
|
176
|
+
for n in nodes
|
|
177
|
+
}
|
|
178
|
+
|
|
179
|
+
def depth(i: int) -> int:
|
|
180
|
+
d, p = 0, nodes[i].parent
|
|
181
|
+
while p is not None:
|
|
182
|
+
d += name.get(p) is not None
|
|
183
|
+
p = nodes[p].parent
|
|
184
|
+
return d
|
|
185
|
+
|
|
186
|
+
events: list[tuple[int, int, Any]] = [(n.id, 1, n) for n in nodes if name.get(n.id)]
|
|
187
|
+
events += [(b.seq, 0, b) for b in ex.branches if name.get(b.node)]
|
|
188
|
+
out = []
|
|
189
|
+
for _, kind, o in sorted(events, key=lambda e: (e[0], e[1])):
|
|
190
|
+
if kind == 0:
|
|
191
|
+
out.append(
|
|
192
|
+
f"{' ' * (depth(o.node) + 1)}branch {o.site} `{preds.get(o.site, '?')[:70]}` = {'T' if o.outcome else 'F'}"
|
|
193
|
+
)
|
|
194
|
+
else:
|
|
195
|
+
args = ", ".join(
|
|
196
|
+
f"{a}={t}" + (f"#{d[:6]}" if d else "") for a, t, d in getattr(o, "args", ())
|
|
197
|
+
)
|
|
198
|
+
res = f" -> #{o.result[:6]}" if getattr(o, "result", "") else ""
|
|
199
|
+
tag = " (stand-in, mocked)" if isinstance(o, SubstitutionNode) else ""
|
|
200
|
+
out.append(
|
|
201
|
+
f"{' ' * depth(o.id)}call {_short(name[o.id] or '')}({args}) [{getattr(o, 'outcome', '?')}]{res}{tag}"
|
|
202
|
+
)
|
|
203
|
+
if len(out) > MAX_LOG_LINES:
|
|
204
|
+
out = [*out[:MAX_LOG_LINES], f"... ({len(out) - MAX_LOG_LINES} more events)"]
|
|
205
|
+
return out
|
|
206
|
+
|
|
207
|
+
|
|
208
|
+
def build_bundle(run: Run) -> tuple[str, list[str], dict[str, Execution]]:
|
|
209
|
+
vocab = vocabulary(run)
|
|
210
|
+
everything = relevant_tests(run)
|
|
211
|
+
required = change_sites(run.mech, Observations(list(everything.values())), run.changed)
|
|
212
|
+
tests = shown_tests(everything, required)
|
|
213
|
+
sk = build_skeleton(list(tests.values()), vocab)
|
|
214
|
+
preds = {s: rec["pred"] for s, rec in run.mech.sites.items()}
|
|
215
|
+
go = run.runtime.startswith("go")
|
|
216
|
+
null_note = "The digest `#5da3a4` is Go's `nil`." if go else ""
|
|
217
|
+
parts = [
|
|
218
|
+
HEAD.format(
|
|
219
|
+
spec=run.spec or "(unspecified)",
|
|
220
|
+
repo=run.root.name,
|
|
221
|
+
runtime=run.runtime,
|
|
222
|
+
symbols=", ".join(_short(s) for s in run.changed),
|
|
223
|
+
null_note=null_note,
|
|
224
|
+
)
|
|
225
|
+
+ FORMAT.replace('"lang": "go"', f'"lang": "{"go" if go else "python"}"')
|
|
226
|
+
]
|
|
227
|
+
seen = {b.site for ex in tests.values() for b in ex.branches}
|
|
228
|
+
parts.append(
|
|
229
|
+
"## Required sites\n\nEvery OBSERVED evaluation of these sites is scored, in every test.\n\n"
|
|
230
|
+
+ "\n".join(
|
|
231
|
+
f"- {s} {run.mech.sites[s]['symbol']} line {run.mech.sites[s]['line']}: `{run.mech.sites[s]['pred']}`"
|
|
232
|
+
+ ("" if s in seen else " (not evaluated by any test)")
|
|
233
|
+
for s in sorted(
|
|
234
|
+
(s for s in required if s in run.mech.sites),
|
|
235
|
+
key=lambda s: (run.mech.sites[s]["symbol"], run.mech.sites[s]["line"]),
|
|
236
|
+
)
|
|
237
|
+
)
|
|
238
|
+
+ "\n"
|
|
239
|
+
)
|
|
240
|
+
runtime_only: dict[str, set[str]] = {}
|
|
241
|
+
for ex in tests.values():
|
|
242
|
+
for b in ex.branches:
|
|
243
|
+
if b.site in required and b.site not in run.mech.sites:
|
|
244
|
+
runtime_only.setdefault(b.site, set()).add(getattr(ex.nodes[b.node], "symbol", "?"))
|
|
245
|
+
if runtime_only:
|
|
246
|
+
parts.append(
|
|
247
|
+
"Runtime-only required sites (observed; no static facts, e.g. in closures): "
|
|
248
|
+
+ "; ".join(f"{s} in {', '.join(sorted(v))}" for s, v in sorted(runtime_only.items()))
|
|
249
|
+
+ "\n"
|
|
250
|
+
)
|
|
251
|
+
parts.append(
|
|
252
|
+
"## Observed execution structure (deterministic)\n\n```\n" + render(sk) + "\n```\n"
|
|
253
|
+
)
|
|
254
|
+
lines = []
|
|
255
|
+
files: list[str] = []
|
|
256
|
+
for f in run.mech_fns:
|
|
257
|
+
if canonical(vocab, f["symbol"]) is None:
|
|
258
|
+
continue
|
|
259
|
+
if f["file"] not in files:
|
|
260
|
+
files.append(f["file"])
|
|
261
|
+
lines.append(f"### {_short(f['symbol'])} ({f['file']})")
|
|
262
|
+
for s in f["sites"]:
|
|
263
|
+
ops = "; ".join(f"{o['path']} ← {', '.join(o['origins'])}" for o in s["operands"])
|
|
264
|
+
req = " & ".join(
|
|
265
|
+
f"{r[0]}={'T' if r[1] else 'F'}" if r[1] is not None else r[0]
|
|
266
|
+
for r in s["requires"]
|
|
267
|
+
)
|
|
268
|
+
lines.append(
|
|
269
|
+
f"- site {s['site']} line {s['line']}: `{s['pred']}` | operands: {ops} | requires: {req or '—'} | then exits: {s['then_exits']}, else exits: {s['else_exits']}"
|
|
270
|
+
)
|
|
271
|
+
for c in f["calls"]:
|
|
272
|
+
req = " & ".join(
|
|
273
|
+
f"{r[0]}={'T' if r[1] else 'F'}" if r[1] is not None else r[0]
|
|
274
|
+
for r in c["requires"]
|
|
275
|
+
)
|
|
276
|
+
lines.append(f"- call `{c['callee'][:80]}` line {c['line']} requires: {req or '—'}")
|
|
277
|
+
lines.append("")
|
|
278
|
+
parts.append("## Mechanics (deterministic, intra-procedural)\n\n" + "\n".join(lines))
|
|
279
|
+
logs = []
|
|
280
|
+
entries = set(run.executed_changed)
|
|
281
|
+
for ref, ex in tests.items():
|
|
282
|
+
exits = [
|
|
283
|
+
f"{_short(n.symbol)} {n.outcome}"
|
|
284
|
+
for n in ex.nodes
|
|
285
|
+
if isinstance(n, CallNode) and n.symbol in entries
|
|
286
|
+
][:6]
|
|
287
|
+
logs.append(
|
|
288
|
+
f"### {ref} test: {ex.outcome} exits: {'; '.join(exits)}\n"
|
|
289
|
+
+ "\n".join(event_log(ex, vocab, preds))
|
|
290
|
+
)
|
|
291
|
+
parts.append(
|
|
292
|
+
"## Observed event logs\n\nCalls of the listed functions in chronological order, indented by depth among them, "
|
|
293
|
+
"with argument shapes and value digests, results, exits, and the observed outcome of every decision site "
|
|
294
|
+
"evaluated in those calls.\n\n```\n" + "\n\n".join(logs) + "\n```\n"
|
|
295
|
+
)
|
|
296
|
+
rng = run.spec.split()[-1] if run.spec.startswith("git diff") else ""
|
|
297
|
+
if rng:
|
|
298
|
+
# the change's non-test files beside the vocabulary's (constants and literals live there)
|
|
299
|
+
touched = [
|
|
300
|
+
f
|
|
301
|
+
for f in run.git("diff", "--name-only", rng).split()
|
|
302
|
+
if f not in files and not _is_test_path(f) and f.endswith(_SOURCE_EXT)
|
|
303
|
+
]
|
|
304
|
+
# smallest first (constants files are small), generated code skipped
|
|
305
|
+
heads = {f: run.git("show", f"{run.revision}:{f}") for f in touched}
|
|
306
|
+
touched = sorted(
|
|
307
|
+
(f for f in touched if heads[f] and "Code generated" not in heads[f][:400]),
|
|
308
|
+
key=lambda f: len(heads[f]),
|
|
309
|
+
)
|
|
310
|
+
files += touched[:MAX_EXTRA_FILES]
|
|
311
|
+
diff = run.git("diff", "-U25", rng, "--", *files)
|
|
312
|
+
parts.append(f"## The change ({run.spec}, the files shown below)\n\n```diff\n{diff}```\n")
|
|
313
|
+
test_files = sorted({f for f in (_test_file(run, r) for r in tests) if f})
|
|
314
|
+
marks: dict[str, set[int]] = {}
|
|
315
|
+
for f in run.mech_fns:
|
|
316
|
+
if canonical(vocab, f["symbol"]) is not None:
|
|
317
|
+
marks.setdefault(f["file"], set()).update(
|
|
318
|
+
x["line"] for x in (*f["sites"], *f["calls"]) if x.get("line")
|
|
319
|
+
)
|
|
320
|
+
test_names = {r.split("::")[-1].split("/")[0] for r in tests}
|
|
321
|
+
for src in files + test_files:
|
|
322
|
+
text = run.git("show", f"{run.revision}:{src}").splitlines()
|
|
323
|
+
if not text:
|
|
324
|
+
continue
|
|
325
|
+
keep = _windows(text, marks.get(src, set()), test_names)
|
|
326
|
+
body, prev = [], 0
|
|
327
|
+
for i in keep:
|
|
328
|
+
if i != prev + 1:
|
|
329
|
+
body.append(" ...")
|
|
330
|
+
body.append(f"{i:4d} {text[i - 1]}")
|
|
331
|
+
prev = i
|
|
332
|
+
parts.append(f"## Source: {src}\n\n```\n" + "\n".join(body) + "\n```\n")
|
|
333
|
+
if len(tests) < len(everything):
|
|
334
|
+
parts.append(
|
|
335
|
+
f"({len(everything) - len(tests)} further tests executed the change and are checked too; not listed.)\n"
|
|
336
|
+
)
|
|
337
|
+
return "\n".join(parts), vocab, everything
|
|
338
|
+
|
|
339
|
+
|
|
340
|
+
def _windows(text: list[str], marks: set[int], test_names: set[str]) -> list[int]:
|
|
341
|
+
"""All lines of a small file; of a large one, windows around the marked lines and the
|
|
342
|
+
bodies of the listed tests."""
|
|
343
|
+
n = len(text)
|
|
344
|
+
if n <= MAX_FILE_LINES:
|
|
345
|
+
return list(range(1, n + 1))
|
|
346
|
+
keep: set[int] = set()
|
|
347
|
+
for m in marks:
|
|
348
|
+
keep.update(range(max(1, m - WINDOW), min(n, m + WINDOW) + 1))
|
|
349
|
+
for i, ln in enumerate(text, 1):
|
|
350
|
+
s = ln.lstrip()
|
|
351
|
+
if s.startswith(("def ", "func ", "async def ")) and any(f"{t}(" in s for t in test_names):
|
|
352
|
+
ind = len(ln) - len(s)
|
|
353
|
+
j = i + 1
|
|
354
|
+
while j <= n and (
|
|
355
|
+
not text[j - 1].strip()
|
|
356
|
+
or len(text[j - 1]) - len(text[j - 1].lstrip()) > ind
|
|
357
|
+
or text[j - 1].lstrip().startswith((")", "}"))
|
|
358
|
+
):
|
|
359
|
+
j += 1
|
|
360
|
+
keep.update(range(i, min(n, j) + 1))
|
|
361
|
+
return sorted(keep)
|
|
362
|
+
|
|
363
|
+
|
|
364
|
+
def _test_file(run: Run, ref: str) -> str | None:
|
|
365
|
+
for e in run.artifact.get("executions") or []:
|
|
366
|
+
if e.get("id") == ref:
|
|
367
|
+
return str(e["file"]) if e.get("file") else None
|
|
368
|
+
return None
|
|
369
|
+
|
|
370
|
+
|
|
371
|
+
def propose(bundle: str, writer: str) -> tuple[dict[str, Any], str, dict[str, Any]]:
|
|
372
|
+
"""`recorded:<proposals.json>` replays a proposal; `openai[:<model>]` asks OpenAI."""
|
|
373
|
+
t0 = time.monotonic()
|
|
374
|
+
if writer.startswith("recorded:"):
|
|
375
|
+
path = Path(writer.split(":", 1)[1])
|
|
376
|
+
meta = (
|
|
377
|
+
json.loads(path.with_suffix(".meta.json").read_text())
|
|
378
|
+
if path.with_suffix(".meta.json").exists()
|
|
379
|
+
else {}
|
|
380
|
+
)
|
|
381
|
+
return (
|
|
382
|
+
json.loads(path.read_text()),
|
|
383
|
+
str(meta.get("model", f"recorded:{path.name}")),
|
|
384
|
+
{"seconds": 0.0},
|
|
385
|
+
)
|
|
386
|
+
if writer.startswith("openai"):
|
|
387
|
+
from diffgenome.llm import OpenAIProbeWriter
|
|
388
|
+
|
|
389
|
+
model = writer.split(":", 1)[1] if ":" in writer else None
|
|
390
|
+
w = OpenAIProbeWriter(model=model, timeout=3600)
|
|
391
|
+
effort = os.environ.get("DIFFGENOME_OPENAI_REASONING_EFFORT", "").strip()
|
|
392
|
+
data = w._post(
|
|
393
|
+
"/chat/completions",
|
|
394
|
+
{
|
|
395
|
+
**({"reasoning_effort": effort} if effort else {}),
|
|
396
|
+
"model": w.model,
|
|
397
|
+
"messages": [
|
|
398
|
+
{
|
|
399
|
+
"role": "system",
|
|
400
|
+
"content": "You write Behavioral Genome proposals. Output one JSON object only.",
|
|
401
|
+
},
|
|
402
|
+
{"role": "user", "content": bundle},
|
|
403
|
+
],
|
|
404
|
+
"response_format": {"type": "json_object"},
|
|
405
|
+
},
|
|
406
|
+
)
|
|
407
|
+
choices = data.get("choices")
|
|
408
|
+
assert isinstance(choices, list) and choices
|
|
409
|
+
usage = data.get("usage") or {}
|
|
410
|
+
return (
|
|
411
|
+
json.loads(choices[0]["message"]["content"]),
|
|
412
|
+
w.model,
|
|
413
|
+
{"seconds": round(time.monotonic() - t0, 1), "usage": usage},
|
|
414
|
+
)
|
|
415
|
+
raise ValueError(f"unknown writer {writer!r}")
|
|
416
|
+
|
|
417
|
+
|
|
418
|
+
def _resolve(key: str, refs: list[str]) -> str | None:
|
|
419
|
+
"""A scenario key names a test by its full id, or by a unique suffix after '/'."""
|
|
420
|
+
if key in refs:
|
|
421
|
+
return key
|
|
422
|
+
hits = [r for r in refs if r.endswith(("/" + key, "::" + key))]
|
|
423
|
+
return hits[0] if len(hits) == 1 else None
|
|
424
|
+
|
|
425
|
+
|
|
426
|
+
def changed_lines(run: Any) -> dict[str, set[int]]:
|
|
427
|
+
"""Head-side line numbers the change added or modified, per file."""
|
|
428
|
+
rng = run.spec.split()[-1] if getattr(run, "spec", "").startswith("git diff") else ""
|
|
429
|
+
out: dict[str, set[int]] = {}
|
|
430
|
+
if not rng:
|
|
431
|
+
return out
|
|
432
|
+
cur = ""
|
|
433
|
+
for ln in run.git("diff", "-U0", rng).splitlines():
|
|
434
|
+
if ln.startswith("+++ "):
|
|
435
|
+
cur = ln[6:] if ln.startswith("+++ b/") else ""
|
|
436
|
+
elif ln.startswith("@@") and cur:
|
|
437
|
+
new = ln.split("+", 1)[1].split(" ", 1)[0]
|
|
438
|
+
start, _, count = new.partition(",")
|
|
439
|
+
n = int(count) if count else 1
|
|
440
|
+
out.setdefault(cur, set()).update(range(int(start), int(start) + n))
|
|
441
|
+
return out
|
|
442
|
+
|
|
443
|
+
|
|
444
|
+
def _site_changed(run: Any, site: str | None, touched: dict[str, set[int]]) -> bool:
|
|
445
|
+
s = run.mech.sites.get(site or "")
|
|
446
|
+
return bool(s) and s["line"] in touched.get(s.get("file", ""), set())
|
|
447
|
+
|
|
448
|
+
|
|
449
|
+
def eligible(it: Item) -> bool:
|
|
450
|
+
return it.status == VERIFIED or (it.status == SUPPORTED and not it.contradicted)
|
|
451
|
+
|
|
452
|
+
|
|
453
|
+
def establish(
|
|
454
|
+
run: Run, proposals: dict[str, Any], model: str, tests: dict[str, Execution], tree: Path
|
|
455
|
+
) -> tuple[Genome, dict[str, Any]]:
|
|
456
|
+
refs = list(tests)
|
|
457
|
+
scen_in = proposals.get("scenarios") or {}
|
|
458
|
+
scenarios = {r: s for k, s in scen_in.items() if (r := _resolve(k, refs))}
|
|
459
|
+
obs = Observations(list(tests.values()))
|
|
460
|
+
g = genome_from_proposals(proposals, {"repo": run.root.name, "change": run.spec}, model)
|
|
461
|
+
agree: dict[str, list[tuple[str, bool]]] = {}
|
|
462
|
+
for ref, sc in scenarios.items():
|
|
463
|
+
ob = obs.get(ref)
|
|
464
|
+
assert ob is not None
|
|
465
|
+
pred = predict_sequence(g, sc, min_status="hypothesis")
|
|
466
|
+
for d in g.decisions:
|
|
467
|
+
if not d.site:
|
|
468
|
+
continue
|
|
469
|
+
p = [e[2] for e in pred.events if e[0] == "branch" and e[1] == d.site]
|
|
470
|
+
o = [b.outcome for b in ob.branches if b.site == d.site]
|
|
471
|
+
if (not p and not o) or (pred.indeterminate and len(p) < len(o)):
|
|
472
|
+
continue
|
|
473
|
+
agree.setdefault(d.id, []).append((ref, p == o))
|
|
474
|
+
sub = StateSubstrate(Substrate([], set(), tree, set()), run.mech, obs)
|
|
475
|
+
establish_state(g, sub, agreement_seq=agree, scenarios=scenarios)
|
|
476
|
+
sk = build_skeleton(list(tests.values()), genome_vocabulary(g))
|
|
477
|
+
required = change_sites(run.mech, obs, run.changed)
|
|
478
|
+
sites = {d.site for d in g.decisions if d.site}
|
|
479
|
+
per_test: list[dict[str, Any]] = []
|
|
480
|
+
for ref, sc in sorted(scenarios.items()):
|
|
481
|
+
ob = obs.get(ref)
|
|
482
|
+
assert ob is not None
|
|
483
|
+
cmp = compare_sequence(g, predict_sequence(g, sc), ob, sites, required, sk, sc)
|
|
484
|
+
per_test.append(
|
|
485
|
+
{"test": ref, "match": bool(cmp["match"]), "indeterminate": cmp["indeterminate"] or ""}
|
|
486
|
+
)
|
|
487
|
+
ev = {
|
|
488
|
+
"tests": len(tests),
|
|
489
|
+
"scenarios": len(scenarios),
|
|
490
|
+
"unresolved_scenarios": sorted(k for k in scen_in if not _resolve(k, refs)),
|
|
491
|
+
"consistent": sum(r["match"] for r in per_test),
|
|
492
|
+
"indeterminate": sum(1 for r in per_test if r["indeterminate"]),
|
|
493
|
+
"contradicted": sum(1 for r in per_test if not r["match"] and not r["indeterminate"]),
|
|
494
|
+
"per_test": per_test,
|
|
495
|
+
"site_agreement": {k: [ok for _, ok in v] for k, v in agree.items()},
|
|
496
|
+
}
|
|
497
|
+
return g, ev
|
|
498
|
+
|
|
499
|
+
|
|
500
|
+
def summary(
|
|
501
|
+
run: Run,
|
|
502
|
+
g: Genome,
|
|
503
|
+
ev: dict[str, Any],
|
|
504
|
+
model: str,
|
|
505
|
+
proposals: dict[str, Any],
|
|
506
|
+
bundle_digest: str,
|
|
507
|
+
cost: dict[str, Any],
|
|
508
|
+
) -> dict[str, Any]:
|
|
509
|
+
"""The artifact's `genome` section: checked claims only."""
|
|
510
|
+
|
|
511
|
+
def site_ref(site: str | None) -> dict[str, Any]:
|
|
512
|
+
s = run.mech.sites.get(site or "")
|
|
513
|
+
return {"file": s.get("file"), "line": s["line"], "source": s["pred"]} if s else {}
|
|
514
|
+
|
|
515
|
+
def outcome(b: Any) -> str | None:
|
|
516
|
+
o = b.outcome or {}
|
|
517
|
+
return f"{_short(o.get('entity', ''))} {o['is']}" if o.get("is") else None
|
|
518
|
+
|
|
519
|
+
agreement = ev.get("site_agreement") or {}
|
|
520
|
+
touched = changed_lines(run)
|
|
521
|
+
rules = []
|
|
522
|
+
for d in g.decisions:
|
|
523
|
+
# a supported rule is shown only when its predicted outcomes matched the observed
|
|
524
|
+
# ones in every test that exercised it (a checked meaning, not only a cited source)
|
|
525
|
+
seen = agreement.get(d.id) or []
|
|
526
|
+
if not eligible(d) or (d.status != VERIFIED and not (seen and all(seen))):
|
|
527
|
+
continue
|
|
528
|
+
rules.append(
|
|
529
|
+
{
|
|
530
|
+
"id": d.id,
|
|
531
|
+
"status": d.status,
|
|
532
|
+
"entity": _short(d.entity),
|
|
533
|
+
"site": d.site,
|
|
534
|
+
**site_ref(d.site),
|
|
535
|
+
"meaning": d.predicate,
|
|
536
|
+
"inputs": d.inputs,
|
|
537
|
+
"when_true": d.true_branch.effect or None,
|
|
538
|
+
"when_false": d.false_branch.effect or None,
|
|
539
|
+
"outcome_true": outcome(d.true_branch),
|
|
540
|
+
"outcome_false": outcome(d.false_branch),
|
|
541
|
+
"basis": d.status_reason[-200:],
|
|
542
|
+
"agreeing_tests": sum(seen),
|
|
543
|
+
"at_changed_line": _site_changed(run, d.site, touched),
|
|
544
|
+
}
|
|
545
|
+
)
|
|
546
|
+
rules.sort(key=lambda r: (not r["at_changed_line"], r["status"] != VERIFIED))
|
|
547
|
+
identities = []
|
|
548
|
+
literals = []
|
|
549
|
+
for v in g.variables:
|
|
550
|
+
bs = bindings_of(v)
|
|
551
|
+
ids = [b for b in bs if b.get("kind") == "identity"]
|
|
552
|
+
if len(ids) >= 2 and eligible(v):
|
|
553
|
+
identities.append(
|
|
554
|
+
{
|
|
555
|
+
"name": v.name,
|
|
556
|
+
"status": v.status,
|
|
557
|
+
"same_value_at": [
|
|
558
|
+
f"{_short(b['at']['entity'])} {b['at']['point']}" for b in ids
|
|
559
|
+
],
|
|
560
|
+
"basis": v.status_reason[-200:],
|
|
561
|
+
}
|
|
562
|
+
)
|
|
563
|
+
for b in bs:
|
|
564
|
+
if b.get("kind") == "equals_literal" and b.get("_literal_ok"):
|
|
565
|
+
lit = b.get("literal") or {}
|
|
566
|
+
src = lit.get("source") or {}
|
|
567
|
+
literals.append(
|
|
568
|
+
{
|
|
569
|
+
"name": v.name,
|
|
570
|
+
"at": f"{_short(b['at']['entity'])} {b['at']['point']}",
|
|
571
|
+
"equals": lit.get("value"),
|
|
572
|
+
"written_at": f"{src.get('file')}:{src.get('line')}",
|
|
573
|
+
}
|
|
574
|
+
)
|
|
575
|
+
transitions = [
|
|
576
|
+
{"id": t.id, "entity": _short(t.entity), "when": t.when, "sets": t.sets, "status": t.status}
|
|
577
|
+
for t in g.transitions
|
|
578
|
+
if t.status == VERIFIED
|
|
579
|
+
]
|
|
580
|
+
kinds = ("variables", "decisions", "transitions", "procedures", "regions", "rules", "regimes")
|
|
581
|
+
counts = {
|
|
582
|
+
st: sum(1 for k in kinds for i in getattr(g, k) if i.status == st)
|
|
583
|
+
for st in ("verified", "supported", "hypothesis", "rejected")
|
|
584
|
+
}
|
|
585
|
+
counts["contradicted_not_exported"] = sum(
|
|
586
|
+
1 for k in kinds for i in getattr(g, k) if i.contradicted and i.status != VERIFIED
|
|
587
|
+
)
|
|
588
|
+
return {
|
|
589
|
+
"format": SUMMARY_FORMAT,
|
|
590
|
+
"proposed_by": model,
|
|
591
|
+
"bundle_digest": bundle_digest,
|
|
592
|
+
"established_from": f"{ev['tests']} executions of existing tests",
|
|
593
|
+
# three different counts; a consumer must not report one as another
|
|
594
|
+
"accounting": {
|
|
595
|
+
"relevant_executions_checked": ev["tests"],
|
|
596
|
+
"scenario_predictions_checked": ev["scenarios"],
|
|
597
|
+
"executions_listed_in_artifact": len(
|
|
598
|
+
getattr(run, "artifact", {}).get("executions") or []
|
|
599
|
+
),
|
|
600
|
+
},
|
|
601
|
+
"consistency": {
|
|
602
|
+
k: ev[k] for k in ("scenarios", "consistent", "indeterminate", "contradicted")
|
|
603
|
+
},
|
|
604
|
+
"decision_rules": rules,
|
|
605
|
+
"identities": identities,
|
|
606
|
+
"literals": literals,
|
|
607
|
+
"transitions": transitions,
|
|
608
|
+
"statuses": counts,
|
|
609
|
+
"unknowns": [u for u in (proposals.get("unknowns") or []) if isinstance(u, dict)][:12],
|
|
610
|
+
"cost": cost,
|
|
611
|
+
"note": "Only verified claims, and supported claims no execution contradicted, are listed. Meanings are the model's words; statuses come from the traces.",
|
|
612
|
+
}
|
|
613
|
+
|
|
614
|
+
|
|
615
|
+
def main(argv: list[str] | None = None) -> int:
|
|
616
|
+
import argparse
|
|
617
|
+
|
|
618
|
+
ap = argparse.ArgumentParser(prog="diffgenome genome")
|
|
619
|
+
ap.add_argument(
|
|
620
|
+
"--run", required=True, type=Path, help="output directory of `diffgenome change`"
|
|
621
|
+
)
|
|
622
|
+
ap.add_argument(
|
|
623
|
+
"--tree",
|
|
624
|
+
type=Path,
|
|
625
|
+
default=None,
|
|
626
|
+
help="source tree at head (default: the artifact's repository root)",
|
|
627
|
+
)
|
|
628
|
+
ap.add_argument(
|
|
629
|
+
"--writer",
|
|
630
|
+
default="bundle-only",
|
|
631
|
+
help="bundle-only | recorded:<proposals.json> | openai[:<model>]",
|
|
632
|
+
)
|
|
633
|
+
ap.add_argument(
|
|
634
|
+
"--out",
|
|
635
|
+
type=Path,
|
|
636
|
+
default=None,
|
|
637
|
+
help="directory for the bundle, proposals, genome and evaluation (default: <run>/genome)",
|
|
638
|
+
)
|
|
639
|
+
ap.add_argument(
|
|
640
|
+
"--attach",
|
|
641
|
+
action="store_true",
|
|
642
|
+
help="write the summary into <run>/diffgenome-change.json as `genome`",
|
|
643
|
+
)
|
|
644
|
+
args = ap.parse_args(argv)
|
|
645
|
+
run = Run(args.run)
|
|
646
|
+
out = args.out or args.run / "genome"
|
|
647
|
+
out.mkdir(parents=True, exist_ok=True)
|
|
648
|
+
bundle, vocab, tests = build_bundle(run)
|
|
649
|
+
digest = hashlib.sha256(bundle.encode()).hexdigest()[:16]
|
|
650
|
+
(out / "context-bundle.md").write_text(bundle)
|
|
651
|
+
print(
|
|
652
|
+
f"bundle {digest}: {len(vocab)} functions, {len(tests)} tests, {len(bundle) // 1024} KiB -> {out / 'context-bundle.md'}"
|
|
653
|
+
)
|
|
654
|
+
if args.writer == "bundle-only":
|
|
655
|
+
return 0
|
|
656
|
+
proposals, model, cost = propose(bundle, args.writer)
|
|
657
|
+
(out / "proposals.json").write_text(json.dumps(proposals, indent=1) + "\n")
|
|
658
|
+
g, ev = establish(run, proposals, model, tests, args.tree or run.root)
|
|
659
|
+
s = summary(run, g, ev, model, proposals, digest, cost)
|
|
660
|
+
(out / "genome.json").write_text(dumps(g))
|
|
661
|
+
(out / "genome.md").write_text(render_markdown(g))
|
|
662
|
+
(out / "evaluation.json").write_text(json.dumps(ev, indent=1) + "\n")
|
|
663
|
+
(out / "genome-summary.json").write_text(json.dumps(s, indent=1) + "\n")
|
|
664
|
+
print(json.dumps({k: s[k] for k in ("consistency", "statuses")}))
|
|
665
|
+
print(
|
|
666
|
+
f"exported: {len(s['decision_rules'])} decision rules, {len(s['identities'])} identities, {len(s['literals'])} literals, {len(s['transitions'])} transitions"
|
|
667
|
+
)
|
|
668
|
+
if args.attach:
|
|
669
|
+
art_path = args.run / "diffgenome-change.json"
|
|
670
|
+
art = json.loads(art_path.read_text())
|
|
671
|
+
art["genome"] = s
|
|
672
|
+
art_path.write_text(json.dumps(art, indent=1) + "\n")
|
|
673
|
+
print(f"attached to {art_path}")
|
|
674
|
+
return 0
|