diffgenome 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- diffgenome/__init__.py +7 -0
- diffgenome/__main__.py +240 -0
- diffgenome/_collectors/go/dg/dg.go +623 -0
- diffgenome/_collectors/go/go.mod +3 -0
- diffgenome/_collectors/go/instrument/facts.go +346 -0
- diffgenome/_collectors/go/instrument/main.go +484 -0
- diffgenome/_collectors/node/instrument.js +289 -0
- diffgenome/_collectors/node/jest-setup.js +40 -0
- diffgenome/_collectors/node/package-lock.json +35 -0
- diffgenome/_collectors/node/package.json +11 -0
- diffgenome/_collectors/node/runtime.js +426 -0
- diffgenome/ambiguity.py +122 -0
- diffgenome/api.py +67 -0
- diffgenome/change.py +86 -0
- diffgenome/change_artifact.py +310 -0
- diffgenome/collect/__init__.py +2 -0
- diffgenome/collect/go_test.py +271 -0
- diffgenome/collect/node_jest.py +319 -0
- diffgenome/collect/py_monitoring.py +985 -0
- diffgenome/collect/py_runtime.py +116 -0
- diffgenome/collect/py_symbols.py +238 -0
- diffgenome/collect/pytest_plugin.py +130 -0
- diffgenome/compose.py +469 -0
- diffgenome/dependence.py +264 -0
- diffgenome/evaluate.py +669 -0
- diffgenome/frontends/__init__.py +0 -0
- diffgenome/frontends/python_ir.py +335 -0
- diffgenome/genome.py +1016 -0
- diffgenome/genome_pipeline.py +674 -0
- diffgenome/genome_prompt.py +33 -0
- diffgenome/genome_state.py +2118 -0
- diffgenome/graph.py +426 -0
- diffgenome/llm.py +189 -0
- diffgenome/model.py +364 -0
- diffgenome/mvp.py +398 -0
- diffgenome/probe.py +509 -0
- diffgenome/projection.py +308 -0
- diffgenome/py.typed +0 -0
- diffgenome/render.py +118 -0
- diffgenome/report.py +363 -0
- diffgenome/resolve.py +37 -0
- diffgenome/runtime.py +74 -0
- diffgenome/runtime_evidence.py +261 -0
- diffgenome/sandbox.py +166 -0
- diffgenome/serialize.py +96 -0
- diffgenome/sites.py +19 -0
- diffgenome/static_types.py +69 -0
- diffgenome/structure.py +462 -0
- diffgenome-0.1.0.dist-info/METADATA +139 -0
- diffgenome-0.1.0.dist-info/RECORD +53 -0
- diffgenome-0.1.0.dist-info/WHEEL +4 -0
- diffgenome-0.1.0.dist-info/entry_points.txt +2 -0
- diffgenome-0.1.0.dist-info/licenses/LICENSE +202 -0
diffgenome/probe.py
ADDED
|
@@ -0,0 +1,509 @@
|
|
|
1
|
+
"""Map-driven probe generation: pick a deficit, ask for one probe, run it confined, verify.
|
|
2
|
+
|
|
3
|
+
Language-neutral above the `SourceContext` adapter (which reads Python source) and the
|
|
4
|
+
runner command (pytest). The verdict on a probe comes from its trace, never from the model:
|
|
5
|
+
it passed, it executed the intended target as a real call, it attempted no egress, and it
|
|
6
|
+
changed the map.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import shutil
|
|
12
|
+
from dataclasses import dataclass, field
|
|
13
|
+
from pathlib import Path
|
|
14
|
+
|
|
15
|
+
from diffgenome.compose import Corpus, JoinAttempt, build_corpus
|
|
16
|
+
from diffgenome.graph import BehavioralGraph, GraphEdge, Metrics, Neighborhood, build_graph
|
|
17
|
+
from diffgenome.llm import ProbeDraft, ProbeRequest, ProbeWriter
|
|
18
|
+
from diffgenome.model import (
|
|
19
|
+
ArgShapes,
|
|
20
|
+
CallNode,
|
|
21
|
+
EvidenceKind,
|
|
22
|
+
Execution,
|
|
23
|
+
JoinStrength,
|
|
24
|
+
NodeRef,
|
|
25
|
+
Origin,
|
|
26
|
+
OsEventNode,
|
|
27
|
+
StateFacts,
|
|
28
|
+
Stimulus,
|
|
29
|
+
SubstitutionNode,
|
|
30
|
+
SymbolId,
|
|
31
|
+
)
|
|
32
|
+
from diffgenome.runtime import RuntimeAdapter, SymbolIndex
|
|
33
|
+
from diffgenome.sandbox import Workspace
|
|
34
|
+
|
|
35
|
+
CONSTRAINTS = """- The probe is a unit-level stimulus, not an integration test.
|
|
36
|
+
- Real in-repo code must execute; genuine external dependencies (network, model/weight
|
|
37
|
+
files, GPU, subprocesses, real filesystem outside tmp_path, clocks/timers) stay substituted.
|
|
38
|
+
- Never contact a real service. The environment has no network; an attempt fails the probe.
|
|
39
|
+
- No long sleeps, no background threads left running, no writes outside tmp_path.
|
|
40
|
+
- One test function is enough. Make it deterministic.
|
|
41
|
+
- Do not modify repository files; the probe is a new file only."""
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
@dataclass
|
|
45
|
+
class ProbeObjective:
|
|
46
|
+
kind: str # "uncovered_symbol" | "internal_gap" | "weak_join" | "state_condition"
|
|
47
|
+
target: SymbolId
|
|
48
|
+
caller: SymbolId
|
|
49
|
+
site: NodeRef
|
|
50
|
+
distance: int
|
|
51
|
+
seam_args: ArgShapes
|
|
52
|
+
seam_outcome: str
|
|
53
|
+
current_join: JoinStrength | None
|
|
54
|
+
supporting_executions: int
|
|
55
|
+
seam_state: StateFacts = () # state facts the seam exposed (state_condition objectives)
|
|
56
|
+
state_conflicts: tuple[str, ...] = () # "self.enabled bool:true≠bool:false" per rejection
|
|
57
|
+
|
|
58
|
+
def describe(self) -> str:
|
|
59
|
+
if self.kind == "uncovered_symbol":
|
|
60
|
+
return (
|
|
61
|
+
f"No existing execution reaches `{self.target}` at all (it is part of the "
|
|
62
|
+
f"change). Write a probe that executes the real `{self.target}` directly. "
|
|
63
|
+
"Substitute its direct in-repo collaborators with stand-ins that keep their "
|
|
64
|
+
"identity (unittest.mock.patch(..., autospec=True) or jest.spyOn) so the seams "
|
|
65
|
+
"are recorded, and substitute any external dependency."
|
|
66
|
+
)
|
|
67
|
+
if self.kind == "state_condition":
|
|
68
|
+
cond = ", ".join(f"{n} is {b}" for n, b, _ in self.seam_state[:6])
|
|
69
|
+
why = "; ".join(self.state_conflicts[:3])
|
|
70
|
+
return (
|
|
71
|
+
f"`{self.caller}` reaches `{self.target}` through a stand-in while the receiver "
|
|
72
|
+
f"or module state is: {cond}. Every existing execution of the real "
|
|
73
|
+
f"`{self.target}` ran under a different state ({why}), so no observed "
|
|
74
|
+
f"continuation is valid for this seam. Write a probe that executes the real "
|
|
75
|
+
f"`{self.target}` with matching arguments ({_fmt_args(self.seam_args)}) and "
|
|
76
|
+
f"under that same state condition, with its external dependencies substituted."
|
|
77
|
+
)
|
|
78
|
+
if self.kind == "internal_gap":
|
|
79
|
+
return (
|
|
80
|
+
f"No existing test executes `{self.target}` for real. `{self.caller}` reaches it "
|
|
81
|
+
f"only through a stand-in (called with {_fmt_args(self.seam_args)}, which "
|
|
82
|
+
f"{self.seam_outcome}). Write a probe that executes the real `{self.target}` "
|
|
83
|
+
f"from a comparable entry, with its own external dependencies substituted."
|
|
84
|
+
)
|
|
85
|
+
return (
|
|
86
|
+
f"`{self.caller}` reaches `{self.target}` through a stand-in (called with "
|
|
87
|
+
f"{_fmt_args(self.seam_args)}); existing executions of the real `{self.target}` only "
|
|
88
|
+
f"match at {self.current_join.name if self.current_join else 'no'} strength. Write a "
|
|
89
|
+
f"probe that executes the real `{self.target}` with arguments matching that call, so "
|
|
90
|
+
f"the seam can be matched on values."
|
|
91
|
+
)
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def _fmt_args(args: ArgShapes) -> str:
|
|
95
|
+
return "(" + ", ".join(f"{n}: {s}" for n, s, _ in args) + ")" if args else "no arguments"
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def select_objectives(
|
|
99
|
+
nb: Neighborhood,
|
|
100
|
+
graph: BehavioralGraph,
|
|
101
|
+
limit: int,
|
|
102
|
+
index: SymbolIndex | None = None,
|
|
103
|
+
skipped: list[str] | None = None,
|
|
104
|
+
) -> list[ProbeObjective]:
|
|
105
|
+
"""Deterministic priority: gaps before weak joins, nearer the change first, then the
|
|
106
|
+
seams with the most upstream support. Targets must be repository code that a probe can
|
|
107
|
+
make *execute*: a class with no in-repo __init__ never appears as a call, so a gap on
|
|
108
|
+
it is recorded as not probeable rather than attempted."""
|
|
109
|
+
out: list[ProbeObjective] = []
|
|
110
|
+
seen: set[SymbolId] = set()
|
|
111
|
+
|
|
112
|
+
def objective(kind: str, d: int, e: GraphEdge) -> ProbeObjective | None:
|
|
113
|
+
if graph.origin(e.callee) is not Origin.REPO or e.callee in seen:
|
|
114
|
+
return None
|
|
115
|
+
sym = graph.symbols.get(e.callee)
|
|
116
|
+
is_declaration = sym is not None and sym.kind == "declaration"
|
|
117
|
+
if index is not None and not is_declaration:
|
|
118
|
+
d_ = index.find(e.callee)
|
|
119
|
+
is_declaration = d_ is not None and d_.kind == "class"
|
|
120
|
+
if is_declaration:
|
|
121
|
+
seen.add(e.callee)
|
|
122
|
+
if skipped is not None:
|
|
123
|
+
skipped.append(f"{e.callee}: declaration with no in-repo executable body")
|
|
124
|
+
return None
|
|
125
|
+
ev = e.evidence[0]
|
|
126
|
+
assert graph.corpus is not None
|
|
127
|
+
ex = graph.corpus.executions[ev.site.execution]
|
|
128
|
+
node = ex.nodes[ev.site.node]
|
|
129
|
+
args: ArgShapes = node.args if isinstance(node, SubstitutionNode | CallNode) else ()
|
|
130
|
+
outcome = getattr(node, "outcome", "unknown")
|
|
131
|
+
seen.add(e.callee)
|
|
132
|
+
return ProbeObjective(
|
|
133
|
+
kind, e.callee, e.caller, ev.site, d, args, outcome, e.best_join, len(e.executions)
|
|
134
|
+
)
|
|
135
|
+
|
|
136
|
+
# Changed symbols nothing executes: the change itself is dark. Distance 0.
|
|
137
|
+
for seed in nb.seeds:
|
|
138
|
+
if seed in graph.tests_by_symbol or seed in seen or graph.origin(seed) is Origin.TEST:
|
|
139
|
+
continue
|
|
140
|
+
if index is not None:
|
|
141
|
+
d_ = index.find(seed)
|
|
142
|
+
if d_ is None or d_.kind == "class":
|
|
143
|
+
continue
|
|
144
|
+
seen.add(seed)
|
|
145
|
+
out.append(
|
|
146
|
+
ProbeObjective(
|
|
147
|
+
"uncovered_symbol", seed, seed, NodeRef("", 0), 0, (), "unknown", None, 0
|
|
148
|
+
)
|
|
149
|
+
)
|
|
150
|
+
if len(out) >= limit:
|
|
151
|
+
return out
|
|
152
|
+
gaps = sorted(
|
|
153
|
+
nb.edges_of(EvidenceKind.INTERNAL_GAP), key=lambda x: (x[0], -len(x[1].executions))
|
|
154
|
+
)
|
|
155
|
+
weak = sorted(
|
|
156
|
+
((d, e) for d, e in nb.edges_of(EvidenceKind.COMPOSED) if not e.strong),
|
|
157
|
+
key=lambda x: (x[0], -len(x[1].executions)),
|
|
158
|
+
)
|
|
159
|
+
|
|
160
|
+
def state_rejections(callee: SymbolId) -> list[JoinAttempt]:
|
|
161
|
+
return [
|
|
162
|
+
a
|
|
163
|
+
for a in graph.attempts
|
|
164
|
+
if a.target == callee and not a.accepted and "state conflict" in a.note
|
|
165
|
+
]
|
|
166
|
+
|
|
167
|
+
for kind, edges in (("internal_gap", gaps), ("weak_join", weak)):
|
|
168
|
+
for d, e in edges:
|
|
169
|
+
rejected = state_rejections(e.callee) if kind == "internal_gap" else []
|
|
170
|
+
# The real target ran, but every candidate fragment was rejected on state alone:
|
|
171
|
+
# not a coverage gap but a state gap. "Execute X under state condition Y."
|
|
172
|
+
actual_kind = (
|
|
173
|
+
"state_condition" if rejected and e.callee in graph.tests_by_symbol else kind
|
|
174
|
+
)
|
|
175
|
+
o = objective(actual_kind, d, e)
|
|
176
|
+
if o:
|
|
177
|
+
if actual_kind == "state_condition":
|
|
178
|
+
assert graph.corpus is not None
|
|
179
|
+
node = graph.corpus.executions[o.site.execution].nodes[o.site.node]
|
|
180
|
+
o.seam_state = getattr(node, "state", ())
|
|
181
|
+
o.state_conflicts = tuple(
|
|
182
|
+
sorted(
|
|
183
|
+
{a.note.split("state conflict: ", 1)[1].split(";")[0] for a in rejected}
|
|
184
|
+
)
|
|
185
|
+
)
|
|
186
|
+
out.append(o)
|
|
187
|
+
if len(out) >= limit:
|
|
188
|
+
return out
|
|
189
|
+
return out
|
|
190
|
+
# Seams every candidate fragment was rejected at on state alone: the real target ran,
|
|
191
|
+
# but never under the state the seam exposes. "Execute X under state condition Y."
|
|
192
|
+
for d, e in gaps:
|
|
193
|
+
if e.callee in seen or e.callee not in graph.tests_by_symbol:
|
|
194
|
+
continue
|
|
195
|
+
rejected = [
|
|
196
|
+
a
|
|
197
|
+
for a in graph.attempts
|
|
198
|
+
if a.target == e.callee and not a.accepted and "state conflict" in a.note
|
|
199
|
+
]
|
|
200
|
+
if not rejected:
|
|
201
|
+
continue
|
|
202
|
+
o = objective("state_condition", d, e)
|
|
203
|
+
if o is None:
|
|
204
|
+
continue
|
|
205
|
+
assert graph.corpus is not None
|
|
206
|
+
node = graph.corpus.executions[o.site.execution].nodes[o.site.node]
|
|
207
|
+
o.seam_state = getattr(node, "state", ())
|
|
208
|
+
o.state_conflicts = tuple(
|
|
209
|
+
sorted({a.note.split("state conflict: ", 1)[1].split(";")[0] for a in rejected})
|
|
210
|
+
)
|
|
211
|
+
out.append(o)
|
|
212
|
+
if len(out) >= limit:
|
|
213
|
+
return out
|
|
214
|
+
return out
|
|
215
|
+
|
|
216
|
+
|
|
217
|
+
# --------------------------------------------------------------------------- context
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
class SourceContext:
|
|
221
|
+
"""Bounded probe context from the symbol index and the graph. Runtime-neutral: it asks
|
|
222
|
+
the index for sources and modules and the runtime adapter for conventions."""
|
|
223
|
+
|
|
224
|
+
def __init__(self, index: SymbolIndex, graph: BehavioralGraph) -> None:
|
|
225
|
+
self.index = index
|
|
226
|
+
self.graph = graph
|
|
227
|
+
|
|
228
|
+
def build(self, o: ProbeObjective) -> str:
|
|
229
|
+
parts: list[str] = []
|
|
230
|
+
target_def = self.index.find(o.target)
|
|
231
|
+
if target_def:
|
|
232
|
+
parts.append(
|
|
233
|
+
f"## Target `{o.target}` ({target_def.path}:{target_def.start}-{target_def.end})"
|
|
234
|
+
)
|
|
235
|
+
parts.append(_clip(self.index.source(o.target) or "", 120))
|
|
236
|
+
cls = self.index.enclosing_class(o.target)
|
|
237
|
+
if cls:
|
|
238
|
+
# The whole class when it is small: the target usually delegates to
|
|
239
|
+
# siblings (private helpers, constructors) the probe must understand.
|
|
240
|
+
parts.append(f"## Enclosing class `{cls.symbol}`")
|
|
241
|
+
parts.append(_clip(self.index.source(cls.symbol) or "", 160))
|
|
242
|
+
parts.append(f"## Imports of {target_def.path}")
|
|
243
|
+
parts.append(_module_head(self.index.repo_root / target_def.path))
|
|
244
|
+
else:
|
|
245
|
+
parts.append(f"## Target `{o.target}` (source not located)")
|
|
246
|
+
if o.kind == "uncovered_symbol":
|
|
247
|
+
parts.append("## Runtime evidence about the target")
|
|
248
|
+
parts.append("- never executed by any existing test or probe")
|
|
249
|
+
mod = self.index.module_of(o.target) or ""
|
|
250
|
+
for tf in self.index.tests_importing(mod, limit=2) if mod else []:
|
|
251
|
+
parts.append(f"## Existing test file {tf} (conventions; first lines)")
|
|
252
|
+
text = (self.index.repo_root / tf).read_text(encoding="utf-8")
|
|
253
|
+
parts.append(_clip(_numbered(text), 70))
|
|
254
|
+
return "\n\n".join(parts)
|
|
255
|
+
parts.append(f"## Caller `{o.caller}` (reaches the target through a stand-in)")
|
|
256
|
+
parts.append(_clip(self.index.source(o.caller) or "(source not located)", 80))
|
|
257
|
+
assert self.graph.corpus is not None
|
|
258
|
+
seed = self.graph.corpus.executions[o.site.execution]
|
|
259
|
+
parts.append(f"## The existing test that produced this seam: {seed.stimulus_ref}")
|
|
260
|
+
test_source = None
|
|
261
|
+
for n in seed.nodes: # the root, or the first test-origin call under it
|
|
262
|
+
if isinstance(n, CallNode) and self.graph.origin(n.symbol) is Origin.TEST:
|
|
263
|
+
test_source = self.index.source(n.symbol)
|
|
264
|
+
if test_source:
|
|
265
|
+
break
|
|
266
|
+
parts.append(_clip(test_source or "(source not located)", 80))
|
|
267
|
+
subs = [n for n in seed.nodes if isinstance(n, SubstitutionNode)]
|
|
268
|
+
if subs:
|
|
269
|
+
parts.append(
|
|
270
|
+
"## Substitutions active in that test (keep the external ones substituted)"
|
|
271
|
+
)
|
|
272
|
+
for s in subs[:12]:
|
|
273
|
+
origin = (
|
|
274
|
+
self.graph.origin(s.claimed_target).value if s.claimed_target else "unknown"
|
|
275
|
+
)
|
|
276
|
+
path = ".".join(s.path) or "<call>"
|
|
277
|
+
parts.append(f"- {s.mechanism.value} .{path} claims={s.claimed_target} ({origin})")
|
|
278
|
+
parts.append("## Runtime evidence about the target")
|
|
279
|
+
outs = self.graph.outcomes_by_symbol.get(o.target)
|
|
280
|
+
parts.append(f"- outcomes observed so far: {dict(outs) if outs else 'never executed'}")
|
|
281
|
+
for e in self.graph.out.get(o.target, [])[:12]:
|
|
282
|
+
parts.append(f"- {o.target.split('.')[-1]} → {e.callee} [{e.kind.value}]")
|
|
283
|
+
module = self.index.module_of(o.target) if target_def else None
|
|
284
|
+
for tf in self.index.tests_importing(module, limit=2) if module else []:
|
|
285
|
+
parts.append(f"## Existing test file {tf} (conventions; first lines)")
|
|
286
|
+
parts.append(
|
|
287
|
+
_clip(_numbered((self.index.repo_root / tf).read_text(encoding="utf-8")), 70)
|
|
288
|
+
)
|
|
289
|
+
return "\n\n".join(parts)
|
|
290
|
+
|
|
291
|
+
|
|
292
|
+
def _numbered(text: str) -> str:
|
|
293
|
+
return "\n".join(f"{i + 1:5d} {line}" for i, line in enumerate(text.splitlines()))
|
|
294
|
+
|
|
295
|
+
|
|
296
|
+
def _module_head(path: Path) -> str:
|
|
297
|
+
try:
|
|
298
|
+
lines = path.read_text(encoding="utf-8").splitlines()
|
|
299
|
+
except OSError:
|
|
300
|
+
return ""
|
|
301
|
+
head = [
|
|
302
|
+
line
|
|
303
|
+
for line in lines[:60]
|
|
304
|
+
if line.startswith(("import ", "from ")) or (line.startswith("_") and "=" in line)
|
|
305
|
+
]
|
|
306
|
+
return "\n".join(head[:30])
|
|
307
|
+
|
|
308
|
+
|
|
309
|
+
def _clip(text: str, max_lines: int) -> str:
|
|
310
|
+
lines = text.splitlines()
|
|
311
|
+
return "\n".join(lines[:max_lines]) + (
|
|
312
|
+
f"\n... ({len(lines) - max_lines} more lines)" if len(lines) > max_lines else ""
|
|
313
|
+
)
|
|
314
|
+
|
|
315
|
+
|
|
316
|
+
# --------------------------------------------------------------------------- execution
|
|
317
|
+
|
|
318
|
+
|
|
319
|
+
@dataclass
|
|
320
|
+
class ProbeAttempt:
|
|
321
|
+
objective: ProbeObjective
|
|
322
|
+
attempt: int
|
|
323
|
+
draft: ProbeDraft | None
|
|
324
|
+
verdict: str # "accepted" | "rejected" | "error"
|
|
325
|
+
reasons: list[str] = field(default_factory=list)
|
|
326
|
+
executions: list[Execution] = field(default_factory=list)
|
|
327
|
+
metrics_before: Metrics | None = None
|
|
328
|
+
metrics_after: Metrics | None = None
|
|
329
|
+
learned: list[str] = field(default_factory=list)
|
|
330
|
+
|
|
331
|
+
|
|
332
|
+
@dataclass
|
|
333
|
+
class ProbeRunner:
|
|
334
|
+
workspace: Workspace
|
|
335
|
+
runtime: RuntimeAdapter
|
|
336
|
+
|
|
337
|
+
def run(self, draft: ProbeDraft, tag: str, out_dir: Path) -> tuple[list[Execution], str, str]:
|
|
338
|
+
rel = self.runtime.probe_relpath(tag)
|
|
339
|
+
probe_file = self.workspace.repo / rel
|
|
340
|
+
probe_file.parent.mkdir(parents=True, exist_ok=True)
|
|
341
|
+
probe_file.write_text(draft.code)
|
|
342
|
+
try:
|
|
343
|
+
executions, stdout, stderr = self.runtime.trace(
|
|
344
|
+
self.workspace, out_dir / f"traces-{tag}", Stimulus.GENERATED_PROBE, [rel]
|
|
345
|
+
)
|
|
346
|
+
finally:
|
|
347
|
+
probe_file.unlink(missing_ok=True)
|
|
348
|
+
return executions, stdout, stderr
|
|
349
|
+
|
|
350
|
+
|
|
351
|
+
def verify(
|
|
352
|
+
o: ProbeObjective,
|
|
353
|
+
executions: list[Execution],
|
|
354
|
+
stdout: str,
|
|
355
|
+
stderr: str,
|
|
356
|
+
corpus: Corpus,
|
|
357
|
+
seeds: list[SymbolId],
|
|
358
|
+
before: Metrics,
|
|
359
|
+
up: int,
|
|
360
|
+
down: int,
|
|
361
|
+
) -> tuple[str, list[str], Metrics | None, list[str], Corpus | None]:
|
|
362
|
+
"""Deterministic verdict. Returns (verdict, reasons, metrics_after, learned, new_corpus)."""
|
|
363
|
+
reasons: list[str] = []
|
|
364
|
+
if not executions:
|
|
365
|
+
tail = (stderr or stdout).strip().splitlines()[-8:]
|
|
366
|
+
return (
|
|
367
|
+
"error",
|
|
368
|
+
["probe produced no executions (collection or import error)", *tail],
|
|
369
|
+
None,
|
|
370
|
+
[],
|
|
371
|
+
None,
|
|
372
|
+
)
|
|
373
|
+
failed = [e for e in executions if e.outcome != "passed"]
|
|
374
|
+
if failed:
|
|
375
|
+
text = stdout if "failed" in stdout or "FAIL" in stdout else stderr
|
|
376
|
+
tail = [line for line in text.strip().splitlines() if line.strip()][-15:]
|
|
377
|
+
reasons.append(f"{len(failed)} of {len(executions)} probe tests failed")
|
|
378
|
+
reasons.extend(tail)
|
|
379
|
+
egress = [
|
|
380
|
+
(e.stimulus_ref, n.target)
|
|
381
|
+
for e in executions
|
|
382
|
+
for n in e.nodes
|
|
383
|
+
if isinstance(n, OsEventNode)
|
|
384
|
+
]
|
|
385
|
+
if egress:
|
|
386
|
+
reasons.append("egress attempted: " + ", ".join(t for _, t in egress[:5]))
|
|
387
|
+
ran_target = any(
|
|
388
|
+
isinstance(n, CallNode) and n.symbol == o.target for e in executions for n in e.nodes
|
|
389
|
+
)
|
|
390
|
+
if not ran_target:
|
|
391
|
+
reasons.append(f"target {o.target} did not execute as a real call")
|
|
392
|
+
if reasons:
|
|
393
|
+
return "rejected", reasons, None, [], None
|
|
394
|
+
new_corpus = build_corpus(list(corpus.executions.values()) + executions)
|
|
395
|
+
graph = build_graph(new_corpus)
|
|
396
|
+
after = graph.neighborhood(seeds, up=up, down=down).metrics()
|
|
397
|
+
learned: list[str] = []
|
|
398
|
+
for e in executions:
|
|
399
|
+
for n in e.nodes:
|
|
400
|
+
if isinstance(n, CallNode) and n.id != 0 and n.parent is not None:
|
|
401
|
+
p = e.nodes[n.parent]
|
|
402
|
+
if isinstance(p, CallNode):
|
|
403
|
+
learned.append(f"{p.symbol} → {n.symbol} [{n.outcome}]")
|
|
404
|
+
learned = sorted(set(learned))
|
|
405
|
+
improved = (
|
|
406
|
+
after.internal_gaps < before.internal_gaps
|
|
407
|
+
or after.strong_joins > before.strong_joins
|
|
408
|
+
or after.composed > before.composed # a new continuation, even a weak one, is evidence
|
|
409
|
+
or after.observed > before.observed
|
|
410
|
+
)
|
|
411
|
+
if not improved:
|
|
412
|
+
return (
|
|
413
|
+
"rejected",
|
|
414
|
+
["probe ran the target but added no evidence to the neighborhood"],
|
|
415
|
+
after,
|
|
416
|
+
learned,
|
|
417
|
+
None,
|
|
418
|
+
)
|
|
419
|
+
return "accepted", [], after, learned, new_corpus
|
|
420
|
+
|
|
421
|
+
|
|
422
|
+
def apply_budget_policy(
|
|
423
|
+
objectives: list[ProbeObjective],
|
|
424
|
+
max_objectives: int,
|
|
425
|
+
max_distance: int | None,
|
|
426
|
+
skipped: list[str] | None = None,
|
|
427
|
+
) -> list[ProbeObjective]:
|
|
428
|
+
"""Keep objectives within `max_distance` hops of a changed symbol, in priority order,
|
|
429
|
+
up to `max_objectives`. What the policy excludes is reported, never silently dropped."""
|
|
430
|
+
if max_distance is not None:
|
|
431
|
+
far = [o for o in objectives if o.distance > max_distance]
|
|
432
|
+
objectives = [o for o in objectives if o.distance <= max_distance]
|
|
433
|
+
if skipped is not None:
|
|
434
|
+
skipped.extend(
|
|
435
|
+
f"{o.target}: {o.kind} at distance {o.distance} > budget policy {max_distance}"
|
|
436
|
+
for o in far
|
|
437
|
+
)
|
|
438
|
+
return objectives[:max_objectives]
|
|
439
|
+
|
|
440
|
+
|
|
441
|
+
def run_probe_loop(
|
|
442
|
+
graph: BehavioralGraph,
|
|
443
|
+
neighborhood: Neighborhood,
|
|
444
|
+
index: SymbolIndex,
|
|
445
|
+
writer: ProbeWriter,
|
|
446
|
+
runner: ProbeRunner,
|
|
447
|
+
out_dir: Path,
|
|
448
|
+
max_objectives: int,
|
|
449
|
+
max_attempts: int,
|
|
450
|
+
up: int,
|
|
451
|
+
down: int,
|
|
452
|
+
skipped: list[str] | None = None,
|
|
453
|
+
max_distance: int | None = None,
|
|
454
|
+
) -> tuple[list[ProbeAttempt], Corpus]:
|
|
455
|
+
"""`max_distance` is the integrated-mode budget policy: only objectives within that
|
|
456
|
+
many hops of a changed symbol are attempted (0: changed symbols nothing executes;
|
|
457
|
+
1: their direct callers and callees). Farther ones are listed in `skipped`, not
|
|
458
|
+
silently dropped."""
|
|
459
|
+
out_dir.mkdir(parents=True, exist_ok=True)
|
|
460
|
+
assert graph.corpus is not None
|
|
461
|
+
corpus = graph.corpus
|
|
462
|
+
seeds = list(neighborhood.seeds)
|
|
463
|
+
attempts: list[ProbeAttempt] = []
|
|
464
|
+
objectives = apply_budget_policy(
|
|
465
|
+
select_objectives(neighborhood, graph, max_objectives * 4, index, skipped),
|
|
466
|
+
max_objectives,
|
|
467
|
+
max_distance,
|
|
468
|
+
skipped,
|
|
469
|
+
)
|
|
470
|
+
for i, o in enumerate(objectives):
|
|
471
|
+
failures: list[str] = []
|
|
472
|
+
current_graph = build_graph(corpus)
|
|
473
|
+
before = current_graph.neighborhood(seeds, up=up, down=down).metrics()
|
|
474
|
+
context = SourceContext(index, current_graph).build(o)
|
|
475
|
+
conventions = runner.runtime.conventions(index)
|
|
476
|
+
for attempt in range(1, max_attempts + 1):
|
|
477
|
+
tag = f"{i}_{attempt}"
|
|
478
|
+
request = ProbeRequest(o.describe(), context, CONSTRAINTS, conventions, failures)
|
|
479
|
+
try:
|
|
480
|
+
draft = writer.write(request)
|
|
481
|
+
except Exception as exc: # the writer is an external service
|
|
482
|
+
attempts.append(ProbeAttempt(o, attempt, None, "error", [f"writer error: {exc}"]))
|
|
483
|
+
break
|
|
484
|
+
(out_dir / f"probe-{tag}.py").write_text(draft.code)
|
|
485
|
+
try:
|
|
486
|
+
executions, stdout, stderr = runner.run(draft, tag, out_dir)
|
|
487
|
+
(out_dir / f"probe-{tag}.log").write_text(stdout + "\n--- stderr ---\n" + stderr)
|
|
488
|
+
except Exception as exc:
|
|
489
|
+
attempts.append(ProbeAttempt(o, attempt, draft, "error", [f"runner error: {exc}"]))
|
|
490
|
+
break
|
|
491
|
+
verdict, reasons, after, learned, new_corpus = verify(
|
|
492
|
+
o, executions, stdout, stderr, corpus, seeds, before, up, down
|
|
493
|
+
)
|
|
494
|
+
attempts.append(
|
|
495
|
+
ProbeAttempt(
|
|
496
|
+
o, attempt, draft, verdict, reasons, executions, before, after, learned
|
|
497
|
+
)
|
|
498
|
+
)
|
|
499
|
+
if verdict == "accepted" and new_corpus is not None:
|
|
500
|
+
corpus = new_corpus
|
|
501
|
+
break
|
|
502
|
+
failures.append(f"attempt {attempt}: " + "; ".join(reasons[:6]))
|
|
503
|
+
return attempts, corpus
|
|
504
|
+
|
|
505
|
+
|
|
506
|
+
def copy_probe_traces(src: Path, dst: Path) -> None:
|
|
507
|
+
dst.mkdir(parents=True, exist_ok=True)
|
|
508
|
+
for f in src.glob("*.json"):
|
|
509
|
+
shutil.copy(f, dst / f.name)
|