diffgenome 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- diffgenome/__init__.py +7 -0
- diffgenome/__main__.py +240 -0
- diffgenome/_collectors/go/dg/dg.go +623 -0
- diffgenome/_collectors/go/go.mod +3 -0
- diffgenome/_collectors/go/instrument/facts.go +346 -0
- diffgenome/_collectors/go/instrument/main.go +484 -0
- diffgenome/_collectors/node/instrument.js +289 -0
- diffgenome/_collectors/node/jest-setup.js +40 -0
- diffgenome/_collectors/node/package-lock.json +35 -0
- diffgenome/_collectors/node/package.json +11 -0
- diffgenome/_collectors/node/runtime.js +426 -0
- diffgenome/ambiguity.py +122 -0
- diffgenome/api.py +67 -0
- diffgenome/change.py +86 -0
- diffgenome/change_artifact.py +310 -0
- diffgenome/collect/__init__.py +2 -0
- diffgenome/collect/go_test.py +271 -0
- diffgenome/collect/node_jest.py +319 -0
- diffgenome/collect/py_monitoring.py +985 -0
- diffgenome/collect/py_runtime.py +116 -0
- diffgenome/collect/py_symbols.py +238 -0
- diffgenome/collect/pytest_plugin.py +130 -0
- diffgenome/compose.py +469 -0
- diffgenome/dependence.py +264 -0
- diffgenome/evaluate.py +669 -0
- diffgenome/frontends/__init__.py +0 -0
- diffgenome/frontends/python_ir.py +335 -0
- diffgenome/genome.py +1016 -0
- diffgenome/genome_pipeline.py +674 -0
- diffgenome/genome_prompt.py +33 -0
- diffgenome/genome_state.py +2118 -0
- diffgenome/graph.py +426 -0
- diffgenome/llm.py +189 -0
- diffgenome/model.py +364 -0
- diffgenome/mvp.py +398 -0
- diffgenome/probe.py +509 -0
- diffgenome/projection.py +308 -0
- diffgenome/py.typed +0 -0
- diffgenome/render.py +118 -0
- diffgenome/report.py +363 -0
- diffgenome/resolve.py +37 -0
- diffgenome/runtime.py +74 -0
- diffgenome/runtime_evidence.py +261 -0
- diffgenome/sandbox.py +166 -0
- diffgenome/serialize.py +96 -0
- diffgenome/sites.py +19 -0
- diffgenome/static_types.py +69 -0
- diffgenome/structure.py +462 -0
- diffgenome-0.1.0.dist-info/METADATA +139 -0
- diffgenome-0.1.0.dist-info/RECORD +53 -0
- diffgenome-0.1.0.dist-info/WHEEL +4 -0
- diffgenome-0.1.0.dist-info/entry_points.txt +2 -0
- diffgenome-0.1.0.dist-info/licenses/LICENSE +202 -0
diffgenome/compose.py
ADDED
|
@@ -0,0 +1,469 @@
|
|
|
1
|
+
"""Composition: join independently observed execution fragments at internal boundaries.
|
|
2
|
+
|
|
3
|
+
Language-agnostic. Input is a corpus of `Execution`s; output is a tree of `Branch`es
|
|
4
|
+
rooted at one seed execution, where every branch carries an `Edge` with explicit
|
|
5
|
+
`Evidence`, plus the join attempts and gaps that the expansion produced. Fragments are
|
|
6
|
+
never flattened into paths: alternatives for one seam are sibling branches.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import re
|
|
12
|
+
from collections import defaultdict
|
|
13
|
+
from dataclasses import dataclass, field
|
|
14
|
+
|
|
15
|
+
from diffgenome.model import (
|
|
16
|
+
ArgShapes,
|
|
17
|
+
BoundaryClass,
|
|
18
|
+
CallNode,
|
|
19
|
+
Edge,
|
|
20
|
+
Evidence,
|
|
21
|
+
EvidenceKind,
|
|
22
|
+
Execution,
|
|
23
|
+
Fidelity,
|
|
24
|
+
IdentityMismatch,
|
|
25
|
+
JoinStrength,
|
|
26
|
+
Node,
|
|
27
|
+
NodeRef,
|
|
28
|
+
Origin,
|
|
29
|
+
OsEventNode,
|
|
30
|
+
Stimulus,
|
|
31
|
+
SubstitutionNode,
|
|
32
|
+
Symbol,
|
|
33
|
+
SymbolId,
|
|
34
|
+
)
|
|
35
|
+
from diffgenome.resolve import resolve
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
@dataclass
|
|
39
|
+
class Corpus:
|
|
40
|
+
executions: dict[str, Execution]
|
|
41
|
+
symbols: dict[SymbolId, Symbol]
|
|
42
|
+
fragments: dict[SymbolId, list[NodeRef]] # real calls by symbol, passing executions only
|
|
43
|
+
children: dict[str, dict[int, list[Node]]]
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def _merge_symbol(table: dict[SymbolId, Symbol], sym: Symbol, execution: str) -> None:
|
|
47
|
+
have = table.get(sym.id)
|
|
48
|
+
if have is None:
|
|
49
|
+
table[sym.id] = sym
|
|
50
|
+
return
|
|
51
|
+
known = {have.origin, sym.origin} - {Origin.UNKNOWN}
|
|
52
|
+
if len(known) > 1:
|
|
53
|
+
raise IdentityMismatch(
|
|
54
|
+
f"{sym.id}: origin {have.origin.value} vs {sym.origin.value} (in {execution})"
|
|
55
|
+
)
|
|
56
|
+
if have.location and sym.location and have.location != sym.location:
|
|
57
|
+
raise IdentityMismatch(f"{sym.id}: {have.location} vs {sym.location} (in {execution})")
|
|
58
|
+
if (
|
|
59
|
+
have.origin is Origin.UNKNOWN
|
|
60
|
+
or (have.location is None and sym.location)
|
|
61
|
+
or (have.kind == "callable" and sym.kind != "callable")
|
|
62
|
+
):
|
|
63
|
+
table[sym.id] = Symbol(
|
|
64
|
+
sym.id, sym.origin if have.origin is Origin.UNKNOWN else have.origin,
|
|
65
|
+
have.location or sym.location,
|
|
66
|
+
sym.kind if sym.kind != "callable" else have.kind,
|
|
67
|
+
) # fmt: skip
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def build_corpus(executions: list[Execution]) -> Corpus:
|
|
71
|
+
symbols: dict[SymbolId, Symbol] = {}
|
|
72
|
+
fragments: dict[SymbolId, list[NodeRef]] = defaultdict(list)
|
|
73
|
+
children: dict[str, dict[int, list[Node]]] = {}
|
|
74
|
+
by_id: dict[str, Execution] = {}
|
|
75
|
+
for ex in executions:
|
|
76
|
+
if ex.id in by_id:
|
|
77
|
+
raise ValueError(f"duplicate execution id {ex.id}")
|
|
78
|
+
by_id[ex.id] = ex
|
|
79
|
+
for sym in ex.symbols:
|
|
80
|
+
_merge_symbol(symbols, sym, ex.id)
|
|
81
|
+
kids: dict[int, list[Node]] = defaultdict(list)
|
|
82
|
+
for node in ex.nodes:
|
|
83
|
+
if node.parent is not None:
|
|
84
|
+
kids[node.parent].append(node)
|
|
85
|
+
if isinstance(node, CallNode) and node.id != 0 and ex.outcome == "passed":
|
|
86
|
+
fragments[node.symbol].append(NodeRef(ex.id, node.id))
|
|
87
|
+
children[ex.id] = kids
|
|
88
|
+
return Corpus(by_id, symbols, dict(fragments), children)
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
@dataclass
|
|
92
|
+
class JoinAttempt:
|
|
93
|
+
site: NodeRef
|
|
94
|
+
fragment: NodeRef
|
|
95
|
+
target: SymbolId
|
|
96
|
+
grade: JoinStrength | None # None: known unsound (exit/type/state conflict), never composable
|
|
97
|
+
accepted: bool
|
|
98
|
+
note: str = ""
|
|
99
|
+
result_compatible: bool | None = None
|
|
100
|
+
exit: str | None = None # exit compatibility: same | kind | unknown | None (conflict)
|
|
101
|
+
"""Whether the fragment returned what the stand-in returned (None: unknown). Entry
|
|
102
|
+
compatibility (the grade) says the fragment could continue from the seam; result
|
|
103
|
+
compatibility says the seed's continuation *after* the seam is supported. They are
|
|
104
|
+
independent, and only the first is a join grade."""
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
@dataclass
|
|
108
|
+
class Gap:
|
|
109
|
+
site: NodeRef
|
|
110
|
+
target: SymbolId
|
|
111
|
+
kind: str # "no-fragment" | "no-compatible-fragment" | "cycle" | "depth"
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
@dataclass
|
|
115
|
+
class Branch:
|
|
116
|
+
edge: Edge
|
|
117
|
+
children: list[Branch] = field(default_factory=list)
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
@dataclass
|
|
121
|
+
class Composition:
|
|
122
|
+
seed: str
|
|
123
|
+
root: SymbolId
|
|
124
|
+
branches: list[Branch]
|
|
125
|
+
attempts: list[JoinAttempt]
|
|
126
|
+
gaps: list[Gap]
|
|
127
|
+
|
|
128
|
+
def edges(self) -> list[Edge]:
|
|
129
|
+
out: list[Edge] = []
|
|
130
|
+
|
|
131
|
+
def walk(bs: list[Branch]) -> None:
|
|
132
|
+
for b in bs:
|
|
133
|
+
out.append(b.edge)
|
|
134
|
+
walk(b.children)
|
|
135
|
+
|
|
136
|
+
walk(self.branches)
|
|
137
|
+
return out
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def _shape_type(shape: str) -> str:
|
|
141
|
+
return shape.split("[", 1)[0]
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
_STRUCTURAL = (
|
|
145
|
+
"NoneType", "null", "undefined", "nil", "int", "float", "str", "bool", "bytes", "number",
|
|
146
|
+
"string", "boolean", "bigint", "list", "tuple", "dict", "set", "frozenset", "array", "object",
|
|
147
|
+
"int8", "int16", "int32", "int64", "uint", "uint8", "uint16", "uint32", "uint64", "float32",
|
|
148
|
+
"float64", "[]", "map[",
|
|
149
|
+
) # fmt: skip
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def _is_structural(shape: str) -> bool:
|
|
153
|
+
"""Primitive, None/nil or container shapes: their kind is the whole story. Anything else
|
|
154
|
+
is an object whose relationship to another object type (interface, subclass, duck
|
|
155
|
+
typing) the collector cannot see."""
|
|
156
|
+
t = _shape_type(shape)
|
|
157
|
+
for k in _STRUCTURAL:
|
|
158
|
+
if t == k or t.startswith(k + "[") or (k.endswith("[") and t.startswith(k)):
|
|
159
|
+
return True
|
|
160
|
+
return t.startswith("[]")
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
def _types_compatible(a: str, b: str) -> str | None:
|
|
164
|
+
"""None when compatible; "conflict" for a decisive structural mismatch (None vs dict is a
|
|
165
|
+
different branch); "object" when two object types differ but may be related (a
|
|
166
|
+
*gin.Context where a context.Context is declared), which is unverified, not wrong.
|
|
167
|
+
A stand-in argument compares as the type it stands in for; a spec-less stand-in is a
|
|
168
|
+
wildcard."""
|
|
169
|
+
ta, tb = _shape_type(a), _shape_type(b)
|
|
170
|
+
if ta == tb:
|
|
171
|
+
return None
|
|
172
|
+
for x, y in ((ta, tb), (tb, ta)):
|
|
173
|
+
if x == "stand-in":
|
|
174
|
+
return None
|
|
175
|
+
if x.startswith("stand-in:") and x[len("stand-in:") :] == y:
|
|
176
|
+
return None
|
|
177
|
+
if _is_structural(ta) or _is_structural(tb):
|
|
178
|
+
return "conflict"
|
|
179
|
+
return "object"
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
def fragment_shape(corpus: Corpus, ref: NodeRef) -> tuple[object, ...]:
|
|
183
|
+
"""Structural signature of a fragment subtree: symbols, kinds, outcomes, claims and
|
|
184
|
+
paths, but not argument values. Fragments with one shape are one behavior path and
|
|
185
|
+
are expanded once; their provenance is kept together (Evidence.alternates)."""
|
|
186
|
+
ex = corpus.executions[ref.execution]
|
|
187
|
+
|
|
188
|
+
def walk(node_id: int) -> tuple[object, ...]:
|
|
189
|
+
parts: list[object] = []
|
|
190
|
+
for k in corpus.children[ex.id].get(node_id, []):
|
|
191
|
+
if isinstance(k, CallNode):
|
|
192
|
+
parts.append(("c", k.symbol, k.outcome.split(":")[0], walk(k.id)))
|
|
193
|
+
elif isinstance(k, SubstitutionNode):
|
|
194
|
+
parts.append(("s", k.mechanism.value, k.claimed_target, k.path, walk(k.id)))
|
|
195
|
+
else:
|
|
196
|
+
parts.append(("o", k.kind.value, k.target))
|
|
197
|
+
return tuple(parts)
|
|
198
|
+
|
|
199
|
+
root = ex.nodes[ref.node]
|
|
200
|
+
assert isinstance(root, CallNode)
|
|
201
|
+
return (root.symbol, root.outcome.split(":")[0], walk(ref.node))
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
def outcome_parts(outcome: str) -> tuple[str, str | None]:
|
|
205
|
+
"""``category`` and ``identity`` of an Outcome string (see model.Outcome)."""
|
|
206
|
+
cat, _, ident = outcome.partition(":")
|
|
207
|
+
return cat, (ident or None)
|
|
208
|
+
|
|
209
|
+
|
|
210
|
+
def exit_compatibility(site_outcome: str, fragment_outcome: str) -> str | None:
|
|
211
|
+
"""EXIT half of join validity. ``same``: categories and known identities agree;
|
|
212
|
+
``kind``: categories agree and an identity is unknown; ``unknown``: an outcome was not
|
|
213
|
+
observed; None: conflict (different categories, or same category with two different
|
|
214
|
+
known identities: an error-returning seam is not a panicking fragment, and a seam that
|
|
215
|
+
caught ValueError is not a fragment that raised KeyError)."""
|
|
216
|
+
ca, ia = outcome_parts(site_outcome)
|
|
217
|
+
cb, ib = outcome_parts(fragment_outcome)
|
|
218
|
+
if ca == "unknown" or cb == "unknown":
|
|
219
|
+
return "unknown"
|
|
220
|
+
if ca != cb:
|
|
221
|
+
return None
|
|
222
|
+
if ia and ib:
|
|
223
|
+
return "same" if ia == ib else None
|
|
224
|
+
return "same" if (ia is None and ib is None) else "kind"
|
|
225
|
+
|
|
226
|
+
|
|
227
|
+
def outcomes_compatible(site: SubstitutionNode, fragment: CallNode) -> bool | None:
|
|
228
|
+
"""Legacy view of exit compatibility: None when unknown, False on conflict."""
|
|
229
|
+
e = exit_compatibility(site.outcome, fragment.outcome)
|
|
230
|
+
return None if e == "unknown" else e is not None
|
|
231
|
+
|
|
232
|
+
|
|
233
|
+
def state_compatibility(site: SubstitutionNode, fragment: CallNode) -> tuple[str, str]:
|
|
234
|
+
"""STATE rung. Facts are compared by name over the names both sides observed.
|
|
235
|
+
Returns (verdict, note): ``match`` (>=1 fact compared, all equal), ``conflict`` (a
|
|
236
|
+
compared bucket differs: a different branch condition), ``unavailable`` (no common
|
|
237
|
+
facts, or no facts on a side)."""
|
|
238
|
+
a = {name: (bucket, digest) for name, bucket, digest in site.state}
|
|
239
|
+
b = {name: (bucket, digest) for name, bucket, digest in fragment.state}
|
|
240
|
+
common = sorted(set(a) & set(b))
|
|
241
|
+
if not common:
|
|
242
|
+
return (
|
|
243
|
+
"unavailable",
|
|
244
|
+
"state unavailable on one side" if not a or not b else "no common state facts",
|
|
245
|
+
)
|
|
246
|
+
differ = [n for n in common if a[n] != b[n]]
|
|
247
|
+
if differ:
|
|
248
|
+
return "conflict", "state conflict: " + ", ".join(
|
|
249
|
+
f"{n} {a[n][0]}≠{b[n][0]}" for n in differ
|
|
250
|
+
)
|
|
251
|
+
return "match", f"state matched on {len(common)} fact(s): " + ", ".join(common[:6])
|
|
252
|
+
|
|
253
|
+
|
|
254
|
+
_PLACEHOLDER = re.compile(r"^(arg|_)\d+$|^\*")
|
|
255
|
+
|
|
256
|
+
|
|
257
|
+
def align_arguments(
|
|
258
|
+
site: ArgShapes, fragment: ArgShapes
|
|
259
|
+
) -> tuple[list[tuple[tuple[str, str, str], tuple[str, str, str]]], list[str]]:
|
|
260
|
+
"""Pair the seam's arguments with the fragment's. By name when both sides recorded
|
|
261
|
+
real parameter names (a keyword call at the seam and the signature order in the
|
|
262
|
+
fragment differ in order and in which defaults appear); by position otherwise
|
|
263
|
+
(placeholder names such as ``arg0``). Returns the pairs and the names present on
|
|
264
|
+
one side only (for positional pairing: the surplus positions)."""
|
|
265
|
+
named = all(not _PLACEHOLDER.match(n) for n, _, _ in site) and all(
|
|
266
|
+
not _PLACEHOLDER.match(n) for n, _, _ in fragment
|
|
267
|
+
)
|
|
268
|
+
if named:
|
|
269
|
+
by_name = {n: a for a in fragment for n in [a[0]]}
|
|
270
|
+
pairs = [(a, by_name[a[0]]) for a in site if a[0] in by_name]
|
|
271
|
+
site_names = {a[0] for a in site}
|
|
272
|
+
one_sided = sorted(({a[0] for a in site} - set(by_name)) | (set(by_name) - site_names))
|
|
273
|
+
if pairs:
|
|
274
|
+
return pairs, one_sided
|
|
275
|
+
pairs = list(zip(site, fragment, strict=False))
|
|
276
|
+
longer = site if len(site) > len(fragment) else fragment
|
|
277
|
+
return pairs, [a[0] for a in longer[len(pairs) :]]
|
|
278
|
+
|
|
279
|
+
|
|
280
|
+
def grade_seam(
|
|
281
|
+
site: SubstitutionNode, fragment: CallNode, use_state: bool = True
|
|
282
|
+
) -> tuple[JoinStrength | None, str]:
|
|
283
|
+
"""ENTRY grade of a seam, or ``None`` when the join is known unsound (exit conflict,
|
|
284
|
+
decisive argument-kind conflict, or state conflict). Exit compatibility is computed
|
|
285
|
+
separately by `exit_compatibility`; `use_state=False` ignores state facts entirely
|
|
286
|
+
(the VALUE-only baseline of experiment 07)."""
|
|
287
|
+
exit_ = exit_compatibility(site.outcome, fragment.outcome)
|
|
288
|
+
if exit_ is None:
|
|
289
|
+
return None, f"exit conflict: stand-in {site.outcome}, fragment {fragment.outcome}"
|
|
290
|
+
compatible: bool | None = None if exit_ == "unknown" else True
|
|
291
|
+
notes = [] if compatible else ["outcome unknown on one side"]
|
|
292
|
+
if not site.args and not fragment.args:
|
|
293
|
+
# No arguments on either side: argument compatibility holds vacuously. Entry
|
|
294
|
+
# compatibility then rests entirely on receiver state, which is the STATE rung.
|
|
295
|
+
if compatible is None:
|
|
296
|
+
return JoinStrength.ARG_SHAPE, "; ".join([*notes, "no arguments"])
|
|
297
|
+
if not use_state:
|
|
298
|
+
return JoinStrength.VALUE, "no arguments"
|
|
299
|
+
verdict, snote = state_compatibility(site, fragment)
|
|
300
|
+
if verdict == "conflict":
|
|
301
|
+
return None, "no arguments; " + snote
|
|
302
|
+
if verdict == "match":
|
|
303
|
+
return JoinStrength.STATE, "no arguments; " + snote
|
|
304
|
+
return JoinStrength.VALUE, "no arguments; " + snote
|
|
305
|
+
if not site.args or not fragment.args:
|
|
306
|
+
return JoinStrength.SYMBOL, "; ".join([*notes, "arity differs: arguments on one side only"])
|
|
307
|
+
pairs, one_sided = align_arguments(site.args, fragment.args)
|
|
308
|
+
verdicts = [(n, a, b, _types_compatible(a, b)) for (n, a, _), (_, b, _) in pairs]
|
|
309
|
+
conflicts = [f"{a}≠{b}" for _, a, b, v in verdicts if v == "conflict"]
|
|
310
|
+
if conflicts:
|
|
311
|
+
# A decisive argument-kind conflict is evidence against the seam, not a weak match:
|
|
312
|
+
# the fragment was entered with different kinds of values (Kokoro: None vs dict).
|
|
313
|
+
return None, "; ".join([*notes, "type conflict: " + ", ".join(conflicts)])
|
|
314
|
+
unverified = [f"{n}: {a} vs {b}" for n, a, b, v in verdicts if v == "object"]
|
|
315
|
+
if unverified:
|
|
316
|
+
notes.append("object types differ, relationship unverified: " + ", ".join(unverified))
|
|
317
|
+
return JoinStrength.ARG_SHAPE, "; ".join(notes)
|
|
318
|
+
wild = [n for n, a, b, _ in verdicts if _shape_type(a) != _shape_type(b)]
|
|
319
|
+
if wild:
|
|
320
|
+
notes.append("stand-in argument, type unverified: " + ", ".join(wild))
|
|
321
|
+
return JoinStrength.ARG_SHAPE, "; ".join(notes)
|
|
322
|
+
if one_sided:
|
|
323
|
+
# A parameter passed on one side only: the other side used its default, or a
|
|
324
|
+
# different call shape. Equality of that parameter is not observed, so no VALUE.
|
|
325
|
+
notes.append("argument on one side only: " + ", ".join(one_sided))
|
|
326
|
+
return JoinStrength.ARG_SHAPE, "; ".join(notes)
|
|
327
|
+
missing = [n for (_, _, da), (n, _, db) in pairs if not da or not db]
|
|
328
|
+
if missing:
|
|
329
|
+
notes.append("value unavailable: " + ", ".join(missing))
|
|
330
|
+
return JoinStrength.ARG_SHAPE, "; ".join(notes)
|
|
331
|
+
differ = [n for (_, _, da), (n, _, db) in pairs if da != db]
|
|
332
|
+
if differ:
|
|
333
|
+
notes.append("values differ: " + ", ".join(differ))
|
|
334
|
+
return JoinStrength.ARG_SHAPE, "; ".join(notes)
|
|
335
|
+
if compatible is None:
|
|
336
|
+
return JoinStrength.ARG_SHAPE, "; ".join(notes) # never VALUE with an unknown outcome
|
|
337
|
+
if not use_state:
|
|
338
|
+
return JoinStrength.VALUE, "; ".join(notes)
|
|
339
|
+
verdict, snote = state_compatibility(site, fragment)
|
|
340
|
+
if verdict == "conflict":
|
|
341
|
+
return None, snote
|
|
342
|
+
if verdict == "match":
|
|
343
|
+
return JoinStrength.STATE, snote
|
|
344
|
+
return JoinStrength.VALUE, "; ".join([*notes, snote])
|
|
345
|
+
|
|
346
|
+
|
|
347
|
+
def _node_symbol(node: Node) -> SymbolId:
|
|
348
|
+
if isinstance(node, CallNode):
|
|
349
|
+
return node.symbol
|
|
350
|
+
if isinstance(node, SubstitutionNode):
|
|
351
|
+
return node.substitute or node.claimed_target or "stand-in:?"
|
|
352
|
+
return f"os:{node.kind.value}:{node.target}"
|
|
353
|
+
|
|
354
|
+
|
|
355
|
+
def _invoked(sub: SubstitutionNode, target: SymbolId | None) -> SymbolId:
|
|
356
|
+
"""What the stand-in call invoked, as an edge callee. The claim names the first member
|
|
357
|
+
of the path; a longer path (a call on a return value) is appended so that
|
|
358
|
+
``execute`` and ``execute.().fetchone`` are distinct callees."""
|
|
359
|
+
if target is None:
|
|
360
|
+
return "stand-in:" + ".".join(filter(None, [sub.substitute or "?", *sub.path]))
|
|
361
|
+
return target + "".join(f".{p}" for p in sub.path[1:])
|
|
362
|
+
|
|
363
|
+
|
|
364
|
+
def compose(
|
|
365
|
+
corpus: Corpus,
|
|
366
|
+
seed_id: str,
|
|
367
|
+
min_join: JoinStrength = JoinStrength.SYMBOL,
|
|
368
|
+
max_depth: int = 8,
|
|
369
|
+
use_state: bool = True,
|
|
370
|
+
) -> Composition:
|
|
371
|
+
seed = corpus.executions[seed_id]
|
|
372
|
+
attempts: list[JoinAttempt] = []
|
|
373
|
+
gaps: list[Gap] = []
|
|
374
|
+
|
|
375
|
+
def probe(*ids: str) -> bool:
|
|
376
|
+
return any(corpus.executions[i].stimulus is Stimulus.GENERATED_PROBE for i in ids)
|
|
377
|
+
|
|
378
|
+
def expand(ex: Execution, node_id: int, caller: SymbolId, on_path: frozenset[SymbolId],
|
|
379
|
+
depth: int) -> list[Branch]: # fmt: skip
|
|
380
|
+
out: list[Branch] = []
|
|
381
|
+
for child in corpus.children[ex.id].get(node_id, []):
|
|
382
|
+
site = NodeRef(ex.id, child.id)
|
|
383
|
+
if isinstance(child, CallNode):
|
|
384
|
+
complete = ex.collectors[child.collector].fidelity is Fidelity.COMPLETE
|
|
385
|
+
kind = EvidenceKind.OBSERVED if complete else EvidenceKind.OBSERVED_SAMPLED
|
|
386
|
+
ev = Evidence(kind, site, probe_derived=probe(ex.id))
|
|
387
|
+
out.append(Branch(Edge(caller, child.symbol, ev),
|
|
388
|
+
expand(ex, child.id, child.symbol, on_path, depth))) # fmt: skip
|
|
389
|
+
continue
|
|
390
|
+
if isinstance(child, OsEventNode):
|
|
391
|
+
ev = Evidence(EvidenceKind.OS_BOUNDARY, site, probe_derived=probe(ex.id))
|
|
392
|
+
out.append(Branch(Edge(caller, _node_symbol(child), ev)))
|
|
393
|
+
continue
|
|
394
|
+
res = resolve(child, corpus.symbols)
|
|
395
|
+
executed = expand(ex, child.id, _node_symbol(child), on_path, depth)
|
|
396
|
+
if res.classification is BoundaryClass.EXTERNAL:
|
|
397
|
+
ev = Evidence(EvidenceKind.EXTERNAL_BOUNDARY, site, rule=res.rule,
|
|
398
|
+
probe_derived=probe(ex.id)) # fmt: skip
|
|
399
|
+
out.append(Branch(Edge(caller, _invoked(child, res.target), ev), executed))
|
|
400
|
+
continue
|
|
401
|
+
if res.classification is BoundaryClass.UNRESOLVED:
|
|
402
|
+
ev = Evidence(EvidenceKind.UNRESOLVED_BOUNDARY, site, rule=res.rule,
|
|
403
|
+
probe_derived=probe(ex.id)) # fmt: skip
|
|
404
|
+
out.append(Branch(Edge(caller, _invoked(child, res.target), ev), executed))
|
|
405
|
+
continue
|
|
406
|
+
target = res.target
|
|
407
|
+
assert target is not None
|
|
408
|
+
frags = corpus.fragments.get(target, [])
|
|
409
|
+
accepted_any = False
|
|
410
|
+
groups: dict[
|
|
411
|
+
tuple[object, ...], list[tuple[NodeRef, JoinStrength, str, bool | None]]
|
|
412
|
+
] = {}
|
|
413
|
+
for ref in frags:
|
|
414
|
+
frag_ex = corpus.executions[ref.execution]
|
|
415
|
+
frag_node = frag_ex.nodes[ref.node]
|
|
416
|
+
assert isinstance(frag_node, CallNode)
|
|
417
|
+
grade, note = grade_seam(child, frag_node, use_state)
|
|
418
|
+
exit_ = exit_compatibility(child.outcome, frag_node.outcome)
|
|
419
|
+
ok = grade is not None and grade.value >= min_join.value
|
|
420
|
+
same_result = (
|
|
421
|
+
child.result == frag_node.result if child.result and frag_node.result else None
|
|
422
|
+
)
|
|
423
|
+
if same_result is False:
|
|
424
|
+
note = "; ".join(filter(None, [note, "result differs: seed continued on it"]))
|
|
425
|
+
attempts.append(JoinAttempt(site, ref, target, grade, ok, note, same_result, exit_))
|
|
426
|
+
if ok and grade is not None:
|
|
427
|
+
groups.setdefault(fragment_shape(corpus, ref), []).append(
|
|
428
|
+
(ref, grade, note, same_result)
|
|
429
|
+
)
|
|
430
|
+
for members in groups.values():
|
|
431
|
+
# One branch per distinct behavior path; the best-graded member represents
|
|
432
|
+
# it and the others are kept as alternates, so no provenance is lost.
|
|
433
|
+
members.sort(key=lambda m: (-m[1].value, m[0].execution, m[0].node))
|
|
434
|
+
ref, grade, _, _ = members[0]
|
|
435
|
+
accepted_any = True
|
|
436
|
+
frag_outcome = corpus.executions[ref.execution].nodes[ref.node].outcome
|
|
437
|
+
ev = Evidence(
|
|
438
|
+
EvidenceKind.COMPOSED,
|
|
439
|
+
site,
|
|
440
|
+
rule=res.rule,
|
|
441
|
+
fragment=ref,
|
|
442
|
+
join=grade,
|
|
443
|
+
probe_derived=probe(ex.id, ref.execution),
|
|
444
|
+
alternates=tuple(m[0] for m in members[1:]),
|
|
445
|
+
exit=exit_compatibility(child.outcome, frag_outcome),
|
|
446
|
+
)
|
|
447
|
+
frag_ex = corpus.executions[ref.execution]
|
|
448
|
+
if target in on_path:
|
|
449
|
+
gaps.append(Gap(site, target, "cycle"))
|
|
450
|
+
out.append(Branch(Edge(caller, target, ev), executed))
|
|
451
|
+
elif depth >= max_depth:
|
|
452
|
+
gaps.append(Gap(site, target, "depth"))
|
|
453
|
+
out.append(Branch(Edge(caller, target, ev), executed))
|
|
454
|
+
else:
|
|
455
|
+
kids = expand(frag_ex, ref.node, target, on_path | {target}, depth + 1)
|
|
456
|
+
out.append(Branch(Edge(caller, target, ev), executed + kids))
|
|
457
|
+
if not accepted_any:
|
|
458
|
+
gaps.append(
|
|
459
|
+
Gap(site, target, "no-fragment" if not frags else "no-compatible-fragment")
|
|
460
|
+
)
|
|
461
|
+
ev = Evidence(EvidenceKind.INTERNAL_GAP, site, rule=res.rule,
|
|
462
|
+
probe_derived=probe(ex.id)) # fmt: skip
|
|
463
|
+
out.append(Branch(Edge(caller, target, ev), executed))
|
|
464
|
+
return out
|
|
465
|
+
|
|
466
|
+
root = seed.nodes[0]
|
|
467
|
+
assert isinstance(root, CallNode)
|
|
468
|
+
branches = expand(seed, 0, root.symbol, frozenset(), 0)
|
|
469
|
+
return Composition(seed_id, root.symbol, branches, attempts, gaps)
|