diffgenome 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (53) hide show
  1. diffgenome/__init__.py +7 -0
  2. diffgenome/__main__.py +240 -0
  3. diffgenome/_collectors/go/dg/dg.go +623 -0
  4. diffgenome/_collectors/go/go.mod +3 -0
  5. diffgenome/_collectors/go/instrument/facts.go +346 -0
  6. diffgenome/_collectors/go/instrument/main.go +484 -0
  7. diffgenome/_collectors/node/instrument.js +289 -0
  8. diffgenome/_collectors/node/jest-setup.js +40 -0
  9. diffgenome/_collectors/node/package-lock.json +35 -0
  10. diffgenome/_collectors/node/package.json +11 -0
  11. diffgenome/_collectors/node/runtime.js +426 -0
  12. diffgenome/ambiguity.py +122 -0
  13. diffgenome/api.py +67 -0
  14. diffgenome/change.py +86 -0
  15. diffgenome/change_artifact.py +310 -0
  16. diffgenome/collect/__init__.py +2 -0
  17. diffgenome/collect/go_test.py +271 -0
  18. diffgenome/collect/node_jest.py +319 -0
  19. diffgenome/collect/py_monitoring.py +985 -0
  20. diffgenome/collect/py_runtime.py +116 -0
  21. diffgenome/collect/py_symbols.py +238 -0
  22. diffgenome/collect/pytest_plugin.py +130 -0
  23. diffgenome/compose.py +469 -0
  24. diffgenome/dependence.py +264 -0
  25. diffgenome/evaluate.py +669 -0
  26. diffgenome/frontends/__init__.py +0 -0
  27. diffgenome/frontends/python_ir.py +335 -0
  28. diffgenome/genome.py +1016 -0
  29. diffgenome/genome_pipeline.py +674 -0
  30. diffgenome/genome_prompt.py +33 -0
  31. diffgenome/genome_state.py +2118 -0
  32. diffgenome/graph.py +426 -0
  33. diffgenome/llm.py +189 -0
  34. diffgenome/model.py +364 -0
  35. diffgenome/mvp.py +398 -0
  36. diffgenome/probe.py +509 -0
  37. diffgenome/projection.py +308 -0
  38. diffgenome/py.typed +0 -0
  39. diffgenome/render.py +118 -0
  40. diffgenome/report.py +363 -0
  41. diffgenome/resolve.py +37 -0
  42. diffgenome/runtime.py +74 -0
  43. diffgenome/runtime_evidence.py +261 -0
  44. diffgenome/sandbox.py +166 -0
  45. diffgenome/serialize.py +96 -0
  46. diffgenome/sites.py +19 -0
  47. diffgenome/static_types.py +69 -0
  48. diffgenome/structure.py +462 -0
  49. diffgenome-0.1.0.dist-info/METADATA +139 -0
  50. diffgenome-0.1.0.dist-info/RECORD +53 -0
  51. diffgenome-0.1.0.dist-info/WHEEL +4 -0
  52. diffgenome-0.1.0.dist-info/entry_points.txt +2 -0
  53. diffgenome-0.1.0.dist-info/licenses/LICENSE +202 -0
diffgenome/graph.py ADDED
@@ -0,0 +1,426 @@
1
+ """Repository-level behavioral graph: every execution's evidence merged, nothing flattened.
2
+
3
+ Built by composing from every execution as a seed and merging the resulting edges by
4
+ (caller, callee, kind). An edge keeps every piece of evidence that supports it, so a
5
+ reader can always get back to the executions, seams and fragments behind it. Language-
6
+ agnostic: it consumes the protocol and the composer's output only.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ from collections import Counter, defaultdict, deque
12
+ from dataclasses import dataclass, field
13
+ from typing import Any
14
+
15
+ from diffgenome.compose import Composition, Corpus, Gap, JoinAttempt, compose
16
+ from diffgenome.model import (
17
+ CallNode,
18
+ Evidence,
19
+ EvidenceKind,
20
+ JoinStrength,
21
+ NodeRef,
22
+ Origin,
23
+ Symbol,
24
+ SymbolId,
25
+ )
26
+
27
+ _TERMINAL = {
28
+ EvidenceKind.EXTERNAL_BOUNDARY,
29
+ EvidenceKind.OS_BOUNDARY,
30
+ EvidenceKind.UNRESOLVED_BOUNDARY,
31
+ EvidenceKind.INTERNAL_GAP,
32
+ }
33
+ _TRAVERSABLE = {EvidenceKind.OBSERVED, EvidenceKind.OBSERVED_SAMPLED, EvidenceKind.COMPOSED}
34
+
35
+
36
+ @dataclass
37
+ class GraphEdge:
38
+ caller: SymbolId
39
+ callee: SymbolId
40
+ kind: EvidenceKind
41
+ evidence: list[Evidence] = field(default_factory=list)
42
+ executions: set[str] = field(default_factory=set) # executions whose observations support it
43
+ rules: set[str] = field(default_factory=set)
44
+ joins: Counter[JoinStrength] = field(default_factory=Counter) # composed only
45
+ probe_derived: bool = False
46
+
47
+ @property
48
+ def best_join(self) -> JoinStrength | None:
49
+ return max(self.joins, key=lambda j: j.value) if self.joins else None
50
+
51
+ @property
52
+ def strong(self) -> bool:
53
+ """Observed, or composed at VALUE or better. The bar the north star counts."""
54
+ if self.kind is EvidenceKind.OBSERVED:
55
+ return True
56
+ best = self.best_join
57
+ return self.kind is EvidenceKind.COMPOSED and best is not None and best.value >= 3
58
+
59
+ def add(self, ev: Evidence) -> None:
60
+ key = (ev.site, ev.fragment)
61
+ if any((e.site, e.fragment) == key for e in self.evidence):
62
+ return
63
+ self.evidence.append(ev)
64
+ self.executions.add(ev.site.execution)
65
+ if ev.fragment:
66
+ self.executions.add(ev.fragment.execution)
67
+ self.executions.update(a.execution for a in ev.alternates)
68
+ if ev.rule:
69
+ self.rules.add(ev.rule)
70
+ if ev.join:
71
+ self.joins[ev.join] += 1
72
+ self.probe_derived = self.probe_derived or ev.probe_derived
73
+
74
+
75
+ @dataclass
76
+ class BehavioralGraph:
77
+ """One repository-level graph. Built once from a corpus; serializable to JSON with all
78
+ provenance and loadable back for queries without the corpus."""
79
+
80
+ symbols: dict[SymbolId, Symbol]
81
+ edges: dict[tuple[SymbolId, SymbolId, EvidenceKind], GraphEdge]
82
+ out: dict[SymbolId, list[GraphEdge]]
83
+ inc: dict[SymbolId, list[GraphEdge]]
84
+ tests_by_symbol: dict[SymbolId, set[str]] # executions in which the symbol really ran
85
+ outcomes_by_symbol: dict[SymbolId, Counter[str]]
86
+ gaps: list[Gap]
87
+ attempts: list[JoinAttempt] # every join attempt, accepted or not
88
+ corpus: Corpus | None = None # present when built from traces; None when loaded from JSON
89
+ compositions: dict[str, Composition] = field(default_factory=dict)
90
+ use_state: bool = True # whether the STATE rung was consulted when this graph was built
91
+
92
+ def origin(self, symbol: SymbolId) -> Origin:
93
+ s = self.symbols.get(symbol)
94
+ return s.origin if s else Origin.UNKNOWN
95
+
96
+ # ------------------------------------------------------------------ materialization
97
+
98
+ def to_json(self) -> dict[str, object]:
99
+ def ev(e: Evidence) -> dict[str, object]:
100
+ return {
101
+ "kind": e.kind.value,
102
+ "site": {"execution": e.site.execution, "node": e.site.node},
103
+ "fragment": {"execution": e.fragment.execution, "node": e.fragment.node}
104
+ if e.fragment
105
+ else None,
106
+ "alternates": [{"execution": a.execution, "node": a.node} for a in e.alternates],
107
+ "join": e.join.name if e.join else None,
108
+ "exit": e.exit,
109
+ "rule": e.rule,
110
+ "probe_derived": e.probe_derived,
111
+ }
112
+
113
+ return {
114
+ "format": "diffgenome-graph/1",
115
+ "symbols": [
116
+ {
117
+ "id": s.id,
118
+ "origin": s.origin.value,
119
+ "location": {"path": s.location.path, "line": s.location.line}
120
+ if s.location
121
+ else None,
122
+ "kind": s.kind,
123
+ "executed_by": sorted(self.tests_by_symbol.get(s.id, ())),
124
+ "outcomes": dict(self.outcomes_by_symbol.get(s.id, Counter())),
125
+ }
126
+ for s in sorted(self.symbols.values(), key=lambda x: x.id)
127
+ ],
128
+ "edges": [
129
+ {
130
+ "caller": e.caller,
131
+ "callee": e.callee,
132
+ "kind": e.kind.value,
133
+ "best_join": e.best_join.name if e.best_join else None,
134
+ "joins": {
135
+ j.name: n for j, n in sorted(e.joins.items(), key=lambda x: x[0].value)
136
+ },
137
+ "rules": sorted(e.rules),
138
+ "executions": sorted(e.executions),
139
+ "probe_derived": e.probe_derived,
140
+ "evidence": [ev(x) for x in e.evidence],
141
+ }
142
+ for e in sorted(
143
+ self.edges.values(), key=lambda x: (x.caller, x.callee, x.kind.value)
144
+ )
145
+ ],
146
+ "gaps": [
147
+ {
148
+ "site": {"execution": g.site.execution, "node": g.site.node},
149
+ "target": g.target,
150
+ "kind": g.kind,
151
+ }
152
+ for g in self.gaps
153
+ ],
154
+ "join_attempts": [
155
+ {
156
+ "site": {"execution": a.site.execution, "node": a.site.node},
157
+ "fragment": {"execution": a.fragment.execution, "node": a.fragment.node},
158
+ "target": a.target,
159
+ "grade": a.grade.name if a.grade else None,
160
+ "accepted": a.accepted,
161
+ "note": a.note,
162
+ "result_compatible": a.result_compatible,
163
+ "exit": a.exit,
164
+ }
165
+ for a in self.attempts
166
+ ],
167
+ }
168
+
169
+ @classmethod
170
+ def from_json(cls, doc: dict[str, Any]) -> BehavioralGraph:
171
+ from diffgenome.model import SourceLocation
172
+
173
+ symbols: dict[SymbolId, Symbol] = {}
174
+ tests: dict[SymbolId, set[str]] = {}
175
+ outcomes: dict[SymbolId, Counter[str]] = {}
176
+ for s in doc["symbols"]:
177
+ loc = (
178
+ SourceLocation(s["location"]["path"], s["location"]["line"])
179
+ if s["location"]
180
+ else None
181
+ )
182
+ symbols[s["id"]] = Symbol(s["id"], Origin(s["origin"]), loc, s.get("kind", "callable"))
183
+ tests[s["id"]] = set(s["executed_by"])
184
+ outcomes[s["id"]] = Counter(s["outcomes"])
185
+ edges: dict[tuple[SymbolId, SymbolId, EvidenceKind], GraphEdge] = {}
186
+ for e in doc["edges"]:
187
+ kind = EvidenceKind(e["kind"])
188
+ ge = GraphEdge(e["caller"], e["callee"], kind)
189
+ for x in e["evidence"]:
190
+ ge.add(
191
+ Evidence(
192
+ EvidenceKind(x["kind"]),
193
+ NodeRef(x["site"]["execution"], x["site"]["node"]),
194
+ rule=x["rule"],
195
+ fragment=NodeRef(x["fragment"]["execution"], x["fragment"]["node"])
196
+ if x["fragment"]
197
+ else None,
198
+ join=JoinStrength[x["join"]] if x["join"] else None,
199
+ alternates=tuple(
200
+ NodeRef(a["execution"], a["node"]) for a in x["alternates"]
201
+ ),
202
+ probe_derived=x["probe_derived"],
203
+ )
204
+ )
205
+ edges[(ge.caller, ge.callee, kind)] = ge
206
+ out: dict[SymbolId, list[GraphEdge]] = defaultdict(list)
207
+ inc: dict[SymbolId, list[GraphEdge]] = defaultdict(list)
208
+ for ge in edges.values():
209
+ out[ge.caller].append(ge)
210
+ inc[ge.callee].append(ge)
211
+ gaps = [
212
+ Gap(NodeRef(g["site"]["execution"], g["site"]["node"]), g["target"], g["kind"])
213
+ for g in doc["gaps"]
214
+ ]
215
+ attempts = [
216
+ JoinAttempt(
217
+ NodeRef(a["site"]["execution"], a["site"]["node"]),
218
+ NodeRef(a["fragment"]["execution"], a["fragment"]["node"]),
219
+ a["target"],
220
+ JoinStrength[a["grade"]] if a["grade"] else None,
221
+ a["accepted"],
222
+ a["note"],
223
+ a["result_compatible"],
224
+ a.get("exit"),
225
+ )
226
+ for a in doc["join_attempts"]
227
+ ]
228
+ return cls(symbols, edges, dict(out), dict(inc), tests, outcomes, gaps, attempts)
229
+
230
+ # ------------------------------------------------------------------ queries
231
+
232
+ def inspect(self, symbol: SymbolId, depth: int = 3) -> Inspection:
233
+ up = self._walk(symbol, depth, upstream=True)
234
+ down = self._walk(symbol, depth, upstream=False)
235
+ boundaries = [e for e in self.out.get(symbol, []) if e.kind in _TERMINAL]
236
+ return Inspection(
237
+ symbol=symbol,
238
+ upstream=up,
239
+ downstream=down,
240
+ tests=sorted(self.tests_by_symbol.get(symbol, ())),
241
+ boundaries=boundaries,
242
+ gaps=[g for g in self.gaps if g.target == symbol],
243
+ outcomes=self.outcomes_by_symbol.get(symbol, Counter()),
244
+ )
245
+
246
+ def _walk(self, start: SymbolId, depth: int, upstream: bool) -> list[tuple[int, GraphEdge]]:
247
+ """Bounded BFS over traversable edges, returning (distance, edge). Terminal edges at
248
+ the frontier are included at their distance so boundaries are visible."""
249
+ index = self.inc if upstream else self.out
250
+ seen: set[tuple[SymbolId, SymbolId, EvidenceKind]] = set()
251
+ result: list[tuple[int, GraphEdge]] = []
252
+ frontier: deque[tuple[SymbolId, int]] = deque([(start, 0)])
253
+ visited = {start}
254
+ while frontier:
255
+ sym, d = frontier.popleft()
256
+ if d >= depth:
257
+ continue
258
+ for e in index.get(sym, []):
259
+ key = (e.caller, e.callee, e.kind)
260
+ if key in seen:
261
+ continue
262
+ seen.add(key)
263
+ result.append((d + 1, e))
264
+ nxt = e.caller if upstream else e.callee
265
+ if e.kind in _TRAVERSABLE and nxt not in visited:
266
+ visited.add(nxt)
267
+ frontier.append((nxt, d + 1))
268
+ return result
269
+
270
+ def neighborhood(self, seeds: list[SymbolId], up: int = 3, down: int = 4) -> Neighborhood:
271
+ edges: dict[tuple[SymbolId, SymbolId, EvidenceKind], tuple[int, GraphEdge]] = {}
272
+ for s in seeds:
273
+ for d, e in self._walk(s, up, upstream=True) + self._walk(s, down, upstream=False):
274
+ key = (e.caller, e.callee, e.kind)
275
+ if key not in edges or d < edges[key][0]:
276
+ edges[key] = (d, e)
277
+ symbols = set(seeds)
278
+ for _, e in edges.values():
279
+ symbols.update((e.caller, e.callee))
280
+ tests: set[str] = set()
281
+ for s in symbols:
282
+ tests.update(self.tests_by_symbol.get(s, ()))
283
+ return Neighborhood(seeds, edges, symbols, tests, self)
284
+
285
+
286
+ @dataclass
287
+ class Inspection:
288
+ symbol: SymbolId
289
+ upstream: list[tuple[int, GraphEdge]]
290
+ downstream: list[tuple[int, GraphEdge]]
291
+ tests: list[str]
292
+ boundaries: list[GraphEdge]
293
+ gaps: list[Gap]
294
+ outcomes: Counter[str]
295
+
296
+
297
+ @dataclass
298
+ class Metrics:
299
+ observed: int
300
+ composed: int
301
+ strong_joins: int # composed edges whose best join is VALUE or better
302
+ weak_joins: int # composed edges at SYMBOL / ARG_SHAPE only
303
+ internal_gaps: int
304
+ unresolved_boundaries: int
305
+ external_boundaries: int
306
+ os_boundaries: int
307
+ unsound_attempts: int
308
+ probe_derived_edges: int
309
+ symbols: int
310
+ tests: int
311
+
312
+ @property
313
+ def denominator(self) -> int:
314
+ """Provisional: the seams we *know about* in the neighborhood. Observed and composed
315
+ edges plus the places where knowledge stops inside the repository (gaps, weak
316
+ joins, unresolved stand-ins). External and OS boundaries are terminal by design
317
+ and are not counted as missing. This is not the true set of locally realizable
318
+ behavior; it undercounts anything no execution has come near."""
319
+ return self.observed + self.composed + self.internal_gaps + self.unresolved_boundaries
320
+
321
+ @property
322
+ def reconstructed(self) -> int:
323
+ return self.observed + self.strong_joins
324
+
325
+ @property
326
+ def ratio(self) -> float:
327
+ return self.reconstructed / self.denominator if self.denominator else 0.0
328
+
329
+ def as_dict(self) -> dict[str, object]:
330
+ return {
331
+ "observed_edges": self.observed,
332
+ "composed_edges": self.composed,
333
+ "strong_joins": self.strong_joins,
334
+ "weak_joins": self.weak_joins,
335
+ "internal_gaps": self.internal_gaps,
336
+ "unresolved_boundaries": self.unresolved_boundaries,
337
+ "external_boundaries": self.external_boundaries,
338
+ "os_boundaries": self.os_boundaries,
339
+ "unsound_attempts": self.unsound_attempts,
340
+ "probe_derived_edges": self.probe_derived_edges,
341
+ "symbols": self.symbols,
342
+ "tests": self.tests,
343
+ "reconstructed": self.reconstructed,
344
+ "denominator_provisional": self.denominator,
345
+ "reconstruction_ratio_provisional": round(self.ratio, 3),
346
+ }
347
+
348
+
349
+ @dataclass
350
+ class Neighborhood:
351
+ seeds: list[SymbolId]
352
+ edges: dict[tuple[SymbolId, SymbolId, EvidenceKind], tuple[int, GraphEdge]]
353
+ symbols: set[SymbolId]
354
+ tests: set[str]
355
+ graph: BehavioralGraph
356
+
357
+ def edges_of(self, kind: EvidenceKind) -> list[tuple[int, GraphEdge]]:
358
+ return sorted(
359
+ ((d, e) for d, e in self.edges.values() if e.kind is kind),
360
+ key=lambda x: (x[0], x[1].caller, x[1].callee),
361
+ )
362
+
363
+ def metrics(self) -> Metrics:
364
+ es = [e for _, e in self.edges.values()]
365
+ composed = [e for e in es if e.kind is EvidenceKind.COMPOSED]
366
+ strong = sum(1 for e in composed if e.strong)
367
+ sites = {ev.site for e in es for ev in e.evidence}
368
+ unsound = sum(1 for a in self.graph.attempts if a.grade is None and a.site in sites)
369
+ return Metrics(
370
+ observed=sum(1 for e in es if e.kind is EvidenceKind.OBSERVED),
371
+ composed=len(composed),
372
+ strong_joins=strong,
373
+ weak_joins=len(composed) - strong,
374
+ internal_gaps=sum(1 for e in es if e.kind is EvidenceKind.INTERNAL_GAP),
375
+ unresolved_boundaries=sum(1 for e in es if e.kind is EvidenceKind.UNRESOLVED_BOUNDARY),
376
+ external_boundaries=sum(1 for e in es if e.kind is EvidenceKind.EXTERNAL_BOUNDARY),
377
+ os_boundaries=sum(1 for e in es if e.kind is EvidenceKind.OS_BOUNDARY),
378
+ unsound_attempts=unsound,
379
+ probe_derived_edges=sum(1 for e in es if e.probe_derived),
380
+ symbols=len(self.symbols),
381
+ tests=len(self.tests),
382
+ )
383
+
384
+
385
+ def build_graph(
386
+ corpus: Corpus, min_join: JoinStrength = JoinStrength.SYMBOL, use_state: bool = True
387
+ ) -> BehavioralGraph:
388
+ edges: dict[tuple[SymbolId, SymbolId, EvidenceKind], GraphEdge] = {}
389
+ gaps: list[Gap] = []
390
+ attempts: list[JoinAttempt] = []
391
+ compositions: dict[str, Composition] = {}
392
+ seen_gaps: set[tuple[NodeRef, SymbolId, str]] = set()
393
+ seen_attempts: set[tuple[NodeRef, NodeRef]] = set()
394
+ for ex_id in corpus.executions:
395
+ c = compose(corpus, ex_id, min_join=min_join, use_state=use_state)
396
+ compositions[ex_id] = c
397
+ for edge in c.edges():
398
+ key = (edge.caller, edge.callee, edge.evidence.kind)
399
+ edges.setdefault(key, GraphEdge(*key)).add(edge.evidence)
400
+ for g in c.gaps:
401
+ k = (g.site, g.target, g.kind)
402
+ if k not in seen_gaps:
403
+ seen_gaps.add(k)
404
+ gaps.append(g)
405
+ for a in c.attempts:
406
+ k2 = (a.site, a.fragment)
407
+ if k2 not in seen_attempts:
408
+ seen_attempts.add(k2)
409
+ attempts.append(a)
410
+ out: dict[SymbolId, list[GraphEdge]] = defaultdict(list)
411
+ inc: dict[SymbolId, list[GraphEdge]] = defaultdict(list)
412
+ for e in edges.values():
413
+ out[e.caller].append(e)
414
+ inc[e.callee].append(e)
415
+ tests_by_symbol: dict[SymbolId, set[str]] = defaultdict(set)
416
+ outcomes: dict[SymbolId, Counter[str]] = defaultdict(Counter)
417
+ for ex in corpus.executions.values():
418
+ for n in ex.nodes:
419
+ if isinstance(n, CallNode) and n.id != 0:
420
+ tests_by_symbol[n.symbol].add(ex.id)
421
+ outcomes[n.symbol][n.outcome.split(":")[0]] += 1
422
+ return BehavioralGraph(
423
+ dict(corpus.symbols), edges, dict(out), dict(inc), dict(tests_by_symbol), dict(outcomes),
424
+ gaps, attempts, corpus, compositions,
425
+ use_state=use_state,
426
+ ) # fmt: skip
diffgenome/llm.py ADDED
@@ -0,0 +1,189 @@
1
+ """The replaceable LLM seam. One narrow job: write a probe for a stated objective.
2
+
3
+ `ProbeWriter` is the whole interface. `OpenAIProbeWriter` speaks to the chat completions
4
+ endpoint over urllib; `RecordedProbeWriter` replays canned answers for tests. No provider
5
+ framework: a second provider is a second class implementing `write`.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import json
11
+ import os
12
+ import subprocess
13
+ import urllib.error
14
+ import urllib.request
15
+ from dataclasses import dataclass, field
16
+ from pathlib import Path
17
+ from typing import Protocol
18
+
19
+
20
+ @dataclass
21
+ class ProbeRequest:
22
+ objective: str # one paragraph: what the probe must make execute, and why
23
+ context: str # bounded: sources, seam facts, nearby tests
24
+ constraints: str # the safety and unit-level rules, verbatim
25
+ conventions: str = "" # runtime/framework conventions from the runtime adapter
26
+ previous_failures: list[str] = field(default_factory=list)
27
+
28
+
29
+ @dataclass
30
+ class ProbeDraft:
31
+ filename: str
32
+ code: str
33
+ rationale: str
34
+ model: str # which model/recording produced it (provenance)
35
+
36
+
37
+ class ProbeWriter(Protocol):
38
+ def write(self, request: ProbeRequest) -> ProbeDraft: ...
39
+
40
+
41
+ _SYSTEM = """You write one minimal, isolated unit-level test ("probe") for a repository.
42
+ The probe exists so that a specific in-repo function executes under tracing. It must:
43
+ - be a test file for the repository's own test framework (stated in the request); import
44
+ the target through the repository's own module paths;
45
+ - execute the real in-repo target and, where possible, the real in-repo code it calls;
46
+ - keep every genuine external dependency substituted (network, model files, GPUs, disk
47
+ outside a tmp_path, subprocesses, time-based waits): use unittest.mock / pytest fixtures;
48
+ - never contact any real service, never sleep for long, never spawn processes;
49
+ - follow the conventions visible in the nearby tests (fixtures, async style, imports);
50
+ - be deterministic and fast; prefer the smallest stimulus that reaches the target.
51
+ Reply with a single JSON object:
52
+ {"filename": "<test file name>", "code": "<file>", "rationale": "<one paragraph>"}.
53
+ The code must be a complete file; the filename is advisory (the harness places the file)."""
54
+
55
+
56
+ _KEY_FILE = Path.home() / ".config" / "diffgenome" / "openai_api_key"
57
+ _KEYCHAIN_SERVICE = "diffgenome-openai"
58
+ _PREFERRED_MODELS = ("gpt-5", "gpt-4.1", "gpt-4o")
59
+
60
+
61
+ def load_api_key() -> str:
62
+ """Environment, then a user-only file, then the macOS Keychain. The key stays in this
63
+ process: probes run with a scrubbed environment and no network."""
64
+ key = os.environ.get("OPENAI_API_KEY", "").strip()
65
+ if key:
66
+ return key
67
+ try:
68
+ if _KEY_FILE.is_file():
69
+ key = _KEY_FILE.read_text().strip()
70
+ if key:
71
+ return key
72
+ except OSError:
73
+ pass
74
+ try:
75
+ out = subprocess.run(
76
+ ["security", "find-generic-password", "-s", _KEYCHAIN_SERVICE, "-w"],
77
+ capture_output=True,
78
+ text=True,
79
+ timeout=10,
80
+ check=False,
81
+ )
82
+ if out.returncode == 0 and out.stdout.strip():
83
+ return out.stdout.strip()
84
+ except (OSError, subprocess.TimeoutExpired):
85
+ pass
86
+ raise RuntimeError(
87
+ f"no OpenAI key: set OPENAI_API_KEY, write {_KEY_FILE} (mode 600), "
88
+ f"or add a Keychain item '{_KEYCHAIN_SERVICE}'"
89
+ )
90
+
91
+
92
+ class OpenAIProbeWriter:
93
+ def __init__(
94
+ self, model: str | None = None, api_key: str | None = None, timeout: int = 180
95
+ ) -> None:
96
+ self.api_key = api_key or load_api_key()
97
+ self.timeout = timeout
98
+ self.base = os.environ.get("OPENAI_BASE_URL", "https://api.openai.com/v1").rstrip("/")
99
+ self.model = model or os.environ.get("DIFFGENOME_OPENAI_MODEL") or self._pick_model()
100
+
101
+ def _get(self, path: str) -> dict[str, object]:
102
+ req = urllib.request.Request(
103
+ f"{self.base}{path}", headers={"Authorization": f"Bearer {self.api_key}"}
104
+ )
105
+ with urllib.request.urlopen(req, timeout=self.timeout) as resp:
106
+ data: dict[str, object] = json.load(resp)
107
+ return data
108
+
109
+ def _pick_model(self) -> str:
110
+ """Choose the first preferred family the account actually lists (exact id, else the
111
+ newest dated variant); never guess a model name blindly."""
112
+ try:
113
+ listed = self._get("/models").get("data")
114
+ except (urllib.error.URLError, RuntimeError, OSError):
115
+ return _PREFERRED_MODELS[1]
116
+ ids = sorted(str(m["id"]) for m in listed) if isinstance(listed, list) else []
117
+ for family in _PREFERRED_MODELS:
118
+ if family in ids:
119
+ return family
120
+ dated = [
121
+ i for i in ids if i.startswith(family + "-") and i[len(family) + 1 :][:1].isdigit()
122
+ ]
123
+ if dated:
124
+ return sorted(dated)[-1]
125
+ return _PREFERRED_MODELS[1]
126
+
127
+ def _post(self, path: str, body: dict[str, object]) -> dict[str, object]:
128
+ req = urllib.request.Request(
129
+ f"{self.base}{path}",
130
+ data=json.dumps(body).encode(),
131
+ headers={"Authorization": f"Bearer {self.api_key}", "Content-Type": "application/json"},
132
+ method="POST",
133
+ )
134
+ try:
135
+ with urllib.request.urlopen(req, timeout=self.timeout) as resp:
136
+ data: dict[str, object] = json.load(resp)
137
+ return data
138
+ except urllib.error.HTTPError as exc:
139
+ detail = exc.read().decode(errors="replace")[:500]
140
+ raise RuntimeError(f"OpenAI HTTP {exc.code}: {detail}") from exc
141
+
142
+ def write(self, request: ProbeRequest) -> ProbeDraft:
143
+ user = (
144
+ f"OBJECTIVE\n{request.objective}\n\nCONSTRAINTS\n{request.constraints}\n\n"
145
+ f"RUNTIME AND CONVENTIONS\n{request.conventions}\n\nCONTEXT\n{request.context}\n"
146
+ )
147
+ if request.previous_failures:
148
+ user += "\nPREVIOUS ATTEMPTS FAILED VERIFICATION\n" + "\n---\n".join(
149
+ request.previous_failures
150
+ )
151
+ user += "\nWrite a different probe that avoids these failures.\n"
152
+ data = self._post(
153
+ "/chat/completions",
154
+ {
155
+ "model": self.model,
156
+ "messages": [
157
+ {"role": "system", "content": _SYSTEM},
158
+ {"role": "user", "content": user},
159
+ ],
160
+ "response_format": {"type": "json_object"},
161
+ },
162
+ )
163
+ choices = data.get("choices")
164
+ assert isinstance(choices, list) and choices
165
+ content = choices[0]["message"]["content"]
166
+ try:
167
+ obj = json.loads(content)
168
+ except json.JSONDecodeError as exc:
169
+ raise RuntimeError(f"model returned non-JSON: {content[:200]}") from exc
170
+ filename = str(obj.get("filename", "probe"))
171
+ return ProbeDraft(filename, str(obj["code"]), str(obj.get("rationale", "")), self.model)
172
+
173
+
174
+ class RecordedProbeWriter:
175
+ """Replays drafts from a directory of ``<n>.py`` files in order; for tests and for
176
+ re-running a report without spending tokens. Records nothing itself."""
177
+
178
+ def __init__(self, directory: Path) -> None:
179
+ self.files = sorted(directory.glob("*.py"))
180
+ self.calls = 0
181
+
182
+ def write(self, request: ProbeRequest) -> ProbeDraft:
183
+ if self.calls >= len(self.files):
184
+ raise RuntimeError("recorded writer has no more drafts")
185
+ file = self.files[self.calls]
186
+ self.calls += 1
187
+ return ProbeDraft(
188
+ f"test_probe_{file.stem}.py", file.read_text(), f"recorded: {file.name}", "recorded"
189
+ )