fuzzprep 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- fuzzprep/__init__.py +3 -0
- fuzzprep/__main__.py +4 -0
- fuzzprep/agents/harness_builder/SKILL.md +156 -0
- fuzzprep/agents/library_builder/SKILL.md +147 -0
- fuzzprep/agents/scripts/check_build.sh +87 -0
- fuzzprep/agents/scripts/check_build_in_container.sh +73 -0
- fuzzprep/agents/scripts/check_dockerfile_from_scratch.sh +33 -0
- fuzzprep/cli.py +1109 -0
- fuzzprep/core/__init__.py +0 -0
- fuzzprep/core/agent_stream.py +304 -0
- fuzzprep/core/files.py +35 -0
- fuzzprep/core/paths.py +25 -0
- fuzzprep/core/reporting.py +218 -0
- fuzzprep/core/repos.py +217 -0
- fuzzprep/core/resources.py +41 -0
- fuzzprep/core/subprocesses.py +197 -0
- fuzzprep/feature_extractor/__init__.py +0 -0
- fuzzprep/feature_extractor/benchmark_yaml.py +92 -0
- fuzzprep/feature_extractor/extraction.py +184 -0
- fuzzprep/feature_extractor/models.py +96 -0
- fuzzprep/feature_extractor/native/.clang-format +1 -0
- fuzzprep/feature_extractor/native/CMakeLists.txt +90 -0
- fuzzprep/feature_extractor/native/include/feature_extractor.hpp +136 -0
- fuzzprep/feature_extractor/native/src/extraction_action.cpp +294 -0
- fuzzprep/feature_extractor/native/src/json_writer.cpp +154 -0
- fuzzprep/feature_extractor/native/src/macro_callbacks.cpp +75 -0
- fuzzprep/feature_extractor/native/src/main.cpp +130 -0
- fuzzprep/feature_extractor/native_build.py +99 -0
- fuzzprep/library_builder/__init__.py +0 -0
- fuzzprep/library_builder/agents.py +472 -0
- fuzzprep/library_builder/analysis.py +145 -0
- fuzzprep/library_builder/build_parameters.py +158 -0
- fuzzprep/library_builder/dependency_resolution.py +139 -0
- fuzzprep/library_builder/environments/__init__.py +0 -0
- fuzzprep/library_builder/environments/base.py +89 -0
- fuzzprep/library_builder/environments/gate.py +96 -0
- fuzzprep/library_builder/environments/local.py +205 -0
- fuzzprep/library_builder/environments/oss_fuzz.py +485 -0
- fuzzprep/library_builder/environments/verification.py +125 -0
- fuzzprep/library_builder/exploration.py +217 -0
- fuzzprep/library_builder/generation.py +325 -0
- fuzzprep/library_builder/harness_explorer.py +257 -0
- fuzzprep/library_builder/models.py +169 -0
- fuzzprep/library_builder/package_names.json +33 -0
- fuzzprep/library_builder/package_names.py +40 -0
- fuzzprep/library_builder/scripts.py +389 -0
- fuzzprep/library_builder/stats.py +102 -0
- fuzzprep/library_builder/symbol_patterns.json +65 -0
- fuzzprep/library_builder/timeouts.py +14 -0
- fuzzprep/library_builder/workspace.py +250 -0
- fuzzprep-0.1.0.dist-info/METADATA +255 -0
- fuzzprep-0.1.0.dist-info/RECORD +56 -0
- fuzzprep-0.1.0.dist-info/WHEEL +4 -0
- fuzzprep-0.1.0.dist-info/entry_points.txt +3 -0
- fuzzprep-0.1.0.dist-info/licenses/LICENSE +202 -0
- fuzzprep-0.1.0.dist-info/licenses/THIRD_PARTY_NOTICES.md +52 -0
|
File without changes
|
|
@@ -0,0 +1,304 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
import subprocess
|
|
5
|
+
import time
|
|
6
|
+
from collections.abc import Callable
|
|
7
|
+
from dataclasses import dataclass
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
from typing import Any, Literal
|
|
10
|
+
|
|
11
|
+
AgentActivityKind = Literal[
|
|
12
|
+
"model_text", "status", "file_read", "file_edit", "command_run", "tool_result", "raw_fallback"
|
|
13
|
+
]
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
@dataclass
|
|
17
|
+
class AgentActivityEvent:
|
|
18
|
+
kind: AgentActivityKind
|
|
19
|
+
text: str
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
@dataclass
|
|
23
|
+
class AgentStreamResult:
|
|
24
|
+
"""One agent invocation's output, split into two channels.
|
|
25
|
+
|
|
26
|
+
`combined_text` is the whole rendered transcript, including tool results and narration
|
|
27
|
+
lines. It is what gets persisted for a human to read.
|
|
28
|
+
|
|
29
|
+
`model_text` is only what the model itself wrote: Claude `text` blocks and Codex
|
|
30
|
+
`agent_message` items. Match on this when looking for a marker the model was told to
|
|
31
|
+
print, such as `ACTION REQUIRED` — searching the whole transcript trips on any file that
|
|
32
|
+
merely quotes it. Thinking blocks are excluded too, so an agent weighing whether to print
|
|
33
|
+
a marker is not read as having printed it.
|
|
34
|
+
"""
|
|
35
|
+
|
|
36
|
+
combined_text: str
|
|
37
|
+
exit_code: int
|
|
38
|
+
duration_seconds: float
|
|
39
|
+
cost_usd: float | None = None
|
|
40
|
+
input_tokens: int | None = None
|
|
41
|
+
output_tokens: int | None = None
|
|
42
|
+
model_text: str = ""
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
_CLAUDE_READ_TOOLS = {"Read"}
|
|
46
|
+
_CLAUDE_EDIT_TOOLS = {"Edit", "Write", "MultiEdit", "NotebookEdit"}
|
|
47
|
+
_CLAUDE_COMMAND_TOOLS = {"Bash"}
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def _raw_fallback(line: str) -> list[AgentActivityEvent]:
|
|
51
|
+
return [AgentActivityEvent("raw_fallback", line.rstrip("\n"))]
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def _claude_tool_event(name: str, tool_input: dict[str, Any]) -> AgentActivityEvent:
|
|
55
|
+
if name in _CLAUDE_READ_TOOLS:
|
|
56
|
+
return AgentActivityEvent("file_read", f"Reading {tool_input.get('file_path', '?')}")
|
|
57
|
+
if name in _CLAUDE_EDIT_TOOLS:
|
|
58
|
+
return AgentActivityEvent("file_edit", f"Editing {tool_input.get('file_path', '?')}")
|
|
59
|
+
if name in _CLAUDE_COMMAND_TOOLS:
|
|
60
|
+
return AgentActivityEvent("command_run", f"Running: {tool_input.get('command', '?')}")
|
|
61
|
+
return AgentActivityEvent("status", f"Using tool {name}")
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _tool_result_text(content: Any) -> str:
|
|
65
|
+
if isinstance(content, str):
|
|
66
|
+
return content
|
|
67
|
+
if isinstance(content, list):
|
|
68
|
+
return "\n".join(
|
|
69
|
+
part.get("text", "")
|
|
70
|
+
for part in content
|
|
71
|
+
if isinstance(part, dict) and part.get("type") == "text"
|
|
72
|
+
)
|
|
73
|
+
return json.dumps(content)
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def _claude_content_block_event(block: dict[str, Any]) -> AgentActivityEvent | None:
|
|
77
|
+
block_type = block.get("type")
|
|
78
|
+
if block_type == "text":
|
|
79
|
+
return AgentActivityEvent("model_text", block.get("text", ""))
|
|
80
|
+
if block_type == "thinking":
|
|
81
|
+
thinking = block.get("thinking") or ""
|
|
82
|
+
return AgentActivityEvent("status", f"Thinking: {thinking}") if thinking else None
|
|
83
|
+
if block_type == "tool_use":
|
|
84
|
+
return _claude_tool_event(block.get("name", ""), block.get("input", {}))
|
|
85
|
+
if block_type == "tool_result":
|
|
86
|
+
return AgentActivityEvent("tool_result", _tool_result_text(block.get("content")))
|
|
87
|
+
return None
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def _parse_claude_line(line: str) -> list[AgentActivityEvent]:
|
|
91
|
+
"""Parse one line of `claude --output-format stream-json` output.
|
|
92
|
+
|
|
93
|
+
New top-level event types appear over time, so they are not enumerated. A well-formed JSON
|
|
94
|
+
object with a `type` string is a legitimate event this simply renders no narration for.
|
|
95
|
+
`raw_fallback` is reserved for a line that doesn't look like a structured event at all, so
|
|
96
|
+
garbled input stays visible without every future event type dumping raw JSON.
|
|
97
|
+
"""
|
|
98
|
+
try:
|
|
99
|
+
data = json.loads(line)
|
|
100
|
+
except json.JSONDecodeError:
|
|
101
|
+
return _raw_fallback(line)
|
|
102
|
+
if not isinstance(data, dict) or not isinstance(data.get("type"), str):
|
|
103
|
+
return _raw_fallback(line)
|
|
104
|
+
if data["type"] not in ("assistant", "user"):
|
|
105
|
+
return []
|
|
106
|
+
content = data.get("message", {}).get("content", [])
|
|
107
|
+
if not isinstance(content, list):
|
|
108
|
+
return []
|
|
109
|
+
events: list[AgentActivityEvent] = []
|
|
110
|
+
for block in content:
|
|
111
|
+
if not isinstance(block, dict):
|
|
112
|
+
continue
|
|
113
|
+
event = _claude_content_block_event(block)
|
|
114
|
+
if event is not None:
|
|
115
|
+
events.append(event)
|
|
116
|
+
return events
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def _codex_item_event(item: dict[str, Any]) -> AgentActivityEvent | None:
|
|
120
|
+
item_type = item.get("type")
|
|
121
|
+
if item_type == "command_execution":
|
|
122
|
+
return AgentActivityEvent("command_run", f"Running: {item.get('command', '?')}")
|
|
123
|
+
if item_type == "file_change":
|
|
124
|
+
return AgentActivityEvent("file_edit", f"Editing {item.get('path', '?')}")
|
|
125
|
+
if item_type == "agent_message":
|
|
126
|
+
# Codex's equivalent of a Claude text content block: the model's response text.
|
|
127
|
+
return AgentActivityEvent("model_text", item.get("text", ""))
|
|
128
|
+
if item_type == "reasoning":
|
|
129
|
+
# Codex's equivalent of Claude's "thinking" block.
|
|
130
|
+
text = item.get("text") or ""
|
|
131
|
+
return AgentActivityEvent("status", f"Thinking: {text}") if text else None
|
|
132
|
+
if item_type == "error":
|
|
133
|
+
return AgentActivityEvent("status", f"Warning: {item.get('message', '?')}")
|
|
134
|
+
return None
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
def _parse_codex_line(line: str) -> list[AgentActivityEvent]:
|
|
138
|
+
"""Parse one line of `codex exec --json` output.
|
|
139
|
+
|
|
140
|
+
Same silent-skip vs `raw_fallback` distinction as `_parse_claude_line`.
|
|
141
|
+
"""
|
|
142
|
+
try:
|
|
143
|
+
data = json.loads(line)
|
|
144
|
+
except json.JSONDecodeError:
|
|
145
|
+
return _raw_fallback(line)
|
|
146
|
+
if not isinstance(data, dict) or not isinstance(data.get("type"), str):
|
|
147
|
+
return _raw_fallback(line)
|
|
148
|
+
if data["type"] not in ("item.started", "item.updated", "item.completed"):
|
|
149
|
+
return []
|
|
150
|
+
item = data.get("item")
|
|
151
|
+
if not isinstance(item, dict):
|
|
152
|
+
return []
|
|
153
|
+
event = _codex_item_event(item)
|
|
154
|
+
return [event] if event is not None else []
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
_LINE_PARSERS: dict[str, Callable[[str], list[AgentActivityEvent]]] = {
|
|
158
|
+
"claude": _parse_claude_line,
|
|
159
|
+
"codex": _parse_codex_line,
|
|
160
|
+
}
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
def _claude_result_cost(line: str) -> tuple[float | None, int | None, int | None]:
|
|
164
|
+
try:
|
|
165
|
+
data = json.loads(line)
|
|
166
|
+
except json.JSONDecodeError:
|
|
167
|
+
return None, None, None
|
|
168
|
+
if not isinstance(data, dict) or data.get("type") != "result":
|
|
169
|
+
return None, None, None
|
|
170
|
+
cost = data.get("total_cost_usd")
|
|
171
|
+
# An error result -- the shape a 529 leaves behind -- can omit `usage` entirely, and this
|
|
172
|
+
# runs inside the stdout loop, where an AttributeError would abort the whole invocation.
|
|
173
|
+
usage = data.get("usage")
|
|
174
|
+
if not isinstance(usage, dict):
|
|
175
|
+
usage = {}
|
|
176
|
+
input_tokens = usage.get("input_tokens")
|
|
177
|
+
output_tokens = usage.get("output_tokens")
|
|
178
|
+
return (
|
|
179
|
+
cost if isinstance(cost, float) else None,
|
|
180
|
+
input_tokens if isinstance(input_tokens, int) else None,
|
|
181
|
+
output_tokens if isinstance(output_tokens, int) else None,
|
|
182
|
+
)
|
|
183
|
+
|
|
184
|
+
|
|
185
|
+
def _codex_result_cost(line: str) -> tuple[None, int | None, int | None]:
|
|
186
|
+
try:
|
|
187
|
+
data = json.loads(line)
|
|
188
|
+
except json.JSONDecodeError:
|
|
189
|
+
return None, None, None
|
|
190
|
+
if not isinstance(data, dict) or data.get("type") != "turn.completed":
|
|
191
|
+
return None, None, None
|
|
192
|
+
usage = data.get("usage")
|
|
193
|
+
if not isinstance(usage, dict):
|
|
194
|
+
return None, None, None
|
|
195
|
+
# Codex reports tokens but no cost.
|
|
196
|
+
return None, usage.get("input_tokens"), usage.get("output_tokens")
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
def _extract_stats(tool: str, line: str) -> tuple[float | None, int | None, int | None]:
|
|
200
|
+
if tool == "claude":
|
|
201
|
+
return _claude_result_cost(line)
|
|
202
|
+
return _codex_result_cost(line)
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
@dataclass
|
|
206
|
+
class _StreamAccumulator:
|
|
207
|
+
cost_usd: float | None = None
|
|
208
|
+
input_tokens: int | None = None
|
|
209
|
+
output_tokens: int | None = None
|
|
210
|
+
|
|
211
|
+
|
|
212
|
+
def _apply_line_stats(acc: _StreamAccumulator, tool: str, line: str) -> None:
|
|
213
|
+
"""Update the running stats accumulator from one stdout line, last-value-wins."""
|
|
214
|
+
cost, input_tokens, output_tokens = _extract_stats(tool, line)
|
|
215
|
+
if cost is not None:
|
|
216
|
+
acc.cost_usd = cost
|
|
217
|
+
if input_tokens is not None:
|
|
218
|
+
acc.input_tokens = input_tokens
|
|
219
|
+
if output_tokens is not None:
|
|
220
|
+
acc.output_tokens = output_tokens
|
|
221
|
+
|
|
222
|
+
|
|
223
|
+
def run_agent_streaming(
|
|
224
|
+
command: list[str], cwd: Path, timeout: int, tool: str
|
|
225
|
+
) -> AgentStreamResult:
|
|
226
|
+
"""Run an agent CLI, rendering its structured event stream as readable lines.
|
|
227
|
+
|
|
228
|
+
Mirrors run_command_streaming's structure, but parses each line as a backend-specific
|
|
229
|
+
event instead of treating it as opaque text.
|
|
230
|
+
"""
|
|
231
|
+
if tool not in _LINE_PARSERS:
|
|
232
|
+
raise ValueError(f"unknown agent tool: {tool!r}")
|
|
233
|
+
parse_line = _LINE_PARSERS[tool]
|
|
234
|
+
|
|
235
|
+
start = time.monotonic()
|
|
236
|
+
texts: list[str] = []
|
|
237
|
+
model_texts: list[str] = []
|
|
238
|
+
acc = _StreamAccumulator()
|
|
239
|
+
proc = subprocess.Popen(
|
|
240
|
+
command,
|
|
241
|
+
cwd=cwd,
|
|
242
|
+
stdout=subprocess.PIPE,
|
|
243
|
+
stderr=subprocess.STDOUT,
|
|
244
|
+
text=True,
|
|
245
|
+
bufsize=1,
|
|
246
|
+
)
|
|
247
|
+
assert proc.stdout is not None
|
|
248
|
+
try:
|
|
249
|
+
for line in proc.stdout:
|
|
250
|
+
for event in parse_line(line):
|
|
251
|
+
print(event.text, flush=True)
|
|
252
|
+
texts.append(event.text)
|
|
253
|
+
if event.kind == "model_text":
|
|
254
|
+
model_texts.append(event.text)
|
|
255
|
+
_apply_line_stats(acc, tool, line)
|
|
256
|
+
proc.wait(timeout=timeout)
|
|
257
|
+
exit_code = proc.returncode
|
|
258
|
+
except subprocess.TimeoutExpired:
|
|
259
|
+
proc.kill()
|
|
260
|
+
proc.wait()
|
|
261
|
+
exit_code = -1
|
|
262
|
+
|
|
263
|
+
return AgentStreamResult(
|
|
264
|
+
combined_text="\n".join(texts),
|
|
265
|
+
exit_code=exit_code,
|
|
266
|
+
duration_seconds=time.monotonic() - start,
|
|
267
|
+
cost_usd=acc.cost_usd,
|
|
268
|
+
input_tokens=acc.input_tokens,
|
|
269
|
+
output_tokens=acc.output_tokens,
|
|
270
|
+
model_text="\n".join(model_texts),
|
|
271
|
+
)
|
|
272
|
+
|
|
273
|
+
|
|
274
|
+
@dataclass
|
|
275
|
+
class AgentRunSummary:
|
|
276
|
+
backend: str
|
|
277
|
+
outcome: str
|
|
278
|
+
duration_seconds: float
|
|
279
|
+
cost_usd: float | None
|
|
280
|
+
input_tokens: int | None = None
|
|
281
|
+
output_tokens: int | None = None
|
|
282
|
+
|
|
283
|
+
|
|
284
|
+
def format_agent_summary(summary: AgentRunSummary) -> str:
|
|
285
|
+
"""Render the fixed-format '=== Agent Run Summary ===' trailer block."""
|
|
286
|
+
if summary.cost_usd is not None:
|
|
287
|
+
stats_line = f"cost: ${summary.cost_usd:.4f}"
|
|
288
|
+
elif summary.input_tokens is not None and summary.output_tokens is not None:
|
|
289
|
+
stats_line = f"tokens: input={summary.input_tokens} output={summary.output_tokens}"
|
|
290
|
+
else:
|
|
291
|
+
stats_line = "cost: unavailable"
|
|
292
|
+
|
|
293
|
+
return (
|
|
294
|
+
"=== Agent Run Summary ===\n"
|
|
295
|
+
f"backend: {summary.backend}\n"
|
|
296
|
+
f"outcome: {summary.outcome}\n"
|
|
297
|
+
f"duration: {summary.duration_seconds:.1f}s\n"
|
|
298
|
+
f"{stats_line}\n"
|
|
299
|
+
)
|
|
300
|
+
|
|
301
|
+
|
|
302
|
+
def write_agent_report(path: Path, combined_text: str, summary: AgentRunSummary) -> None:
|
|
303
|
+
"""Persist an invocation's transcript and time/cost summary to a report file."""
|
|
304
|
+
path.write_text(f"{combined_text}\n{format_agent_summary(summary)}")
|
fuzzprep/core/files.py
ADDED
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
"""Writing files that are meant to be run."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import shutil
|
|
6
|
+
import stat
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
|
|
9
|
+
_EXECUTABLE_BITS = stat.S_IXUSR | stat.S_IXGRP | stat.S_IXOTH
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def write_executable(path: Path, text: str) -> Path:
|
|
13
|
+
"""Write text to path and make it executable, returning path.
|
|
14
|
+
|
|
15
|
+
Every generated shell script goes through here. FuzzPrep runs them, repair agents run
|
|
16
|
+
them, and users run them, so one written without the executable bit is broken three ways.
|
|
17
|
+
"""
|
|
18
|
+
path.write_text(text)
|
|
19
|
+
return make_executable(path)
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def copy_executable(source: Path, destination: Path) -> Path:
|
|
23
|
+
"""Copy source to destination and make it executable, returning destination.
|
|
24
|
+
|
|
25
|
+
For publishing a validated script verbatim rather than regenerating it, so any repair an
|
|
26
|
+
agent applied survives into the output.
|
|
27
|
+
"""
|
|
28
|
+
shutil.copy2(source, destination)
|
|
29
|
+
return make_executable(destination)
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def make_executable(path: Path) -> Path:
|
|
33
|
+
"""Add the executable bits to an existing file, returning path."""
|
|
34
|
+
path.chmod(path.stat().st_mode | _EXECUTABLE_BITS)
|
|
35
|
+
return path
|
fuzzprep/core/paths.py
ADDED
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from pathlib import Path
|
|
4
|
+
|
|
5
|
+
_STATE_DIR_NAME = ".fuzzprep"
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def default_state_dir() -> Path:
|
|
9
|
+
"""Return the repo-local FuzzPrep state directory path."""
|
|
10
|
+
return Path(_STATE_DIR_NAME)
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def project_dir(state_dir: Path, project_name: str) -> Path:
|
|
14
|
+
"""Return the project workspace directory (.fuzzprep/<project>/)."""
|
|
15
|
+
return state_dir / project_name
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def project_state_file(state_dir: Path, project_name: str) -> Path:
|
|
19
|
+
"""Return the per-project persistent state file (.fuzzprep/<project>/state.json)."""
|
|
20
|
+
return state_dir / project_name / "state.json"
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def project_logs_dir(state_dir: Path, project_name: str) -> Path:
|
|
24
|
+
"""Return the per-phase log directory (.fuzzprep/<project>/logs/)."""
|
|
25
|
+
return state_dir / project_name / "logs"
|
|
@@ -0,0 +1,218 @@
|
|
|
1
|
+
"""Phase reporting and failure diagnostics for the `generate` console output.
|
|
2
|
+
|
|
3
|
+
`Phase`/`PhaseExecution` track which pipeline stage is running and how it ended.
|
|
4
|
+
`PhaseReporter` brackets a phase's output with start/end banners. `FailureDiagnostic`
|
|
5
|
+
turns a phase's failure into a short, located summary.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import time
|
|
11
|
+
from dataclasses import dataclass
|
|
12
|
+
from enum import Enum
|
|
13
|
+
from pathlib import Path
|
|
14
|
+
from types import TracebackType
|
|
15
|
+
from typing import Literal
|
|
16
|
+
|
|
17
|
+
_BANNER_WIDTH = 70
|
|
18
|
+
_DEFAULT_MESSAGE_LINES = 2
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
class Phase(Enum):
|
|
22
|
+
"""Ordered, fixed set of pipeline stages a `generate` run passes through."""
|
|
23
|
+
|
|
24
|
+
INGESTION = "ingestion"
|
|
25
|
+
STATIC_ANALYSIS = "static_analysis"
|
|
26
|
+
DETERMINISTIC_LIBRARY_BUILD = "deterministic_library_build"
|
|
27
|
+
AGENT_LIBRARY_REPAIR = "agent_library_repair"
|
|
28
|
+
HARNESS_COMPILE_PROBE = "harness_compile_probe"
|
|
29
|
+
AGENT_HARNESS_REPAIR = "agent_harness_repair"
|
|
30
|
+
DOCKERFILE_VERIFICATION = "dockerfile_verification"
|
|
31
|
+
OUTPUT_GENERATION = "output_generation"
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
_PHASE_LABELS: dict[Phase, str] = {
|
|
35
|
+
Phase.INGESTION: "Repository ingestion",
|
|
36
|
+
Phase.STATIC_ANALYSIS: "Static analysis",
|
|
37
|
+
Phase.DETERMINISTIC_LIBRARY_BUILD: "Deterministic library build",
|
|
38
|
+
Phase.AGENT_LIBRARY_REPAIR: "Agent-assisted library repair",
|
|
39
|
+
Phase.HARNESS_COMPILE_PROBE: "Harness compile probe",
|
|
40
|
+
Phase.AGENT_HARNESS_REPAIR: "Agent-assisted harness repair",
|
|
41
|
+
Phase.DOCKERFILE_VERIFICATION: "From-scratch Dockerfile verification",
|
|
42
|
+
Phase.OUTPUT_GENERATION: "Output generation",
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
_AGENT_PHASES = frozenset({Phase.AGENT_LIBRARY_REPAIR, Phase.AGENT_HARNESS_REPAIR})
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def phase_label(phase: Phase) -> str:
|
|
49
|
+
"""Return the console label for phase."""
|
|
50
|
+
return _PHASE_LABELS[phase]
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def is_agent_phase(phase: Phase) -> bool:
|
|
54
|
+
"""Whether phase is one of the agent-assisted repair phases."""
|
|
55
|
+
return phase in _AGENT_PHASES
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
PhaseStatus = Literal["running", "succeeded", "failed"]
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
@dataclass
|
|
62
|
+
class PhaseExecution:
|
|
63
|
+
"""One instance per phase actually run in a given `generate` invocation."""
|
|
64
|
+
|
|
65
|
+
phase: Phase
|
|
66
|
+
status: PhaseStatus = "running"
|
|
67
|
+
started_at: float = 0.0
|
|
68
|
+
ended_at: float | None = None
|
|
69
|
+
|
|
70
|
+
def mark_succeeded(self) -> None:
|
|
71
|
+
self._transition("succeeded")
|
|
72
|
+
|
|
73
|
+
def mark_failed(self) -> None:
|
|
74
|
+
self._transition("failed")
|
|
75
|
+
|
|
76
|
+
def _transition(self, status: PhaseStatus) -> None:
|
|
77
|
+
if self.status != "running":
|
|
78
|
+
raise ValueError(f"cannot transition a {self.status!r} PhaseExecution to {status!r}")
|
|
79
|
+
self.status = status
|
|
80
|
+
self.ended_at = time.monotonic()
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
@dataclass
|
|
84
|
+
class FailureDiagnostic:
|
|
85
|
+
"""Shown to the user when a phase fails."""
|
|
86
|
+
|
|
87
|
+
phase: Phase
|
|
88
|
+
step: str
|
|
89
|
+
message: str
|
|
90
|
+
origin: Literal["deterministic", "agent"]
|
|
91
|
+
log_path: Path | None = None
|
|
92
|
+
exit_code: int | None = None
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def summarize_message(text: str, *, max_lines: int = _DEFAULT_MESSAGE_LINES) -> str:
|
|
96
|
+
"""Collapse raw command output down to a short, human-readable diagnostic message:
|
|
97
|
+
the last max_lines non-blank lines, which is usually where the actual error is."""
|
|
98
|
+
non_blank = [line for line in text.splitlines() if line.strip()]
|
|
99
|
+
if not non_blank:
|
|
100
|
+
return "(no output captured)"
|
|
101
|
+
return "\n".join(non_blank[-max_lines:])
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def build_diagnostic( # noqa: PLR0913 -- one param per FailureDiagnostic field
|
|
105
|
+
phase: Phase,
|
|
106
|
+
*,
|
|
107
|
+
step: str,
|
|
108
|
+
message: str,
|
|
109
|
+
origin: Literal["deterministic", "agent"],
|
|
110
|
+
log_path: Path | None = None,
|
|
111
|
+
exit_code: int | None = None,
|
|
112
|
+
) -> FailureDiagnostic:
|
|
113
|
+
"""Construct a `FailureDiagnostic` — the single place these are assembled, so every
|
|
114
|
+
call site produces the same shape."""
|
|
115
|
+
return FailureDiagnostic(
|
|
116
|
+
phase=phase,
|
|
117
|
+
step=step,
|
|
118
|
+
message=message,
|
|
119
|
+
origin=origin,
|
|
120
|
+
log_path=log_path,
|
|
121
|
+
exit_code=exit_code,
|
|
122
|
+
)
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def _banner_line(phase: Phase, marker: str) -> str:
|
|
126
|
+
fill = "#" if is_agent_phase(phase) else "="
|
|
127
|
+
prefix = "AGENT: " if is_agent_phase(phase) else ""
|
|
128
|
+
core = f" {prefix}{phase_label(phase)} [{marker}] "
|
|
129
|
+
pad = max(3, (_BANNER_WIDTH - len(core)) // 2)
|
|
130
|
+
return f"{fill * pad}{core}{fill * pad}"
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def format_phase_start_banner(phase: Phase) -> str:
|
|
134
|
+
"""The line printed before a phase's own output begins."""
|
|
135
|
+
return _banner_line(phase, "START")
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
def format_phase_end_banner(phase: Phase, status: Literal["succeeded", "failed"]) -> str:
|
|
139
|
+
"""The line printed once a phase concludes."""
|
|
140
|
+
return _banner_line(phase, status.upper())
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def format_diagnostic(
|
|
144
|
+
diagnostic: FailureDiagnostic, *, debug: bool = False, raw_output: str | None = None
|
|
145
|
+
) -> str:
|
|
146
|
+
"""Render a `FailureDiagnostic` for the console.
|
|
147
|
+
|
|
148
|
+
When debug is set and raw_output is non-empty, the failing step's full raw output is
|
|
149
|
+
inlined here as well, whether or not it also streamed live.
|
|
150
|
+
"""
|
|
151
|
+
origin_text = (
|
|
152
|
+
"agent repair attempt failed" if diagnostic.origin == "agent" else "build step failed"
|
|
153
|
+
)
|
|
154
|
+
lines = [
|
|
155
|
+
f"--- FAILURE: {phase_label(diagnostic.phase)} ---",
|
|
156
|
+
f"Step: {diagnostic.step}",
|
|
157
|
+
f"Message: {diagnostic.message}",
|
|
158
|
+
f"Origin: {origin_text}",
|
|
159
|
+
]
|
|
160
|
+
if diagnostic.exit_code is not None:
|
|
161
|
+
lines.append(f"Exit code: {diagnostic.exit_code}")
|
|
162
|
+
log_text = str(diagnostic.log_path) if diagnostic.log_path is not None else "(not captured)"
|
|
163
|
+
lines.append(f"Full output: {log_text}")
|
|
164
|
+
if debug and raw_output:
|
|
165
|
+
lines.append("--- Full raw output (--log-level debug) ---")
|
|
166
|
+
lines.append(raw_output)
|
|
167
|
+
lines.append("--- end raw output ---")
|
|
168
|
+
return "\n".join(lines)
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
def format_startup_failure(message: str) -> str:
|
|
172
|
+
"""A failure before any Phase has started still needs an actionable message."""
|
|
173
|
+
return f"--- STARTUP FAILURE ---\n{message}"
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
class PhaseReporter:
|
|
177
|
+
"""Brackets one phase's console output with a start/end banner.
|
|
178
|
+
|
|
179
|
+
Use as a context manager::
|
|
180
|
+
|
|
181
|
+
with PhaseReporter(Phase.DETERMINISTIC_LIBRARY_BUILD) as reporter:
|
|
182
|
+
result = do_the_thing()
|
|
183
|
+
if result.succeeded:
|
|
184
|
+
reporter.succeed()
|
|
185
|
+
else:
|
|
186
|
+
reporter.fail()
|
|
187
|
+
|
|
188
|
+
If the `with` block raises before `.succeed()`/`.fail()` is called, `__exit__`
|
|
189
|
+
marks the phase failed and prints the end banner automatically, without
|
|
190
|
+
swallowing the exception — so callers that fail via exception don't have to
|
|
191
|
+
remember to call `.fail()` themselves.
|
|
192
|
+
"""
|
|
193
|
+
|
|
194
|
+
def __init__(self, phase: Phase) -> None:
|
|
195
|
+
self.phase = phase
|
|
196
|
+
self.execution = PhaseExecution(phase=phase, started_at=time.monotonic())
|
|
197
|
+
|
|
198
|
+
def __enter__(self) -> PhaseReporter:
|
|
199
|
+
print(format_phase_start_banner(self.phase))
|
|
200
|
+
return self
|
|
201
|
+
|
|
202
|
+
def succeed(self) -> None:
|
|
203
|
+
self.execution.mark_succeeded()
|
|
204
|
+
print(format_phase_end_banner(self.phase, "succeeded"))
|
|
205
|
+
|
|
206
|
+
def fail(self) -> None:
|
|
207
|
+
self.execution.mark_failed()
|
|
208
|
+
print(format_phase_end_banner(self.phase, "failed"))
|
|
209
|
+
|
|
210
|
+
def __exit__(
|
|
211
|
+
self,
|
|
212
|
+
exc_type: type[BaseException] | None,
|
|
213
|
+
exc: BaseException | None,
|
|
214
|
+
tb: TracebackType | None,
|
|
215
|
+
) -> Literal[False]:
|
|
216
|
+
if exc_type is not None and self.execution.status == "running":
|
|
217
|
+
self.fail()
|
|
218
|
+
return False
|