fuzzprep 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- fuzzprep/__init__.py +3 -0
- fuzzprep/__main__.py +4 -0
- fuzzprep/agents/harness_builder/SKILL.md +156 -0
- fuzzprep/agents/library_builder/SKILL.md +147 -0
- fuzzprep/agents/scripts/check_build.sh +87 -0
- fuzzprep/agents/scripts/check_build_in_container.sh +73 -0
- fuzzprep/agents/scripts/check_dockerfile_from_scratch.sh +33 -0
- fuzzprep/cli.py +1109 -0
- fuzzprep/core/__init__.py +0 -0
- fuzzprep/core/agent_stream.py +304 -0
- fuzzprep/core/files.py +35 -0
- fuzzprep/core/paths.py +25 -0
- fuzzprep/core/reporting.py +218 -0
- fuzzprep/core/repos.py +217 -0
- fuzzprep/core/resources.py +41 -0
- fuzzprep/core/subprocesses.py +197 -0
- fuzzprep/feature_extractor/__init__.py +0 -0
- fuzzprep/feature_extractor/benchmark_yaml.py +92 -0
- fuzzprep/feature_extractor/extraction.py +184 -0
- fuzzprep/feature_extractor/models.py +96 -0
- fuzzprep/feature_extractor/native/.clang-format +1 -0
- fuzzprep/feature_extractor/native/CMakeLists.txt +90 -0
- fuzzprep/feature_extractor/native/include/feature_extractor.hpp +136 -0
- fuzzprep/feature_extractor/native/src/extraction_action.cpp +294 -0
- fuzzprep/feature_extractor/native/src/json_writer.cpp +154 -0
- fuzzprep/feature_extractor/native/src/macro_callbacks.cpp +75 -0
- fuzzprep/feature_extractor/native/src/main.cpp +130 -0
- fuzzprep/feature_extractor/native_build.py +99 -0
- fuzzprep/library_builder/__init__.py +0 -0
- fuzzprep/library_builder/agents.py +472 -0
- fuzzprep/library_builder/analysis.py +145 -0
- fuzzprep/library_builder/build_parameters.py +158 -0
- fuzzprep/library_builder/dependency_resolution.py +139 -0
- fuzzprep/library_builder/environments/__init__.py +0 -0
- fuzzprep/library_builder/environments/base.py +89 -0
- fuzzprep/library_builder/environments/gate.py +96 -0
- fuzzprep/library_builder/environments/local.py +205 -0
- fuzzprep/library_builder/environments/oss_fuzz.py +485 -0
- fuzzprep/library_builder/environments/verification.py +125 -0
- fuzzprep/library_builder/exploration.py +217 -0
- fuzzprep/library_builder/generation.py +325 -0
- fuzzprep/library_builder/harness_explorer.py +257 -0
- fuzzprep/library_builder/models.py +169 -0
- fuzzprep/library_builder/package_names.json +33 -0
- fuzzprep/library_builder/package_names.py +40 -0
- fuzzprep/library_builder/scripts.py +389 -0
- fuzzprep/library_builder/stats.py +102 -0
- fuzzprep/library_builder/symbol_patterns.json +65 -0
- fuzzprep/library_builder/timeouts.py +14 -0
- fuzzprep/library_builder/workspace.py +250 -0
- fuzzprep-0.1.0.dist-info/METADATA +255 -0
- fuzzprep-0.1.0.dist-info/RECORD +56 -0
- fuzzprep-0.1.0.dist-info/WHEEL +4 -0
- fuzzprep-0.1.0.dist-info/entry_points.txt +3 -0
- fuzzprep-0.1.0.dist-info/licenses/LICENSE +202 -0
- fuzzprep-0.1.0.dist-info/licenses/THIRD_PARTY_NOTICES.md +52 -0
|
@@ -0,0 +1,99 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import hashlib
|
|
4
|
+
import logging
|
|
5
|
+
import os
|
|
6
|
+
import shlex
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
|
|
9
|
+
from fuzzprep.core.paths import default_state_dir
|
|
10
|
+
from fuzzprep.core.subprocesses import run_command, run_command_streaming
|
|
11
|
+
|
|
12
|
+
logger = logging.getLogger(__name__)
|
|
13
|
+
|
|
14
|
+
_NATIVE_SRC_DIR = Path(__file__).parent / "native"
|
|
15
|
+
_BINARY_NAME = "feature_extractor"
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class NativeBuildError(Exception):
|
|
19
|
+
"""The native feature_extractor tool failed to configure or build."""
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def build_native_tool(*, force_rebuild: bool = False) -> Path:
|
|
23
|
+
"""Build (or reuse a cached build of) the native feature_extractor binary.
|
|
24
|
+
|
|
25
|
+
Cached under .fuzzprep/native-build/build/, keyed to the `clang --version` string and a
|
|
26
|
+
hash of native/'s own sources. A new LLVM version means a rebuild rather than a silent
|
|
27
|
+
reuse against a mismatched LibTooling ABI, and an edit to the tool invalidates the cache
|
|
28
|
+
without a caller having to pass force_rebuild.
|
|
29
|
+
"""
|
|
30
|
+
build_dir = default_state_dir() / "native-build" / "build"
|
|
31
|
+
binary_path = build_dir / _BINARY_NAME
|
|
32
|
+
build_key_marker = build_dir / ".build_key"
|
|
33
|
+
current_build_key = f"{_detect_llvm_version()}:{_hash_native_sources()}"
|
|
34
|
+
if (
|
|
35
|
+
not force_rebuild
|
|
36
|
+
and binary_path.exists()
|
|
37
|
+
and build_key_marker.exists()
|
|
38
|
+
and build_key_marker.read_text().strip() == current_build_key
|
|
39
|
+
):
|
|
40
|
+
return binary_path
|
|
41
|
+
|
|
42
|
+
build_dir.mkdir(parents=True, exist_ok=True)
|
|
43
|
+
_configure(build_dir)
|
|
44
|
+
_build(build_dir)
|
|
45
|
+
|
|
46
|
+
if not binary_path.exists():
|
|
47
|
+
raise NativeBuildError(
|
|
48
|
+
f"Feature extraction build reported success but {binary_path} was not produced."
|
|
49
|
+
)
|
|
50
|
+
logger.info("Built the native feature_extractor tool at %s", binary_path)
|
|
51
|
+
build_key_marker.write_text(current_build_key)
|
|
52
|
+
return binary_path
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def _hash_native_sources() -> str:
|
|
56
|
+
"""Hash every file under native/, so any edit to the tool's sources invalidates the
|
|
57
|
+
cached binary."""
|
|
58
|
+
hasher = hashlib.sha256()
|
|
59
|
+
for path in sorted(p for p in _NATIVE_SRC_DIR.rglob("*") if p.is_file()):
|
|
60
|
+
hasher.update(str(path.relative_to(_NATIVE_SRC_DIR)).encode())
|
|
61
|
+
hasher.update(path.read_bytes())
|
|
62
|
+
return hasher.hexdigest()
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def _configure(build_dir: Path) -> None:
|
|
66
|
+
command = ["cmake", "-S", str(_NATIVE_SRC_DIR), "-B", str(build_dir)]
|
|
67
|
+
result = run_command_streaming(command, Path.cwd(), timeout=300)
|
|
68
|
+
if result.exit_code != 0:
|
|
69
|
+
raise NativeBuildError(
|
|
70
|
+
f"cmake failed to configure the native feature_extractor tool in {build_dir} "
|
|
71
|
+
f"(exit {result.exit_code}). See the cmake output above.\n"
|
|
72
|
+
"The tool needs LLVM's and Clang's CMake packages (LLVMConfig.cmake, "
|
|
73
|
+
"ClangConfig.cmake). On Debian or Ubuntu, install llvm-dev and libclang-dev. If "
|
|
74
|
+
"more than one LLVM version is installed, set CMAKE_PREFIX_PATH to the tree the "
|
|
75
|
+
"clang on PATH belongs to, because cmake can otherwise find a version whose "
|
|
76
|
+
"libraries are absent.\n"
|
|
77
|
+
f"Reproduce with:\n cd {Path.cwd()} && {shlex.join(command)}"
|
|
78
|
+
)
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def _build(build_dir: Path) -> None:
|
|
82
|
+
# process_cpu_count rather than cpu_count: it honours the CPU affinity and cgroup limit a
|
|
83
|
+
# container gives us, and this build links clangTooling, which is memory-hungry per job.
|
|
84
|
+
jobs = str(os.process_cpu_count() or 1)
|
|
85
|
+
command = ["cmake", "--build", str(build_dir), "--target", _BINARY_NAME, "-j", jobs]
|
|
86
|
+
result = run_command_streaming(command, Path.cwd(), timeout=1800)
|
|
87
|
+
if result.exit_code != 0:
|
|
88
|
+
raise NativeBuildError(
|
|
89
|
+
f"cmake failed to build the native feature_extractor tool in {build_dir} "
|
|
90
|
+
f"(exit {result.exit_code}). See the compiler output above.\n"
|
|
91
|
+
f"Reproduce with:\n cd {Path.cwd()} && {shlex.join(command)}"
|
|
92
|
+
)
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def _detect_llvm_version() -> str:
|
|
96
|
+
result = run_command(["clang", "--version"], Path.cwd(), timeout=10)
|
|
97
|
+
if result.exit_code == 0 and result.stdout:
|
|
98
|
+
return result.stdout.splitlines()[0].strip()
|
|
99
|
+
return "unknown"
|
|
File without changes
|
|
@@ -0,0 +1,472 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import re
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
|
|
6
|
+
from fuzzprep.core.agent_stream import (
|
|
7
|
+
AgentRunSummary,
|
|
8
|
+
AgentStreamResult,
|
|
9
|
+
format_agent_summary,
|
|
10
|
+
run_agent_streaming,
|
|
11
|
+
write_agent_report,
|
|
12
|
+
)
|
|
13
|
+
from fuzzprep.core.resources import skill_instructions
|
|
14
|
+
|
|
15
|
+
# Imported as modules, not by name, so a test that patches an artifact check patches the
|
|
16
|
+
# one this module calls too.
|
|
17
|
+
from fuzzprep.library_builder import exploration, harness_explorer
|
|
18
|
+
from fuzzprep.library_builder.build_parameters import neutral_compiler_environment
|
|
19
|
+
from fuzzprep.library_builder.environments import verification
|
|
20
|
+
from fuzzprep.library_builder.environments.base import Environment
|
|
21
|
+
from fuzzprep.library_builder.exploration import read_agent_report
|
|
22
|
+
from fuzzprep.library_builder.models import (
|
|
23
|
+
AgentStopReason,
|
|
24
|
+
AnalysisResult,
|
|
25
|
+
BuildExplorationResult,
|
|
26
|
+
HarnessExplorationResult,
|
|
27
|
+
HarnessPaths,
|
|
28
|
+
)
|
|
29
|
+
from fuzzprep.library_builder.scripts import HARNESS_SOURCE_DIR
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def _verification_command(
|
|
33
|
+
environment: Environment,
|
|
34
|
+
*,
|
|
35
|
+
workdir: Path,
|
|
36
|
+
project_name: str,
|
|
37
|
+
keep_artifacts: bool = False,
|
|
38
|
+
) -> str:
|
|
39
|
+
"""The command that proves a fix works in the selected environment.
|
|
40
|
+
|
|
41
|
+
Delegates to environments/verification.py, so an agent verifies its fix with the exact
|
|
42
|
+
command the pipeline gates on -- including whether that gate rebuilds the library.
|
|
43
|
+
"""
|
|
44
|
+
return " ".join(
|
|
45
|
+
verification.verification_command(
|
|
46
|
+
workdir,
|
|
47
|
+
environment=environment,
|
|
48
|
+
project_name=project_name,
|
|
49
|
+
keep_artifacts=keep_artifacts,
|
|
50
|
+
)
|
|
51
|
+
)
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
_KEEP_ARTIFACTS_NOTE = (
|
|
55
|
+
"`--keep-artifacts` reuses the install/ tree the library build already produced, so the "
|
|
56
|
+
"gate compiles the harnesses without rebuilding the library. Keep the option: the library "
|
|
57
|
+
"build is not what failed here, and rebuilding it costs the whole run's build time again. "
|
|
58
|
+
"If your fix changes build_library.sh or the Dockerfile, run the command again without "
|
|
59
|
+
"the option, because a change to the library build has to survive a cold build.\n"
|
|
60
|
+
)
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
_LOCAL_PACKAGE_POLICY = (
|
|
64
|
+
"workdir is a directory on this actual host machine. Do not run apt-get/dnf "
|
|
65
|
+
"install yourself here, even for a package you're fully confident about — that would "
|
|
66
|
+
"make an irreversible change to the user's system. Follow the missing-package steps "
|
|
67
|
+
"above: disable the optional feature if possible, otherwise report it in "
|
|
68
|
+
"agent_report.json and stop. This isn't specific to package managers: don't make any "
|
|
69
|
+
"other global change to this host either (editing ~/.bashrc or another shell rc "
|
|
70
|
+
"file, a global `pip install`/`npm install -g`, writing outside workdir) even if it "
|
|
71
|
+
"would fix the build faster — nothing outside workdir is reflected in this "
|
|
72
|
+
"project's output, so the fix wouldn't reproduce on a different machine. If the fix "
|
|
73
|
+
"requires it, put it in a script inside workdir instead."
|
|
74
|
+
)
|
|
75
|
+
|
|
76
|
+
_OSS_FUZZ_PACKAGE_POLICY = (
|
|
77
|
+
"workdir is an OSS-Fuzz project directory containing a Dockerfile. The verification "
|
|
78
|
+
"command rebuilds this Dockerfile from scratch into a fresh, disposable container "
|
|
79
|
+
"every time — it never touches this host machine. Unlike the local environment, you "
|
|
80
|
+
"MAY add packages directly: append to (or add a new) `RUN apt-get install -y "
|
|
81
|
+
"--no-install-recommends ...` line in workdir/Dockerfile, then re-run the verification "
|
|
82
|
+
"command yourself to confirm it works. Still report every package you add via "
|
|
83
|
+
"missing_apt_packages in agent_report.json — FuzzPrep needs every package "
|
|
84
|
+
"recorded even though you resolved it yourself, so the generated setup.sh installs it "
|
|
85
|
+
"too. Only stop and request human action if you cannot determine a correct apt "
|
|
86
|
+
"package name at all."
|
|
87
|
+
)
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def _package_policy_note(environment: Environment) -> str:
|
|
91
|
+
"""The environment-specific policy on installing missing system packages.
|
|
92
|
+
|
|
93
|
+
Locally, installing would mutate the user's real machine, so the agent stops and
|
|
94
|
+
reports. In the container it can edit the Dockerfile and verify the result itself.
|
|
95
|
+
"""
|
|
96
|
+
policy = (
|
|
97
|
+
_OSS_FUZZ_PACKAGE_POLICY if environment is Environment.OSS_FUZZ else _LOCAL_PACKAGE_POLICY
|
|
98
|
+
)
|
|
99
|
+
return f"### Package installation policy\n\n{policy}\n"
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
_ACTION_REQUIRED = "ACTION REQUIRED"
|
|
103
|
+
|
|
104
|
+
_BUDGET_PATTERN = re.compile(
|
|
105
|
+
"|".join(
|
|
106
|
+
(
|
|
107
|
+
# Claude: 5-hour session limit
|
|
108
|
+
r"reached the 5 hour limit",
|
|
109
|
+
r"session time limit",
|
|
110
|
+
# Codex/OpenAI: quota and rate limit errors
|
|
111
|
+
r"usage limit (?:reached|exceeded)",
|
|
112
|
+
r"reached (?:your|the).{0,80}usage limit",
|
|
113
|
+
r"rate limit (?:reached|exceeded)",
|
|
114
|
+
r"quota (?:exceeded|reached)",
|
|
115
|
+
r"exceeded your current quota",
|
|
116
|
+
r"too many requests",
|
|
117
|
+
r"\b429\b",
|
|
118
|
+
r"try again (?:after|in) \d+",
|
|
119
|
+
)
|
|
120
|
+
),
|
|
121
|
+
re.IGNORECASE | re.DOTALL,
|
|
122
|
+
)
|
|
123
|
+
|
|
124
|
+
# A transient server-side error, distinct from _BUDGET_PATTERN's 429: nothing about this run
|
|
125
|
+
# provoked it and nothing about the build has to change for the next attempt to work. The CLI
|
|
126
|
+
# prints its "API Error: ..." line as plain text rather than a stream event, so it reaches
|
|
127
|
+
# combined_text through the raw-line fallback.
|
|
128
|
+
_SERVICE_ERROR_PATTERN = re.compile(
|
|
129
|
+
"|".join(
|
|
130
|
+
(
|
|
131
|
+
r"API Error: 5\d\d",
|
|
132
|
+
r"\b529\b",
|
|
133
|
+
r"\boverloaded\b",
|
|
134
|
+
r"\b(?:502|503|504)\b.{0,40}(?:bad gateway|unavailable|timeout)",
|
|
135
|
+
r"(?:service|server) (?:is )?(?:temporarily )?unavailable",
|
|
136
|
+
r"internal server error",
|
|
137
|
+
)
|
|
138
|
+
),
|
|
139
|
+
re.IGNORECASE | re.DOTALL,
|
|
140
|
+
)
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def _service_unavailable(result: AgentStreamResult) -> bool:
|
|
144
|
+
"""Whether this invocation died on a transient API error instead of running.
|
|
145
|
+
|
|
146
|
+
Both conditions are required. The pattern alone matches a build log that merely prints
|
|
147
|
+
"overloaded", and a non-zero exit alone is the ordinary "agent tried and failed" case.
|
|
148
|
+
Together they identify the run that never reached the model -- which is corroborated by
|
|
149
|
+
the accounting: these invocations report a few tenths of a cent and zero output tokens.
|
|
150
|
+
"""
|
|
151
|
+
return bool(_SERVICE_ERROR_PATTERN.search(result.combined_text)) and result.exit_code != 0
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
def _stop_reason(result: AgentStreamResult) -> AgentStopReason | None:
|
|
155
|
+
"""The reason the agent stopped without a fix, or None if it did not stop that way.
|
|
156
|
+
|
|
157
|
+
The two conditions come from different layers, so they are read from different channels.
|
|
158
|
+
|
|
159
|
+
A budget/rate limit comes from the agent CLI and can surface anywhere in the transcript,
|
|
160
|
+
so it is matched against combined_text. It also fails the process, so it stays gated on a
|
|
161
|
+
non-zero exit code — which is what keeps the looser patterns (a bare "429") from matching
|
|
162
|
+
a build log that merely mentions them.
|
|
163
|
+
|
|
164
|
+
ACTION REQUIRED comes from the model, which cannot set the exit code: `claude --print`
|
|
165
|
+
exits 0 whenever the CLI ran. So the marker is honored on its own, and read only from
|
|
166
|
+
model_text. Matching combined_text instead failed runs where the agent had merely read a
|
|
167
|
+
file quoting the marker — SKILL.md documents it four times.
|
|
168
|
+
|
|
169
|
+
A transient server error is read the same way as a budget limit, and checked first: it is
|
|
170
|
+
the more specific of the two, and an overloaded API is the reason the run stopped even if
|
|
171
|
+
the transcript happens to also mention a rate limit.
|
|
172
|
+
"""
|
|
173
|
+
if _service_unavailable(result):
|
|
174
|
+
return AgentStopReason.SERVICE_UNAVAILABLE
|
|
175
|
+
if _BUDGET_PATTERN.search(result.combined_text) and result.exit_code != 0:
|
|
176
|
+
return AgentStopReason.BUDGET_LIMITED
|
|
177
|
+
if _ACTION_REQUIRED in result.model_text:
|
|
178
|
+
return AgentStopReason.ACTION_REQUIRED
|
|
179
|
+
return None
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
def _determine_outcome(result: AgentStreamResult) -> str:
|
|
183
|
+
"""Classify an agent invocation's outcome for the persisted/printed summary.
|
|
184
|
+
|
|
185
|
+
Reads each signal through _stop_reason, so the printed outcome can't disagree with the
|
|
186
|
+
stop reason recorded on the result.
|
|
187
|
+
"""
|
|
188
|
+
if _service_unavailable(result):
|
|
189
|
+
return AgentStopReason.SERVICE_UNAVAILABLE.value
|
|
190
|
+
if _BUDGET_PATTERN.search(result.combined_text):
|
|
191
|
+
return AgentStopReason.BUDGET_LIMITED.value
|
|
192
|
+
if result.exit_code == -1:
|
|
193
|
+
return "timed_out"
|
|
194
|
+
stop_reason = _stop_reason(result)
|
|
195
|
+
if stop_reason is not None:
|
|
196
|
+
return stop_reason.value
|
|
197
|
+
return "succeeded" if result.exit_code == 0 else "failed"
|
|
198
|
+
|
|
199
|
+
|
|
200
|
+
def _report_agent_run(report_path: Path, tool: str, result: AgentStreamResult) -> AgentRunSummary:
|
|
201
|
+
"""Write the persisted transcript+summary report and print the summary to the terminal."""
|
|
202
|
+
summary = AgentRunSummary(
|
|
203
|
+
backend=tool,
|
|
204
|
+
outcome=_determine_outcome(result),
|
|
205
|
+
duration_seconds=result.duration_seconds,
|
|
206
|
+
cost_usd=result.cost_usd,
|
|
207
|
+
input_tokens=result.input_tokens,
|
|
208
|
+
output_tokens=result.output_tokens,
|
|
209
|
+
)
|
|
210
|
+
write_agent_report(report_path, result.combined_text, summary)
|
|
211
|
+
print(format_agent_summary(summary))
|
|
212
|
+
return summary
|
|
213
|
+
|
|
214
|
+
|
|
215
|
+
def build_library_prompt(
|
|
216
|
+
analysis: AnalysisResult,
|
|
217
|
+
build_result: BuildExplorationResult,
|
|
218
|
+
workdir: Path,
|
|
219
|
+
environment: Environment,
|
|
220
|
+
) -> str:
|
|
221
|
+
"""Construct a Claude prompt for diagnosing and fixing a failed library build."""
|
|
222
|
+
instructions = skill_instructions("library_builder")
|
|
223
|
+
stdout_tail = "\n".join(build_result.stdout.splitlines()[-200:])
|
|
224
|
+
verify_command = _verification_command(
|
|
225
|
+
environment,
|
|
226
|
+
workdir=workdir,
|
|
227
|
+
project_name=analysis.project_name,
|
|
228
|
+
)
|
|
229
|
+
return (
|
|
230
|
+
f"{instructions}\n\n"
|
|
231
|
+
f"## Build failure context\n\n"
|
|
232
|
+
f"- source_dir: {analysis.source_path}\n"
|
|
233
|
+
f"- build_system: {analysis.build_system.value}\n"
|
|
234
|
+
f"- command: {' '.join(build_result.command)}\n"
|
|
235
|
+
f"- exit_code: {build_result.exit_code}\n"
|
|
236
|
+
f"- build_library.sh: {workdir / 'build_library.sh'}\n"
|
|
237
|
+
f"- install_dir: {workdir / 'install'}\n"
|
|
238
|
+
f"- build_dir: {workdir / 'build'}\n\n"
|
|
239
|
+
f"### Build output (last 200 lines)\n\n"
|
|
240
|
+
f"```\n{stdout_tail}\n```\n\n"
|
|
241
|
+
f"### Verification\n\n"
|
|
242
|
+
f"After applying a fix, verify it works by running this exact command:\n\n"
|
|
243
|
+
f" {verify_command}\n\n"
|
|
244
|
+
f"{_package_policy_note(environment)}"
|
|
245
|
+
)
|
|
246
|
+
|
|
247
|
+
|
|
248
|
+
def construct_claude_command(prompt: str) -> list[str]:
|
|
249
|
+
cmd = [
|
|
250
|
+
"claude",
|
|
251
|
+
"--print",
|
|
252
|
+
"--permission-mode",
|
|
253
|
+
"auto",
|
|
254
|
+
"--output-format=stream-json",
|
|
255
|
+
"--verbose",
|
|
256
|
+
prompt,
|
|
257
|
+
]
|
|
258
|
+
return cmd
|
|
259
|
+
|
|
260
|
+
|
|
261
|
+
def construct_codex_command(prompt: str) -> list[str]:
|
|
262
|
+
cmd = ["codex", "exec", "--sandbox", "workspace-write", "--json", prompt]
|
|
263
|
+
return cmd
|
|
264
|
+
|
|
265
|
+
|
|
266
|
+
def invoke_library_builder_agent( # noqa: PLR0913 -- public API; all 6 params are distinct required inputs
|
|
267
|
+
analysis: AnalysisResult,
|
|
268
|
+
build_result: BuildExplorationResult,
|
|
269
|
+
workdir: Path,
|
|
270
|
+
*,
|
|
271
|
+
tool: str = "claude",
|
|
272
|
+
timeout: int = 600,
|
|
273
|
+
environment: Environment = Environment.LOCAL,
|
|
274
|
+
) -> BuildExplorationResult:
|
|
275
|
+
"""Spawn a Claude Code or Codex subprocess to diagnose and fix a failed build.
|
|
276
|
+
|
|
277
|
+
Streams agent output to the terminal. CWD is set to workdir, where build_library.sh
|
|
278
|
+
lives; the agent can still read and modify the repo's build files via source_dir.
|
|
279
|
+
|
|
280
|
+
Runs with no compiler environment exported, for the reason given on
|
|
281
|
+
neutral_compiler_environment.
|
|
282
|
+
"""
|
|
283
|
+
prompt = build_library_prompt(analysis, build_result, workdir, environment)
|
|
284
|
+
if tool == "claude":
|
|
285
|
+
cmd = construct_claude_command(prompt)
|
|
286
|
+
elif tool == "codex":
|
|
287
|
+
cmd = construct_codex_command(prompt)
|
|
288
|
+
else:
|
|
289
|
+
raise ValueError(f"unknown agent tool: {tool!r}")
|
|
290
|
+
|
|
291
|
+
with neutral_compiler_environment():
|
|
292
|
+
result = run_agent_streaming(cmd, workdir, timeout, tool)
|
|
293
|
+
_report_agent_run(workdir / "agent_library_build.log", tool, result)
|
|
294
|
+
report = read_agent_report(workdir)
|
|
295
|
+
stop_reason = _stop_reason(result)
|
|
296
|
+
|
|
297
|
+
succeeded = result.exit_code == 0 and stop_reason is None
|
|
298
|
+
stderr = ""
|
|
299
|
+
validation_errors: list[str] = []
|
|
300
|
+
if succeeded:
|
|
301
|
+
validation_errors = exploration.validate_install_artifacts(workdir / "install")
|
|
302
|
+
if validation_errors:
|
|
303
|
+
succeeded = False
|
|
304
|
+
stderr += "\n" + "\n".join(validation_errors)
|
|
305
|
+
|
|
306
|
+
return BuildExplorationResult(
|
|
307
|
+
build_system=analysis.build_system,
|
|
308
|
+
succeeded=succeeded,
|
|
309
|
+
command=cmd,
|
|
310
|
+
stdout=result.combined_text,
|
|
311
|
+
stderr=stderr,
|
|
312
|
+
exit_code=result.exit_code,
|
|
313
|
+
duration_seconds=result.duration_seconds,
|
|
314
|
+
llm_used=True,
|
|
315
|
+
# Recorded on the agent lane too, not just the deterministic one: generation publishes
|
|
316
|
+
# this tree, and dropping it here shipped an output whose compile_harness.sh had no
|
|
317
|
+
# install/ to link against.
|
|
318
|
+
install_dir=(workdir / "install") if succeeded else None,
|
|
319
|
+
script_path=(workdir / "build_library.sh") if succeeded else None,
|
|
320
|
+
agent_stop_reason=stop_reason,
|
|
321
|
+
validation_errors=validation_errors,
|
|
322
|
+
cost_usd=result.cost_usd,
|
|
323
|
+
input_tokens=result.input_tokens,
|
|
324
|
+
output_tokens=result.output_tokens,
|
|
325
|
+
transcript_path=workdir / "agent_library_build.log",
|
|
326
|
+
agent_summary=report.summary if report else None,
|
|
327
|
+
missing_apt_packages=report.missing_apt_packages if report else [],
|
|
328
|
+
extra_include_paths=report.extra_include_paths if report else [],
|
|
329
|
+
extra_library_paths=report.extra_library_paths if report else [],
|
|
330
|
+
environment=environment,
|
|
331
|
+
)
|
|
332
|
+
|
|
333
|
+
|
|
334
|
+
def build_harness_prompt(
|
|
335
|
+
analysis: AnalysisResult,
|
|
336
|
+
harness: HarnessExplorationResult,
|
|
337
|
+
install_dir: Path,
|
|
338
|
+
workdir: Path,
|
|
339
|
+
environment: Environment,
|
|
340
|
+
) -> str:
|
|
341
|
+
"""Construct a Claude prompt for diagnosing and fixing a failed harness link probe.
|
|
342
|
+
|
|
343
|
+
The verification command carries the gate's own keep-artifacts decision, from
|
|
344
|
+
harness.gate_keeps_artifacts: the library tree this harness links against was built by an
|
|
345
|
+
earlier phase, and a harness fix does not invalidate it.
|
|
346
|
+
"""
|
|
347
|
+
instructions = skill_instructions("harness_builder")
|
|
348
|
+
# Streaming Runners merge stderr into stdout and leave .stderr empty, so .output is what
|
|
349
|
+
# reliably carries the diagnostic text.
|
|
350
|
+
output_tail = "\n".join(harness.output.splitlines()[-200:])
|
|
351
|
+
verify_command = _verification_command(
|
|
352
|
+
environment,
|
|
353
|
+
workdir=workdir,
|
|
354
|
+
project_name=analysis.project_name,
|
|
355
|
+
keep_artifacts=harness.gate_keeps_artifacts,
|
|
356
|
+
)
|
|
357
|
+
keep_artifacts_note = _KEEP_ARTIFACTS_NOTE if harness.gate_keeps_artifacts else ""
|
|
358
|
+
return (
|
|
359
|
+
f"{instructions}\n\n"
|
|
360
|
+
f"## Harness compilation failure context\n\n"
|
|
361
|
+
f"- source_dir: {analysis.source_path}\n"
|
|
362
|
+
f"- install_dir: {install_dir}\n"
|
|
363
|
+
f"- workdir: {workdir}\n"
|
|
364
|
+
f"- compile_harness.sh: {workdir / 'compile_harness.sh'}\n"
|
|
365
|
+
f"- harness_source: {workdir / HARNESS_SOURCE_DIR}\n"
|
|
366
|
+
f"- static_libs: {', '.join(p.name for p in harness.static_libs) or '(none)'}\n"
|
|
367
|
+
f"- auto_resolved_link_flags: {' '.join(harness.transitive_link_flags) or '(none)'}\n"
|
|
368
|
+
f"- missing_system_libs (linker-reported): "
|
|
369
|
+
f"{', '.join(harness.missing_system_libs) or '(none detected)'}\n"
|
|
370
|
+
f"- exit_code: {harness.exit_code}\n\n"
|
|
371
|
+
f"### Linker/compiler output (last 200 lines)\n\n"
|
|
372
|
+
f"```\n{output_tail}\n```\n\n"
|
|
373
|
+
f"### Verification\n\n"
|
|
374
|
+
f"After applying a fix, verify it works by running this exact command:\n\n"
|
|
375
|
+
f" {verify_command}\n\n"
|
|
376
|
+
f"{keep_artifacts_note}\n"
|
|
377
|
+
f"{_package_policy_note(environment)}"
|
|
378
|
+
)
|
|
379
|
+
|
|
380
|
+
|
|
381
|
+
def invoke_harness_builder_agent( # noqa: PLR0913 -- public API; all 6 params are distinct required inputs
|
|
382
|
+
analysis: AnalysisResult,
|
|
383
|
+
harness: HarnessExplorationResult,
|
|
384
|
+
paths: HarnessPaths,
|
|
385
|
+
*,
|
|
386
|
+
tool: str = "claude",
|
|
387
|
+
timeout: int = 600,
|
|
388
|
+
environment: Environment = Environment.LOCAL,
|
|
389
|
+
) -> HarnessExplorationResult:
|
|
390
|
+
"""Spawn a Claude Code or Codex subprocess to diagnose and fix a failed harness link probe.
|
|
391
|
+
|
|
392
|
+
Streams agent output to the terminal. CWD is set to paths.workdir so the agent can read
|
|
393
|
+
and modify compile_harness.sh and harness_source/ directly.
|
|
394
|
+
|
|
395
|
+
Runs with no compiler environment exported, for the reason given on
|
|
396
|
+
neutral_compiler_environment.
|
|
397
|
+
"""
|
|
398
|
+
prompt = build_harness_prompt(analysis, harness, paths.install_dir, paths.workdir, environment)
|
|
399
|
+
if tool == "claude":
|
|
400
|
+
cmd = construct_claude_command(prompt)
|
|
401
|
+
elif tool == "codex":
|
|
402
|
+
cmd = construct_codex_command(prompt)
|
|
403
|
+
else:
|
|
404
|
+
raise ValueError(f"unknown agent tool: {tool!r}")
|
|
405
|
+
|
|
406
|
+
with neutral_compiler_environment():
|
|
407
|
+
result = run_agent_streaming(cmd, paths.workdir, timeout, tool)
|
|
408
|
+
_report_agent_run(paths.workdir / "agent_harness_build.log", tool, result)
|
|
409
|
+
report = read_agent_report(paths.workdir)
|
|
410
|
+
stop_reason = _stop_reason(result)
|
|
411
|
+
|
|
412
|
+
succeeded = result.exit_code == 0 and stop_reason is None
|
|
413
|
+
stderr = ""
|
|
414
|
+
missing_system_libs = harness.missing_system_libs
|
|
415
|
+
# The agent edits compile_harness.sh directly and that script is what ships, so its text
|
|
416
|
+
# is the link configuration of record. These lists only describe what was tried.
|
|
417
|
+
static_libs = harness.static_libs
|
|
418
|
+
transitive_link_flags = harness.transitive_link_flags
|
|
419
|
+
extra_library_paths = harness.extra_library_paths
|
|
420
|
+
script_path = paths.workdir / "compile_harness.sh"
|
|
421
|
+
validation_errors: list[str] = []
|
|
422
|
+
if succeeded:
|
|
423
|
+
validation_errors = harness_explorer.validate_harness_artifacts(paths.workdir)
|
|
424
|
+
if validation_errors:
|
|
425
|
+
succeeded = False
|
|
426
|
+
stderr += "\n" + "\n".join(validation_errors)
|
|
427
|
+
if not succeeded:
|
|
428
|
+
# A validation-only failure (agent claimed done but out/ is empty) has no new linker
|
|
429
|
+
# stderr to re-parse, so keep the pre-agent list rather than discarding it.
|
|
430
|
+
missing_system_libs = list(
|
|
431
|
+
dict.fromkeys(
|
|
432
|
+
missing_system_libs + harness_explorer.extract_missing_system_libs(stderr)
|
|
433
|
+
)
|
|
434
|
+
)
|
|
435
|
+
|
|
436
|
+
# The agent reports the bare library name it could not resolve (e.g. "ldap"). Add the
|
|
437
|
+
# matching -l flag so the generated script links it once the package is installed.
|
|
438
|
+
report_missing_libs = report.missing_libs if report else []
|
|
439
|
+
if report_missing_libs:
|
|
440
|
+
missing_system_libs = list(dict.fromkeys(missing_system_libs + report_missing_libs))
|
|
441
|
+
new_flags = [f"-l{lib}" for lib in report_missing_libs]
|
|
442
|
+
transitive_link_flags = list(dict.fromkeys(transitive_link_flags + new_flags))
|
|
443
|
+
|
|
444
|
+
report_library_paths = report.extra_library_paths if report else []
|
|
445
|
+
extra_library_paths = list(dict.fromkeys(extra_library_paths + report_library_paths))
|
|
446
|
+
|
|
447
|
+
return HarnessExplorationResult(
|
|
448
|
+
succeeded=succeeded,
|
|
449
|
+
command=cmd,
|
|
450
|
+
static_libs=static_libs,
|
|
451
|
+
include_dir=harness.include_dir,
|
|
452
|
+
transitive_link_flags=transitive_link_flags,
|
|
453
|
+
stdout=result.combined_text,
|
|
454
|
+
stderr=stderr,
|
|
455
|
+
exit_code=result.exit_code,
|
|
456
|
+
missing_system_libs=missing_system_libs if not succeeded else [],
|
|
457
|
+
llm_used=True,
|
|
458
|
+
script_path=script_path if succeeded else None,
|
|
459
|
+
agent_stop_reason=stop_reason,
|
|
460
|
+
validation_errors=validation_errors,
|
|
461
|
+
duration_seconds=result.duration_seconds,
|
|
462
|
+
cost_usd=result.cost_usd,
|
|
463
|
+
input_tokens=result.input_tokens,
|
|
464
|
+
output_tokens=result.output_tokens,
|
|
465
|
+
transcript_path=paths.workdir / "agent_harness_build.log",
|
|
466
|
+
agent_summary=report.summary if report else None,
|
|
467
|
+
missing_apt_packages=report.missing_apt_packages if report else [],
|
|
468
|
+
extra_include_paths=report.extra_include_paths if report else [],
|
|
469
|
+
extra_library_paths=extra_library_paths,
|
|
470
|
+
environment=environment,
|
|
471
|
+
gate_keeps_artifacts=harness.gate_keeps_artifacts,
|
|
472
|
+
)
|