fuzzprep 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (56) hide show
  1. fuzzprep/__init__.py +3 -0
  2. fuzzprep/__main__.py +4 -0
  3. fuzzprep/agents/harness_builder/SKILL.md +156 -0
  4. fuzzprep/agents/library_builder/SKILL.md +147 -0
  5. fuzzprep/agents/scripts/check_build.sh +87 -0
  6. fuzzprep/agents/scripts/check_build_in_container.sh +73 -0
  7. fuzzprep/agents/scripts/check_dockerfile_from_scratch.sh +33 -0
  8. fuzzprep/cli.py +1109 -0
  9. fuzzprep/core/__init__.py +0 -0
  10. fuzzprep/core/agent_stream.py +304 -0
  11. fuzzprep/core/files.py +35 -0
  12. fuzzprep/core/paths.py +25 -0
  13. fuzzprep/core/reporting.py +218 -0
  14. fuzzprep/core/repos.py +217 -0
  15. fuzzprep/core/resources.py +41 -0
  16. fuzzprep/core/subprocesses.py +197 -0
  17. fuzzprep/feature_extractor/__init__.py +0 -0
  18. fuzzprep/feature_extractor/benchmark_yaml.py +92 -0
  19. fuzzprep/feature_extractor/extraction.py +184 -0
  20. fuzzprep/feature_extractor/models.py +96 -0
  21. fuzzprep/feature_extractor/native/.clang-format +1 -0
  22. fuzzprep/feature_extractor/native/CMakeLists.txt +90 -0
  23. fuzzprep/feature_extractor/native/include/feature_extractor.hpp +136 -0
  24. fuzzprep/feature_extractor/native/src/extraction_action.cpp +294 -0
  25. fuzzprep/feature_extractor/native/src/json_writer.cpp +154 -0
  26. fuzzprep/feature_extractor/native/src/macro_callbacks.cpp +75 -0
  27. fuzzprep/feature_extractor/native/src/main.cpp +130 -0
  28. fuzzprep/feature_extractor/native_build.py +99 -0
  29. fuzzprep/library_builder/__init__.py +0 -0
  30. fuzzprep/library_builder/agents.py +472 -0
  31. fuzzprep/library_builder/analysis.py +145 -0
  32. fuzzprep/library_builder/build_parameters.py +158 -0
  33. fuzzprep/library_builder/dependency_resolution.py +139 -0
  34. fuzzprep/library_builder/environments/__init__.py +0 -0
  35. fuzzprep/library_builder/environments/base.py +89 -0
  36. fuzzprep/library_builder/environments/gate.py +96 -0
  37. fuzzprep/library_builder/environments/local.py +205 -0
  38. fuzzprep/library_builder/environments/oss_fuzz.py +485 -0
  39. fuzzprep/library_builder/environments/verification.py +125 -0
  40. fuzzprep/library_builder/exploration.py +217 -0
  41. fuzzprep/library_builder/generation.py +325 -0
  42. fuzzprep/library_builder/harness_explorer.py +257 -0
  43. fuzzprep/library_builder/models.py +169 -0
  44. fuzzprep/library_builder/package_names.json +33 -0
  45. fuzzprep/library_builder/package_names.py +40 -0
  46. fuzzprep/library_builder/scripts.py +389 -0
  47. fuzzprep/library_builder/stats.py +102 -0
  48. fuzzprep/library_builder/symbol_patterns.json +65 -0
  49. fuzzprep/library_builder/timeouts.py +14 -0
  50. fuzzprep/library_builder/workspace.py +250 -0
  51. fuzzprep-0.1.0.dist-info/METADATA +255 -0
  52. fuzzprep-0.1.0.dist-info/RECORD +56 -0
  53. fuzzprep-0.1.0.dist-info/WHEEL +4 -0
  54. fuzzprep-0.1.0.dist-info/entry_points.txt +3 -0
  55. fuzzprep-0.1.0.dist-info/licenses/LICENSE +202 -0
  56. fuzzprep-0.1.0.dist-info/licenses/THIRD_PARTY_NOTICES.md +52 -0
@@ -0,0 +1,99 @@
1
+ from __future__ import annotations
2
+
3
+ import hashlib
4
+ import logging
5
+ import os
6
+ import shlex
7
+ from pathlib import Path
8
+
9
+ from fuzzprep.core.paths import default_state_dir
10
+ from fuzzprep.core.subprocesses import run_command, run_command_streaming
11
+
12
+ logger = logging.getLogger(__name__)
13
+
14
+ _NATIVE_SRC_DIR = Path(__file__).parent / "native"
15
+ _BINARY_NAME = "feature_extractor"
16
+
17
+
18
+ class NativeBuildError(Exception):
19
+ """The native feature_extractor tool failed to configure or build."""
20
+
21
+
22
+ def build_native_tool(*, force_rebuild: bool = False) -> Path:
23
+ """Build (or reuse a cached build of) the native feature_extractor binary.
24
+
25
+ Cached under .fuzzprep/native-build/build/, keyed to the `clang --version` string and a
26
+ hash of native/'s own sources. A new LLVM version means a rebuild rather than a silent
27
+ reuse against a mismatched LibTooling ABI, and an edit to the tool invalidates the cache
28
+ without a caller having to pass force_rebuild.
29
+ """
30
+ build_dir = default_state_dir() / "native-build" / "build"
31
+ binary_path = build_dir / _BINARY_NAME
32
+ build_key_marker = build_dir / ".build_key"
33
+ current_build_key = f"{_detect_llvm_version()}:{_hash_native_sources()}"
34
+ if (
35
+ not force_rebuild
36
+ and binary_path.exists()
37
+ and build_key_marker.exists()
38
+ and build_key_marker.read_text().strip() == current_build_key
39
+ ):
40
+ return binary_path
41
+
42
+ build_dir.mkdir(parents=True, exist_ok=True)
43
+ _configure(build_dir)
44
+ _build(build_dir)
45
+
46
+ if not binary_path.exists():
47
+ raise NativeBuildError(
48
+ f"Feature extraction build reported success but {binary_path} was not produced."
49
+ )
50
+ logger.info("Built the native feature_extractor tool at %s", binary_path)
51
+ build_key_marker.write_text(current_build_key)
52
+ return binary_path
53
+
54
+
55
+ def _hash_native_sources() -> str:
56
+ """Hash every file under native/, so any edit to the tool's sources invalidates the
57
+ cached binary."""
58
+ hasher = hashlib.sha256()
59
+ for path in sorted(p for p in _NATIVE_SRC_DIR.rglob("*") if p.is_file()):
60
+ hasher.update(str(path.relative_to(_NATIVE_SRC_DIR)).encode())
61
+ hasher.update(path.read_bytes())
62
+ return hasher.hexdigest()
63
+
64
+
65
+ def _configure(build_dir: Path) -> None:
66
+ command = ["cmake", "-S", str(_NATIVE_SRC_DIR), "-B", str(build_dir)]
67
+ result = run_command_streaming(command, Path.cwd(), timeout=300)
68
+ if result.exit_code != 0:
69
+ raise NativeBuildError(
70
+ f"cmake failed to configure the native feature_extractor tool in {build_dir} "
71
+ f"(exit {result.exit_code}). See the cmake output above.\n"
72
+ "The tool needs LLVM's and Clang's CMake packages (LLVMConfig.cmake, "
73
+ "ClangConfig.cmake). On Debian or Ubuntu, install llvm-dev and libclang-dev. If "
74
+ "more than one LLVM version is installed, set CMAKE_PREFIX_PATH to the tree the "
75
+ "clang on PATH belongs to, because cmake can otherwise find a version whose "
76
+ "libraries are absent.\n"
77
+ f"Reproduce with:\n cd {Path.cwd()} && {shlex.join(command)}"
78
+ )
79
+
80
+
81
+ def _build(build_dir: Path) -> None:
82
+ # process_cpu_count rather than cpu_count: it honours the CPU affinity and cgroup limit a
83
+ # container gives us, and this build links clangTooling, which is memory-hungry per job.
84
+ jobs = str(os.process_cpu_count() or 1)
85
+ command = ["cmake", "--build", str(build_dir), "--target", _BINARY_NAME, "-j", jobs]
86
+ result = run_command_streaming(command, Path.cwd(), timeout=1800)
87
+ if result.exit_code != 0:
88
+ raise NativeBuildError(
89
+ f"cmake failed to build the native feature_extractor tool in {build_dir} "
90
+ f"(exit {result.exit_code}). See the compiler output above.\n"
91
+ f"Reproduce with:\n cd {Path.cwd()} && {shlex.join(command)}"
92
+ )
93
+
94
+
95
+ def _detect_llvm_version() -> str:
96
+ result = run_command(["clang", "--version"], Path.cwd(), timeout=10)
97
+ if result.exit_code == 0 and result.stdout:
98
+ return result.stdout.splitlines()[0].strip()
99
+ return "unknown"
File without changes
@@ -0,0 +1,472 @@
1
+ from __future__ import annotations
2
+
3
+ import re
4
+ from pathlib import Path
5
+
6
+ from fuzzprep.core.agent_stream import (
7
+ AgentRunSummary,
8
+ AgentStreamResult,
9
+ format_agent_summary,
10
+ run_agent_streaming,
11
+ write_agent_report,
12
+ )
13
+ from fuzzprep.core.resources import skill_instructions
14
+
15
+ # Imported as modules, not by name, so a test that patches an artifact check patches the
16
+ # one this module calls too.
17
+ from fuzzprep.library_builder import exploration, harness_explorer
18
+ from fuzzprep.library_builder.build_parameters import neutral_compiler_environment
19
+ from fuzzprep.library_builder.environments import verification
20
+ from fuzzprep.library_builder.environments.base import Environment
21
+ from fuzzprep.library_builder.exploration import read_agent_report
22
+ from fuzzprep.library_builder.models import (
23
+ AgentStopReason,
24
+ AnalysisResult,
25
+ BuildExplorationResult,
26
+ HarnessExplorationResult,
27
+ HarnessPaths,
28
+ )
29
+ from fuzzprep.library_builder.scripts import HARNESS_SOURCE_DIR
30
+
31
+
32
+ def _verification_command(
33
+ environment: Environment,
34
+ *,
35
+ workdir: Path,
36
+ project_name: str,
37
+ keep_artifacts: bool = False,
38
+ ) -> str:
39
+ """The command that proves a fix works in the selected environment.
40
+
41
+ Delegates to environments/verification.py, so an agent verifies its fix with the exact
42
+ command the pipeline gates on -- including whether that gate rebuilds the library.
43
+ """
44
+ return " ".join(
45
+ verification.verification_command(
46
+ workdir,
47
+ environment=environment,
48
+ project_name=project_name,
49
+ keep_artifacts=keep_artifacts,
50
+ )
51
+ )
52
+
53
+
54
+ _KEEP_ARTIFACTS_NOTE = (
55
+ "`--keep-artifacts` reuses the install/ tree the library build already produced, so the "
56
+ "gate compiles the harnesses without rebuilding the library. Keep the option: the library "
57
+ "build is not what failed here, and rebuilding it costs the whole run's build time again. "
58
+ "If your fix changes build_library.sh or the Dockerfile, run the command again without "
59
+ "the option, because a change to the library build has to survive a cold build.\n"
60
+ )
61
+
62
+
63
+ _LOCAL_PACKAGE_POLICY = (
64
+ "workdir is a directory on this actual host machine. Do not run apt-get/dnf "
65
+ "install yourself here, even for a package you're fully confident about — that would "
66
+ "make an irreversible change to the user's system. Follow the missing-package steps "
67
+ "above: disable the optional feature if possible, otherwise report it in "
68
+ "agent_report.json and stop. This isn't specific to package managers: don't make any "
69
+ "other global change to this host either (editing ~/.bashrc or another shell rc "
70
+ "file, a global `pip install`/`npm install -g`, writing outside workdir) even if it "
71
+ "would fix the build faster — nothing outside workdir is reflected in this "
72
+ "project's output, so the fix wouldn't reproduce on a different machine. If the fix "
73
+ "requires it, put it in a script inside workdir instead."
74
+ )
75
+
76
+ _OSS_FUZZ_PACKAGE_POLICY = (
77
+ "workdir is an OSS-Fuzz project directory containing a Dockerfile. The verification "
78
+ "command rebuilds this Dockerfile from scratch into a fresh, disposable container "
79
+ "every time — it never touches this host machine. Unlike the local environment, you "
80
+ "MAY add packages directly: append to (or add a new) `RUN apt-get install -y "
81
+ "--no-install-recommends ...` line in workdir/Dockerfile, then re-run the verification "
82
+ "command yourself to confirm it works. Still report every package you add via "
83
+ "missing_apt_packages in agent_report.json — FuzzPrep needs every package "
84
+ "recorded even though you resolved it yourself, so the generated setup.sh installs it "
85
+ "too. Only stop and request human action if you cannot determine a correct apt "
86
+ "package name at all."
87
+ )
88
+
89
+
90
+ def _package_policy_note(environment: Environment) -> str:
91
+ """The environment-specific policy on installing missing system packages.
92
+
93
+ Locally, installing would mutate the user's real machine, so the agent stops and
94
+ reports. In the container it can edit the Dockerfile and verify the result itself.
95
+ """
96
+ policy = (
97
+ _OSS_FUZZ_PACKAGE_POLICY if environment is Environment.OSS_FUZZ else _LOCAL_PACKAGE_POLICY
98
+ )
99
+ return f"### Package installation policy\n\n{policy}\n"
100
+
101
+
102
+ _ACTION_REQUIRED = "ACTION REQUIRED"
103
+
104
+ _BUDGET_PATTERN = re.compile(
105
+ "|".join(
106
+ (
107
+ # Claude: 5-hour session limit
108
+ r"reached the 5 hour limit",
109
+ r"session time limit",
110
+ # Codex/OpenAI: quota and rate limit errors
111
+ r"usage limit (?:reached|exceeded)",
112
+ r"reached (?:your|the).{0,80}usage limit",
113
+ r"rate limit (?:reached|exceeded)",
114
+ r"quota (?:exceeded|reached)",
115
+ r"exceeded your current quota",
116
+ r"too many requests",
117
+ r"\b429\b",
118
+ r"try again (?:after|in) \d+",
119
+ )
120
+ ),
121
+ re.IGNORECASE | re.DOTALL,
122
+ )
123
+
124
+ # A transient server-side error, distinct from _BUDGET_PATTERN's 429: nothing about this run
125
+ # provoked it and nothing about the build has to change for the next attempt to work. The CLI
126
+ # prints its "API Error: ..." line as plain text rather than a stream event, so it reaches
127
+ # combined_text through the raw-line fallback.
128
+ _SERVICE_ERROR_PATTERN = re.compile(
129
+ "|".join(
130
+ (
131
+ r"API Error: 5\d\d",
132
+ r"\b529\b",
133
+ r"\boverloaded\b",
134
+ r"\b(?:502|503|504)\b.{0,40}(?:bad gateway|unavailable|timeout)",
135
+ r"(?:service|server) (?:is )?(?:temporarily )?unavailable",
136
+ r"internal server error",
137
+ )
138
+ ),
139
+ re.IGNORECASE | re.DOTALL,
140
+ )
141
+
142
+
143
+ def _service_unavailable(result: AgentStreamResult) -> bool:
144
+ """Whether this invocation died on a transient API error instead of running.
145
+
146
+ Both conditions are required. The pattern alone matches a build log that merely prints
147
+ "overloaded", and a non-zero exit alone is the ordinary "agent tried and failed" case.
148
+ Together they identify the run that never reached the model -- which is corroborated by
149
+ the accounting: these invocations report a few tenths of a cent and zero output tokens.
150
+ """
151
+ return bool(_SERVICE_ERROR_PATTERN.search(result.combined_text)) and result.exit_code != 0
152
+
153
+
154
+ def _stop_reason(result: AgentStreamResult) -> AgentStopReason | None:
155
+ """The reason the agent stopped without a fix, or None if it did not stop that way.
156
+
157
+ The two conditions come from different layers, so they are read from different channels.
158
+
159
+ A budget/rate limit comes from the agent CLI and can surface anywhere in the transcript,
160
+ so it is matched against combined_text. It also fails the process, so it stays gated on a
161
+ non-zero exit code — which is what keeps the looser patterns (a bare "429") from matching
162
+ a build log that merely mentions them.
163
+
164
+ ACTION REQUIRED comes from the model, which cannot set the exit code: `claude --print`
165
+ exits 0 whenever the CLI ran. So the marker is honored on its own, and read only from
166
+ model_text. Matching combined_text instead failed runs where the agent had merely read a
167
+ file quoting the marker — SKILL.md documents it four times.
168
+
169
+ A transient server error is read the same way as a budget limit, and checked first: it is
170
+ the more specific of the two, and an overloaded API is the reason the run stopped even if
171
+ the transcript happens to also mention a rate limit.
172
+ """
173
+ if _service_unavailable(result):
174
+ return AgentStopReason.SERVICE_UNAVAILABLE
175
+ if _BUDGET_PATTERN.search(result.combined_text) and result.exit_code != 0:
176
+ return AgentStopReason.BUDGET_LIMITED
177
+ if _ACTION_REQUIRED in result.model_text:
178
+ return AgentStopReason.ACTION_REQUIRED
179
+ return None
180
+
181
+
182
+ def _determine_outcome(result: AgentStreamResult) -> str:
183
+ """Classify an agent invocation's outcome for the persisted/printed summary.
184
+
185
+ Reads each signal through _stop_reason, so the printed outcome can't disagree with the
186
+ stop reason recorded on the result.
187
+ """
188
+ if _service_unavailable(result):
189
+ return AgentStopReason.SERVICE_UNAVAILABLE.value
190
+ if _BUDGET_PATTERN.search(result.combined_text):
191
+ return AgentStopReason.BUDGET_LIMITED.value
192
+ if result.exit_code == -1:
193
+ return "timed_out"
194
+ stop_reason = _stop_reason(result)
195
+ if stop_reason is not None:
196
+ return stop_reason.value
197
+ return "succeeded" if result.exit_code == 0 else "failed"
198
+
199
+
200
+ def _report_agent_run(report_path: Path, tool: str, result: AgentStreamResult) -> AgentRunSummary:
201
+ """Write the persisted transcript+summary report and print the summary to the terminal."""
202
+ summary = AgentRunSummary(
203
+ backend=tool,
204
+ outcome=_determine_outcome(result),
205
+ duration_seconds=result.duration_seconds,
206
+ cost_usd=result.cost_usd,
207
+ input_tokens=result.input_tokens,
208
+ output_tokens=result.output_tokens,
209
+ )
210
+ write_agent_report(report_path, result.combined_text, summary)
211
+ print(format_agent_summary(summary))
212
+ return summary
213
+
214
+
215
+ def build_library_prompt(
216
+ analysis: AnalysisResult,
217
+ build_result: BuildExplorationResult,
218
+ workdir: Path,
219
+ environment: Environment,
220
+ ) -> str:
221
+ """Construct a Claude prompt for diagnosing and fixing a failed library build."""
222
+ instructions = skill_instructions("library_builder")
223
+ stdout_tail = "\n".join(build_result.stdout.splitlines()[-200:])
224
+ verify_command = _verification_command(
225
+ environment,
226
+ workdir=workdir,
227
+ project_name=analysis.project_name,
228
+ )
229
+ return (
230
+ f"{instructions}\n\n"
231
+ f"## Build failure context\n\n"
232
+ f"- source_dir: {analysis.source_path}\n"
233
+ f"- build_system: {analysis.build_system.value}\n"
234
+ f"- command: {' '.join(build_result.command)}\n"
235
+ f"- exit_code: {build_result.exit_code}\n"
236
+ f"- build_library.sh: {workdir / 'build_library.sh'}\n"
237
+ f"- install_dir: {workdir / 'install'}\n"
238
+ f"- build_dir: {workdir / 'build'}\n\n"
239
+ f"### Build output (last 200 lines)\n\n"
240
+ f"```\n{stdout_tail}\n```\n\n"
241
+ f"### Verification\n\n"
242
+ f"After applying a fix, verify it works by running this exact command:\n\n"
243
+ f" {verify_command}\n\n"
244
+ f"{_package_policy_note(environment)}"
245
+ )
246
+
247
+
248
+ def construct_claude_command(prompt: str) -> list[str]:
249
+ cmd = [
250
+ "claude",
251
+ "--print",
252
+ "--permission-mode",
253
+ "auto",
254
+ "--output-format=stream-json",
255
+ "--verbose",
256
+ prompt,
257
+ ]
258
+ return cmd
259
+
260
+
261
+ def construct_codex_command(prompt: str) -> list[str]:
262
+ cmd = ["codex", "exec", "--sandbox", "workspace-write", "--json", prompt]
263
+ return cmd
264
+
265
+
266
+ def invoke_library_builder_agent( # noqa: PLR0913 -- public API; all 6 params are distinct required inputs
267
+ analysis: AnalysisResult,
268
+ build_result: BuildExplorationResult,
269
+ workdir: Path,
270
+ *,
271
+ tool: str = "claude",
272
+ timeout: int = 600,
273
+ environment: Environment = Environment.LOCAL,
274
+ ) -> BuildExplorationResult:
275
+ """Spawn a Claude Code or Codex subprocess to diagnose and fix a failed build.
276
+
277
+ Streams agent output to the terminal. CWD is set to workdir, where build_library.sh
278
+ lives; the agent can still read and modify the repo's build files via source_dir.
279
+
280
+ Runs with no compiler environment exported, for the reason given on
281
+ neutral_compiler_environment.
282
+ """
283
+ prompt = build_library_prompt(analysis, build_result, workdir, environment)
284
+ if tool == "claude":
285
+ cmd = construct_claude_command(prompt)
286
+ elif tool == "codex":
287
+ cmd = construct_codex_command(prompt)
288
+ else:
289
+ raise ValueError(f"unknown agent tool: {tool!r}")
290
+
291
+ with neutral_compiler_environment():
292
+ result = run_agent_streaming(cmd, workdir, timeout, tool)
293
+ _report_agent_run(workdir / "agent_library_build.log", tool, result)
294
+ report = read_agent_report(workdir)
295
+ stop_reason = _stop_reason(result)
296
+
297
+ succeeded = result.exit_code == 0 and stop_reason is None
298
+ stderr = ""
299
+ validation_errors: list[str] = []
300
+ if succeeded:
301
+ validation_errors = exploration.validate_install_artifacts(workdir / "install")
302
+ if validation_errors:
303
+ succeeded = False
304
+ stderr += "\n" + "\n".join(validation_errors)
305
+
306
+ return BuildExplorationResult(
307
+ build_system=analysis.build_system,
308
+ succeeded=succeeded,
309
+ command=cmd,
310
+ stdout=result.combined_text,
311
+ stderr=stderr,
312
+ exit_code=result.exit_code,
313
+ duration_seconds=result.duration_seconds,
314
+ llm_used=True,
315
+ # Recorded on the agent lane too, not just the deterministic one: generation publishes
316
+ # this tree, and dropping it here shipped an output whose compile_harness.sh had no
317
+ # install/ to link against.
318
+ install_dir=(workdir / "install") if succeeded else None,
319
+ script_path=(workdir / "build_library.sh") if succeeded else None,
320
+ agent_stop_reason=stop_reason,
321
+ validation_errors=validation_errors,
322
+ cost_usd=result.cost_usd,
323
+ input_tokens=result.input_tokens,
324
+ output_tokens=result.output_tokens,
325
+ transcript_path=workdir / "agent_library_build.log",
326
+ agent_summary=report.summary if report else None,
327
+ missing_apt_packages=report.missing_apt_packages if report else [],
328
+ extra_include_paths=report.extra_include_paths if report else [],
329
+ extra_library_paths=report.extra_library_paths if report else [],
330
+ environment=environment,
331
+ )
332
+
333
+
334
+ def build_harness_prompt(
335
+ analysis: AnalysisResult,
336
+ harness: HarnessExplorationResult,
337
+ install_dir: Path,
338
+ workdir: Path,
339
+ environment: Environment,
340
+ ) -> str:
341
+ """Construct a Claude prompt for diagnosing and fixing a failed harness link probe.
342
+
343
+ The verification command carries the gate's own keep-artifacts decision, from
344
+ harness.gate_keeps_artifacts: the library tree this harness links against was built by an
345
+ earlier phase, and a harness fix does not invalidate it.
346
+ """
347
+ instructions = skill_instructions("harness_builder")
348
+ # Streaming Runners merge stderr into stdout and leave .stderr empty, so .output is what
349
+ # reliably carries the diagnostic text.
350
+ output_tail = "\n".join(harness.output.splitlines()[-200:])
351
+ verify_command = _verification_command(
352
+ environment,
353
+ workdir=workdir,
354
+ project_name=analysis.project_name,
355
+ keep_artifacts=harness.gate_keeps_artifacts,
356
+ )
357
+ keep_artifacts_note = _KEEP_ARTIFACTS_NOTE if harness.gate_keeps_artifacts else ""
358
+ return (
359
+ f"{instructions}\n\n"
360
+ f"## Harness compilation failure context\n\n"
361
+ f"- source_dir: {analysis.source_path}\n"
362
+ f"- install_dir: {install_dir}\n"
363
+ f"- workdir: {workdir}\n"
364
+ f"- compile_harness.sh: {workdir / 'compile_harness.sh'}\n"
365
+ f"- harness_source: {workdir / HARNESS_SOURCE_DIR}\n"
366
+ f"- static_libs: {', '.join(p.name for p in harness.static_libs) or '(none)'}\n"
367
+ f"- auto_resolved_link_flags: {' '.join(harness.transitive_link_flags) or '(none)'}\n"
368
+ f"- missing_system_libs (linker-reported): "
369
+ f"{', '.join(harness.missing_system_libs) or '(none detected)'}\n"
370
+ f"- exit_code: {harness.exit_code}\n\n"
371
+ f"### Linker/compiler output (last 200 lines)\n\n"
372
+ f"```\n{output_tail}\n```\n\n"
373
+ f"### Verification\n\n"
374
+ f"After applying a fix, verify it works by running this exact command:\n\n"
375
+ f" {verify_command}\n\n"
376
+ f"{keep_artifacts_note}\n"
377
+ f"{_package_policy_note(environment)}"
378
+ )
379
+
380
+
381
+ def invoke_harness_builder_agent( # noqa: PLR0913 -- public API; all 6 params are distinct required inputs
382
+ analysis: AnalysisResult,
383
+ harness: HarnessExplorationResult,
384
+ paths: HarnessPaths,
385
+ *,
386
+ tool: str = "claude",
387
+ timeout: int = 600,
388
+ environment: Environment = Environment.LOCAL,
389
+ ) -> HarnessExplorationResult:
390
+ """Spawn a Claude Code or Codex subprocess to diagnose and fix a failed harness link probe.
391
+
392
+ Streams agent output to the terminal. CWD is set to paths.workdir so the agent can read
393
+ and modify compile_harness.sh and harness_source/ directly.
394
+
395
+ Runs with no compiler environment exported, for the reason given on
396
+ neutral_compiler_environment.
397
+ """
398
+ prompt = build_harness_prompt(analysis, harness, paths.install_dir, paths.workdir, environment)
399
+ if tool == "claude":
400
+ cmd = construct_claude_command(prompt)
401
+ elif tool == "codex":
402
+ cmd = construct_codex_command(prompt)
403
+ else:
404
+ raise ValueError(f"unknown agent tool: {tool!r}")
405
+
406
+ with neutral_compiler_environment():
407
+ result = run_agent_streaming(cmd, paths.workdir, timeout, tool)
408
+ _report_agent_run(paths.workdir / "agent_harness_build.log", tool, result)
409
+ report = read_agent_report(paths.workdir)
410
+ stop_reason = _stop_reason(result)
411
+
412
+ succeeded = result.exit_code == 0 and stop_reason is None
413
+ stderr = ""
414
+ missing_system_libs = harness.missing_system_libs
415
+ # The agent edits compile_harness.sh directly and that script is what ships, so its text
416
+ # is the link configuration of record. These lists only describe what was tried.
417
+ static_libs = harness.static_libs
418
+ transitive_link_flags = harness.transitive_link_flags
419
+ extra_library_paths = harness.extra_library_paths
420
+ script_path = paths.workdir / "compile_harness.sh"
421
+ validation_errors: list[str] = []
422
+ if succeeded:
423
+ validation_errors = harness_explorer.validate_harness_artifacts(paths.workdir)
424
+ if validation_errors:
425
+ succeeded = False
426
+ stderr += "\n" + "\n".join(validation_errors)
427
+ if not succeeded:
428
+ # A validation-only failure (agent claimed done but out/ is empty) has no new linker
429
+ # stderr to re-parse, so keep the pre-agent list rather than discarding it.
430
+ missing_system_libs = list(
431
+ dict.fromkeys(
432
+ missing_system_libs + harness_explorer.extract_missing_system_libs(stderr)
433
+ )
434
+ )
435
+
436
+ # The agent reports the bare library name it could not resolve (e.g. "ldap"). Add the
437
+ # matching -l flag so the generated script links it once the package is installed.
438
+ report_missing_libs = report.missing_libs if report else []
439
+ if report_missing_libs:
440
+ missing_system_libs = list(dict.fromkeys(missing_system_libs + report_missing_libs))
441
+ new_flags = [f"-l{lib}" for lib in report_missing_libs]
442
+ transitive_link_flags = list(dict.fromkeys(transitive_link_flags + new_flags))
443
+
444
+ report_library_paths = report.extra_library_paths if report else []
445
+ extra_library_paths = list(dict.fromkeys(extra_library_paths + report_library_paths))
446
+
447
+ return HarnessExplorationResult(
448
+ succeeded=succeeded,
449
+ command=cmd,
450
+ static_libs=static_libs,
451
+ include_dir=harness.include_dir,
452
+ transitive_link_flags=transitive_link_flags,
453
+ stdout=result.combined_text,
454
+ stderr=stderr,
455
+ exit_code=result.exit_code,
456
+ missing_system_libs=missing_system_libs if not succeeded else [],
457
+ llm_used=True,
458
+ script_path=script_path if succeeded else None,
459
+ agent_stop_reason=stop_reason,
460
+ validation_errors=validation_errors,
461
+ duration_seconds=result.duration_seconds,
462
+ cost_usd=result.cost_usd,
463
+ input_tokens=result.input_tokens,
464
+ output_tokens=result.output_tokens,
465
+ transcript_path=paths.workdir / "agent_harness_build.log",
466
+ agent_summary=report.summary if report else None,
467
+ missing_apt_packages=report.missing_apt_packages if report else [],
468
+ extra_include_paths=report.extra_include_paths if report else [],
469
+ extra_library_paths=extra_library_paths,
470
+ environment=environment,
471
+ gate_keeps_artifacts=harness.gate_keeps_artifacts,
472
+ )