fuzzprep 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (56) hide show
  1. fuzzprep/__init__.py +3 -0
  2. fuzzprep/__main__.py +4 -0
  3. fuzzprep/agents/harness_builder/SKILL.md +156 -0
  4. fuzzprep/agents/library_builder/SKILL.md +147 -0
  5. fuzzprep/agents/scripts/check_build.sh +87 -0
  6. fuzzprep/agents/scripts/check_build_in_container.sh +73 -0
  7. fuzzprep/agents/scripts/check_dockerfile_from_scratch.sh +33 -0
  8. fuzzprep/cli.py +1109 -0
  9. fuzzprep/core/__init__.py +0 -0
  10. fuzzprep/core/agent_stream.py +304 -0
  11. fuzzprep/core/files.py +35 -0
  12. fuzzprep/core/paths.py +25 -0
  13. fuzzprep/core/reporting.py +218 -0
  14. fuzzprep/core/repos.py +217 -0
  15. fuzzprep/core/resources.py +41 -0
  16. fuzzprep/core/subprocesses.py +197 -0
  17. fuzzprep/feature_extractor/__init__.py +0 -0
  18. fuzzprep/feature_extractor/benchmark_yaml.py +92 -0
  19. fuzzprep/feature_extractor/extraction.py +184 -0
  20. fuzzprep/feature_extractor/models.py +96 -0
  21. fuzzprep/feature_extractor/native/.clang-format +1 -0
  22. fuzzprep/feature_extractor/native/CMakeLists.txt +90 -0
  23. fuzzprep/feature_extractor/native/include/feature_extractor.hpp +136 -0
  24. fuzzprep/feature_extractor/native/src/extraction_action.cpp +294 -0
  25. fuzzprep/feature_extractor/native/src/json_writer.cpp +154 -0
  26. fuzzprep/feature_extractor/native/src/macro_callbacks.cpp +75 -0
  27. fuzzprep/feature_extractor/native/src/main.cpp +130 -0
  28. fuzzprep/feature_extractor/native_build.py +99 -0
  29. fuzzprep/library_builder/__init__.py +0 -0
  30. fuzzprep/library_builder/agents.py +472 -0
  31. fuzzprep/library_builder/analysis.py +145 -0
  32. fuzzprep/library_builder/build_parameters.py +158 -0
  33. fuzzprep/library_builder/dependency_resolution.py +139 -0
  34. fuzzprep/library_builder/environments/__init__.py +0 -0
  35. fuzzprep/library_builder/environments/base.py +89 -0
  36. fuzzprep/library_builder/environments/gate.py +96 -0
  37. fuzzprep/library_builder/environments/local.py +205 -0
  38. fuzzprep/library_builder/environments/oss_fuzz.py +485 -0
  39. fuzzprep/library_builder/environments/verification.py +125 -0
  40. fuzzprep/library_builder/exploration.py +217 -0
  41. fuzzprep/library_builder/generation.py +325 -0
  42. fuzzprep/library_builder/harness_explorer.py +257 -0
  43. fuzzprep/library_builder/models.py +169 -0
  44. fuzzprep/library_builder/package_names.json +33 -0
  45. fuzzprep/library_builder/package_names.py +40 -0
  46. fuzzprep/library_builder/scripts.py +389 -0
  47. fuzzprep/library_builder/stats.py +102 -0
  48. fuzzprep/library_builder/symbol_patterns.json +65 -0
  49. fuzzprep/library_builder/timeouts.py +14 -0
  50. fuzzprep/library_builder/workspace.py +250 -0
  51. fuzzprep-0.1.0.dist-info/METADATA +255 -0
  52. fuzzprep-0.1.0.dist-info/RECORD +56 -0
  53. fuzzprep-0.1.0.dist-info/WHEEL +4 -0
  54. fuzzprep-0.1.0.dist-info/entry_points.txt +3 -0
  55. fuzzprep-0.1.0.dist-info/licenses/LICENSE +202 -0
  56. fuzzprep-0.1.0.dist-info/licenses/THIRD_PARTY_NOTICES.md +52 -0
fuzzprep/core/repos.py ADDED
@@ -0,0 +1,217 @@
1
+ from __future__ import annotations
2
+
3
+ import logging
4
+ import re
5
+ import shutil
6
+ import subprocess
7
+ from dataclasses import dataclass
8
+ from pathlib import Path
9
+
10
+ from fuzzprep.core.paths import project_state_file
11
+
12
+ logger = logging.getLogger(__name__)
13
+
14
+ _PROJECT_NAME_PATTERN = re.compile(r"[a-z0-9][a-z0-9_.+-]*")
15
+
16
+
17
+ class RepositoryNotFoundError(Exception):
18
+ """Local path does not exist or is not a directory."""
19
+
20
+
21
+ class NoCloneableOriginError(Exception):
22
+ """Local repository has no cloneable git remote origin."""
23
+
24
+
25
+ class CloneFailedError(Exception):
26
+ """A git operation against the remote repository failed, carrying git's own stderr."""
27
+
28
+
29
+ class LocalRepoRefError(Exception):
30
+ """--repo-ref was combined with a local path, where it cannot be honoured."""
31
+
32
+
33
+ class InvalidProjectNameError(Exception):
34
+ """The project name is not a single safe path component, or names a workspace outside
35
+ the state directory."""
36
+
37
+
38
+ @dataclass
39
+ class RepoSource:
40
+ source_path: Path
41
+ clone_url: str
42
+ project_name: str
43
+ repo_ref: str | None = None
44
+
45
+
46
+ def name_from_url(url: str) -> str:
47
+ """Infer a project name from a repository URL basename."""
48
+ name = url.rstrip("/").split("/")[-1]
49
+ if name.endswith(".git"):
50
+ name = name[:-4]
51
+ return name or "project"
52
+
53
+
54
+ def _validated_project_name(name: str) -> str:
55
+ """Return name if it is a single path component safe to join onto the state directory.
56
+
57
+ The name becomes a directory under the state directory, and that directory is emptied on
58
+ every re-run. `Path("state") / "/tmp/x"` is `/tmp/x` and `Path("state") / ".."` is the state
59
+ directory's parent, so an unchecked name aims the reset at an arbitrary directory. It also
60
+ becomes a Docker image name component, which is why the character set is the one Docker
61
+ accepts rather than merely everything a filesystem tolerates.
62
+
63
+ Raises InvalidProjectNameError if the name is empty, `.`, `..`, contains a path separator,
64
+ or holds a character outside [a-z0-9_.+-].
65
+ """
66
+ if not _PROJECT_NAME_PATTERN.fullmatch(name):
67
+ raise InvalidProjectNameError(
68
+ f"Invalid project name {name!r}: it becomes a directory under the state directory "
69
+ "and a Docker image name, so it must be a single component starting with a letter "
70
+ "or digit and holding only [a-z0-9_.+-] — no path separators, `.`, or `..`. "
71
+ "Pass --project-name NAME to choose one."
72
+ )
73
+ return name
74
+
75
+
76
+ def clean_project_dir(project_dir: Path, keep: set[Path], *, state_dir: Path) -> None:
77
+ """Empty project_dir, preserving the paths in keep and any child that contains one.
78
+
79
+ A kept path is not always a direct child, so containment is matched as well as equality:
80
+ deleting the child that holds a nested kept path would take it along too.
81
+
82
+ Raises InvalidProjectNameError if project_dir is not a direct child of state_dir. Name
83
+ validation already rejects the names that would break that, so this is the backstop for
84
+ the next caller that builds a project directory some other way — including through a
85
+ symlink planted where the workspace goes.
86
+ """
87
+ if project_dir.resolve().parent != state_dir.resolve():
88
+ raise InvalidProjectNameError(
89
+ f"Refusing to reset {project_dir}: it is not a direct child of the state directory "
90
+ f"{state_dir}, and resetting it would delete files FuzzPrep does not own."
91
+ )
92
+ keep_resolved = {p.resolve() for p in keep}
93
+ for child in project_dir.iterdir():
94
+ resolved = child.resolve()
95
+ if any(kept == resolved or kept.is_relative_to(resolved) for kept in keep_resolved):
96
+ continue
97
+ if child.is_dir() and not child.is_symlink():
98
+ shutil.rmtree(child)
99
+ else:
100
+ child.unlink()
101
+
102
+
103
+ def _run_git(command: list[str], *, cwd: Path | None = None) -> None:
104
+ """Run a git command, raising CloneFailedError with git's own stderr on failure.
105
+
106
+ `check=True` would surface an unreachable host or a bad ref as a CalledProcessError
107
+ traceback; callers need a typed failure to report as an ingestion diagnostic.
108
+ """
109
+ result = subprocess.run(command, cwd=cwd, capture_output=True, text=True)
110
+ if result.returncode != 0:
111
+ raise CloneFailedError(
112
+ f"git {' '.join(command[1:])} failed (exit {result.returncode}):\n"
113
+ f"{result.stderr.strip()}"
114
+ )
115
+
116
+
117
+ def ingest_url(
118
+ url: str,
119
+ *,
120
+ project_name: str | None = None,
121
+ repo_ref: str | None = None,
122
+ state_dir: Path,
123
+ ) -> RepoSource:
124
+ """Clone a remote repository into state_dir and return a RepoSource.
125
+
126
+ The repo_ref checkout and the submodule update run on the fresh-clone and already-cloned
127
+ paths alike: `git clean -fdx` wipes untracked submodule content and a checkout can move
128
+ submodule pointers, so without the update a re-run builds against empty or stale trees.
129
+
130
+ state.json survives the workspace wipe, so the dependencies learned across prior runs are
131
+ not discarded on every re-run.
132
+
133
+ Raises InvalidProjectNameError if the project name — given or inferred from the URL — is not
134
+ a single safe path component.
135
+ Raises CloneFailedError if a git operation fails.
136
+ """
137
+ given_or_inferred = project_name if project_name is not None else name_from_url(url)
138
+ name = _validated_project_name(given_or_inferred.lower())
139
+ project_dir = state_dir / name
140
+ dest = project_dir / "src"
141
+ repo_source = RepoSource(source_path=dest, clone_url=url, project_name=name, repo_ref=repo_ref)
142
+ state_file = project_state_file(state_dir, name)
143
+
144
+ if not dest.exists():
145
+ _run_git(["git", "clone", "--recursive", url, str(dest)])
146
+ else:
147
+ clean_project_dir(project_dir, keep={dest, state_file}, state_dir=state_dir)
148
+ _run_git(["git", "reset", "--hard"], cwd=dest)
149
+ _run_git(["git", "clean", "-fdx"], cwd=dest)
150
+ if repo_ref:
151
+ _run_git(["git", "checkout", repo_ref], cwd=dest)
152
+ _run_git(["git", "submodule", "update", "--init", "--recursive"], cwd=dest)
153
+ return repo_source
154
+
155
+
156
+ def ingest_local(
157
+ path: Path,
158
+ *,
159
+ project_name: str | None = None,
160
+ repo_ref: str | None = None,
161
+ state_dir: Path,
162
+ ) -> RepoSource:
163
+ """Validate a local repository path and return a RepoSource.
164
+
165
+ Resets the project workspace as ingest_url does, so a previous run's Dockerfile, scripts,
166
+ and agent report cannot be mistaken for this run's. The user's source directory is never
167
+ touched, including when it sits inside the workspace being reset — a deliberate layout,
168
+ since staging a copy at <state_dir>/<project>/src is what earns $SCRIPT_DIR/src-relative
169
+ scripts in the output. Passing it to `keep` is what makes that safe.
170
+
171
+ Raises RepositoryNotFoundError if the path does not exist or is not a directory.
172
+ Raises InvalidProjectNameError if the project name is not a single safe path component.
173
+ Raises NoCloneableOriginError if the repository has no cloneable git remote origin.
174
+ Raises LocalRepoRefError if repo_ref is set: honouring it would check out a ref in a
175
+ working tree the user owns, and ignoring it would ship a setup.sh and Dockerfile pinning a
176
+ ref that was never built.
177
+ """
178
+ if not path.exists() or not path.is_dir():
179
+ raise RepositoryNotFoundError(f"Local path does not exist or is not a directory: {path}")
180
+ if repo_ref is not None:
181
+ raise LocalRepoRefError(
182
+ f"--repo-ref {repo_ref!r} cannot be applied to the local path {path}: "
183
+ "FuzzPrep will not check out a ref in a working tree you own, and the "
184
+ "generated output would otherwise claim a ref it never built. "
185
+ f"Check out {repo_ref!r} yourself first, or pass the repository URL instead."
186
+ )
187
+ given_or_inferred = project_name if project_name is not None else path.name
188
+ name = _validated_project_name(given_or_inferred.lower())
189
+ origin = _get_git_origin(path)
190
+ if origin is None:
191
+ raise NoCloneableOriginError(
192
+ f"Local repository has no cloneable git origin: {path}. "
193
+ "Generated Dockerfiles require a cloneable URL. "
194
+ "Add a remote with: git remote add origin <url>"
195
+ )
196
+ project_directory = state_dir / name
197
+ if project_directory.is_dir():
198
+ clean_project_dir(
199
+ project_directory,
200
+ keep={project_state_file(state_dir, name), path.resolve()},
201
+ state_dir=state_dir,
202
+ )
203
+ return RepoSource(source_path=path, clone_url=origin, project_name=name, repo_ref=repo_ref)
204
+
205
+
206
+ def _get_git_origin(path: Path) -> str | None:
207
+ """Return the git remote origin URL, or None if unavailable."""
208
+ result = subprocess.run(
209
+ ["git", "remote", "get-url", "origin"],
210
+ cwd=path,
211
+ capture_output=True,
212
+ text=True,
213
+ )
214
+ if result.returncode == 0:
215
+ url = result.stdout.strip()
216
+ return url if url else None
217
+ return None
@@ -0,0 +1,41 @@
1
+ """Locating the data files FuzzPrep ships: agent skills and the build gate scripts.
2
+
3
+ These live under `src/fuzzprep/agents/` and resolve through `importlib.resources`, so a
4
+ git checkout and an installed wheel find them the same way. A missing file is an error, not a
5
+ degraded mode: without the gate scripts no build can be verified, and without a skill file a
6
+ repair agent gets a one-sentence prompt instead of its instructions.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ from importlib.resources import as_file, files
12
+ from pathlib import Path
13
+
14
+ _AGENTS_PACKAGE = "fuzzprep.agents"
15
+
16
+
17
+ def agent_script(name: str) -> Path:
18
+ """Return the filesystem path of agents/scripts/<name>.
19
+
20
+ A path rather than a stream: these are shell scripts passed as subprocess arguments, and
21
+ one of them is bind-mounted into a container.
22
+ """
23
+ return _resource_path(f"scripts/{name}")
24
+
25
+
26
+ def skill_instructions(agent_name: str) -> str:
27
+ """Return the text of agents/<agent_name>/SKILL.md — a repair agent's instructions."""
28
+ return _resource_path(f"{agent_name}/SKILL.md").read_text()
29
+
30
+
31
+ def _resource_path(relative_path: str) -> Path:
32
+ resource = files(_AGENTS_PACKAGE).joinpath(relative_path)
33
+ # as_file materializes the resource if the distribution is zipped. FuzzPrep installs
34
+ # unzipped, so the path stays valid after the context exits.
35
+ with as_file(resource) as path:
36
+ if not path.is_file():
37
+ raise FileNotFoundError(
38
+ f"FuzzPrep resource {relative_path!r} is missing from the "
39
+ f"{_AGENTS_PACKAGE} package — the installation is incomplete."
40
+ )
41
+ return path
@@ -0,0 +1,197 @@
1
+ from __future__ import annotations
2
+
3
+ import contextlib
4
+ import contextvars
5
+ import os
6
+ import signal
7
+ import subprocess
8
+ import threading
9
+ import time
10
+ from collections.abc import Callable, Iterator
11
+ from dataclasses import dataclass
12
+ from pathlib import Path
13
+ from typing import IO
14
+
15
+
16
+ class MergedOutput:
17
+ """Adds `output` to a result type that carries a command's two output streams.
18
+
19
+ Whether `stderr` holds anything depends on the runner: `run_command_streaming` merges it
20
+ into stdout and leaves it empty, while `run_command` keeps the two apart. A caller
21
+ scanning for diagnostics wants both and should not have to know which runner it got.
22
+ """
23
+
24
+ stdout: str
25
+ stderr: str
26
+
27
+ @property
28
+ def output(self) -> str:
29
+ return self.stdout + self.stderr
30
+
31
+
32
+ @dataclass
33
+ class RunResult(MergedOutput):
34
+ stdout: str
35
+ stderr: str
36
+ exit_code: int
37
+ duration_seconds: float
38
+
39
+
40
+ # (command, cwd, timeout) -> RunResult — the one shape host-subprocess and docker-wrapped
41
+ # execution share, so a caller can swap how a command runs without touching its own logic.
42
+ Runner = Callable[[list[str], Path, int], RunResult]
43
+
44
+ _quiet: contextvars.ContextVar[bool] = contextvars.ContextVar("fuzzprep_quiet", default=False)
45
+ _log_path: contextvars.ContextVar[Path | None] = contextvars.ContextVar(
46
+ "fuzzprep_log_path", default=None
47
+ )
48
+
49
+
50
+ @contextlib.contextmanager
51
+ def streaming_context(*, quiet: bool = False, log_path: Path | None = None) -> Iterator[None]:
52
+ """Scope `run_command_streaming`'s live-printing and log-file destination for every call
53
+ made while this context is active.
54
+
55
+ Set once at a phase boundary in cli.py and read implicitly wherever
56
+ `run_command_streaming` is invoked, so the modules in between need no quiet/log_path
57
+ parameter of their own.
58
+ """
59
+ quiet_token = _quiet.set(quiet)
60
+ log_path_token = _log_path.set(log_path)
61
+ try:
62
+ yield
63
+ finally:
64
+ _quiet.reset(quiet_token)
65
+ _log_path.reset(log_path_token)
66
+
67
+
68
+ def _write_log(log_path: Path | None, text: str) -> None:
69
+ if log_path is None:
70
+ return
71
+ log_path.parent.mkdir(parents=True, exist_ok=True)
72
+ log_path.write_text(text)
73
+
74
+
75
+ # How long to wait for the reader thread to notice the pipe closed after a kill. The thread
76
+ # is a daemon, so an unresponsive grandchild holding stdout costs this much delay and is then
77
+ # abandoned rather than deadlocking the run.
78
+ _DRAIN_GRACE_SECONDS = 5
79
+
80
+
81
+ def _drain(stream: IO[str], lines: list[str], *, quiet: bool) -> None:
82
+ """Read stream to EOF, collecting each line and optionally echoing it live."""
83
+ for line in stream:
84
+ if not quiet:
85
+ print(line, end="", flush=True)
86
+ lines.append(line)
87
+
88
+
89
+ def _terminate_process_group(proc: subprocess.Popen[str]) -> None:
90
+ """Kill the child and anything it spawned.
91
+
92
+ A build runs `bash build_library.sh`, which spawns make, which spawns compilers. Killing
93
+ only the shell leaves that tree running and holding the stdout pipe open, so a timed-out
94
+ run never returns. `run_command_streaming` starts the child in its own session so the
95
+ whole tree can be signalled by process group here.
96
+ """
97
+ try:
98
+ os.killpg(os.getpgid(proc.pid), signal.SIGKILL)
99
+ except (ProcessLookupError, PermissionError):
100
+ proc.kill()
101
+
102
+
103
+ def run_command_streaming(command: list[str], cwd: Path, timeout: int) -> RunResult:
104
+ """Run command, printing each line in real-time while capturing combined output.
105
+
106
+ `timeout` bounds the whole call, not just the wait after the child closes stdout, so a
107
+ command that hangs while producing no output is killed at the deadline and reported as
108
+ `exit_code=-1`, matching `run_command`. Output from before the deadline is still returned.
109
+
110
+ The active `streaming_context` decides whether the live per-line printing happens. The
111
+ full output always goes to that context's log_path, on success, failure, or timeout.
112
+ """
113
+ quiet = _quiet.get()
114
+ log_path = _log_path.get()
115
+ start = time.monotonic()
116
+ lines: list[str] = []
117
+ proc = subprocess.Popen(
118
+ command,
119
+ cwd=cwd,
120
+ stdout=subprocess.PIPE,
121
+ stderr=subprocess.STDOUT,
122
+ text=True,
123
+ bufsize=1,
124
+ start_new_session=True,
125
+ )
126
+ assert proc.stdout is not None
127
+ reader = threading.Thread(
128
+ target=_drain, args=(proc.stdout, lines), kwargs={"quiet": quiet}, daemon=True
129
+ )
130
+ reader.start()
131
+ try:
132
+ exit_code = _await_exit(proc, reader, timeout=timeout, start=start)
133
+ except KeyboardInterrupt:
134
+ # start_new_session detached the child, so the terminal's Ctrl-C never reached it.
135
+ # Pass it on rather than orphaning a running build.
136
+ _terminate_process_group(proc)
137
+ raise
138
+ _write_log(log_path, "".join(lines))
139
+ return RunResult(
140
+ stdout="".join(lines),
141
+ stderr="",
142
+ exit_code=exit_code,
143
+ duration_seconds=time.monotonic() - start,
144
+ )
145
+
146
+
147
+ def _await_exit(
148
+ proc: subprocess.Popen[str], reader: threading.Thread, *, timeout: int, start: float
149
+ ) -> int:
150
+ """Wait for proc to finish within the deadline, killing it if it doesn't.
151
+
152
+ Returns its exit code, or -1 on timeout.
153
+ """
154
+ reader.join(timeout)
155
+ if not reader.is_alive():
156
+ remaining = max(timeout - (time.monotonic() - start), 0.0)
157
+ with contextlib.suppress(subprocess.TimeoutExpired):
158
+ return proc.wait(timeout=remaining)
159
+ _terminate_process_group(proc)
160
+ reader.join(_DRAIN_GRACE_SECONDS)
161
+ proc.wait()
162
+ return -1
163
+
164
+
165
+ def run_command(command: list[str], cwd: Path, timeout: int) -> RunResult:
166
+ start = time.monotonic()
167
+ try:
168
+ completed = subprocess.run(
169
+ command,
170
+ cwd=cwd,
171
+ timeout=timeout,
172
+ capture_output=True,
173
+ text=True,
174
+ )
175
+ duration = time.monotonic() - start
176
+ return RunResult(
177
+ stdout=completed.stdout,
178
+ stderr=completed.stderr,
179
+ exit_code=completed.returncode,
180
+ duration_seconds=duration,
181
+ )
182
+ except subprocess.TimeoutExpired as exc:
183
+ duration = time.monotonic() - start
184
+ return RunResult(
185
+ stdout=_decode_output(exc.stdout),
186
+ stderr=_decode_output(exc.stderr),
187
+ exit_code=-1,
188
+ duration_seconds=duration,
189
+ )
190
+
191
+
192
+ def _decode_output(output: bytes | str | None) -> str:
193
+ if output is None:
194
+ return ""
195
+ if isinstance(output, bytes):
196
+ return output.decode(errors="replace")
197
+ return output
File without changes
@@ -0,0 +1,92 @@
1
+ from __future__ import annotations
2
+
3
+ from pathlib import Path
4
+
5
+ import yaml
6
+
7
+ from fuzzprep.feature_extractor.extraction import (
8
+ MissingFeatureArtifactError,
9
+ load_feature_artifact,
10
+ )
11
+ from fuzzprep.feature_extractor.models import (
12
+ BenchmarkFunction,
13
+ BenchmarkYaml,
14
+ FunctionSignature,
15
+ )
16
+ from fuzzprep.library_builder.models import Language
17
+
18
+ _DEFAULT_TARGET_NAME = "default_fuzzer"
19
+ _INCLUDE_HEADERS = []
20
+ _EXT_BY_LANGUAGE = {Language.C: "c", Language.CPP: "cc"}
21
+
22
+
23
+ def select_public(f: FunctionSignature) -> bool:
24
+ return f.is_public_api
25
+
26
+
27
+ def select_by_header(f: FunctionSignature) -> bool:
28
+ return Path(f.header_path).name in _INCLUDE_HEADERS
29
+
30
+
31
+ def generate_benchmark(
32
+ output_dir: Path,
33
+ headers: list[str],
34
+ target_name: str | None = None,
35
+ target_path: str | None = None,
36
+ ) -> BenchmarkYaml:
37
+ """Convert <output_dir>/features.json into an oss-fuzz-gen-compatible YAML benchmark.
38
+
39
+ Keeps only public-API functions, or only those declared in `headers` when it is non-empty,
40
+ and overwrites <output_dir>/<project_name>.yaml.
41
+ """
42
+ check_include_func = select_public
43
+ if len(headers) > 0:
44
+ global _INCLUDE_HEADERS
45
+ _INCLUDE_HEADERS = headers
46
+ check_include_func = select_by_header
47
+ features_json = output_dir / "features.json"
48
+ if not features_json.is_file():
49
+ raise MissingFeatureArtifactError(
50
+ f"{features_json} not found. Run 'fuzzprep extract-features' first."
51
+ )
52
+ artifact = load_feature_artifact(features_json)
53
+
54
+ resolved_target_name = target_name or _DEFAULT_TARGET_NAME
55
+ ext = _EXT_BY_LANGUAGE.get(artifact.language, "c")
56
+ resolved_target_path = target_path or f"/src/harness_source/{resolved_target_name}.{ext}"
57
+
58
+ benchmark = BenchmarkYaml(
59
+ project=artifact.project_name,
60
+ language=artifact.language.value,
61
+ target_name=resolved_target_name,
62
+ target_path=resolved_target_path,
63
+ functions=[
64
+ BenchmarkFunction(
65
+ name=f.name, signature=f.signature, return_type=f.return_type, params=f.params
66
+ )
67
+ for f in artifact.functions
68
+ if check_include_func(f)
69
+ ],
70
+ )
71
+
72
+ output_path = output_dir / f"{artifact.project_name}.yaml"
73
+ output_path.write_text(yaml.safe_dump(_to_yaml_dict(benchmark), sort_keys=False))
74
+ return benchmark
75
+
76
+
77
+ def _to_yaml_dict(benchmark: BenchmarkYaml) -> dict:
78
+ return {
79
+ "project": benchmark.project,
80
+ "language": benchmark.language,
81
+ "target_name": benchmark.target_name,
82
+ "target_path": benchmark.target_path,
83
+ "functions": [
84
+ {
85
+ "name": f.name,
86
+ "signature": f.signature,
87
+ "return_type": f.return_type,
88
+ "params": [{"name": p.name, "type": p.type} for p in f.params],
89
+ }
90
+ for f in benchmark.functions
91
+ ],
92
+ }