ferrum-cli 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ferrum/__init__.py +6 -0
- ferrum/agent.py +387 -0
- ferrum/cli.py +479 -0
- ferrum/config.py +178 -0
- ferrum/context.py +379 -0
- ferrum/model.py +302 -0
- ferrum/patch.py +65 -0
- ferrum/prompts/system.md +89 -0
- ferrum/safety.py +124 -0
- ferrum/tools.py +296 -0
- ferrum/verifier.py +227 -0
- ferrum_cli-0.1.0.dist-info/METADATA +74 -0
- ferrum_cli-0.1.0.dist-info/RECORD +17 -0
- ferrum_cli-0.1.0.dist-info/WHEEL +5 -0
- ferrum_cli-0.1.0.dist-info/entry_points.txt +2 -0
- ferrum_cli-0.1.0.dist-info/licenses/LICENSE +211 -0
- ferrum_cli-0.1.0.dist-info/top_level.txt +1 -0
ferrum/__init__.py
ADDED
ferrum/agent.py
ADDED
|
@@ -0,0 +1,387 @@
|
|
|
1
|
+
"""The agent loop: the model reads (and, in fix mode, patches) until done."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
import logging
|
|
7
|
+
from collections.abc import Callable, Mapping
|
|
8
|
+
from dataclasses import dataclass
|
|
9
|
+
from pathlib import Path, PurePosixPath
|
|
10
|
+
from typing import Any, ClassVar
|
|
11
|
+
|
|
12
|
+
from ferrum.config import Config
|
|
13
|
+
from ferrum.context import ProjectContext
|
|
14
|
+
from ferrum.model import (
|
|
15
|
+
MALFORMED_JSON,
|
|
16
|
+
ModelProvider,
|
|
17
|
+
ModelResponse,
|
|
18
|
+
extract_tool_calls,
|
|
19
|
+
)
|
|
20
|
+
from ferrum.patch import Patch, format_unified_diff
|
|
21
|
+
from ferrum.safety import PathEscapeError, is_denied, safe_join
|
|
22
|
+
from ferrum.tools import Tool, ToolError, ToolRegistry, ToolResult, _require_string
|
|
23
|
+
from ferrum.verifier import Verifier, VerifyOutcome
|
|
24
|
+
|
|
25
|
+
log = logging.getLogger(__name__)
|
|
26
|
+
|
|
27
|
+
# Inside the package, so wheel installs ship the real prompt too.
|
|
28
|
+
PROMPT_PATH = Path(__file__).resolve().parent / "prompts" / "system.md"
|
|
29
|
+
FALLBACK_SYSTEM = (
|
|
30
|
+
"You are Ferrum, a coding harness for systems programming. "
|
|
31
|
+
"Inspect before you modify, cite file:line evidence, make the smallest "
|
|
32
|
+
"change, and never claim a fix is verified unless a build or test "
|
|
33
|
+
"actually passed."
|
|
34
|
+
)
|
|
35
|
+
|
|
36
|
+
READ_ONLY_NOTE = (
|
|
37
|
+
"Read-only mode: investigate thoroughly before answering. A directory "
|
|
38
|
+
"listing alone is not an answer — read the relevant source files and "
|
|
39
|
+
"search for the relevant names; several tool calls are expected. You "
|
|
40
|
+
"cannot modify files, so do not call apply_patch."
|
|
41
|
+
)
|
|
42
|
+
|
|
43
|
+
FIX_NOTE = (
|
|
44
|
+
"Fix mode: inspect thoroughly before patching — read the file you are "
|
|
45
|
+
"about to change and its callers; several tool calls are expected. "
|
|
46
|
+
"Call apply_patch only with text you have read in this session."
|
|
47
|
+
)
|
|
48
|
+
|
|
49
|
+
DRY_RUN_NOTE = (
|
|
50
|
+
"Dry run: apply_patch will display your proposed diff, but the harness "
|
|
51
|
+
"will not modify files, ask for confirmation, or run verification. "
|
|
52
|
+
"Propose the one correct patch and then summarize it for the user."
|
|
53
|
+
)
|
|
54
|
+
|
|
55
|
+
NUDGE = (
|
|
56
|
+
"You have not called any tools yet. Investigate first: call list_files, "
|
|
57
|
+
"read_file or search_code and base your answer on what you read."
|
|
58
|
+
)
|
|
59
|
+
|
|
60
|
+
READ_NUDGE = (
|
|
61
|
+
"A directory listing alone does not answer the question. Use read_file "
|
|
62
|
+
"on the most relevant source file (and search_code for names) before "
|
|
63
|
+
"you answer."
|
|
64
|
+
)
|
|
65
|
+
|
|
66
|
+
MAX_NUDGES = 2
|
|
67
|
+
READ_TOOLS = ("read_file", "search_code", "apply_patch")
|
|
68
|
+
|
|
69
|
+
PROGRESS = {
|
|
70
|
+
"list_files": "inspecting the project",
|
|
71
|
+
"read_file": "reading",
|
|
72
|
+
"search_code": "searching",
|
|
73
|
+
"apply_patch": "proposing a patch to",
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
@dataclass
|
|
78
|
+
class AgentResult:
|
|
79
|
+
text: str
|
|
80
|
+
turns: int
|
|
81
|
+
capped: bool = False
|
|
82
|
+
patched: bool = False
|
|
83
|
+
verified: bool | None = None
|
|
84
|
+
verify_report: str | None = None
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
class ApplyPatch(Tool):
|
|
88
|
+
name = "apply_patch"
|
|
89
|
+
description = (
|
|
90
|
+
"Replace exact text in a project file after showing the diff and "
|
|
91
|
+
"asking the user. The old text must match the file verbatim and "
|
|
92
|
+
"uniquely. After a successful patch the project is built and "
|
|
93
|
+
"tested; the verification output comes back in the result."
|
|
94
|
+
)
|
|
95
|
+
parameters: ClassVar[dict[str, Any]] = {
|
|
96
|
+
"type": "object",
|
|
97
|
+
"properties": {
|
|
98
|
+
"path": {
|
|
99
|
+
"type": "string",
|
|
100
|
+
"description": "File path relative to the project root.",
|
|
101
|
+
},
|
|
102
|
+
"old": {
|
|
103
|
+
"type": "string",
|
|
104
|
+
"description": "Exact text currently in the file.",
|
|
105
|
+
},
|
|
106
|
+
"new": {
|
|
107
|
+
"type": "string",
|
|
108
|
+
"description": "Replacement text.",
|
|
109
|
+
},
|
|
110
|
+
"description": {
|
|
111
|
+
"type": "string",
|
|
112
|
+
"description": "One line: why this change fixes the problem.",
|
|
113
|
+
},
|
|
114
|
+
},
|
|
115
|
+
"required": ["path", "old", "new"],
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
def __init__(
|
|
119
|
+
self,
|
|
120
|
+
root: Path,
|
|
121
|
+
confirm: Callable[[str], bool],
|
|
122
|
+
ui: Callable[[str], None] = print,
|
|
123
|
+
verifier: Verifier | None = None,
|
|
124
|
+
dry_run: bool = False,
|
|
125
|
+
) -> None:
|
|
126
|
+
self.root = Path(root)
|
|
127
|
+
self.confirm = confirm
|
|
128
|
+
self.ui = ui
|
|
129
|
+
self.verifier = verifier
|
|
130
|
+
self.dry_run = dry_run
|
|
131
|
+
self.applied = False
|
|
132
|
+
self.outcome: VerifyOutcome | None = None
|
|
133
|
+
|
|
134
|
+
def execute(self, arguments: Mapping[str, Any]) -> ToolResult:
|
|
135
|
+
rel = _require_string(arguments, "path")
|
|
136
|
+
old = _require_string(arguments, "old")
|
|
137
|
+
new = arguments.get("new")
|
|
138
|
+
if not isinstance(new, str):
|
|
139
|
+
raise ToolError("new is required and must be a string")
|
|
140
|
+
description = arguments.get("description") or ""
|
|
141
|
+
if not isinstance(description, str):
|
|
142
|
+
description = str(description)
|
|
143
|
+
|
|
144
|
+
if is_denied(PurePosixPath(rel.replace("\\", "/"))):
|
|
145
|
+
raise ToolError(f"refused: {rel} is excluded (secret, binary, or ignored)")
|
|
146
|
+
try:
|
|
147
|
+
target = safe_join(self.root, rel)
|
|
148
|
+
except PathEscapeError as exc:
|
|
149
|
+
raise ToolError(str(exc)) from exc
|
|
150
|
+
if target.is_dir():
|
|
151
|
+
raise ToolError(f"is a directory: {rel}")
|
|
152
|
+
if not target.exists():
|
|
153
|
+
raise ToolError(f"file not found: {rel} (read it first)")
|
|
154
|
+
|
|
155
|
+
try:
|
|
156
|
+
data = target.read_bytes()
|
|
157
|
+
except OSError as exc:
|
|
158
|
+
raise ToolError(f"cannot read {rel}: {exc}") from exc
|
|
159
|
+
if b"\x00" in data[:8192]:
|
|
160
|
+
raise ToolError(f"refused: {rel} looks like a binary file")
|
|
161
|
+
try:
|
|
162
|
+
text = data.decode("utf-8")
|
|
163
|
+
except UnicodeDecodeError as exc:
|
|
164
|
+
raise ToolError(f"not valid UTF-8 text: {rel}") from exc
|
|
165
|
+
|
|
166
|
+
crlf = "\r\n" in text
|
|
167
|
+
current = text.replace("\r\n", "\n").replace("\r", "\n")
|
|
168
|
+
patch = Patch(rel=rel, old=old, new=new, description=description)
|
|
169
|
+
try:
|
|
170
|
+
updated = patch.apply(current)
|
|
171
|
+
except Exception as exc:
|
|
172
|
+
raise ToolError(f"Patch rejected: {exc}. Re-read the file and retry.") from exc
|
|
173
|
+
|
|
174
|
+
self.ui(format_unified_diff(current, updated, rel))
|
|
175
|
+
if self.dry_run:
|
|
176
|
+
return ToolResult.success(
|
|
177
|
+
f"Dry run: the patch was NOT applied to {rel}"
|
|
178
|
+
" — the file on disk is unchanged and no verification ran. "
|
|
179
|
+
"Summarize this proposed change for the user and stop."
|
|
180
|
+
)
|
|
181
|
+
if not self.confirm("Apply patch? [y/N] "):
|
|
182
|
+
return ToolResult.failure(
|
|
183
|
+
"The user declined this patch. Do not apply it. "
|
|
184
|
+
"Ask what they would prefer or stop."
|
|
185
|
+
)
|
|
186
|
+
|
|
187
|
+
out = updated.replace("\n", "\r\n") if crlf else updated
|
|
188
|
+
try:
|
|
189
|
+
target.write_bytes(out.encode("utf-8"))
|
|
190
|
+
except OSError as exc:
|
|
191
|
+
raise ToolError(f"cannot write {rel}: {exc}") from exc
|
|
192
|
+
self.applied = True
|
|
193
|
+
|
|
194
|
+
report = f"Patch applied to {rel}."
|
|
195
|
+
if description:
|
|
196
|
+
report += f" ({description})"
|
|
197
|
+
if self.verifier is not None:
|
|
198
|
+
self.outcome = self.verifier.verify()
|
|
199
|
+
report += "\n\n--- verification ---\n" + self.outcome.report()
|
|
200
|
+
else:
|
|
201
|
+
report += "\n(verification unavailable)"
|
|
202
|
+
return ToolResult.success(report)
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
class Agent:
|
|
206
|
+
def __init__(
|
|
207
|
+
self,
|
|
208
|
+
config: Config,
|
|
209
|
+
provider: ModelProvider,
|
|
210
|
+
registry: ToolRegistry,
|
|
211
|
+
printer: Callable[[str], None] | None = None,
|
|
212
|
+
verifier: Verifier | None = None,
|
|
213
|
+
) -> None:
|
|
214
|
+
self.config = config
|
|
215
|
+
self.provider = provider
|
|
216
|
+
self.registry = registry
|
|
217
|
+
self.printer = printer
|
|
218
|
+
self.verifier = verifier
|
|
219
|
+
self.apply_tool: ApplyPatch | None = None
|
|
220
|
+
|
|
221
|
+
@staticmethod
|
|
222
|
+
def system_prompt() -> str:
|
|
223
|
+
try:
|
|
224
|
+
return PROMPT_PATH.read_text(encoding="utf-8")
|
|
225
|
+
except OSError:
|
|
226
|
+
return FALLBACK_SYSTEM
|
|
227
|
+
|
|
228
|
+
def run(
|
|
229
|
+
self,
|
|
230
|
+
task: str,
|
|
231
|
+
context: ProjectContext,
|
|
232
|
+
*,
|
|
233
|
+
edit: bool = False,
|
|
234
|
+
confirm: Callable[[str], bool] | None = None,
|
|
235
|
+
dry_run: bool = False,
|
|
236
|
+
) -> AgentResult:
|
|
237
|
+
if edit and "apply_patch" not in self.registry:
|
|
238
|
+
if confirm is None:
|
|
239
|
+
raise ValueError("edit mode requires a confirm callback")
|
|
240
|
+
self.apply_tool = ApplyPatch(
|
|
241
|
+
context.root,
|
|
242
|
+
confirm,
|
|
243
|
+
ui=self.printer or print,
|
|
244
|
+
verifier=self.verifier,
|
|
245
|
+
dry_run=dry_run,
|
|
246
|
+
)
|
|
247
|
+
self.registry.register(self.apply_tool)
|
|
248
|
+
|
|
249
|
+
user = task
|
|
250
|
+
user += "\n\n" + (FIX_NOTE if edit else READ_ONLY_NOTE)
|
|
251
|
+
if dry_run:
|
|
252
|
+
user += "\n\n" + DRY_RUN_NOTE
|
|
253
|
+
user += "\n\nProject:\n" + context.summary()
|
|
254
|
+
messages: list[dict[str, Any]] = [
|
|
255
|
+
{"role": "system", "content": self.system_prompt()},
|
|
256
|
+
{"role": "user", "content": user},
|
|
257
|
+
]
|
|
258
|
+
|
|
259
|
+
turns = 0
|
|
260
|
+
nudges = 0
|
|
261
|
+
used_tools = False
|
|
262
|
+
used_read = False
|
|
263
|
+
read_nudges = 0
|
|
264
|
+
for turn in range(1, self.config.max_iterations + 1):
|
|
265
|
+
turns = turn
|
|
266
|
+
response = self.provider.complete(messages, self.registry.schemas())
|
|
267
|
+
calls = _merge_tool_calls(
|
|
268
|
+
response.tool_calls, extract_tool_calls(response.content)
|
|
269
|
+
)
|
|
270
|
+
if calls:
|
|
271
|
+
response = ModelResponse(response.content, calls)
|
|
272
|
+
if not response.tool_calls:
|
|
273
|
+
if not used_tools and nudges < MAX_NUDGES:
|
|
274
|
+
# Small models love answering from memory; force one look.
|
|
275
|
+
nudges += 1
|
|
276
|
+
messages.append(_assistant_message(response))
|
|
277
|
+
messages.append({"role": "user", "content": NUDGE})
|
|
278
|
+
continue
|
|
279
|
+
if (
|
|
280
|
+
used_tools
|
|
281
|
+
and not used_read
|
|
282
|
+
and read_nudges < MAX_NUDGES
|
|
283
|
+
):
|
|
284
|
+
# Listing is not reading; push for the actual source.
|
|
285
|
+
read_nudges += 1
|
|
286
|
+
messages.append(_assistant_message(response))
|
|
287
|
+
messages.append({"role": "user", "content": READ_NUDGE})
|
|
288
|
+
continue
|
|
289
|
+
return self._finish(response.content, turns, capped=False, edit=edit)
|
|
290
|
+
messages.append(_assistant_message(response))
|
|
291
|
+
used_tools = True
|
|
292
|
+
for call in response.tool_calls:
|
|
293
|
+
if call.name in READ_TOOLS:
|
|
294
|
+
used_read = True
|
|
295
|
+
self._progress(call)
|
|
296
|
+
result = self._execute(call)
|
|
297
|
+
messages.append(
|
|
298
|
+
{
|
|
299
|
+
"role": "tool",
|
|
300
|
+
"tool_call_id": call.id,
|
|
301
|
+
"content": result.to_model_message(),
|
|
302
|
+
}
|
|
303
|
+
)
|
|
304
|
+
return self._finish(
|
|
305
|
+
f"Stopped after {self.config.max_iterations} iterations "
|
|
306
|
+
"without a final answer.",
|
|
307
|
+
turns,
|
|
308
|
+
capped=True,
|
|
309
|
+
edit=edit,
|
|
310
|
+
)
|
|
311
|
+
|
|
312
|
+
def _execute(self, call) -> ToolResult:
|
|
313
|
+
if MALFORMED_JSON in call.arguments:
|
|
314
|
+
raw = str(call.arguments[MALFORMED_JSON])[:500]
|
|
315
|
+
return ToolResult.failure(
|
|
316
|
+
f"invalid JSON arguments for {call.name!r}: {raw!r}. "
|
|
317
|
+
"Send only the argument values as a JSON object "
|
|
318
|
+
'(e.g. {"path": "main.c"}); escape newlines as \\n.'
|
|
319
|
+
)
|
|
320
|
+
if not call.name:
|
|
321
|
+
return ToolResult.failure("tool call had no name")
|
|
322
|
+
return self.registry.execute(call.name, call.arguments)
|
|
323
|
+
|
|
324
|
+
def _progress(self, call) -> None:
|
|
325
|
+
if self.printer is None:
|
|
326
|
+
return
|
|
327
|
+
label = PROGRESS.get(call.name)
|
|
328
|
+
if label is None:
|
|
329
|
+
self.printer(f"Ferrum > {call.name}...")
|
|
330
|
+
return
|
|
331
|
+
if call.name in ("read_file", "apply_patch"):
|
|
332
|
+
detail = f" {call.arguments.get('path', '')}"
|
|
333
|
+
elif call.name == "search_code":
|
|
334
|
+
detail = f" {call.arguments.get('pattern', '')!r}"
|
|
335
|
+
else:
|
|
336
|
+
detail = ""
|
|
337
|
+
self.printer(f"Ferrum > {label}{detail}...")
|
|
338
|
+
|
|
339
|
+
def _finish(self, text: str, turns: int, *, capped: bool, edit: bool) -> AgentResult:
|
|
340
|
+
if not text:
|
|
341
|
+
text = "The model returned no answer."
|
|
342
|
+
patched = bool(self.apply_tool and self.apply_tool.applied)
|
|
343
|
+
verified: bool | None = None
|
|
344
|
+
report: str | None = None
|
|
345
|
+
if patched and self.apply_tool and self.apply_tool.outcome:
|
|
346
|
+
verified = self.apply_tool.outcome.verified
|
|
347
|
+
report = self.apply_tool.outcome.report()
|
|
348
|
+
return AgentResult(
|
|
349
|
+
text=text,
|
|
350
|
+
turns=turns,
|
|
351
|
+
capped=capped,
|
|
352
|
+
patched=patched,
|
|
353
|
+
verified=verified,
|
|
354
|
+
verify_report=report,
|
|
355
|
+
)
|
|
356
|
+
|
|
357
|
+
|
|
358
|
+
def _merge_tool_calls(native: list, extracted: list) -> list:
|
|
359
|
+
"""Native calls first, then JSON-from-text calls that are not duplicates."""
|
|
360
|
+
seen = {(c.name, json.dumps(c.arguments, sort_keys=True)) for c in native}
|
|
361
|
+
merged = list(native)
|
|
362
|
+
for call in extracted:
|
|
363
|
+
key = (call.name, json.dumps(call.arguments, sort_keys=True))
|
|
364
|
+
if key not in seen:
|
|
365
|
+
seen.add(key)
|
|
366
|
+
merged.append(call)
|
|
367
|
+
return merged
|
|
368
|
+
|
|
369
|
+
|
|
370
|
+
def _assistant_message(response: ModelResponse) -> dict[str, Any]:
|
|
371
|
+
message: dict[str, Any] = {
|
|
372
|
+
"role": "assistant",
|
|
373
|
+
"content": response.content or "",
|
|
374
|
+
}
|
|
375
|
+
if response.tool_calls:
|
|
376
|
+
message["tool_calls"] = [
|
|
377
|
+
{
|
|
378
|
+
"id": call.id,
|
|
379
|
+
"type": "function",
|
|
380
|
+
"function": {
|
|
381
|
+
"name": call.name,
|
|
382
|
+
"arguments": json.dumps(call.arguments),
|
|
383
|
+
},
|
|
384
|
+
}
|
|
385
|
+
for call in response.tool_calls
|
|
386
|
+
]
|
|
387
|
+
return message
|