devagent-ai 0.3.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent/__init__.py +5 -0
- agent/llm.py +6 -0
- agent/loop.py +16 -0
- agent/memory.py +5 -0
- agent/prompts.py +3 -0
- agent/tools.py +6 -0
- devagent/__init__.py +3 -0
- devagent/__main__.py +4 -0
- devagent/artifacts.py +46 -0
- devagent/cli.py +164 -0
- devagent/config.py +72 -0
- devagent/discovery.py +504 -0
- devagent/evaluation.py +388 -0
- devagent/memory.py +51 -0
- devagent/models.py +251 -0
- devagent/orchestrator.py +887 -0
- devagent/providers.py +344 -0
- devagent/report.py +52 -0
- devagent/retrieval.py +478 -0
- devagent/safety.py +131 -0
- devagent/state_machine.py +45 -0
- devagent/tasking.py +83 -0
- devagent/workspace.py +199 -0
- devagent/worktree.py +151 -0
- devagent_ai-0.3.1.dist-info/METADATA +415 -0
- devagent_ai-0.3.1.dist-info/RECORD +31 -0
- devagent_ai-0.3.1.dist-info/WHEEL +5 -0
- devagent_ai-0.3.1.dist-info/entry_points.txt +2 -0
- devagent_ai-0.3.1.dist-info/licenses/LICENSE +21 -0
- devagent_ai-0.3.1.dist-info/licenses/NOTICE +10 -0
- devagent_ai-0.3.1.dist-info/top_level.txt +2 -0
devagent/orchestrator.py
ADDED
|
@@ -0,0 +1,887 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
import difflib
|
|
5
|
+
import subprocess
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
from typing import Any, Callable, Sequence
|
|
8
|
+
|
|
9
|
+
from devagent.artifacts import RunArtifacts
|
|
10
|
+
from devagent.discovery import discover_repository
|
|
11
|
+
from devagent.memory import RepositoryMemory
|
|
12
|
+
from devagent.models import (
|
|
13
|
+
AgentState,
|
|
14
|
+
ChangeMetrics,
|
|
15
|
+
EngineeringPlan,
|
|
16
|
+
Evidence,
|
|
17
|
+
Outcome,
|
|
18
|
+
RiskLevel,
|
|
19
|
+
ReviewDecision,
|
|
20
|
+
ReviewIssue,
|
|
21
|
+
RunResult,
|
|
22
|
+
TaskType,
|
|
23
|
+
Understanding,
|
|
24
|
+
VerificationResult,
|
|
25
|
+
jsonable,
|
|
26
|
+
)
|
|
27
|
+
from devagent.providers import ModelProvider, ProviderError
|
|
28
|
+
from devagent.report import render_report
|
|
29
|
+
from devagent.retrieval import retrieve_context
|
|
30
|
+
from devagent.safety import SafetyError
|
|
31
|
+
from devagent.state_machine import InvalidTransition, Lifecycle
|
|
32
|
+
from devagent.tasking import compile_task
|
|
33
|
+
from devagent.workspace import Workspace
|
|
34
|
+
from devagent.worktree import select_worktree
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
_BUILTIN_GIT_VERIFICATION: dict[tuple[str, ...], str] = {
|
|
38
|
+
("git", "diff", "--check"): "harness",
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
_NON_EMPTY_STRING = {"type": "string", "minLength": 1, "pattern": r"\S"}
|
|
42
|
+
_STRING_LIST = {"type": "array", "items": _NON_EMPTY_STRING}
|
|
43
|
+
_CONFIDENCE = {"type": "number", "minimum": 0.0, "maximum": 1.0}
|
|
44
|
+
|
|
45
|
+
_REPLACE_ACTION = {
|
|
46
|
+
"type": "object",
|
|
47
|
+
"properties": {
|
|
48
|
+
"tool": {"type": "string", "const": "replace_text"},
|
|
49
|
+
"arguments": {
|
|
50
|
+
"type": "object",
|
|
51
|
+
"properties": {
|
|
52
|
+
"path": _NON_EMPTY_STRING,
|
|
53
|
+
"old": _NON_EMPTY_STRING,
|
|
54
|
+
"new": {"type": "string"},
|
|
55
|
+
},
|
|
56
|
+
"required": ["path", "old", "new"],
|
|
57
|
+
"additionalProperties": False,
|
|
58
|
+
},
|
|
59
|
+
},
|
|
60
|
+
"required": ["tool", "arguments"],
|
|
61
|
+
"additionalProperties": False,
|
|
62
|
+
}
|
|
63
|
+
_COUNTED_REPLACE_ACTION = {
|
|
64
|
+
"type": "object",
|
|
65
|
+
"properties": {
|
|
66
|
+
"tool": {"type": "string", "const": "replace_text"},
|
|
67
|
+
"arguments": {
|
|
68
|
+
"type": "object",
|
|
69
|
+
"properties": {
|
|
70
|
+
"path": _NON_EMPTY_STRING,
|
|
71
|
+
"old": _NON_EMPTY_STRING,
|
|
72
|
+
"new": {"type": "string"},
|
|
73
|
+
"count": {"type": "integer", "minimum": 1},
|
|
74
|
+
},
|
|
75
|
+
"required": ["path", "old", "new", "count"],
|
|
76
|
+
"additionalProperties": False,
|
|
77
|
+
},
|
|
78
|
+
},
|
|
79
|
+
"required": ["tool", "arguments"],
|
|
80
|
+
"additionalProperties": False,
|
|
81
|
+
}
|
|
82
|
+
_WRITE_ACTION = {
|
|
83
|
+
"type": "object",
|
|
84
|
+
"properties": {
|
|
85
|
+
"tool": {"type": "string", "const": "write_file"},
|
|
86
|
+
"arguments": {
|
|
87
|
+
"type": "object",
|
|
88
|
+
"properties": {"path": _NON_EMPTY_STRING, "content": {"type": "string"}},
|
|
89
|
+
"required": ["path", "content"],
|
|
90
|
+
"additionalProperties": False,
|
|
91
|
+
},
|
|
92
|
+
},
|
|
93
|
+
"required": ["tool", "arguments"],
|
|
94
|
+
"additionalProperties": False,
|
|
95
|
+
}
|
|
96
|
+
_ACTION = {"anyOf": [_REPLACE_ACTION, _COUNTED_REPLACE_ACTION, _WRITE_ACTION]}
|
|
97
|
+
|
|
98
|
+
UNDERSTANDING_SCHEMA = {
|
|
99
|
+
"type": "object",
|
|
100
|
+
"properties": {
|
|
101
|
+
"problem": _NON_EMPTY_STRING,
|
|
102
|
+
"expected_behavior": _NON_EMPTY_STRING,
|
|
103
|
+
"affected_paths": {**_STRING_LIST, "minItems": 1},
|
|
104
|
+
"root_cause": _NON_EMPTY_STRING,
|
|
105
|
+
"evidence": {
|
|
106
|
+
"type": "array",
|
|
107
|
+
"minItems": 1,
|
|
108
|
+
"items": {
|
|
109
|
+
"type": "object",
|
|
110
|
+
"properties": {
|
|
111
|
+
"statement": _NON_EMPTY_STRING,
|
|
112
|
+
"paths": {**_STRING_LIST, "minItems": 1},
|
|
113
|
+
"confidence": _CONFIDENCE,
|
|
114
|
+
},
|
|
115
|
+
"required": ["statement", "paths", "confidence"],
|
|
116
|
+
"additionalProperties": False,
|
|
117
|
+
},
|
|
118
|
+
},
|
|
119
|
+
"proposed_solution": {**_STRING_LIST, "minItems": 1},
|
|
120
|
+
"confidence": _CONFIDENCE,
|
|
121
|
+
},
|
|
122
|
+
"required": ["problem", "expected_behavior", "affected_paths", "root_cause", "evidence", "proposed_solution", "confidence"],
|
|
123
|
+
"additionalProperties": False,
|
|
124
|
+
}
|
|
125
|
+
PLAN_SCHEMA = {
|
|
126
|
+
"type": "object",
|
|
127
|
+
"properties": {
|
|
128
|
+
"files_to_inspect": {**_STRING_LIST, "minItems": 1},
|
|
129
|
+
"implementation": {**_STRING_LIST, "minItems": 1},
|
|
130
|
+
"verification": {
|
|
131
|
+
"type": "array",
|
|
132
|
+
"items": {"type": "array", "minItems": 1, "items": _NON_EMPTY_STRING},
|
|
133
|
+
},
|
|
134
|
+
"rationale": _NON_EMPTY_STRING,
|
|
135
|
+
},
|
|
136
|
+
"required": ["files_to_inspect", "implementation", "verification", "rationale"],
|
|
137
|
+
"additionalProperties": False,
|
|
138
|
+
}
|
|
139
|
+
IMPLEMENT_SCHEMA = {
|
|
140
|
+
"type": "object",
|
|
141
|
+
"properties": {
|
|
142
|
+
"actions": {"type": "array", "minItems": 1, "items": _ACTION},
|
|
143
|
+
"summary": {"anyOf": [_NON_EMPTY_STRING, {**_STRING_LIST, "minItems": 1}]},
|
|
144
|
+
},
|
|
145
|
+
"required": ["actions", "summary"],
|
|
146
|
+
"additionalProperties": False,
|
|
147
|
+
}
|
|
148
|
+
DIAGNOSE_SCHEMA = {
|
|
149
|
+
"type": "object",
|
|
150
|
+
"properties": {
|
|
151
|
+
"decision": {"type": "string", "enum": ["fix", "replan", "block"]},
|
|
152
|
+
"updated_hypothesis": _NON_EMPTY_STRING,
|
|
153
|
+
"actions": {"type": "array", "items": _ACTION},
|
|
154
|
+
},
|
|
155
|
+
"required": ["decision", "updated_hypothesis", "actions"],
|
|
156
|
+
"additionalProperties": False,
|
|
157
|
+
}
|
|
158
|
+
REVIEW_SCHEMA = {
|
|
159
|
+
"type": "object",
|
|
160
|
+
"properties": {
|
|
161
|
+
"approved": {"type": "boolean"},
|
|
162
|
+
"issues": {
|
|
163
|
+
"type": "array",
|
|
164
|
+
"items": {
|
|
165
|
+
"type": "object",
|
|
166
|
+
"properties": {
|
|
167
|
+
"severity": {"type": "string", "enum": ["low", "medium", "high", "critical"]},
|
|
168
|
+
"reason": _NON_EMPTY_STRING,
|
|
169
|
+
"path": {"anyOf": [_NON_EMPTY_STRING, {"type": "null"}]},
|
|
170
|
+
},
|
|
171
|
+
"required": ["severity", "reason", "path"],
|
|
172
|
+
"additionalProperties": False,
|
|
173
|
+
},
|
|
174
|
+
},
|
|
175
|
+
"summary": _NON_EMPTY_STRING,
|
|
176
|
+
},
|
|
177
|
+
"required": ["approved", "issues", "summary"],
|
|
178
|
+
"additionalProperties": False,
|
|
179
|
+
}
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
class OrchestrationError(RuntimeError):
|
|
183
|
+
pass
|
|
184
|
+
|
|
185
|
+
|
|
186
|
+
def _strings(value: Any, field: str, role: str) -> list[str]:
|
|
187
|
+
if not isinstance(value, list) or not all(isinstance(item, str) and item.strip() for item in value):
|
|
188
|
+
raise OrchestrationError(
|
|
189
|
+
f"Invalid {role} response field '{field}': expected array of non-empty strings; received {type(value).__name__}"
|
|
190
|
+
)
|
|
191
|
+
return [item.strip() for item in value]
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
def _non_empty_string(value: Any, role: str, field: str) -> str:
|
|
195
|
+
if not isinstance(value, str) or not value.strip():
|
|
196
|
+
raise OrchestrationError(f"Invalid {role} response field '{field}': expected non-empty string; received {type(value).__name__}")
|
|
197
|
+
return value.strip()
|
|
198
|
+
|
|
199
|
+
|
|
200
|
+
def _bounded_confidence(value: Any, field: str) -> float:
|
|
201
|
+
if isinstance(value, bool) or not isinstance(value, (int, float)):
|
|
202
|
+
safe_actual = repr(value) if isinstance(value, str) and value.lower() in {"low", "medium", "high"} else type(value).__name__
|
|
203
|
+
raise OrchestrationError(
|
|
204
|
+
f"Invalid understand response field '{field}': expected number from 0.0 through 1.0; received {safe_actual}"
|
|
205
|
+
)
|
|
206
|
+
confidence = float(value)
|
|
207
|
+
if not 0.0 <= confidence <= 1.0:
|
|
208
|
+
raise OrchestrationError(
|
|
209
|
+
f"Invalid understand response field '{field}': expected number from 0.0 through 1.0; received {value!r}"
|
|
210
|
+
)
|
|
211
|
+
return confidence
|
|
212
|
+
|
|
213
|
+
|
|
214
|
+
def _understanding(response: dict[str, Any]) -> Understanding:
|
|
215
|
+
try:
|
|
216
|
+
expected = {"problem", "expected_behavior", "affected_paths", "root_cause", "evidence", "proposed_solution", "confidence"}
|
|
217
|
+
if set(response) != expected:
|
|
218
|
+
raise TypeError(f"expected exactly fields {sorted(expected)!r}")
|
|
219
|
+
raw_evidence = response["evidence"]
|
|
220
|
+
if not isinstance(raw_evidence, list) or not raw_evidence:
|
|
221
|
+
raise TypeError("evidence must be a non-empty array")
|
|
222
|
+
evidence: list[Evidence] = []
|
|
223
|
+
for index, item in enumerate(raw_evidence):
|
|
224
|
+
if not isinstance(item, dict) or set(item) != {"statement", "paths", "confidence"}:
|
|
225
|
+
raise TypeError(f"evidence[{index}] must contain exactly statement, paths, and confidence")
|
|
226
|
+
evidence.append(
|
|
227
|
+
Evidence(
|
|
228
|
+
_non_empty_string(item["statement"], "understand", f"evidence[{index}].statement"),
|
|
229
|
+
tuple(_strings(item["paths"], f"evidence[{index}].paths", "understand")),
|
|
230
|
+
_bounded_confidence(item["confidence"], f"evidence[{index}].confidence"),
|
|
231
|
+
)
|
|
232
|
+
)
|
|
233
|
+
return Understanding(
|
|
234
|
+
problem=_non_empty_string(response["problem"], "understand", "problem"),
|
|
235
|
+
expected_behavior=_non_empty_string(response["expected_behavior"], "understand", "expected_behavior"),
|
|
236
|
+
affected_paths=_strings(response["affected_paths"], "affected_paths", "understand"),
|
|
237
|
+
root_cause=_non_empty_string(response["root_cause"], "understand", "root_cause"),
|
|
238
|
+
evidence=evidence,
|
|
239
|
+
proposed_solution=_strings(response["proposed_solution"], "proposed_solution", "understand"),
|
|
240
|
+
confidence=_bounded_confidence(response["confidence"], "confidence"),
|
|
241
|
+
)
|
|
242
|
+
except OrchestrationError:
|
|
243
|
+
raise
|
|
244
|
+
except (KeyError, TypeError, ValueError) as exc:
|
|
245
|
+
raise OrchestrationError(f"Invalid understanding response: {exc}") from exc
|
|
246
|
+
|
|
247
|
+
|
|
248
|
+
def _plan(response: dict[str, Any], fallback: Sequence[tuple[str, ...]]) -> EngineeringPlan:
|
|
249
|
+
try:
|
|
250
|
+
expected = {"files_to_inspect", "implementation", "verification", "rationale"}
|
|
251
|
+
if set(response) != expected:
|
|
252
|
+
raise TypeError(f"expected exactly fields {sorted(expected)!r}")
|
|
253
|
+
commands: list[tuple[str, ...]] = []
|
|
254
|
+
raw_commands = response["verification"]
|
|
255
|
+
if not isinstance(raw_commands, list):
|
|
256
|
+
raise TypeError("verification must be a list")
|
|
257
|
+
for command in raw_commands:
|
|
258
|
+
if not isinstance(command, list) or not all(isinstance(token, str) and token for token in command):
|
|
259
|
+
raise TypeError("verification commands must be argv arrays")
|
|
260
|
+
commands.append(tuple(command))
|
|
261
|
+
return EngineeringPlan(
|
|
262
|
+
_strings(response["files_to_inspect"], "files_to_inspect", "plan"),
|
|
263
|
+
_strings(response["implementation"], "implementation", "plan"),
|
|
264
|
+
commands or list(fallback),
|
|
265
|
+
_non_empty_string(response["rationale"], "plan", "rationale"),
|
|
266
|
+
)
|
|
267
|
+
except OrchestrationError:
|
|
268
|
+
raise
|
|
269
|
+
except (KeyError, TypeError) as exc:
|
|
270
|
+
raise OrchestrationError(f"Invalid plan response: {exc}") from exc
|
|
271
|
+
|
|
272
|
+
|
|
273
|
+
def _review(response: dict[str, Any]) -> ReviewDecision:
|
|
274
|
+
try:
|
|
275
|
+
if set(response) != {"approved", "issues", "summary"}:
|
|
276
|
+
raise TypeError("expected exactly approved, issues, and summary")
|
|
277
|
+
if not isinstance(response["approved"], bool) or not isinstance(response["issues"], list):
|
|
278
|
+
raise TypeError("approved must be boolean and issues must be an array")
|
|
279
|
+
issues: list[ReviewIssue] = []
|
|
280
|
+
for index, item in enumerate(response["issues"]):
|
|
281
|
+
if not isinstance(item, dict) or set(item) != {"severity", "reason", "path"}:
|
|
282
|
+
raise TypeError(f"issues[{index}] must contain exactly severity, reason, and path")
|
|
283
|
+
severity = _non_empty_string(item["severity"], "review", f"issues[{index}].severity")
|
|
284
|
+
if severity not in {"low", "medium", "high", "critical"}:
|
|
285
|
+
raise TypeError(f"issues[{index}].severity is unsupported")
|
|
286
|
+
path = item["path"]
|
|
287
|
+
if path is not None:
|
|
288
|
+
path = _non_empty_string(path, "review", f"issues[{index}].path")
|
|
289
|
+
issues.append(ReviewIssue(severity, _non_empty_string(item["reason"], "review", f"issues[{index}].reason"), path))
|
|
290
|
+
approved = response["approved"]
|
|
291
|
+
if approved and issues:
|
|
292
|
+
raise OrchestrationError("Reviewer cannot approve while returning issues")
|
|
293
|
+
return ReviewDecision(approved, issues, _non_empty_string(response["summary"], "review", "summary"))
|
|
294
|
+
except OrchestrationError:
|
|
295
|
+
raise
|
|
296
|
+
except (KeyError, TypeError) as exc:
|
|
297
|
+
raise OrchestrationError(f"Invalid review response: {exc}") from exc
|
|
298
|
+
|
|
299
|
+
|
|
300
|
+
def _execute_actions(workspace: Workspace, response: dict[str, Any], allowed_paths: set[str]) -> list[str]:
|
|
301
|
+
if not isinstance(response, dict) or set(response) != {"actions", "summary"}:
|
|
302
|
+
raise OrchestrationError("Invalid implement response: expected exactly actions and summary")
|
|
303
|
+
actions = response.get("actions")
|
|
304
|
+
if not isinstance(actions, list) or not actions:
|
|
305
|
+
raise OrchestrationError("Implementation requires at least one structured action")
|
|
306
|
+
changed: list[str] = []
|
|
307
|
+
for action in actions:
|
|
308
|
+
if not isinstance(action, dict) or set(action) != {"tool", "arguments"} or not isinstance(action["arguments"], dict):
|
|
309
|
+
raise OrchestrationError("Every action must contain only tool and arguments")
|
|
310
|
+
tool, arguments = action["tool"], action["arguments"]
|
|
311
|
+
if tool not in {"replace_text", "write_file"}:
|
|
312
|
+
raise OrchestrationError(f"Implementation tool is not allowed: {tool}")
|
|
313
|
+
path = arguments.get("path")
|
|
314
|
+
if not isinstance(path, str) or path not in allowed_paths:
|
|
315
|
+
raise OrchestrationError(f"Action path was not inspected/planned: {path}")
|
|
316
|
+
if tool == "replace_text":
|
|
317
|
+
required = {"path", "old", "new"}
|
|
318
|
+
if frozenset(arguments) not in {frozenset(required), frozenset({*required, "count"})}:
|
|
319
|
+
raise OrchestrationError("replace_text arguments must contain only path, old, new, and optional count")
|
|
320
|
+
if not all(isinstance(arguments[key], str) for key in required) or not arguments["old"]:
|
|
321
|
+
raise OrchestrationError("replace_text requires string path, non-empty old, and string new")
|
|
322
|
+
count = arguments.get("count", 1)
|
|
323
|
+
if isinstance(count, bool) or not isinstance(count, int) or count < 1:
|
|
324
|
+
raise OrchestrationError("replace_text count must be an integer greater than zero")
|
|
325
|
+
workspace.replace_text(path, arguments["old"], arguments["new"], count)
|
|
326
|
+
else:
|
|
327
|
+
if set(arguments) != {"path", "content"} or not isinstance(arguments.get("content"), str):
|
|
328
|
+
raise OrchestrationError("write_file requires only string path and content")
|
|
329
|
+
workspace.write_file(path, arguments["content"])
|
|
330
|
+
changed.append(path)
|
|
331
|
+
return changed
|
|
332
|
+
|
|
333
|
+
|
|
334
|
+
def _diff(root: Path, workspace: Workspace, max_chars: int = 40_000) -> str:
|
|
335
|
+
completed = subprocess.run(["git", "diff", "--no-ext-diff", "--"], cwd=root, capture_output=True, text=True, timeout=20, check=False)
|
|
336
|
+
text = completed.stdout if completed.returncode == 0 else "(Git diff unavailable)\n"
|
|
337
|
+
tracked = {
|
|
338
|
+
line[3:].split(" -> ")[-1]
|
|
339
|
+
for line in subprocess.run(["git", "status", "--porcelain=v1"], cwd=root, capture_output=True, text=True, timeout=20, check=False).stdout.splitlines()
|
|
340
|
+
if len(line) > 3 and not line.startswith("??")
|
|
341
|
+
}
|
|
342
|
+
for relative in sorted(workspace.modified_paths - tracked):
|
|
343
|
+
target = root / relative
|
|
344
|
+
if not target.is_file():
|
|
345
|
+
continue
|
|
346
|
+
content = target.read_text(encoding="utf-8", errors="replace").splitlines(keepends=True)
|
|
347
|
+
text += "".join(difflib.unified_diff([], content, fromfile="/dev/null", tofile=f"b/{relative}"))
|
|
348
|
+
return text[:max_chars]
|
|
349
|
+
|
|
350
|
+
|
|
351
|
+
def _metrics(root: Path, modified_paths: set[str]) -> ChangeMetrics:
|
|
352
|
+
completed = subprocess.run(["git", "diff", "--numstat", "--"], cwd=root, capture_output=True, text=True, timeout=20, check=False)
|
|
353
|
+
added = deleted = 0
|
|
354
|
+
paths: set[str] = set(modified_paths)
|
|
355
|
+
if completed.returncode == 0:
|
|
356
|
+
for line in completed.stdout.splitlines():
|
|
357
|
+
parts = line.split("\t", 2)
|
|
358
|
+
if len(parts) != 3:
|
|
359
|
+
continue
|
|
360
|
+
if parts[0].isdigit():
|
|
361
|
+
added += int(parts[0])
|
|
362
|
+
if parts[1].isdigit():
|
|
363
|
+
deleted += int(parts[1])
|
|
364
|
+
paths.add(parts[2])
|
|
365
|
+
for relative in modified_paths:
|
|
366
|
+
tracked = subprocess.run(
|
|
367
|
+
["git", "cat-file", "-e", f"HEAD:{relative}"], cwd=root, capture_output=True, timeout=10, check=False
|
|
368
|
+
).returncode == 0
|
|
369
|
+
target = root / relative
|
|
370
|
+
if not tracked and target.is_file():
|
|
371
|
+
added += len(target.read_text(encoding="utf-8", errors="replace").splitlines())
|
|
372
|
+
return ChangeMetrics(len(paths), added, deleted, sorted(paths))
|
|
373
|
+
|
|
374
|
+
|
|
375
|
+
def _default_commands(repository: Any) -> tuple[list[tuple[str, ...]], list[tuple[str, ...]]]:
|
|
376
|
+
targeted: list[tuple[str, ...]] = []
|
|
377
|
+
broad: list[tuple[str, ...]] = []
|
|
378
|
+
for capability in repository.capabilities:
|
|
379
|
+
if not capability.trusted:
|
|
380
|
+
continue
|
|
381
|
+
destination = broad if capability.broad or capability.kind in {"build", "lint", "typecheck", "integration"} else targeted
|
|
382
|
+
if capability.command not in destination:
|
|
383
|
+
destination.append(capability.command)
|
|
384
|
+
if not targeted and any("python" in component.languages for component in repository.components):
|
|
385
|
+
targeted.append(("python", "-m", "compileall", "-q", "."))
|
|
386
|
+
return targeted[:3], broad[:5]
|
|
387
|
+
|
|
388
|
+
|
|
389
|
+
def _builtin_verification_commands(repository: Any) -> list[tuple[str, ...]]:
|
|
390
|
+
if not repository.git_head:
|
|
391
|
+
return []
|
|
392
|
+
return list(_BUILTIN_GIT_VERIFICATION)
|
|
393
|
+
|
|
394
|
+
|
|
395
|
+
def _command_kind(command: tuple[str, ...], repository: Any) -> str | None:
|
|
396
|
+
if repository.git_head:
|
|
397
|
+
builtin_kind = _BUILTIN_GIT_VERIFICATION.get(command)
|
|
398
|
+
if builtin_kind is not None:
|
|
399
|
+
return builtin_kind
|
|
400
|
+
for capability in repository.capabilities:
|
|
401
|
+
if not capability.trusted:
|
|
402
|
+
continue
|
|
403
|
+
if command == capability.command:
|
|
404
|
+
return capability.kind
|
|
405
|
+
if capability.command[:3] in {("python", "-m", "pytest"), ("python3", "-m", "pytest")} and command[:3] == capability.command[:3]:
|
|
406
|
+
return capability.kind
|
|
407
|
+
if capability.command and command and capability.command[0] in {"pytest", "cargo", "go", "mvn", "ctest", "dotnet"} and command[0] == capability.command[0]:
|
|
408
|
+
return capability.kind
|
|
409
|
+
if command[:3] in {("python", "-m", "compileall"), ("python3", "-m", "compileall")} and any(
|
|
410
|
+
"python" in component.languages for component in repository.components
|
|
411
|
+
):
|
|
412
|
+
return "build"
|
|
413
|
+
return None
|
|
414
|
+
|
|
415
|
+
|
|
416
|
+
def _support_acceptance_criteria(task: Any, changes: ChangeMetrics, final_results: list[VerificationResult], review: ReviewDecision) -> None:
|
|
417
|
+
passing_commands = [" ".join(result.command) for result in final_results if result.passed]
|
|
418
|
+
for criterion in task.acceptance_criteria:
|
|
419
|
+
lowered = criterion.description.lower()
|
|
420
|
+
if "review" in lowered and review.approved:
|
|
421
|
+
criterion.evidence.append("Independent reviewer approved the final diff")
|
|
422
|
+
elif "test" in lowered or "coverage" in lowered:
|
|
423
|
+
criterion.evidence.extend(path for path in changes.paths if _is_test_path(path))
|
|
424
|
+
criterion.evidence.extend(passing_commands)
|
|
425
|
+
elif "unrelated" in lowered:
|
|
426
|
+
criterion.evidence.append(f"Minimal-diff gate accepted {changes.files_changed} changed file(s)")
|
|
427
|
+
elif changes.paths:
|
|
428
|
+
criterion.evidence.extend(changes.paths)
|
|
429
|
+
|
|
430
|
+
|
|
431
|
+
def _is_test_path(path: str) -> bool:
|
|
432
|
+
lowered = path.lower()
|
|
433
|
+
parts = Path(lowered).parts
|
|
434
|
+
return (
|
|
435
|
+
any(part in {"test", "tests", "spec", "specs", "__tests__"} for part in parts)
|
|
436
|
+
or Path(lowered).name.startswith("test_")
|
|
437
|
+
or ".test." in lowered
|
|
438
|
+
or ".spec." in lowered
|
|
439
|
+
)
|
|
440
|
+
|
|
441
|
+
|
|
442
|
+
def _run_commands(workspace: Workspace, commands: Sequence[tuple[str, ...]], phase: str, *, baseline: bool = False) -> list[VerificationResult]:
|
|
443
|
+
results: list[VerificationResult] = []
|
|
444
|
+
for command in commands:
|
|
445
|
+
try:
|
|
446
|
+
results.append(workspace.run(command, phase=phase, baseline=baseline))
|
|
447
|
+
except (SafetyError, OSError) as exc:
|
|
448
|
+
raise OrchestrationError(f"Cannot run {' '.join(command)}: {exc}") from exc
|
|
449
|
+
return results
|
|
450
|
+
|
|
451
|
+
|
|
452
|
+
def _context_supports_understanding(
|
|
453
|
+
understanding: Understanding, context: dict[str, object], working_root: Path
|
|
454
|
+
) -> bool:
|
|
455
|
+
if not understanding.implementation_ready(working_root):
|
|
456
|
+
return False
|
|
457
|
+
snippets = context.get("snippets", {})
|
|
458
|
+
supplied_paths = set(snippets) if isinstance(snippets, dict) else set()
|
|
459
|
+
claimed_paths = {
|
|
460
|
+
path
|
|
461
|
+
for evidence in understanding.evidence
|
|
462
|
+
for path in evidence.paths
|
|
463
|
+
}
|
|
464
|
+
existing_affected_paths = {
|
|
465
|
+
path for path in understanding.affected_paths if (working_root / path).is_file()
|
|
466
|
+
}
|
|
467
|
+
return (
|
|
468
|
+
existing_affected_paths.issubset(supplied_paths)
|
|
469
|
+
and bool(claimed_paths)
|
|
470
|
+
and claimed_paths.issubset(supplied_paths)
|
|
471
|
+
and all((working_root / path).is_file() for path in claimed_paths)
|
|
472
|
+
)
|
|
473
|
+
|
|
474
|
+
|
|
475
|
+
def _baseline_test_count_regressed(
|
|
476
|
+
baseline_results: Sequence[VerificationResult],
|
|
477
|
+
final_results: Sequence[VerificationResult],
|
|
478
|
+
) -> tuple[VerificationResult, VerificationResult] | None:
|
|
479
|
+
for baseline in baseline_results:
|
|
480
|
+
if baseline.tests_run is None:
|
|
481
|
+
continue
|
|
482
|
+
for final in final_results:
|
|
483
|
+
if final.command == baseline.command and final.tests_run is not None:
|
|
484
|
+
if final.tests_run < baseline.tests_run:
|
|
485
|
+
return baseline, final
|
|
486
|
+
break
|
|
487
|
+
return None
|
|
488
|
+
|
|
489
|
+
|
|
490
|
+
class DevAgent:
|
|
491
|
+
"""Deterministic orchestration; model reasoning is bounded inside named states."""
|
|
492
|
+
|
|
493
|
+
def __init__(self, provider: ModelProvider, *, max_corrections: int = 2, isolate: bool = True, verbose: bool = False, status: Callable[[str], None] | None = None) -> None:
|
|
494
|
+
self.provider = provider
|
|
495
|
+
self.max_corrections = max(0, min(max_corrections, 5))
|
|
496
|
+
self.isolate = isolate
|
|
497
|
+
self.verbose = verbose
|
|
498
|
+
self.status = status or (lambda message: None)
|
|
499
|
+
|
|
500
|
+
def _announce(self, artifacts: RunArtifacts, lifecycle: Lifecycle) -> None:
|
|
501
|
+
artifacts.record("state", state=lifecycle.state)
|
|
502
|
+
if self.verbose:
|
|
503
|
+
self.status(f"[{lifecycle.state.value}]")
|
|
504
|
+
|
|
505
|
+
def _diagnostic(self, artifacts: RunArtifacts, category: str, message: str) -> None:
|
|
506
|
+
artifacts.record("diagnostic", category=category, message=message)
|
|
507
|
+
if self.verbose:
|
|
508
|
+
self.status(f"[{category}] {message}")
|
|
509
|
+
|
|
510
|
+
def _show_retrieval(self, artifacts: RunArtifacts, context: dict[str, object]) -> None:
|
|
511
|
+
diagnostics = context.get("diagnostics", {})
|
|
512
|
+
if not isinstance(diagnostics, dict):
|
|
513
|
+
return
|
|
514
|
+
self._diagnostic(
|
|
515
|
+
artifacts,
|
|
516
|
+
"RETRIEVAL",
|
|
517
|
+
(
|
|
518
|
+
f"lexical matches: {diagnostics.get('exact_lexical_matches', 0)} exact, "
|
|
519
|
+
f"{diagnostics.get('normalized_lexical_matches', 0)} normalized"
|
|
520
|
+
),
|
|
521
|
+
)
|
|
522
|
+
self._diagnostic(
|
|
523
|
+
artifacts,
|
|
524
|
+
"RETRIEVAL",
|
|
525
|
+
f"fallback: {diagnostics.get('fallback', 'none')}",
|
|
526
|
+
)
|
|
527
|
+
selected = diagnostics.get("selected", [])
|
|
528
|
+
if isinstance(selected, list):
|
|
529
|
+
for path in selected:
|
|
530
|
+
self._diagnostic(artifacts, "RETRIEVAL", f"selected: {path}")
|
|
531
|
+
|
|
532
|
+
def run(self, repository_root: Path | str, requirement: str) -> RunResult:
|
|
533
|
+
root = Path(repository_root).expanduser().resolve()
|
|
534
|
+
task = compile_task(requirement)
|
|
535
|
+
artifacts = RunArtifacts(root)
|
|
536
|
+
lifecycle = Lifecycle()
|
|
537
|
+
verification: list[VerificationResult] = []
|
|
538
|
+
not_run: list[str] = []
|
|
539
|
+
implementation: list[str] = []
|
|
540
|
+
recommendations: list[str] = []
|
|
541
|
+
understanding = Understanding("", "", [], "", [], [], 0.0)
|
|
542
|
+
review: ReviewDecision | None = None
|
|
543
|
+
source_repository = discover_repository(root, probe_capabilities=False)
|
|
544
|
+
selection = select_worktree(root, artifacts.run_id, enabled=self.isolate, git_head=source_repository.git_head, dirty_files=source_repository.dirty_files)
|
|
545
|
+
working_root = selection.root
|
|
546
|
+
repository = discover_repository(working_root)
|
|
547
|
+
repository.root = str(root)
|
|
548
|
+
repository.git_branch = source_repository.git_branch
|
|
549
|
+
repository.git_head = source_repository.git_head
|
|
550
|
+
repository.dirty_files = source_repository.dirty_files
|
|
551
|
+
workspace = Workspace(working_root, artifacts, source_repository.dirty_files if not selection.isolated else ())
|
|
552
|
+
artifacts.write_json("metadata.json", {"run_id": artifacts.run_id, "task": task, "repository": repository})
|
|
553
|
+
artifacts.record("worktree", source=str(root), working_root=str(working_root), isolated=selection.isolated, reason=selection.reason)
|
|
554
|
+
self._announce(artifacts, lifecycle)
|
|
555
|
+
self._diagnostic(artifacts, "WORKTREE", f"source: {root}")
|
|
556
|
+
self._diagnostic(artifacts, "WORKTREE", f"isolated: {str(selection.isolated).lower()}")
|
|
557
|
+
self._diagnostic(artifacts, "WORKTREE", f"working_root: {working_root}")
|
|
558
|
+
|
|
559
|
+
try:
|
|
560
|
+
if selection.creation_failed:
|
|
561
|
+
raise OrchestrationError(selection.reason)
|
|
562
|
+
lifecycle.transition(AgentState.DISCOVER)
|
|
563
|
+
self._announce(artifacts, lifecycle)
|
|
564
|
+
self._diagnostic(
|
|
565
|
+
artifacts,
|
|
566
|
+
"DISCOVER",
|
|
567
|
+
f"repository files: {repository.inventory_file_count}",
|
|
568
|
+
)
|
|
569
|
+
self._diagnostic(
|
|
570
|
+
artifacts,
|
|
571
|
+
"DISCOVER",
|
|
572
|
+
f"components: {len(repository.components)}",
|
|
573
|
+
)
|
|
574
|
+
for diagnostic in repository.capability_diagnostics:
|
|
575
|
+
self._diagnostic(artifacts, "VERIFICATION_DISCOVERY", diagnostic)
|
|
576
|
+
memory = RepositoryMemory(root)
|
|
577
|
+
memory.load_facts()
|
|
578
|
+
memory.store_facts(repository.facts)
|
|
579
|
+
|
|
580
|
+
lifecycle.transition(AgentState.UNDERSTAND)
|
|
581
|
+
self._announce(artifacts, lifecycle)
|
|
582
|
+
context = retrieve_context(
|
|
583
|
+
workspace,
|
|
584
|
+
repository,
|
|
585
|
+
task.goal,
|
|
586
|
+
requires_tests=task.requires_tests,
|
|
587
|
+
)
|
|
588
|
+
self._show_retrieval(artifacts, context)
|
|
589
|
+
for context_attempt in range(2):
|
|
590
|
+
response = self.provider.request(role="understand", payload={"task": jsonable(task), "repository": context}, schema=UNDERSTANDING_SCHEMA)
|
|
591
|
+
understanding = _understanding(response)
|
|
592
|
+
if _context_supports_understanding(understanding, context, working_root):
|
|
593
|
+
break
|
|
594
|
+
if context_attempt == 0:
|
|
595
|
+
lifecycle.transition(AgentState.GATHER_CONTEXT)
|
|
596
|
+
self._announce(artifacts, lifecycle)
|
|
597
|
+
context = retrieve_context(
|
|
598
|
+
workspace,
|
|
599
|
+
repository,
|
|
600
|
+
task.goal + " " + " ".join(understanding.affected_paths),
|
|
601
|
+
max_chars=32_000,
|
|
602
|
+
requires_tests=task.requires_tests,
|
|
603
|
+
)
|
|
604
|
+
self._show_retrieval(artifacts, context)
|
|
605
|
+
lifecycle.transition(AgentState.UNDERSTAND)
|
|
606
|
+
self._announce(artifacts, lifecycle)
|
|
607
|
+
else:
|
|
608
|
+
raise OrchestrationError("Evidence gate rejected implementation: root cause or source evidence is insufficient")
|
|
609
|
+
|
|
610
|
+
lifecycle.transition(AgentState.TASK_SPEC)
|
|
611
|
+
self._announce(artifacts, lifecycle)
|
|
612
|
+
targeted, broad = _default_commands(repository)
|
|
613
|
+
lifecycle.transition(AgentState.BASELINE)
|
|
614
|
+
self._announce(artifacts, lifecycle)
|
|
615
|
+
if targeted:
|
|
616
|
+
verification.extend(_run_commands(workspace, targeted[:1], "baseline", baseline=True))
|
|
617
|
+
else:
|
|
618
|
+
not_run.append("Baseline: no evidence-backed local test or compile command was discovered")
|
|
619
|
+
|
|
620
|
+
lifecycle.transition(AgentState.PLAN)
|
|
621
|
+
self._announce(artifacts, lifecycle)
|
|
622
|
+
plan_response = self.provider.request(
|
|
623
|
+
role="plan",
|
|
624
|
+
payload={
|
|
625
|
+
"task": jsonable(task),
|
|
626
|
+
"understanding": jsonable(understanding),
|
|
627
|
+
"capabilities": jsonable(repository.capabilities),
|
|
628
|
+
"builtin_verification": jsonable(
|
|
629
|
+
_builtin_verification_commands(repository)
|
|
630
|
+
),
|
|
631
|
+
},
|
|
632
|
+
schema=PLAN_SCHEMA,
|
|
633
|
+
)
|
|
634
|
+
plan = _plan(plan_response, targeted)
|
|
635
|
+
unsupported = [command for command in plan.verification if _command_kind(command, repository) is None]
|
|
636
|
+
if unsupported:
|
|
637
|
+
raise OrchestrationError(
|
|
638
|
+
"Plan requested verification not supported by repository evidence "
|
|
639
|
+
f"or DevAgent built-ins: {' '.join(unsupported[0])}"
|
|
640
|
+
)
|
|
641
|
+
if task.requires_tests and not any(_command_kind(command, repository) in {"test", "integration"} for command in plan.verification):
|
|
642
|
+
raise OrchestrationError("A task requiring tests must use an evidence-backed test command")
|
|
643
|
+
allowed_paths = set(plan.files_to_inspect)
|
|
644
|
+
if not set(understanding.affected_paths).issubset(allowed_paths):
|
|
645
|
+
raise OrchestrationError("Plan does not include all evidence-backed affected paths")
|
|
646
|
+
for path in allowed_paths:
|
|
647
|
+
if (working_root / path).exists():
|
|
648
|
+
workspace.read_file(path)
|
|
649
|
+
|
|
650
|
+
lifecycle.transition(AgentState.GATHER_CONTEXT)
|
|
651
|
+
self._announce(artifacts, lifecycle)
|
|
652
|
+
lifecycle.transition(AgentState.REPRODUCE)
|
|
653
|
+
self._announce(artifacts, lifecycle)
|
|
654
|
+
baseline_failed = any(result.baseline and not result.passed for result in verification)
|
|
655
|
+
if task.task_type in {TaskType.BUG_FIX, TaskType.RUNTIME_ERROR, TaskType.TEST_FAILURE} and not baseline_failed:
|
|
656
|
+
not_run.append("Pre-change failure was not independently reproduced; implementation is based on source evidence and final regression verification")
|
|
657
|
+
|
|
658
|
+
lifecycle.transition(AgentState.IMPLEMENT)
|
|
659
|
+
self._announce(artifacts, lifecycle)
|
|
660
|
+
implement_response = self.provider.request(
|
|
661
|
+
role="implement",
|
|
662
|
+
payload={"task": jsonable(task), "understanding": jsonable(understanding), "plan": jsonable(plan), "files": {path: workspace.read_file(path) for path in allowed_paths if (working_root / path).is_file()}},
|
|
663
|
+
schema=IMPLEMENT_SCHEMA,
|
|
664
|
+
)
|
|
665
|
+
_execute_actions(workspace, implement_response, allowed_paths)
|
|
666
|
+
implementation.extend(_strings(implement_response.get("summary", []), "summary", "implement") if isinstance(implement_response.get("summary"), list) else [str(implement_response.get("summary", "Implemented planned change"))])
|
|
667
|
+
|
|
668
|
+
corrections = 0
|
|
669
|
+
failure_signatures: list[tuple[object, ...]] = []
|
|
670
|
+
while True:
|
|
671
|
+
lifecycle.transition(AgentState.VERIFY_TARGETED)
|
|
672
|
+
self._announce(artifacts, lifecycle)
|
|
673
|
+
current_targeted = plan.verification or targeted
|
|
674
|
+
if not current_targeted:
|
|
675
|
+
raise OrchestrationError("No targeted verification command is available")
|
|
676
|
+
target_results = _run_commands(workspace, current_targeted[:3], "targeted")
|
|
677
|
+
verification.extend(target_results)
|
|
678
|
+
if all(result.passed for result in target_results):
|
|
679
|
+
break
|
|
680
|
+
lifecycle.transition(AgentState.DIAGNOSE)
|
|
681
|
+
self._announce(artifacts, lifecycle)
|
|
682
|
+
if corrections >= self.max_corrections:
|
|
683
|
+
raise OrchestrationError("Targeted verification still fails after the bounded correction budget")
|
|
684
|
+
diagnosis = self.provider.request(
|
|
685
|
+
role="diagnose",
|
|
686
|
+
payload={"task": jsonable(task), "failures": jsonable([result for result in target_results if not result.passed]), "failed_hypotheses": recommendations},
|
|
687
|
+
schema=DIAGNOSE_SCHEMA,
|
|
688
|
+
)
|
|
689
|
+
decision = diagnosis.get("decision")
|
|
690
|
+
signature = tuple((result.command, result.exit_code, result.classification, result.stderr[-1000:]) for result in target_results if not result.passed)
|
|
691
|
+
failure_signatures.append(signature)
|
|
692
|
+
if failure_signatures.count(signature) >= 2:
|
|
693
|
+
decision = "replan"
|
|
694
|
+
hypothesis = str(diagnosis.get("updated_hypothesis", "")).strip()
|
|
695
|
+
if hypothesis:
|
|
696
|
+
recommendations.append(f"Diagnosis {corrections + 1}: {hypothesis}")
|
|
697
|
+
if decision == "block":
|
|
698
|
+
raise OrchestrationError(hypothesis or "Diagnosis could not identify a safe correction")
|
|
699
|
+
if decision == "replan":
|
|
700
|
+
lifecycle.transition(AgentState.PLAN)
|
|
701
|
+
self._announce(artifacts, lifecycle)
|
|
702
|
+
replanned = self.provider.request(
|
|
703
|
+
role="replan",
|
|
704
|
+
payload={
|
|
705
|
+
"task": jsonable(task),
|
|
706
|
+
"understanding": jsonable(understanding),
|
|
707
|
+
"failures": jsonable(target_results),
|
|
708
|
+
"failed_hypotheses": recommendations,
|
|
709
|
+
"capabilities": jsonable(repository.capabilities),
|
|
710
|
+
"builtin_verification": jsonable(
|
|
711
|
+
_builtin_verification_commands(repository)
|
|
712
|
+
),
|
|
713
|
+
},
|
|
714
|
+
schema=PLAN_SCHEMA,
|
|
715
|
+
)
|
|
716
|
+
plan = _plan(replanned, targeted)
|
|
717
|
+
if any(_command_kind(command, repository) is None for command in plan.verification):
|
|
718
|
+
raise OrchestrationError(
|
|
719
|
+
"Replan selected verification unsupported by repository "
|
|
720
|
+
"evidence or DevAgent built-ins"
|
|
721
|
+
)
|
|
722
|
+
allowed_paths.update(plan.files_to_inspect)
|
|
723
|
+
lifecycle.transition(AgentState.GATHER_CONTEXT)
|
|
724
|
+
self._announce(artifacts, lifecycle)
|
|
725
|
+
for path in plan.files_to_inspect:
|
|
726
|
+
if (working_root / path).is_file():
|
|
727
|
+
workspace.read_file(path)
|
|
728
|
+
lifecycle.transition(AgentState.REPRODUCE)
|
|
729
|
+
self._announce(artifacts, lifecycle)
|
|
730
|
+
lifecycle.transition(AgentState.IMPLEMENT)
|
|
731
|
+
self._announce(artifacts, lifecycle)
|
|
732
|
+
correction_response = self.provider.request(
|
|
733
|
+
role="implement_replan",
|
|
734
|
+
payload={"plan": jsonable(plan), "diagnosis": hypothesis, "files": {path: workspace.read_file(path) for path in allowed_paths if (working_root / path).is_file()}},
|
|
735
|
+
schema=IMPLEMENT_SCHEMA,
|
|
736
|
+
)
|
|
737
|
+
else:
|
|
738
|
+
lifecycle.transition(AgentState.IMPLEMENT)
|
|
739
|
+
self._announce(artifacts, lifecycle)
|
|
740
|
+
correction_response = {"actions": diagnosis.get("actions"), "summary": [hypothesis or "Applied focused correction"]}
|
|
741
|
+
_execute_actions(workspace, correction_response, allowed_paths)
|
|
742
|
+
implementation.append(hypothesis or "Applied focused correction")
|
|
743
|
+
corrections += 1
|
|
744
|
+
|
|
745
|
+
lifecycle.transition(AgentState.VERIFY_BROAD)
|
|
746
|
+
self._announce(artifacts, lifecycle)
|
|
747
|
+
broad_results = _run_commands(workspace, broad, "broad") if broad else []
|
|
748
|
+
verification.extend(broad_results)
|
|
749
|
+
if any(not result.passed for result in broad_results):
|
|
750
|
+
not_run.append("One or more broader checks did not pass in the local environment")
|
|
751
|
+
|
|
752
|
+
lifecycle.transition(AgentState.REVIEW)
|
|
753
|
+
self._announce(artifacts, lifecycle)
|
|
754
|
+
review = _review(self.provider.request(
|
|
755
|
+
role="review",
|
|
756
|
+
payload={
|
|
757
|
+
"task": jsonable(task),
|
|
758
|
+
"acceptance_criteria": jsonable(task.acceptance_criteria),
|
|
759
|
+
"conventions": jsonable(repository.facts),
|
|
760
|
+
"diff": _diff(working_root, workspace),
|
|
761
|
+
"verification": jsonable(verification),
|
|
762
|
+
},
|
|
763
|
+
schema=REVIEW_SCHEMA,
|
|
764
|
+
))
|
|
765
|
+
if not review.approved:
|
|
766
|
+
if corrections >= self.max_corrections:
|
|
767
|
+
raise OrchestrationError("Independent review rejected the patch after the correction budget")
|
|
768
|
+
lifecycle.transition(AgentState.IMPLEMENT)
|
|
769
|
+
self._announce(artifacts, lifecycle)
|
|
770
|
+
revision_response = self.provider.request(
|
|
771
|
+
role="implement_review_fixes",
|
|
772
|
+
payload={"issues": jsonable(review.issues), "plan": jsonable(plan), "diff": _diff(working_root, workspace)},
|
|
773
|
+
schema=IMPLEMENT_SCHEMA,
|
|
774
|
+
)
|
|
775
|
+
_execute_actions(workspace, revision_response, allowed_paths)
|
|
776
|
+
implementation.append(str(revision_response.get("summary", "Addressed review findings")))
|
|
777
|
+
corrections += 1
|
|
778
|
+
lifecycle.transition(AgentState.VERIFY_TARGETED)
|
|
779
|
+
self._announce(artifacts, lifecycle)
|
|
780
|
+
review_fix_results = _run_commands(workspace, plan.verification or targeted, "targeted_after_review")
|
|
781
|
+
verification.extend(review_fix_results)
|
|
782
|
+
if any(not result.passed for result in review_fix_results):
|
|
783
|
+
raise OrchestrationError("Verification failed after review corrections")
|
|
784
|
+
lifecycle.transition(AgentState.VERIFY_BROAD)
|
|
785
|
+
self._announce(artifacts, lifecycle)
|
|
786
|
+
verification.extend(_run_commands(workspace, broad, "broad_after_review") if broad else [])
|
|
787
|
+
lifecycle.transition(AgentState.REVIEW)
|
|
788
|
+
self._announce(artifacts, lifecycle)
|
|
789
|
+
review = _review(self.provider.request(role="review", payload={"task": jsonable(task), "diff": _diff(working_root, workspace), "verification": jsonable(verification)}, schema=REVIEW_SCHEMA))
|
|
790
|
+
if not review.approved:
|
|
791
|
+
raise OrchestrationError("Independent reviewer still rejects the corrected patch")
|
|
792
|
+
|
|
793
|
+
lifecycle.transition(AgentState.QUALITY_CHECK)
|
|
794
|
+
self._announce(artifacts, lifecycle)
|
|
795
|
+
changes = _metrics(working_root, workspace.modified_paths)
|
|
796
|
+
if changes.files_changed > 8 or changes.lines_added + changes.lines_deleted > 500:
|
|
797
|
+
raise OrchestrationError("Minimal-diff gate rejected unexpectedly broad scope")
|
|
798
|
+
if task.requires_tests and task.task_type not in {TaskType.TEST_FAILURE, TaskType.UNIT_TEST} and not any(
|
|
799
|
+
_is_test_path(path) for path in changes.paths
|
|
800
|
+
):
|
|
801
|
+
raise OrchestrationError("Required regression/feature coverage was not added or updated")
|
|
802
|
+
quality_commands = _builtin_verification_commands(repository)
|
|
803
|
+
quality_results = _run_commands(workspace, quality_commands, "quality")
|
|
804
|
+
verification.extend(quality_results)
|
|
805
|
+
if any(not result.passed for result in quality_results):
|
|
806
|
+
raise OrchestrationError("git diff --check rejected the patch")
|
|
807
|
+
|
|
808
|
+
lifecycle.transition(AgentState.FINAL_VERIFY)
|
|
809
|
+
self._announce(artifacts, lifecycle)
|
|
810
|
+
final_commands = list(dict.fromkeys([*(plan.verification or targeted), *broad, *quality_commands]))
|
|
811
|
+
final_results = _run_commands(workspace, final_commands, "final")
|
|
812
|
+
verification.extend(final_results)
|
|
813
|
+
regression = _baseline_test_count_regressed(
|
|
814
|
+
[result for result in verification if result.baseline],
|
|
815
|
+
final_results,
|
|
816
|
+
)
|
|
817
|
+
if regression is not None:
|
|
818
|
+
baseline_result, final_result = regression
|
|
819
|
+
raise OrchestrationError(
|
|
820
|
+
"Final verification collected fewer tests than baseline "
|
|
821
|
+
f"for {' '.join(final_result.command)}: "
|
|
822
|
+
f"{baseline_result.tests_run} -> {final_result.tests_run}"
|
|
823
|
+
)
|
|
824
|
+
current_results = [result for result in final_results if result.revision == workspace.revision]
|
|
825
|
+
if review is not None:
|
|
826
|
+
_support_acceptance_criteria(task, changes, current_results, review)
|
|
827
|
+
unsupported_criteria = [criterion.description for criterion in task.acceptance_criteria if criterion.required and not criterion.evidence]
|
|
828
|
+
if not current_results or any(not result.passed for result in current_results):
|
|
829
|
+
lifecycle.transition(AgentState.PARTIALLY_VERIFIED)
|
|
830
|
+
outcome = Outcome.PARTIALLY_VERIFIED
|
|
831
|
+
elif broad_results and any(not result.passed for result in broad_results):
|
|
832
|
+
lifecycle.transition(AgentState.PARTIALLY_VERIFIED)
|
|
833
|
+
outcome = Outcome.PARTIALLY_VERIFIED
|
|
834
|
+
elif task.risk is RiskLevel.HIGH and not broad:
|
|
835
|
+
not_run.append("High-risk change has no discovered broad/static verification capability")
|
|
836
|
+
lifecycle.transition(AgentState.PARTIALLY_VERIFIED)
|
|
837
|
+
outcome = Outcome.PARTIALLY_VERIFIED
|
|
838
|
+
elif unsupported_criteria:
|
|
839
|
+
not_run.append("Acceptance criteria without final evidence: " + "; ".join(unsupported_criteria))
|
|
840
|
+
lifecycle.transition(AgentState.PARTIALLY_VERIFIED)
|
|
841
|
+
outcome = Outcome.PARTIALLY_VERIFIED
|
|
842
|
+
else:
|
|
843
|
+
lifecycle.transition(AgentState.LEARN)
|
|
844
|
+
self._announce(artifacts, lifecycle)
|
|
845
|
+
for capability in repository.capabilities:
|
|
846
|
+
memory.store_strategy(f"Use {' '.join(capability.command)} for {capability.kind}", [capability.source])
|
|
847
|
+
lifecycle.transition(AgentState.REPORT)
|
|
848
|
+
self._announce(artifacts, lifecycle)
|
|
849
|
+
lifecycle.transition(AgentState.SUCCESS)
|
|
850
|
+
outcome = Outcome.VERIFIED
|
|
851
|
+
except (OrchestrationError, ProviderError, SafetyError, OSError, ValueError) as exc:
|
|
852
|
+
artifacts.record("blocked", reason=str(exc), state=lifecycle.state)
|
|
853
|
+
not_run.append(str(exc))
|
|
854
|
+
if lifecycle.state not in {AgentState.BLOCKED, AgentState.SUCCESS, AgentState.PARTIALLY_VERIFIED}:
|
|
855
|
+
try:
|
|
856
|
+
lifecycle.transition(AgentState.BLOCKED)
|
|
857
|
+
except InvalidTransition:
|
|
858
|
+
lifecycle.state = AgentState.BLOCKED
|
|
859
|
+
lifecycle.history.append(AgentState.BLOCKED)
|
|
860
|
+
outcome = Outcome.BLOCKED
|
|
861
|
+
changes = _metrics(working_root, workspace.modified_paths)
|
|
862
|
+
|
|
863
|
+
result = RunResult(
|
|
864
|
+
outcome=outcome,
|
|
865
|
+
task=task,
|
|
866
|
+
repository=repository,
|
|
867
|
+
run_id=artifacts.run_id,
|
|
868
|
+
run_dir=str(artifacts.root),
|
|
869
|
+
root_cause=understanding.root_cause,
|
|
870
|
+
implementation=implementation,
|
|
871
|
+
changes=changes,
|
|
872
|
+
verification=verification,
|
|
873
|
+
review=review,
|
|
874
|
+
not_run=not_run,
|
|
875
|
+
recommendations=recommendations,
|
|
876
|
+
state_history=lifecycle.history,
|
|
877
|
+
working_root=str(working_root),
|
|
878
|
+
)
|
|
879
|
+
artifacts.verification.write_text(json.dumps(jsonable(verification), indent=2) + "\n", encoding="utf-8")
|
|
880
|
+
artifacts.report.write_text(json.dumps(jsonable(result), indent=2) + "\n", encoding="utf-8")
|
|
881
|
+
artifacts.record("report", outcome=outcome)
|
|
882
|
+
return result
|
|
883
|
+
|
|
884
|
+
|
|
885
|
+
def run_devagent(repository_root: Path | str, requirement: str, provider: ModelProvider, *, verbose: bool = False, status: Callable[[str], None] | None = None) -> tuple[RunResult, str]:
|
|
886
|
+
result = DevAgent(provider, verbose=verbose, status=status).run(repository_root, requirement)
|
|
887
|
+
return result, render_report(result)
|