devagent-ai 0.3.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent/__init__.py +5 -0
- agent/llm.py +6 -0
- agent/loop.py +16 -0
- agent/memory.py +5 -0
- agent/prompts.py +3 -0
- agent/tools.py +6 -0
- devagent/__init__.py +3 -0
- devagent/__main__.py +4 -0
- devagent/artifacts.py +46 -0
- devagent/cli.py +164 -0
- devagent/config.py +72 -0
- devagent/discovery.py +504 -0
- devagent/evaluation.py +388 -0
- devagent/memory.py +51 -0
- devagent/models.py +251 -0
- devagent/orchestrator.py +887 -0
- devagent/providers.py +344 -0
- devagent/report.py +52 -0
- devagent/retrieval.py +478 -0
- devagent/safety.py +131 -0
- devagent/state_machine.py +45 -0
- devagent/tasking.py +83 -0
- devagent/workspace.py +199 -0
- devagent/worktree.py +151 -0
- devagent_ai-0.3.1.dist-info/METADATA +415 -0
- devagent_ai-0.3.1.dist-info/RECORD +31 -0
- devagent_ai-0.3.1.dist-info/WHEEL +5 -0
- devagent_ai-0.3.1.dist-info/entry_points.txt +2 -0
- devagent_ai-0.3.1.dist-info/licenses/LICENSE +21 -0
- devagent_ai-0.3.1.dist-info/licenses/NOTICE +10 -0
- devagent_ai-0.3.1.dist-info/top_level.txt +2 -0
devagent/evaluation.py
ADDED
|
@@ -0,0 +1,388 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
import subprocess
|
|
5
|
+
import time
|
|
6
|
+
from dataclasses import dataclass
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
from typing import Any, Iterable
|
|
9
|
+
|
|
10
|
+
from devagent.models import Outcome, RunResult, jsonable
|
|
11
|
+
from devagent.orchestrator import DevAgent
|
|
12
|
+
from devagent.providers import ModelProvider
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
@dataclass(frozen=True)
|
|
16
|
+
class EvaluationMetrics:
|
|
17
|
+
task_success: bool
|
|
18
|
+
acceptance_criteria_supported: int
|
|
19
|
+
acceptance_criteria_total: int
|
|
20
|
+
acceptance_coverage: float
|
|
21
|
+
new_regressions: int | None
|
|
22
|
+
files_changed: int
|
|
23
|
+
lines_changed: int
|
|
24
|
+
iterations: int
|
|
25
|
+
model_calls: int
|
|
26
|
+
tool_calls: int
|
|
27
|
+
runtime_seconds: float
|
|
28
|
+
outcome: Outcome
|
|
29
|
+
final_verification_passed: bool
|
|
30
|
+
review_approved: bool
|
|
31
|
+
source_head_unchanged: bool
|
|
32
|
+
source_status_unchanged: bool
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
@dataclass(frozen=True)
|
|
36
|
+
class EvaluationExpectation:
|
|
37
|
+
expected_outcomes: tuple[Outcome, ...] = (Outcome.VERIFIED,)
|
|
38
|
+
max_files_changed: int | None = None
|
|
39
|
+
max_lines_changed: int | None = None
|
|
40
|
+
require_acceptance_evidence: bool = True
|
|
41
|
+
require_review_approval: bool = True
|
|
42
|
+
require_final_verification: bool = True
|
|
43
|
+
require_no_new_regressions: bool = True
|
|
44
|
+
require_source_unchanged: bool = True
|
|
45
|
+
|
|
46
|
+
def __post_init__(self) -> None:
|
|
47
|
+
if not self.expected_outcomes:
|
|
48
|
+
raise ValueError("expected_outcomes must contain at least one outcome")
|
|
49
|
+
if self.max_files_changed is not None and self.max_files_changed < 0:
|
|
50
|
+
raise ValueError("max_files_changed must be >= 0")
|
|
51
|
+
if self.max_lines_changed is not None and self.max_lines_changed < 0:
|
|
52
|
+
raise ValueError("max_lines_changed must be >= 0")
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
@dataclass(frozen=True)
|
|
56
|
+
class EvaluationCaseResult:
|
|
57
|
+
name: str
|
|
58
|
+
category: str
|
|
59
|
+
expected_outcomes: tuple[Outcome, ...]
|
|
60
|
+
metrics: EvaluationMetrics
|
|
61
|
+
passed: bool
|
|
62
|
+
false_verified: bool
|
|
63
|
+
unexpected_blocked: bool
|
|
64
|
+
violations: tuple[str, ...]
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
@dataclass(frozen=True)
|
|
68
|
+
class EvaluationSuiteMetrics:
|
|
69
|
+
cases_total: int
|
|
70
|
+
cases_passed: int
|
|
71
|
+
pass_rate: float
|
|
72
|
+
verified: int
|
|
73
|
+
partially_verified: int
|
|
74
|
+
blocked: int
|
|
75
|
+
false_verified: int
|
|
76
|
+
false_verified_rate: float
|
|
77
|
+
unexpected_blocked: int
|
|
78
|
+
acceptance_criteria_supported: int
|
|
79
|
+
acceptance_criteria_total: int
|
|
80
|
+
acceptance_coverage: float
|
|
81
|
+
total_model_calls: int
|
|
82
|
+
total_tool_calls: int
|
|
83
|
+
total_runtime_seconds: float
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
@dataclass(frozen=True)
|
|
87
|
+
class _RepositorySnapshot:
|
|
88
|
+
head: str | None
|
|
89
|
+
status: tuple[str, ...]
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
class _CountingProvider:
|
|
93
|
+
def __init__(self, inner: ModelProvider) -> None:
|
|
94
|
+
self.inner = inner
|
|
95
|
+
self.calls = 0
|
|
96
|
+
|
|
97
|
+
def request(
|
|
98
|
+
self,
|
|
99
|
+
*,
|
|
100
|
+
role: str,
|
|
101
|
+
payload: dict[str, Any],
|
|
102
|
+
schema: dict[str, Any],
|
|
103
|
+
) -> dict[str, Any]:
|
|
104
|
+
self.calls += 1
|
|
105
|
+
return self.inner.request(role=role, payload=payload, schema=schema)
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def _generated_status_path(path: str) -> bool:
|
|
109
|
+
normalized = path.replace("\\", "/").strip('"')
|
|
110
|
+
parts = tuple(part for part in normalized.split("/") if part)
|
|
111
|
+
return (
|
|
112
|
+
normalized == ".devagent"
|
|
113
|
+
or normalized.startswith(".devagent/")
|
|
114
|
+
or normalized.endswith(".pyc")
|
|
115
|
+
or any(
|
|
116
|
+
part in {"__pycache__", ".pytest_cache", ".mypy_cache", ".ruff_cache"}
|
|
117
|
+
for part in parts
|
|
118
|
+
)
|
|
119
|
+
)
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def _repository_snapshot(repository: Path) -> _RepositorySnapshot:
|
|
123
|
+
repository = repository.resolve()
|
|
124
|
+
try:
|
|
125
|
+
head_result = subprocess.run(
|
|
126
|
+
["git", "rev-parse", "HEAD"],
|
|
127
|
+
cwd=repository,
|
|
128
|
+
capture_output=True,
|
|
129
|
+
text=True,
|
|
130
|
+
timeout=10,
|
|
131
|
+
check=False,
|
|
132
|
+
)
|
|
133
|
+
status_result = subprocess.run(
|
|
134
|
+
["git", "status", "--porcelain=v1", "--untracked-files=all"],
|
|
135
|
+
cwd=repository,
|
|
136
|
+
capture_output=True,
|
|
137
|
+
text=True,
|
|
138
|
+
timeout=10,
|
|
139
|
+
check=False,
|
|
140
|
+
)
|
|
141
|
+
except (OSError, subprocess.TimeoutExpired):
|
|
142
|
+
return _RepositorySnapshot(None, ())
|
|
143
|
+
|
|
144
|
+
head = head_result.stdout.strip() if head_result.returncode == 0 else None
|
|
145
|
+
status: list[str] = []
|
|
146
|
+
if status_result.returncode == 0:
|
|
147
|
+
for line in status_result.stdout.splitlines():
|
|
148
|
+
rendered = line.rstrip()
|
|
149
|
+
if not rendered:
|
|
150
|
+
continue
|
|
151
|
+
path = rendered[3:] if len(rendered) >= 4 else rendered
|
|
152
|
+
if " -> " in path:
|
|
153
|
+
path = path.split(" -> ", 1)[1]
|
|
154
|
+
if _generated_status_path(path):
|
|
155
|
+
continue
|
|
156
|
+
status.append(rendered)
|
|
157
|
+
return _RepositorySnapshot(head, tuple(sorted(status)))
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
def _tool_call_count(run_dir: str) -> int:
|
|
161
|
+
observations = Path(run_dir) / "observations.jsonl"
|
|
162
|
+
if not observations.is_file():
|
|
163
|
+
return 0
|
|
164
|
+
count = 0
|
|
165
|
+
for line in observations.read_text(encoding="utf-8").splitlines():
|
|
166
|
+
try:
|
|
167
|
+
event = json.loads(line).get("event")
|
|
168
|
+
except json.JSONDecodeError:
|
|
169
|
+
continue
|
|
170
|
+
if event in {"command_finished", "file_written", "text_replaced"}:
|
|
171
|
+
count += 1
|
|
172
|
+
return count
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
def _final_verification_passed(result: RunResult) -> bool:
|
|
176
|
+
final = [item for item in result.verification if item.phase == "final"]
|
|
177
|
+
return bool(final) and all(item.passed for item in final)
|
|
178
|
+
|
|
179
|
+
|
|
180
|
+
def _known_new_regressions(result: RunResult) -> int | None:
|
|
181
|
+
"""Count only regressions supported by comparable baseline/final evidence."""
|
|
182
|
+
baseline_by_command = {
|
|
183
|
+
item.command: item.passed
|
|
184
|
+
for item in result.verification
|
|
185
|
+
if item.baseline or item.phase == "baseline"
|
|
186
|
+
}
|
|
187
|
+
final = [item for item in result.verification if item.phase == "final"]
|
|
188
|
+
if not final:
|
|
189
|
+
return None
|
|
190
|
+
|
|
191
|
+
regressions = 0
|
|
192
|
+
unknown_failure = False
|
|
193
|
+
for item in final:
|
|
194
|
+
if item.passed:
|
|
195
|
+
continue
|
|
196
|
+
baseline_passed = baseline_by_command.get(item.command)
|
|
197
|
+
if baseline_passed is True:
|
|
198
|
+
regressions += 1
|
|
199
|
+
elif baseline_passed is None:
|
|
200
|
+
unknown_failure = True
|
|
201
|
+
|
|
202
|
+
if unknown_failure and regressions == 0:
|
|
203
|
+
return None
|
|
204
|
+
return regressions
|
|
205
|
+
|
|
206
|
+
|
|
207
|
+
def evaluate(
|
|
208
|
+
repository: Path,
|
|
209
|
+
requirement: str,
|
|
210
|
+
provider: ModelProvider,
|
|
211
|
+
*,
|
|
212
|
+
isolate: bool = True,
|
|
213
|
+
) -> tuple[RunResult, EvaluationMetrics]:
|
|
214
|
+
"""Run one evidence-backed evaluation and capture deterministic production metrics."""
|
|
215
|
+
repository = repository.resolve()
|
|
216
|
+
source_before = _repository_snapshot(repository)
|
|
217
|
+
counted_provider = _CountingProvider(provider)
|
|
218
|
+
started = time.monotonic()
|
|
219
|
+
result = DevAgent(counted_provider, isolate=isolate).run(repository, requirement)
|
|
220
|
+
runtime = time.monotonic() - started
|
|
221
|
+
source_after = _repository_snapshot(repository)
|
|
222
|
+
|
|
223
|
+
supported = sum(bool(criterion.evidence) for criterion in result.task.acceptance_criteria)
|
|
224
|
+
total = len(result.task.acceptance_criteria)
|
|
225
|
+
coverage = supported / total if total else 1.0
|
|
226
|
+
metrics = EvaluationMetrics(
|
|
227
|
+
task_success=result.outcome is Outcome.VERIFIED,
|
|
228
|
+
acceptance_criteria_supported=supported,
|
|
229
|
+
acceptance_criteria_total=total,
|
|
230
|
+
acceptance_coverage=coverage,
|
|
231
|
+
new_regressions=_known_new_regressions(result),
|
|
232
|
+
files_changed=result.changes.files_changed,
|
|
233
|
+
lines_changed=result.changes.lines_added + result.changes.lines_deleted,
|
|
234
|
+
iterations=sum(
|
|
235
|
+
state.value in {"IMPLEMENT", "DIAGNOSE"}
|
|
236
|
+
for state in result.state_history
|
|
237
|
+
),
|
|
238
|
+
model_calls=counted_provider.calls,
|
|
239
|
+
tool_calls=_tool_call_count(result.run_dir),
|
|
240
|
+
runtime_seconds=runtime,
|
|
241
|
+
outcome=result.outcome,
|
|
242
|
+
final_verification_passed=_final_verification_passed(result),
|
|
243
|
+
review_approved=bool(result.review and result.review.approved),
|
|
244
|
+
source_head_unchanged=source_before.head == source_after.head,
|
|
245
|
+
source_status_unchanged=source_before.status == source_after.status,
|
|
246
|
+
)
|
|
247
|
+
return result, metrics
|
|
248
|
+
|
|
249
|
+
|
|
250
|
+
def score_evaluation(
|
|
251
|
+
name: str,
|
|
252
|
+
category: str,
|
|
253
|
+
metrics: EvaluationMetrics,
|
|
254
|
+
expectation: EvaluationExpectation | None = None,
|
|
255
|
+
) -> EvaluationCaseResult:
|
|
256
|
+
"""Score a run without allowing model prose to determine benchmark success."""
|
|
257
|
+
expected = expectation or EvaluationExpectation()
|
|
258
|
+
violations: list[str] = []
|
|
259
|
+
if metrics.outcome not in expected.expected_outcomes:
|
|
260
|
+
allowed = ",".join(item.value for item in expected.expected_outcomes)
|
|
261
|
+
violations.append(f"unexpected_outcome:{metrics.outcome.value}:expected={allowed}")
|
|
262
|
+
|
|
263
|
+
false_verified = (
|
|
264
|
+
metrics.outcome is Outcome.VERIFIED
|
|
265
|
+
and Outcome.VERIFIED not in expected.expected_outcomes
|
|
266
|
+
)
|
|
267
|
+
unexpected_blocked = (
|
|
268
|
+
metrics.outcome is Outcome.BLOCKED
|
|
269
|
+
and Outcome.BLOCKED not in expected.expected_outcomes
|
|
270
|
+
)
|
|
271
|
+
|
|
272
|
+
if metrics.outcome is Outcome.VERIFIED:
|
|
273
|
+
if (
|
|
274
|
+
expected.require_acceptance_evidence
|
|
275
|
+
and metrics.acceptance_criteria_supported < metrics.acceptance_criteria_total
|
|
276
|
+
):
|
|
277
|
+
violations.append("verified_without_complete_acceptance_evidence")
|
|
278
|
+
if expected.require_review_approval and not metrics.review_approved:
|
|
279
|
+
violations.append("verified_without_approved_review")
|
|
280
|
+
if expected.require_final_verification and not metrics.final_verification_passed:
|
|
281
|
+
violations.append("verified_without_final_verification")
|
|
282
|
+
if expected.require_no_new_regressions and metrics.new_regressions is None:
|
|
283
|
+
violations.append("verified_with_unknown_regression_status")
|
|
284
|
+
|
|
285
|
+
if (
|
|
286
|
+
expected.require_no_new_regressions
|
|
287
|
+
and metrics.new_regressions not in {None, 0}
|
|
288
|
+
):
|
|
289
|
+
violations.append(f"new_regressions:{metrics.new_regressions}")
|
|
290
|
+
if expected.require_source_unchanged and not metrics.source_head_unchanged:
|
|
291
|
+
violations.append("source_head_changed")
|
|
292
|
+
if expected.require_source_unchanged and not metrics.source_status_unchanged:
|
|
293
|
+
violations.append("source_status_changed")
|
|
294
|
+
if (
|
|
295
|
+
expected.max_files_changed is not None
|
|
296
|
+
and metrics.files_changed > expected.max_files_changed
|
|
297
|
+
):
|
|
298
|
+
violations.append(
|
|
299
|
+
f"files_changed:{metrics.files_changed}:max={expected.max_files_changed}"
|
|
300
|
+
)
|
|
301
|
+
if (
|
|
302
|
+
expected.max_lines_changed is not None
|
|
303
|
+
and metrics.lines_changed > expected.max_lines_changed
|
|
304
|
+
):
|
|
305
|
+
violations.append(
|
|
306
|
+
f"lines_changed:{metrics.lines_changed}:max={expected.max_lines_changed}"
|
|
307
|
+
)
|
|
308
|
+
|
|
309
|
+
return EvaluationCaseResult(
|
|
310
|
+
name=name,
|
|
311
|
+
category=category,
|
|
312
|
+
expected_outcomes=expected.expected_outcomes,
|
|
313
|
+
metrics=metrics,
|
|
314
|
+
passed=not violations,
|
|
315
|
+
false_verified=false_verified,
|
|
316
|
+
unexpected_blocked=unexpected_blocked,
|
|
317
|
+
violations=tuple(violations),
|
|
318
|
+
)
|
|
319
|
+
|
|
320
|
+
|
|
321
|
+
def evaluate_case(
|
|
322
|
+
name: str,
|
|
323
|
+
category: str,
|
|
324
|
+
repository: Path,
|
|
325
|
+
requirement: str,
|
|
326
|
+
provider: ModelProvider,
|
|
327
|
+
*,
|
|
328
|
+
expectation: EvaluationExpectation | None = None,
|
|
329
|
+
isolate: bool = True,
|
|
330
|
+
) -> tuple[RunResult, EvaluationCaseResult]:
|
|
331
|
+
result, metrics = evaluate(
|
|
332
|
+
repository,
|
|
333
|
+
requirement,
|
|
334
|
+
provider,
|
|
335
|
+
isolate=isolate,
|
|
336
|
+
)
|
|
337
|
+
return result, score_evaluation(name, category, metrics, expectation)
|
|
338
|
+
|
|
339
|
+
|
|
340
|
+
def aggregate_results(results: Iterable[EvaluationCaseResult]) -> EvaluationSuiteMetrics:
|
|
341
|
+
items = list(results)
|
|
342
|
+
total = len(items)
|
|
343
|
+
passed = sum(item.passed for item in items)
|
|
344
|
+
verified = sum(item.metrics.outcome is Outcome.VERIFIED for item in items)
|
|
345
|
+
partially_verified = sum(
|
|
346
|
+
item.metrics.outcome is Outcome.PARTIALLY_VERIFIED for item in items
|
|
347
|
+
)
|
|
348
|
+
blocked = sum(item.metrics.outcome is Outcome.BLOCKED for item in items)
|
|
349
|
+
false_verified = sum(item.false_verified for item in items)
|
|
350
|
+
supported = sum(item.metrics.acceptance_criteria_supported for item in items)
|
|
351
|
+
criteria_total = sum(item.metrics.acceptance_criteria_total for item in items)
|
|
352
|
+
return EvaluationSuiteMetrics(
|
|
353
|
+
cases_total=total,
|
|
354
|
+
cases_passed=passed,
|
|
355
|
+
pass_rate=passed / total if total else 1.0,
|
|
356
|
+
verified=verified,
|
|
357
|
+
partially_verified=partially_verified,
|
|
358
|
+
blocked=blocked,
|
|
359
|
+
false_verified=false_verified,
|
|
360
|
+
false_verified_rate=false_verified / total if total else 0.0,
|
|
361
|
+
unexpected_blocked=sum(item.unexpected_blocked for item in items),
|
|
362
|
+
acceptance_criteria_supported=supported,
|
|
363
|
+
acceptance_criteria_total=criteria_total,
|
|
364
|
+
acceptance_coverage=supported / criteria_total if criteria_total else 1.0,
|
|
365
|
+
total_model_calls=sum(item.metrics.model_calls for item in items),
|
|
366
|
+
total_tool_calls=sum(item.metrics.tool_calls for item in items),
|
|
367
|
+
total_runtime_seconds=sum(item.metrics.runtime_seconds for item in items),
|
|
368
|
+
)
|
|
369
|
+
|
|
370
|
+
|
|
371
|
+
def write_suite_report(
|
|
372
|
+
path: Path,
|
|
373
|
+
results: Iterable[EvaluationCaseResult],
|
|
374
|
+
) -> EvaluationSuiteMetrics:
|
|
375
|
+
"""Write a machine-readable benchmark report suitable for CI artifacts."""
|
|
376
|
+
items = list(results)
|
|
377
|
+
summary = aggregate_results(items)
|
|
378
|
+
payload = {
|
|
379
|
+
"schema_version": 1,
|
|
380
|
+
"summary": jsonable(summary),
|
|
381
|
+
"cases": jsonable(items),
|
|
382
|
+
}
|
|
383
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
384
|
+
path.write_text(
|
|
385
|
+
json.dumps(payload, indent=2, sort_keys=True) + "\n",
|
|
386
|
+
encoding="utf-8",
|
|
387
|
+
)
|
|
388
|
+
return summary
|
devagent/memory.py
ADDED
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
from typing import Iterable
|
|
6
|
+
|
|
7
|
+
from devagent.discovery import facts_are_current
|
|
8
|
+
from devagent.models import RepositoryFact, jsonable
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class RepositoryMemory:
|
|
12
|
+
"""Bounded, evidence-invalidated local repository and strategy memory."""
|
|
13
|
+
|
|
14
|
+
def __init__(self, root: Path | str) -> None:
|
|
15
|
+
self.root = Path(root).resolve()
|
|
16
|
+
self.directory = self.root / ".devagent" / "memory"
|
|
17
|
+
self.directory.mkdir(parents=True, exist_ok=True)
|
|
18
|
+
self.facts_file = self.directory / "repository.json"
|
|
19
|
+
self.strategies_file = self.directory / "strategies.json"
|
|
20
|
+
|
|
21
|
+
def store_facts(self, facts: Iterable[RepositoryFact]) -> None:
|
|
22
|
+
bounded = list(facts)[-250:]
|
|
23
|
+
self.facts_file.write_text(json.dumps(jsonable(bounded), indent=2) + "\n", encoding="utf-8")
|
|
24
|
+
|
|
25
|
+
def load_facts(self) -> list[RepositoryFact]:
|
|
26
|
+
if not self.facts_file.is_file():
|
|
27
|
+
return []
|
|
28
|
+
try:
|
|
29
|
+
values = json.loads(self.facts_file.read_text(encoding="utf-8"))
|
|
30
|
+
facts = [RepositoryFact(item["fact"], item["confidence"], tuple(item["evidence"]), item["fingerprints"], item["learned_at"]) for item in values]
|
|
31
|
+
except (OSError, KeyError, TypeError, json.JSONDecodeError):
|
|
32
|
+
return []
|
|
33
|
+
if not facts_are_current(self.root, facts):
|
|
34
|
+
self.facts_file.unlink(missing_ok=True)
|
|
35
|
+
return []
|
|
36
|
+
return facts
|
|
37
|
+
|
|
38
|
+
def store_strategy(self, statement: str, evidence: list[str]) -> None:
|
|
39
|
+
if not statement.strip() or not evidence:
|
|
40
|
+
return
|
|
41
|
+
values: list[dict[str, object]] = []
|
|
42
|
+
if self.strategies_file.is_file():
|
|
43
|
+
try:
|
|
44
|
+
values = json.loads(self.strategies_file.read_text(encoding="utf-8"))
|
|
45
|
+
except (OSError, json.JSONDecodeError):
|
|
46
|
+
values = []
|
|
47
|
+
entry = {"statement": statement.strip(), "evidence": evidence[:10]}
|
|
48
|
+
if entry not in values:
|
|
49
|
+
values.append(entry)
|
|
50
|
+
self.strategies_file.write_text(json.dumps(values[-100:], indent=2) + "\n", encoding="utf-8")
|
|
51
|
+
|
devagent/models.py
ADDED
|
@@ -0,0 +1,251 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from dataclasses import asdict, dataclass, field
|
|
4
|
+
from enum import Enum
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
class TaskType(str, Enum):
|
|
10
|
+
BUG_FIX = "BUG_FIX"
|
|
11
|
+
FEATURE = "FEATURE"
|
|
12
|
+
RUNTIME_ERROR = "RUNTIME_ERROR"
|
|
13
|
+
TEST_FAILURE = "TEST_FAILURE"
|
|
14
|
+
BUILD_FAILURE = "BUILD_FAILURE"
|
|
15
|
+
UNIT_TEST = "UNIT_TEST"
|
|
16
|
+
REFACTOR = "REFACTOR"
|
|
17
|
+
PERFORMANCE = "PERFORMANCE"
|
|
18
|
+
GENERAL_ENGINEERING_TASK = "GENERAL_ENGINEERING_TASK"
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
class AgentState(str, Enum):
|
|
22
|
+
PREFLIGHT = "PREFLIGHT"
|
|
23
|
+
DISCOVER = "DISCOVER"
|
|
24
|
+
UNDERSTAND = "UNDERSTAND"
|
|
25
|
+
TASK_SPEC = "TASK_SPEC"
|
|
26
|
+
BASELINE = "BASELINE"
|
|
27
|
+
PLAN = "PLAN"
|
|
28
|
+
GATHER_CONTEXT = "GATHER_CONTEXT"
|
|
29
|
+
REPRODUCE = "REPRODUCE"
|
|
30
|
+
IMPLEMENT = "IMPLEMENT"
|
|
31
|
+
VERIFY_TARGETED = "VERIFY_TARGETED"
|
|
32
|
+
DIAGNOSE = "DIAGNOSE"
|
|
33
|
+
VERIFY_BROAD = "VERIFY_BROAD"
|
|
34
|
+
REVIEW = "REVIEW"
|
|
35
|
+
QUALITY_CHECK = "QUALITY_CHECK"
|
|
36
|
+
FINAL_VERIFY = "FINAL_VERIFY"
|
|
37
|
+
LEARN = "LEARN"
|
|
38
|
+
REPORT = "REPORT"
|
|
39
|
+
SUCCESS = "SUCCESS"
|
|
40
|
+
PARTIALLY_VERIFIED = "PARTIALLY_VERIFIED"
|
|
41
|
+
BLOCKED = "BLOCKED"
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
class Outcome(str, Enum):
|
|
45
|
+
VERIFIED = "VERIFIED"
|
|
46
|
+
PARTIALLY_VERIFIED = "PARTIALLY_VERIFIED"
|
|
47
|
+
BLOCKED = "BLOCKED"
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
class RiskLevel(str, Enum):
|
|
51
|
+
LOW = "LOW"
|
|
52
|
+
MEDIUM = "MEDIUM"
|
|
53
|
+
HIGH = "HIGH"
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
class FailureClass(str, Enum):
|
|
57
|
+
ASSERTION_FAILURE = "ASSERTION_FAILURE"
|
|
58
|
+
SYNTAX_ERROR = "SYNTAX_ERROR"
|
|
59
|
+
TYPE_ERROR = "TYPE_ERROR"
|
|
60
|
+
IMPORT_ERROR = "IMPORT_ERROR"
|
|
61
|
+
DEPENDENCY_ERROR = "DEPENDENCY_ERROR"
|
|
62
|
+
BUILD_ERROR = "BUILD_ERROR"
|
|
63
|
+
ENVIRONMENT_ERROR = "ENVIRONMENT_ERROR"
|
|
64
|
+
TIMEOUT = "TIMEOUT"
|
|
65
|
+
FLAKY_TEST = "FLAKY_TEST"
|
|
66
|
+
BASELINE_FAILURE = "BASELINE_FAILURE"
|
|
67
|
+
NEW_REGRESSION = "NEW_REGRESSION"
|
|
68
|
+
UNKNOWN = "UNKNOWN"
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
class CapabilityProvenance(str, Enum):
|
|
72
|
+
EXPLICIT = "EXPLICIT"
|
|
73
|
+
PROBED = "PROBED"
|
|
74
|
+
INFERRED = "INFERRED"
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
@dataclass(frozen=True)
|
|
78
|
+
class Evidence:
|
|
79
|
+
statement: str
|
|
80
|
+
paths: tuple[str, ...]
|
|
81
|
+
confidence: float = 1.0
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
@dataclass(frozen=True)
|
|
85
|
+
class RepositoryFact:
|
|
86
|
+
fact: str
|
|
87
|
+
confidence: float
|
|
88
|
+
evidence: tuple[str, ...]
|
|
89
|
+
fingerprints: dict[str, str]
|
|
90
|
+
learned_at: str
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
@dataclass(frozen=True)
|
|
94
|
+
class Capability:
|
|
95
|
+
kind: str
|
|
96
|
+
command: tuple[str, ...]
|
|
97
|
+
source: str
|
|
98
|
+
component: str = "."
|
|
99
|
+
broad: bool = False
|
|
100
|
+
provenance: CapabilityProvenance = CapabilityProvenance.EXPLICIT
|
|
101
|
+
provenance_detail: str = "declared by repository configuration"
|
|
102
|
+
tests_collected: int | None = None
|
|
103
|
+
|
|
104
|
+
@property
|
|
105
|
+
def trusted(self) -> bool:
|
|
106
|
+
return self.provenance in {CapabilityProvenance.EXPLICIT, CapabilityProvenance.PROBED}
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
@dataclass
|
|
110
|
+
class Component:
|
|
111
|
+
path: str
|
|
112
|
+
languages: list[str] = field(default_factory=list)
|
|
113
|
+
frameworks: list[str] = field(default_factory=list)
|
|
114
|
+
manifests: list[str] = field(default_factory=list)
|
|
115
|
+
test_locations: list[str] = field(default_factory=list)
|
|
116
|
+
capabilities: list[Capability] = field(default_factory=list)
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
@dataclass
|
|
120
|
+
class RepositoryModel:
|
|
121
|
+
root: str
|
|
122
|
+
kind: str
|
|
123
|
+
components: list[Component]
|
|
124
|
+
facts: list[RepositoryFact]
|
|
125
|
+
git_branch: str | None = None
|
|
126
|
+
git_head: str | None = None
|
|
127
|
+
dirty_files: list[str] = field(default_factory=list)
|
|
128
|
+
inventory_file_count: int = 0
|
|
129
|
+
capability_diagnostics: list[str] = field(default_factory=list)
|
|
130
|
+
|
|
131
|
+
@property
|
|
132
|
+
def capabilities(self) -> list[Capability]:
|
|
133
|
+
return [capability for component in self.components for capability in component.capabilities]
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
@dataclass
|
|
137
|
+
class AcceptanceCriterion:
|
|
138
|
+
description: str
|
|
139
|
+
required: bool = True
|
|
140
|
+
evidence: list[str] = field(default_factory=list)
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
@dataclass
|
|
144
|
+
class TaskSpec:
|
|
145
|
+
task_type: TaskType
|
|
146
|
+
goal: str
|
|
147
|
+
requires_code_change: bool
|
|
148
|
+
requires_tests: bool
|
|
149
|
+
acceptance_criteria: list[AcceptanceCriterion]
|
|
150
|
+
risk: RiskLevel
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
@dataclass
|
|
154
|
+
class EngineeringPlan:
|
|
155
|
+
files_to_inspect: list[str]
|
|
156
|
+
implementation: list[str]
|
|
157
|
+
verification: list[tuple[str, ...]]
|
|
158
|
+
rationale: str
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
@dataclass
|
|
162
|
+
class Understanding:
|
|
163
|
+
problem: str
|
|
164
|
+
expected_behavior: str
|
|
165
|
+
affected_paths: list[str]
|
|
166
|
+
root_cause: str
|
|
167
|
+
evidence: list[Evidence]
|
|
168
|
+
proposed_solution: list[str]
|
|
169
|
+
confidence: float
|
|
170
|
+
|
|
171
|
+
def implementation_ready(self, root: Path) -> bool:
|
|
172
|
+
if self.confidence < 0.6 or not self.root_cause.strip() or not self.proposed_solution:
|
|
173
|
+
return False
|
|
174
|
+
if not self.affected_paths or not self.evidence:
|
|
175
|
+
return False
|
|
176
|
+
return all((root / path).resolve().is_relative_to(root.resolve()) for path in self.affected_paths)
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
@dataclass
|
|
180
|
+
class VerificationResult:
|
|
181
|
+
command: tuple[str, ...]
|
|
182
|
+
exit_code: int | None
|
|
183
|
+
duration_seconds: float
|
|
184
|
+
stdout: str
|
|
185
|
+
stderr: str
|
|
186
|
+
classification: FailureClass | None
|
|
187
|
+
revision: int
|
|
188
|
+
phase: str
|
|
189
|
+
timed_out: bool = False
|
|
190
|
+
baseline: bool = False
|
|
191
|
+
tests_run: int | None = None
|
|
192
|
+
tests_passed: int | None = None
|
|
193
|
+
|
|
194
|
+
@property
|
|
195
|
+
def passed(self) -> bool:
|
|
196
|
+
return self.exit_code == 0 and not self.timed_out
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
@dataclass(frozen=True)
|
|
200
|
+
class ReviewIssue:
|
|
201
|
+
severity: str
|
|
202
|
+
reason: str
|
|
203
|
+
path: str | None = None
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
@dataclass
|
|
207
|
+
class ReviewDecision:
|
|
208
|
+
approved: bool
|
|
209
|
+
issues: list[ReviewIssue]
|
|
210
|
+
summary: str
|
|
211
|
+
|
|
212
|
+
|
|
213
|
+
@dataclass
|
|
214
|
+
class ChangeMetrics:
|
|
215
|
+
files_changed: int = 0
|
|
216
|
+
lines_added: int = 0
|
|
217
|
+
lines_deleted: int = 0
|
|
218
|
+
paths: list[str] = field(default_factory=list)
|
|
219
|
+
|
|
220
|
+
|
|
221
|
+
@dataclass
|
|
222
|
+
class RunResult:
|
|
223
|
+
outcome: Outcome
|
|
224
|
+
task: TaskSpec
|
|
225
|
+
repository: RepositoryModel
|
|
226
|
+
run_id: str
|
|
227
|
+
run_dir: str
|
|
228
|
+
root_cause: str
|
|
229
|
+
implementation: list[str]
|
|
230
|
+
changes: ChangeMetrics
|
|
231
|
+
verification: list[VerificationResult]
|
|
232
|
+
review: ReviewDecision | None
|
|
233
|
+
not_run: list[str]
|
|
234
|
+
recommendations: list[str]
|
|
235
|
+
state_history: list[AgentState]
|
|
236
|
+
working_root: str
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
def jsonable(value: Any) -> Any:
|
|
240
|
+
"""Convert nested domain objects to JSON-safe primitive values."""
|
|
241
|
+
if isinstance(value, Enum):
|
|
242
|
+
return value.value
|
|
243
|
+
if hasattr(value, "__dataclass_fields__"):
|
|
244
|
+
return jsonable(asdict(value))
|
|
245
|
+
if isinstance(value, dict):
|
|
246
|
+
return {str(key): jsonable(item) for key, item in value.items()}
|
|
247
|
+
if isinstance(value, (list, tuple)):
|
|
248
|
+
return [jsonable(item) for item in value]
|
|
249
|
+
if isinstance(value, Path):
|
|
250
|
+
return str(value)
|
|
251
|
+
return value
|