devagent-ai 0.3.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
devagent/evaluation.py ADDED
@@ -0,0 +1,388 @@
1
+ from __future__ import annotations
2
+
3
+ import json
4
+ import subprocess
5
+ import time
6
+ from dataclasses import dataclass
7
+ from pathlib import Path
8
+ from typing import Any, Iterable
9
+
10
+ from devagent.models import Outcome, RunResult, jsonable
11
+ from devagent.orchestrator import DevAgent
12
+ from devagent.providers import ModelProvider
13
+
14
+
15
+ @dataclass(frozen=True)
16
+ class EvaluationMetrics:
17
+ task_success: bool
18
+ acceptance_criteria_supported: int
19
+ acceptance_criteria_total: int
20
+ acceptance_coverage: float
21
+ new_regressions: int | None
22
+ files_changed: int
23
+ lines_changed: int
24
+ iterations: int
25
+ model_calls: int
26
+ tool_calls: int
27
+ runtime_seconds: float
28
+ outcome: Outcome
29
+ final_verification_passed: bool
30
+ review_approved: bool
31
+ source_head_unchanged: bool
32
+ source_status_unchanged: bool
33
+
34
+
35
+ @dataclass(frozen=True)
36
+ class EvaluationExpectation:
37
+ expected_outcomes: tuple[Outcome, ...] = (Outcome.VERIFIED,)
38
+ max_files_changed: int | None = None
39
+ max_lines_changed: int | None = None
40
+ require_acceptance_evidence: bool = True
41
+ require_review_approval: bool = True
42
+ require_final_verification: bool = True
43
+ require_no_new_regressions: bool = True
44
+ require_source_unchanged: bool = True
45
+
46
+ def __post_init__(self) -> None:
47
+ if not self.expected_outcomes:
48
+ raise ValueError("expected_outcomes must contain at least one outcome")
49
+ if self.max_files_changed is not None and self.max_files_changed < 0:
50
+ raise ValueError("max_files_changed must be >= 0")
51
+ if self.max_lines_changed is not None and self.max_lines_changed < 0:
52
+ raise ValueError("max_lines_changed must be >= 0")
53
+
54
+
55
+ @dataclass(frozen=True)
56
+ class EvaluationCaseResult:
57
+ name: str
58
+ category: str
59
+ expected_outcomes: tuple[Outcome, ...]
60
+ metrics: EvaluationMetrics
61
+ passed: bool
62
+ false_verified: bool
63
+ unexpected_blocked: bool
64
+ violations: tuple[str, ...]
65
+
66
+
67
+ @dataclass(frozen=True)
68
+ class EvaluationSuiteMetrics:
69
+ cases_total: int
70
+ cases_passed: int
71
+ pass_rate: float
72
+ verified: int
73
+ partially_verified: int
74
+ blocked: int
75
+ false_verified: int
76
+ false_verified_rate: float
77
+ unexpected_blocked: int
78
+ acceptance_criteria_supported: int
79
+ acceptance_criteria_total: int
80
+ acceptance_coverage: float
81
+ total_model_calls: int
82
+ total_tool_calls: int
83
+ total_runtime_seconds: float
84
+
85
+
86
+ @dataclass(frozen=True)
87
+ class _RepositorySnapshot:
88
+ head: str | None
89
+ status: tuple[str, ...]
90
+
91
+
92
+ class _CountingProvider:
93
+ def __init__(self, inner: ModelProvider) -> None:
94
+ self.inner = inner
95
+ self.calls = 0
96
+
97
+ def request(
98
+ self,
99
+ *,
100
+ role: str,
101
+ payload: dict[str, Any],
102
+ schema: dict[str, Any],
103
+ ) -> dict[str, Any]:
104
+ self.calls += 1
105
+ return self.inner.request(role=role, payload=payload, schema=schema)
106
+
107
+
108
+ def _generated_status_path(path: str) -> bool:
109
+ normalized = path.replace("\\", "/").strip('"')
110
+ parts = tuple(part for part in normalized.split("/") if part)
111
+ return (
112
+ normalized == ".devagent"
113
+ or normalized.startswith(".devagent/")
114
+ or normalized.endswith(".pyc")
115
+ or any(
116
+ part in {"__pycache__", ".pytest_cache", ".mypy_cache", ".ruff_cache"}
117
+ for part in parts
118
+ )
119
+ )
120
+
121
+
122
+ def _repository_snapshot(repository: Path) -> _RepositorySnapshot:
123
+ repository = repository.resolve()
124
+ try:
125
+ head_result = subprocess.run(
126
+ ["git", "rev-parse", "HEAD"],
127
+ cwd=repository,
128
+ capture_output=True,
129
+ text=True,
130
+ timeout=10,
131
+ check=False,
132
+ )
133
+ status_result = subprocess.run(
134
+ ["git", "status", "--porcelain=v1", "--untracked-files=all"],
135
+ cwd=repository,
136
+ capture_output=True,
137
+ text=True,
138
+ timeout=10,
139
+ check=False,
140
+ )
141
+ except (OSError, subprocess.TimeoutExpired):
142
+ return _RepositorySnapshot(None, ())
143
+
144
+ head = head_result.stdout.strip() if head_result.returncode == 0 else None
145
+ status: list[str] = []
146
+ if status_result.returncode == 0:
147
+ for line in status_result.stdout.splitlines():
148
+ rendered = line.rstrip()
149
+ if not rendered:
150
+ continue
151
+ path = rendered[3:] if len(rendered) >= 4 else rendered
152
+ if " -> " in path:
153
+ path = path.split(" -> ", 1)[1]
154
+ if _generated_status_path(path):
155
+ continue
156
+ status.append(rendered)
157
+ return _RepositorySnapshot(head, tuple(sorted(status)))
158
+
159
+
160
+ def _tool_call_count(run_dir: str) -> int:
161
+ observations = Path(run_dir) / "observations.jsonl"
162
+ if not observations.is_file():
163
+ return 0
164
+ count = 0
165
+ for line in observations.read_text(encoding="utf-8").splitlines():
166
+ try:
167
+ event = json.loads(line).get("event")
168
+ except json.JSONDecodeError:
169
+ continue
170
+ if event in {"command_finished", "file_written", "text_replaced"}:
171
+ count += 1
172
+ return count
173
+
174
+
175
+ def _final_verification_passed(result: RunResult) -> bool:
176
+ final = [item for item in result.verification if item.phase == "final"]
177
+ return bool(final) and all(item.passed for item in final)
178
+
179
+
180
+ def _known_new_regressions(result: RunResult) -> int | None:
181
+ """Count only regressions supported by comparable baseline/final evidence."""
182
+ baseline_by_command = {
183
+ item.command: item.passed
184
+ for item in result.verification
185
+ if item.baseline or item.phase == "baseline"
186
+ }
187
+ final = [item for item in result.verification if item.phase == "final"]
188
+ if not final:
189
+ return None
190
+
191
+ regressions = 0
192
+ unknown_failure = False
193
+ for item in final:
194
+ if item.passed:
195
+ continue
196
+ baseline_passed = baseline_by_command.get(item.command)
197
+ if baseline_passed is True:
198
+ regressions += 1
199
+ elif baseline_passed is None:
200
+ unknown_failure = True
201
+
202
+ if unknown_failure and regressions == 0:
203
+ return None
204
+ return regressions
205
+
206
+
207
+ def evaluate(
208
+ repository: Path,
209
+ requirement: str,
210
+ provider: ModelProvider,
211
+ *,
212
+ isolate: bool = True,
213
+ ) -> tuple[RunResult, EvaluationMetrics]:
214
+ """Run one evidence-backed evaluation and capture deterministic production metrics."""
215
+ repository = repository.resolve()
216
+ source_before = _repository_snapshot(repository)
217
+ counted_provider = _CountingProvider(provider)
218
+ started = time.monotonic()
219
+ result = DevAgent(counted_provider, isolate=isolate).run(repository, requirement)
220
+ runtime = time.monotonic() - started
221
+ source_after = _repository_snapshot(repository)
222
+
223
+ supported = sum(bool(criterion.evidence) for criterion in result.task.acceptance_criteria)
224
+ total = len(result.task.acceptance_criteria)
225
+ coverage = supported / total if total else 1.0
226
+ metrics = EvaluationMetrics(
227
+ task_success=result.outcome is Outcome.VERIFIED,
228
+ acceptance_criteria_supported=supported,
229
+ acceptance_criteria_total=total,
230
+ acceptance_coverage=coverage,
231
+ new_regressions=_known_new_regressions(result),
232
+ files_changed=result.changes.files_changed,
233
+ lines_changed=result.changes.lines_added + result.changes.lines_deleted,
234
+ iterations=sum(
235
+ state.value in {"IMPLEMENT", "DIAGNOSE"}
236
+ for state in result.state_history
237
+ ),
238
+ model_calls=counted_provider.calls,
239
+ tool_calls=_tool_call_count(result.run_dir),
240
+ runtime_seconds=runtime,
241
+ outcome=result.outcome,
242
+ final_verification_passed=_final_verification_passed(result),
243
+ review_approved=bool(result.review and result.review.approved),
244
+ source_head_unchanged=source_before.head == source_after.head,
245
+ source_status_unchanged=source_before.status == source_after.status,
246
+ )
247
+ return result, metrics
248
+
249
+
250
+ def score_evaluation(
251
+ name: str,
252
+ category: str,
253
+ metrics: EvaluationMetrics,
254
+ expectation: EvaluationExpectation | None = None,
255
+ ) -> EvaluationCaseResult:
256
+ """Score a run without allowing model prose to determine benchmark success."""
257
+ expected = expectation or EvaluationExpectation()
258
+ violations: list[str] = []
259
+ if metrics.outcome not in expected.expected_outcomes:
260
+ allowed = ",".join(item.value for item in expected.expected_outcomes)
261
+ violations.append(f"unexpected_outcome:{metrics.outcome.value}:expected={allowed}")
262
+
263
+ false_verified = (
264
+ metrics.outcome is Outcome.VERIFIED
265
+ and Outcome.VERIFIED not in expected.expected_outcomes
266
+ )
267
+ unexpected_blocked = (
268
+ metrics.outcome is Outcome.BLOCKED
269
+ and Outcome.BLOCKED not in expected.expected_outcomes
270
+ )
271
+
272
+ if metrics.outcome is Outcome.VERIFIED:
273
+ if (
274
+ expected.require_acceptance_evidence
275
+ and metrics.acceptance_criteria_supported < metrics.acceptance_criteria_total
276
+ ):
277
+ violations.append("verified_without_complete_acceptance_evidence")
278
+ if expected.require_review_approval and not metrics.review_approved:
279
+ violations.append("verified_without_approved_review")
280
+ if expected.require_final_verification and not metrics.final_verification_passed:
281
+ violations.append("verified_without_final_verification")
282
+ if expected.require_no_new_regressions and metrics.new_regressions is None:
283
+ violations.append("verified_with_unknown_regression_status")
284
+
285
+ if (
286
+ expected.require_no_new_regressions
287
+ and metrics.new_regressions not in {None, 0}
288
+ ):
289
+ violations.append(f"new_regressions:{metrics.new_regressions}")
290
+ if expected.require_source_unchanged and not metrics.source_head_unchanged:
291
+ violations.append("source_head_changed")
292
+ if expected.require_source_unchanged and not metrics.source_status_unchanged:
293
+ violations.append("source_status_changed")
294
+ if (
295
+ expected.max_files_changed is not None
296
+ and metrics.files_changed > expected.max_files_changed
297
+ ):
298
+ violations.append(
299
+ f"files_changed:{metrics.files_changed}:max={expected.max_files_changed}"
300
+ )
301
+ if (
302
+ expected.max_lines_changed is not None
303
+ and metrics.lines_changed > expected.max_lines_changed
304
+ ):
305
+ violations.append(
306
+ f"lines_changed:{metrics.lines_changed}:max={expected.max_lines_changed}"
307
+ )
308
+
309
+ return EvaluationCaseResult(
310
+ name=name,
311
+ category=category,
312
+ expected_outcomes=expected.expected_outcomes,
313
+ metrics=metrics,
314
+ passed=not violations,
315
+ false_verified=false_verified,
316
+ unexpected_blocked=unexpected_blocked,
317
+ violations=tuple(violations),
318
+ )
319
+
320
+
321
+ def evaluate_case(
322
+ name: str,
323
+ category: str,
324
+ repository: Path,
325
+ requirement: str,
326
+ provider: ModelProvider,
327
+ *,
328
+ expectation: EvaluationExpectation | None = None,
329
+ isolate: bool = True,
330
+ ) -> tuple[RunResult, EvaluationCaseResult]:
331
+ result, metrics = evaluate(
332
+ repository,
333
+ requirement,
334
+ provider,
335
+ isolate=isolate,
336
+ )
337
+ return result, score_evaluation(name, category, metrics, expectation)
338
+
339
+
340
+ def aggregate_results(results: Iterable[EvaluationCaseResult]) -> EvaluationSuiteMetrics:
341
+ items = list(results)
342
+ total = len(items)
343
+ passed = sum(item.passed for item in items)
344
+ verified = sum(item.metrics.outcome is Outcome.VERIFIED for item in items)
345
+ partially_verified = sum(
346
+ item.metrics.outcome is Outcome.PARTIALLY_VERIFIED for item in items
347
+ )
348
+ blocked = sum(item.metrics.outcome is Outcome.BLOCKED for item in items)
349
+ false_verified = sum(item.false_verified for item in items)
350
+ supported = sum(item.metrics.acceptance_criteria_supported for item in items)
351
+ criteria_total = sum(item.metrics.acceptance_criteria_total for item in items)
352
+ return EvaluationSuiteMetrics(
353
+ cases_total=total,
354
+ cases_passed=passed,
355
+ pass_rate=passed / total if total else 1.0,
356
+ verified=verified,
357
+ partially_verified=partially_verified,
358
+ blocked=blocked,
359
+ false_verified=false_verified,
360
+ false_verified_rate=false_verified / total if total else 0.0,
361
+ unexpected_blocked=sum(item.unexpected_blocked for item in items),
362
+ acceptance_criteria_supported=supported,
363
+ acceptance_criteria_total=criteria_total,
364
+ acceptance_coverage=supported / criteria_total if criteria_total else 1.0,
365
+ total_model_calls=sum(item.metrics.model_calls for item in items),
366
+ total_tool_calls=sum(item.metrics.tool_calls for item in items),
367
+ total_runtime_seconds=sum(item.metrics.runtime_seconds for item in items),
368
+ )
369
+
370
+
371
+ def write_suite_report(
372
+ path: Path,
373
+ results: Iterable[EvaluationCaseResult],
374
+ ) -> EvaluationSuiteMetrics:
375
+ """Write a machine-readable benchmark report suitable for CI artifacts."""
376
+ items = list(results)
377
+ summary = aggregate_results(items)
378
+ payload = {
379
+ "schema_version": 1,
380
+ "summary": jsonable(summary),
381
+ "cases": jsonable(items),
382
+ }
383
+ path.parent.mkdir(parents=True, exist_ok=True)
384
+ path.write_text(
385
+ json.dumps(payload, indent=2, sort_keys=True) + "\n",
386
+ encoding="utf-8",
387
+ )
388
+ return summary
devagent/memory.py ADDED
@@ -0,0 +1,51 @@
1
+ from __future__ import annotations
2
+
3
+ import json
4
+ from pathlib import Path
5
+ from typing import Iterable
6
+
7
+ from devagent.discovery import facts_are_current
8
+ from devagent.models import RepositoryFact, jsonable
9
+
10
+
11
+ class RepositoryMemory:
12
+ """Bounded, evidence-invalidated local repository and strategy memory."""
13
+
14
+ def __init__(self, root: Path | str) -> None:
15
+ self.root = Path(root).resolve()
16
+ self.directory = self.root / ".devagent" / "memory"
17
+ self.directory.mkdir(parents=True, exist_ok=True)
18
+ self.facts_file = self.directory / "repository.json"
19
+ self.strategies_file = self.directory / "strategies.json"
20
+
21
+ def store_facts(self, facts: Iterable[RepositoryFact]) -> None:
22
+ bounded = list(facts)[-250:]
23
+ self.facts_file.write_text(json.dumps(jsonable(bounded), indent=2) + "\n", encoding="utf-8")
24
+
25
+ def load_facts(self) -> list[RepositoryFact]:
26
+ if not self.facts_file.is_file():
27
+ return []
28
+ try:
29
+ values = json.loads(self.facts_file.read_text(encoding="utf-8"))
30
+ facts = [RepositoryFact(item["fact"], item["confidence"], tuple(item["evidence"]), item["fingerprints"], item["learned_at"]) for item in values]
31
+ except (OSError, KeyError, TypeError, json.JSONDecodeError):
32
+ return []
33
+ if not facts_are_current(self.root, facts):
34
+ self.facts_file.unlink(missing_ok=True)
35
+ return []
36
+ return facts
37
+
38
+ def store_strategy(self, statement: str, evidence: list[str]) -> None:
39
+ if not statement.strip() or not evidence:
40
+ return
41
+ values: list[dict[str, object]] = []
42
+ if self.strategies_file.is_file():
43
+ try:
44
+ values = json.loads(self.strategies_file.read_text(encoding="utf-8"))
45
+ except (OSError, json.JSONDecodeError):
46
+ values = []
47
+ entry = {"statement": statement.strip(), "evidence": evidence[:10]}
48
+ if entry not in values:
49
+ values.append(entry)
50
+ self.strategies_file.write_text(json.dumps(values[-100:], indent=2) + "\n", encoding="utf-8")
51
+
devagent/models.py ADDED
@@ -0,0 +1,251 @@
1
+ from __future__ import annotations
2
+
3
+ from dataclasses import asdict, dataclass, field
4
+ from enum import Enum
5
+ from pathlib import Path
6
+ from typing import Any
7
+
8
+
9
+ class TaskType(str, Enum):
10
+ BUG_FIX = "BUG_FIX"
11
+ FEATURE = "FEATURE"
12
+ RUNTIME_ERROR = "RUNTIME_ERROR"
13
+ TEST_FAILURE = "TEST_FAILURE"
14
+ BUILD_FAILURE = "BUILD_FAILURE"
15
+ UNIT_TEST = "UNIT_TEST"
16
+ REFACTOR = "REFACTOR"
17
+ PERFORMANCE = "PERFORMANCE"
18
+ GENERAL_ENGINEERING_TASK = "GENERAL_ENGINEERING_TASK"
19
+
20
+
21
+ class AgentState(str, Enum):
22
+ PREFLIGHT = "PREFLIGHT"
23
+ DISCOVER = "DISCOVER"
24
+ UNDERSTAND = "UNDERSTAND"
25
+ TASK_SPEC = "TASK_SPEC"
26
+ BASELINE = "BASELINE"
27
+ PLAN = "PLAN"
28
+ GATHER_CONTEXT = "GATHER_CONTEXT"
29
+ REPRODUCE = "REPRODUCE"
30
+ IMPLEMENT = "IMPLEMENT"
31
+ VERIFY_TARGETED = "VERIFY_TARGETED"
32
+ DIAGNOSE = "DIAGNOSE"
33
+ VERIFY_BROAD = "VERIFY_BROAD"
34
+ REVIEW = "REVIEW"
35
+ QUALITY_CHECK = "QUALITY_CHECK"
36
+ FINAL_VERIFY = "FINAL_VERIFY"
37
+ LEARN = "LEARN"
38
+ REPORT = "REPORT"
39
+ SUCCESS = "SUCCESS"
40
+ PARTIALLY_VERIFIED = "PARTIALLY_VERIFIED"
41
+ BLOCKED = "BLOCKED"
42
+
43
+
44
+ class Outcome(str, Enum):
45
+ VERIFIED = "VERIFIED"
46
+ PARTIALLY_VERIFIED = "PARTIALLY_VERIFIED"
47
+ BLOCKED = "BLOCKED"
48
+
49
+
50
+ class RiskLevel(str, Enum):
51
+ LOW = "LOW"
52
+ MEDIUM = "MEDIUM"
53
+ HIGH = "HIGH"
54
+
55
+
56
+ class FailureClass(str, Enum):
57
+ ASSERTION_FAILURE = "ASSERTION_FAILURE"
58
+ SYNTAX_ERROR = "SYNTAX_ERROR"
59
+ TYPE_ERROR = "TYPE_ERROR"
60
+ IMPORT_ERROR = "IMPORT_ERROR"
61
+ DEPENDENCY_ERROR = "DEPENDENCY_ERROR"
62
+ BUILD_ERROR = "BUILD_ERROR"
63
+ ENVIRONMENT_ERROR = "ENVIRONMENT_ERROR"
64
+ TIMEOUT = "TIMEOUT"
65
+ FLAKY_TEST = "FLAKY_TEST"
66
+ BASELINE_FAILURE = "BASELINE_FAILURE"
67
+ NEW_REGRESSION = "NEW_REGRESSION"
68
+ UNKNOWN = "UNKNOWN"
69
+
70
+
71
+ class CapabilityProvenance(str, Enum):
72
+ EXPLICIT = "EXPLICIT"
73
+ PROBED = "PROBED"
74
+ INFERRED = "INFERRED"
75
+
76
+
77
+ @dataclass(frozen=True)
78
+ class Evidence:
79
+ statement: str
80
+ paths: tuple[str, ...]
81
+ confidence: float = 1.0
82
+
83
+
84
+ @dataclass(frozen=True)
85
+ class RepositoryFact:
86
+ fact: str
87
+ confidence: float
88
+ evidence: tuple[str, ...]
89
+ fingerprints: dict[str, str]
90
+ learned_at: str
91
+
92
+
93
+ @dataclass(frozen=True)
94
+ class Capability:
95
+ kind: str
96
+ command: tuple[str, ...]
97
+ source: str
98
+ component: str = "."
99
+ broad: bool = False
100
+ provenance: CapabilityProvenance = CapabilityProvenance.EXPLICIT
101
+ provenance_detail: str = "declared by repository configuration"
102
+ tests_collected: int | None = None
103
+
104
+ @property
105
+ def trusted(self) -> bool:
106
+ return self.provenance in {CapabilityProvenance.EXPLICIT, CapabilityProvenance.PROBED}
107
+
108
+
109
+ @dataclass
110
+ class Component:
111
+ path: str
112
+ languages: list[str] = field(default_factory=list)
113
+ frameworks: list[str] = field(default_factory=list)
114
+ manifests: list[str] = field(default_factory=list)
115
+ test_locations: list[str] = field(default_factory=list)
116
+ capabilities: list[Capability] = field(default_factory=list)
117
+
118
+
119
+ @dataclass
120
+ class RepositoryModel:
121
+ root: str
122
+ kind: str
123
+ components: list[Component]
124
+ facts: list[RepositoryFact]
125
+ git_branch: str | None = None
126
+ git_head: str | None = None
127
+ dirty_files: list[str] = field(default_factory=list)
128
+ inventory_file_count: int = 0
129
+ capability_diagnostics: list[str] = field(default_factory=list)
130
+
131
+ @property
132
+ def capabilities(self) -> list[Capability]:
133
+ return [capability for component in self.components for capability in component.capabilities]
134
+
135
+
136
+ @dataclass
137
+ class AcceptanceCriterion:
138
+ description: str
139
+ required: bool = True
140
+ evidence: list[str] = field(default_factory=list)
141
+
142
+
143
+ @dataclass
144
+ class TaskSpec:
145
+ task_type: TaskType
146
+ goal: str
147
+ requires_code_change: bool
148
+ requires_tests: bool
149
+ acceptance_criteria: list[AcceptanceCriterion]
150
+ risk: RiskLevel
151
+
152
+
153
+ @dataclass
154
+ class EngineeringPlan:
155
+ files_to_inspect: list[str]
156
+ implementation: list[str]
157
+ verification: list[tuple[str, ...]]
158
+ rationale: str
159
+
160
+
161
+ @dataclass
162
+ class Understanding:
163
+ problem: str
164
+ expected_behavior: str
165
+ affected_paths: list[str]
166
+ root_cause: str
167
+ evidence: list[Evidence]
168
+ proposed_solution: list[str]
169
+ confidence: float
170
+
171
+ def implementation_ready(self, root: Path) -> bool:
172
+ if self.confidence < 0.6 or not self.root_cause.strip() or not self.proposed_solution:
173
+ return False
174
+ if not self.affected_paths or not self.evidence:
175
+ return False
176
+ return all((root / path).resolve().is_relative_to(root.resolve()) for path in self.affected_paths)
177
+
178
+
179
+ @dataclass
180
+ class VerificationResult:
181
+ command: tuple[str, ...]
182
+ exit_code: int | None
183
+ duration_seconds: float
184
+ stdout: str
185
+ stderr: str
186
+ classification: FailureClass | None
187
+ revision: int
188
+ phase: str
189
+ timed_out: bool = False
190
+ baseline: bool = False
191
+ tests_run: int | None = None
192
+ tests_passed: int | None = None
193
+
194
+ @property
195
+ def passed(self) -> bool:
196
+ return self.exit_code == 0 and not self.timed_out
197
+
198
+
199
+ @dataclass(frozen=True)
200
+ class ReviewIssue:
201
+ severity: str
202
+ reason: str
203
+ path: str | None = None
204
+
205
+
206
+ @dataclass
207
+ class ReviewDecision:
208
+ approved: bool
209
+ issues: list[ReviewIssue]
210
+ summary: str
211
+
212
+
213
+ @dataclass
214
+ class ChangeMetrics:
215
+ files_changed: int = 0
216
+ lines_added: int = 0
217
+ lines_deleted: int = 0
218
+ paths: list[str] = field(default_factory=list)
219
+
220
+
221
+ @dataclass
222
+ class RunResult:
223
+ outcome: Outcome
224
+ task: TaskSpec
225
+ repository: RepositoryModel
226
+ run_id: str
227
+ run_dir: str
228
+ root_cause: str
229
+ implementation: list[str]
230
+ changes: ChangeMetrics
231
+ verification: list[VerificationResult]
232
+ review: ReviewDecision | None
233
+ not_run: list[str]
234
+ recommendations: list[str]
235
+ state_history: list[AgentState]
236
+ working_root: str
237
+
238
+
239
+ def jsonable(value: Any) -> Any:
240
+ """Convert nested domain objects to JSON-safe primitive values."""
241
+ if isinstance(value, Enum):
242
+ return value.value
243
+ if hasattr(value, "__dataclass_fields__"):
244
+ return jsonable(asdict(value))
245
+ if isinstance(value, dict):
246
+ return {str(key): jsonable(item) for key, item in value.items()}
247
+ if isinstance(value, (list, tuple)):
248
+ return [jsonable(item) for item in value]
249
+ if isinstance(value, Path):
250
+ return str(value)
251
+ return value