devagent-ai 0.3.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,887 @@
1
+ from __future__ import annotations
2
+
3
+ import json
4
+ import difflib
5
+ import subprocess
6
+ from pathlib import Path
7
+ from typing import Any, Callable, Sequence
8
+
9
+ from devagent.artifacts import RunArtifacts
10
+ from devagent.discovery import discover_repository
11
+ from devagent.memory import RepositoryMemory
12
+ from devagent.models import (
13
+ AgentState,
14
+ ChangeMetrics,
15
+ EngineeringPlan,
16
+ Evidence,
17
+ Outcome,
18
+ RiskLevel,
19
+ ReviewDecision,
20
+ ReviewIssue,
21
+ RunResult,
22
+ TaskType,
23
+ Understanding,
24
+ VerificationResult,
25
+ jsonable,
26
+ )
27
+ from devagent.providers import ModelProvider, ProviderError
28
+ from devagent.report import render_report
29
+ from devagent.retrieval import retrieve_context
30
+ from devagent.safety import SafetyError
31
+ from devagent.state_machine import InvalidTransition, Lifecycle
32
+ from devagent.tasking import compile_task
33
+ from devagent.workspace import Workspace
34
+ from devagent.worktree import select_worktree
35
+
36
+
37
+ _BUILTIN_GIT_VERIFICATION: dict[tuple[str, ...], str] = {
38
+ ("git", "diff", "--check"): "harness",
39
+ }
40
+
41
+ _NON_EMPTY_STRING = {"type": "string", "minLength": 1, "pattern": r"\S"}
42
+ _STRING_LIST = {"type": "array", "items": _NON_EMPTY_STRING}
43
+ _CONFIDENCE = {"type": "number", "minimum": 0.0, "maximum": 1.0}
44
+
45
+ _REPLACE_ACTION = {
46
+ "type": "object",
47
+ "properties": {
48
+ "tool": {"type": "string", "const": "replace_text"},
49
+ "arguments": {
50
+ "type": "object",
51
+ "properties": {
52
+ "path": _NON_EMPTY_STRING,
53
+ "old": _NON_EMPTY_STRING,
54
+ "new": {"type": "string"},
55
+ },
56
+ "required": ["path", "old", "new"],
57
+ "additionalProperties": False,
58
+ },
59
+ },
60
+ "required": ["tool", "arguments"],
61
+ "additionalProperties": False,
62
+ }
63
+ _COUNTED_REPLACE_ACTION = {
64
+ "type": "object",
65
+ "properties": {
66
+ "tool": {"type": "string", "const": "replace_text"},
67
+ "arguments": {
68
+ "type": "object",
69
+ "properties": {
70
+ "path": _NON_EMPTY_STRING,
71
+ "old": _NON_EMPTY_STRING,
72
+ "new": {"type": "string"},
73
+ "count": {"type": "integer", "minimum": 1},
74
+ },
75
+ "required": ["path", "old", "new", "count"],
76
+ "additionalProperties": False,
77
+ },
78
+ },
79
+ "required": ["tool", "arguments"],
80
+ "additionalProperties": False,
81
+ }
82
+ _WRITE_ACTION = {
83
+ "type": "object",
84
+ "properties": {
85
+ "tool": {"type": "string", "const": "write_file"},
86
+ "arguments": {
87
+ "type": "object",
88
+ "properties": {"path": _NON_EMPTY_STRING, "content": {"type": "string"}},
89
+ "required": ["path", "content"],
90
+ "additionalProperties": False,
91
+ },
92
+ },
93
+ "required": ["tool", "arguments"],
94
+ "additionalProperties": False,
95
+ }
96
+ _ACTION = {"anyOf": [_REPLACE_ACTION, _COUNTED_REPLACE_ACTION, _WRITE_ACTION]}
97
+
98
+ UNDERSTANDING_SCHEMA = {
99
+ "type": "object",
100
+ "properties": {
101
+ "problem": _NON_EMPTY_STRING,
102
+ "expected_behavior": _NON_EMPTY_STRING,
103
+ "affected_paths": {**_STRING_LIST, "minItems": 1},
104
+ "root_cause": _NON_EMPTY_STRING,
105
+ "evidence": {
106
+ "type": "array",
107
+ "minItems": 1,
108
+ "items": {
109
+ "type": "object",
110
+ "properties": {
111
+ "statement": _NON_EMPTY_STRING,
112
+ "paths": {**_STRING_LIST, "minItems": 1},
113
+ "confidence": _CONFIDENCE,
114
+ },
115
+ "required": ["statement", "paths", "confidence"],
116
+ "additionalProperties": False,
117
+ },
118
+ },
119
+ "proposed_solution": {**_STRING_LIST, "minItems": 1},
120
+ "confidence": _CONFIDENCE,
121
+ },
122
+ "required": ["problem", "expected_behavior", "affected_paths", "root_cause", "evidence", "proposed_solution", "confidence"],
123
+ "additionalProperties": False,
124
+ }
125
+ PLAN_SCHEMA = {
126
+ "type": "object",
127
+ "properties": {
128
+ "files_to_inspect": {**_STRING_LIST, "minItems": 1},
129
+ "implementation": {**_STRING_LIST, "minItems": 1},
130
+ "verification": {
131
+ "type": "array",
132
+ "items": {"type": "array", "minItems": 1, "items": _NON_EMPTY_STRING},
133
+ },
134
+ "rationale": _NON_EMPTY_STRING,
135
+ },
136
+ "required": ["files_to_inspect", "implementation", "verification", "rationale"],
137
+ "additionalProperties": False,
138
+ }
139
+ IMPLEMENT_SCHEMA = {
140
+ "type": "object",
141
+ "properties": {
142
+ "actions": {"type": "array", "minItems": 1, "items": _ACTION},
143
+ "summary": {"anyOf": [_NON_EMPTY_STRING, {**_STRING_LIST, "minItems": 1}]},
144
+ },
145
+ "required": ["actions", "summary"],
146
+ "additionalProperties": False,
147
+ }
148
+ DIAGNOSE_SCHEMA = {
149
+ "type": "object",
150
+ "properties": {
151
+ "decision": {"type": "string", "enum": ["fix", "replan", "block"]},
152
+ "updated_hypothesis": _NON_EMPTY_STRING,
153
+ "actions": {"type": "array", "items": _ACTION},
154
+ },
155
+ "required": ["decision", "updated_hypothesis", "actions"],
156
+ "additionalProperties": False,
157
+ }
158
+ REVIEW_SCHEMA = {
159
+ "type": "object",
160
+ "properties": {
161
+ "approved": {"type": "boolean"},
162
+ "issues": {
163
+ "type": "array",
164
+ "items": {
165
+ "type": "object",
166
+ "properties": {
167
+ "severity": {"type": "string", "enum": ["low", "medium", "high", "critical"]},
168
+ "reason": _NON_EMPTY_STRING,
169
+ "path": {"anyOf": [_NON_EMPTY_STRING, {"type": "null"}]},
170
+ },
171
+ "required": ["severity", "reason", "path"],
172
+ "additionalProperties": False,
173
+ },
174
+ },
175
+ "summary": _NON_EMPTY_STRING,
176
+ },
177
+ "required": ["approved", "issues", "summary"],
178
+ "additionalProperties": False,
179
+ }
180
+
181
+
182
+ class OrchestrationError(RuntimeError):
183
+ pass
184
+
185
+
186
+ def _strings(value: Any, field: str, role: str) -> list[str]:
187
+ if not isinstance(value, list) or not all(isinstance(item, str) and item.strip() for item in value):
188
+ raise OrchestrationError(
189
+ f"Invalid {role} response field '{field}': expected array of non-empty strings; received {type(value).__name__}"
190
+ )
191
+ return [item.strip() for item in value]
192
+
193
+
194
+ def _non_empty_string(value: Any, role: str, field: str) -> str:
195
+ if not isinstance(value, str) or not value.strip():
196
+ raise OrchestrationError(f"Invalid {role} response field '{field}': expected non-empty string; received {type(value).__name__}")
197
+ return value.strip()
198
+
199
+
200
+ def _bounded_confidence(value: Any, field: str) -> float:
201
+ if isinstance(value, bool) or not isinstance(value, (int, float)):
202
+ safe_actual = repr(value) if isinstance(value, str) and value.lower() in {"low", "medium", "high"} else type(value).__name__
203
+ raise OrchestrationError(
204
+ f"Invalid understand response field '{field}': expected number from 0.0 through 1.0; received {safe_actual}"
205
+ )
206
+ confidence = float(value)
207
+ if not 0.0 <= confidence <= 1.0:
208
+ raise OrchestrationError(
209
+ f"Invalid understand response field '{field}': expected number from 0.0 through 1.0; received {value!r}"
210
+ )
211
+ return confidence
212
+
213
+
214
+ def _understanding(response: dict[str, Any]) -> Understanding:
215
+ try:
216
+ expected = {"problem", "expected_behavior", "affected_paths", "root_cause", "evidence", "proposed_solution", "confidence"}
217
+ if set(response) != expected:
218
+ raise TypeError(f"expected exactly fields {sorted(expected)!r}")
219
+ raw_evidence = response["evidence"]
220
+ if not isinstance(raw_evidence, list) or not raw_evidence:
221
+ raise TypeError("evidence must be a non-empty array")
222
+ evidence: list[Evidence] = []
223
+ for index, item in enumerate(raw_evidence):
224
+ if not isinstance(item, dict) or set(item) != {"statement", "paths", "confidence"}:
225
+ raise TypeError(f"evidence[{index}] must contain exactly statement, paths, and confidence")
226
+ evidence.append(
227
+ Evidence(
228
+ _non_empty_string(item["statement"], "understand", f"evidence[{index}].statement"),
229
+ tuple(_strings(item["paths"], f"evidence[{index}].paths", "understand")),
230
+ _bounded_confidence(item["confidence"], f"evidence[{index}].confidence"),
231
+ )
232
+ )
233
+ return Understanding(
234
+ problem=_non_empty_string(response["problem"], "understand", "problem"),
235
+ expected_behavior=_non_empty_string(response["expected_behavior"], "understand", "expected_behavior"),
236
+ affected_paths=_strings(response["affected_paths"], "affected_paths", "understand"),
237
+ root_cause=_non_empty_string(response["root_cause"], "understand", "root_cause"),
238
+ evidence=evidence,
239
+ proposed_solution=_strings(response["proposed_solution"], "proposed_solution", "understand"),
240
+ confidence=_bounded_confidence(response["confidence"], "confidence"),
241
+ )
242
+ except OrchestrationError:
243
+ raise
244
+ except (KeyError, TypeError, ValueError) as exc:
245
+ raise OrchestrationError(f"Invalid understanding response: {exc}") from exc
246
+
247
+
248
+ def _plan(response: dict[str, Any], fallback: Sequence[tuple[str, ...]]) -> EngineeringPlan:
249
+ try:
250
+ expected = {"files_to_inspect", "implementation", "verification", "rationale"}
251
+ if set(response) != expected:
252
+ raise TypeError(f"expected exactly fields {sorted(expected)!r}")
253
+ commands: list[tuple[str, ...]] = []
254
+ raw_commands = response["verification"]
255
+ if not isinstance(raw_commands, list):
256
+ raise TypeError("verification must be a list")
257
+ for command in raw_commands:
258
+ if not isinstance(command, list) or not all(isinstance(token, str) and token for token in command):
259
+ raise TypeError("verification commands must be argv arrays")
260
+ commands.append(tuple(command))
261
+ return EngineeringPlan(
262
+ _strings(response["files_to_inspect"], "files_to_inspect", "plan"),
263
+ _strings(response["implementation"], "implementation", "plan"),
264
+ commands or list(fallback),
265
+ _non_empty_string(response["rationale"], "plan", "rationale"),
266
+ )
267
+ except OrchestrationError:
268
+ raise
269
+ except (KeyError, TypeError) as exc:
270
+ raise OrchestrationError(f"Invalid plan response: {exc}") from exc
271
+
272
+
273
+ def _review(response: dict[str, Any]) -> ReviewDecision:
274
+ try:
275
+ if set(response) != {"approved", "issues", "summary"}:
276
+ raise TypeError("expected exactly approved, issues, and summary")
277
+ if not isinstance(response["approved"], bool) or not isinstance(response["issues"], list):
278
+ raise TypeError("approved must be boolean and issues must be an array")
279
+ issues: list[ReviewIssue] = []
280
+ for index, item in enumerate(response["issues"]):
281
+ if not isinstance(item, dict) or set(item) != {"severity", "reason", "path"}:
282
+ raise TypeError(f"issues[{index}] must contain exactly severity, reason, and path")
283
+ severity = _non_empty_string(item["severity"], "review", f"issues[{index}].severity")
284
+ if severity not in {"low", "medium", "high", "critical"}:
285
+ raise TypeError(f"issues[{index}].severity is unsupported")
286
+ path = item["path"]
287
+ if path is not None:
288
+ path = _non_empty_string(path, "review", f"issues[{index}].path")
289
+ issues.append(ReviewIssue(severity, _non_empty_string(item["reason"], "review", f"issues[{index}].reason"), path))
290
+ approved = response["approved"]
291
+ if approved and issues:
292
+ raise OrchestrationError("Reviewer cannot approve while returning issues")
293
+ return ReviewDecision(approved, issues, _non_empty_string(response["summary"], "review", "summary"))
294
+ except OrchestrationError:
295
+ raise
296
+ except (KeyError, TypeError) as exc:
297
+ raise OrchestrationError(f"Invalid review response: {exc}") from exc
298
+
299
+
300
+ def _execute_actions(workspace: Workspace, response: dict[str, Any], allowed_paths: set[str]) -> list[str]:
301
+ if not isinstance(response, dict) or set(response) != {"actions", "summary"}:
302
+ raise OrchestrationError("Invalid implement response: expected exactly actions and summary")
303
+ actions = response.get("actions")
304
+ if not isinstance(actions, list) or not actions:
305
+ raise OrchestrationError("Implementation requires at least one structured action")
306
+ changed: list[str] = []
307
+ for action in actions:
308
+ if not isinstance(action, dict) or set(action) != {"tool", "arguments"} or not isinstance(action["arguments"], dict):
309
+ raise OrchestrationError("Every action must contain only tool and arguments")
310
+ tool, arguments = action["tool"], action["arguments"]
311
+ if tool not in {"replace_text", "write_file"}:
312
+ raise OrchestrationError(f"Implementation tool is not allowed: {tool}")
313
+ path = arguments.get("path")
314
+ if not isinstance(path, str) or path not in allowed_paths:
315
+ raise OrchestrationError(f"Action path was not inspected/planned: {path}")
316
+ if tool == "replace_text":
317
+ required = {"path", "old", "new"}
318
+ if frozenset(arguments) not in {frozenset(required), frozenset({*required, "count"})}:
319
+ raise OrchestrationError("replace_text arguments must contain only path, old, new, and optional count")
320
+ if not all(isinstance(arguments[key], str) for key in required) or not arguments["old"]:
321
+ raise OrchestrationError("replace_text requires string path, non-empty old, and string new")
322
+ count = arguments.get("count", 1)
323
+ if isinstance(count, bool) or not isinstance(count, int) or count < 1:
324
+ raise OrchestrationError("replace_text count must be an integer greater than zero")
325
+ workspace.replace_text(path, arguments["old"], arguments["new"], count)
326
+ else:
327
+ if set(arguments) != {"path", "content"} or not isinstance(arguments.get("content"), str):
328
+ raise OrchestrationError("write_file requires only string path and content")
329
+ workspace.write_file(path, arguments["content"])
330
+ changed.append(path)
331
+ return changed
332
+
333
+
334
+ def _diff(root: Path, workspace: Workspace, max_chars: int = 40_000) -> str:
335
+ completed = subprocess.run(["git", "diff", "--no-ext-diff", "--"], cwd=root, capture_output=True, text=True, timeout=20, check=False)
336
+ text = completed.stdout if completed.returncode == 0 else "(Git diff unavailable)\n"
337
+ tracked = {
338
+ line[3:].split(" -> ")[-1]
339
+ for line in subprocess.run(["git", "status", "--porcelain=v1"], cwd=root, capture_output=True, text=True, timeout=20, check=False).stdout.splitlines()
340
+ if len(line) > 3 and not line.startswith("??")
341
+ }
342
+ for relative in sorted(workspace.modified_paths - tracked):
343
+ target = root / relative
344
+ if not target.is_file():
345
+ continue
346
+ content = target.read_text(encoding="utf-8", errors="replace").splitlines(keepends=True)
347
+ text += "".join(difflib.unified_diff([], content, fromfile="/dev/null", tofile=f"b/{relative}"))
348
+ return text[:max_chars]
349
+
350
+
351
+ def _metrics(root: Path, modified_paths: set[str]) -> ChangeMetrics:
352
+ completed = subprocess.run(["git", "diff", "--numstat", "--"], cwd=root, capture_output=True, text=True, timeout=20, check=False)
353
+ added = deleted = 0
354
+ paths: set[str] = set(modified_paths)
355
+ if completed.returncode == 0:
356
+ for line in completed.stdout.splitlines():
357
+ parts = line.split("\t", 2)
358
+ if len(parts) != 3:
359
+ continue
360
+ if parts[0].isdigit():
361
+ added += int(parts[0])
362
+ if parts[1].isdigit():
363
+ deleted += int(parts[1])
364
+ paths.add(parts[2])
365
+ for relative in modified_paths:
366
+ tracked = subprocess.run(
367
+ ["git", "cat-file", "-e", f"HEAD:{relative}"], cwd=root, capture_output=True, timeout=10, check=False
368
+ ).returncode == 0
369
+ target = root / relative
370
+ if not tracked and target.is_file():
371
+ added += len(target.read_text(encoding="utf-8", errors="replace").splitlines())
372
+ return ChangeMetrics(len(paths), added, deleted, sorted(paths))
373
+
374
+
375
+ def _default_commands(repository: Any) -> tuple[list[tuple[str, ...]], list[tuple[str, ...]]]:
376
+ targeted: list[tuple[str, ...]] = []
377
+ broad: list[tuple[str, ...]] = []
378
+ for capability in repository.capabilities:
379
+ if not capability.trusted:
380
+ continue
381
+ destination = broad if capability.broad or capability.kind in {"build", "lint", "typecheck", "integration"} else targeted
382
+ if capability.command not in destination:
383
+ destination.append(capability.command)
384
+ if not targeted and any("python" in component.languages for component in repository.components):
385
+ targeted.append(("python", "-m", "compileall", "-q", "."))
386
+ return targeted[:3], broad[:5]
387
+
388
+
389
+ def _builtin_verification_commands(repository: Any) -> list[tuple[str, ...]]:
390
+ if not repository.git_head:
391
+ return []
392
+ return list(_BUILTIN_GIT_VERIFICATION)
393
+
394
+
395
+ def _command_kind(command: tuple[str, ...], repository: Any) -> str | None:
396
+ if repository.git_head:
397
+ builtin_kind = _BUILTIN_GIT_VERIFICATION.get(command)
398
+ if builtin_kind is not None:
399
+ return builtin_kind
400
+ for capability in repository.capabilities:
401
+ if not capability.trusted:
402
+ continue
403
+ if command == capability.command:
404
+ return capability.kind
405
+ if capability.command[:3] in {("python", "-m", "pytest"), ("python3", "-m", "pytest")} and command[:3] == capability.command[:3]:
406
+ return capability.kind
407
+ if capability.command and command and capability.command[0] in {"pytest", "cargo", "go", "mvn", "ctest", "dotnet"} and command[0] == capability.command[0]:
408
+ return capability.kind
409
+ if command[:3] in {("python", "-m", "compileall"), ("python3", "-m", "compileall")} and any(
410
+ "python" in component.languages for component in repository.components
411
+ ):
412
+ return "build"
413
+ return None
414
+
415
+
416
+ def _support_acceptance_criteria(task: Any, changes: ChangeMetrics, final_results: list[VerificationResult], review: ReviewDecision) -> None:
417
+ passing_commands = [" ".join(result.command) for result in final_results if result.passed]
418
+ for criterion in task.acceptance_criteria:
419
+ lowered = criterion.description.lower()
420
+ if "review" in lowered and review.approved:
421
+ criterion.evidence.append("Independent reviewer approved the final diff")
422
+ elif "test" in lowered or "coverage" in lowered:
423
+ criterion.evidence.extend(path for path in changes.paths if _is_test_path(path))
424
+ criterion.evidence.extend(passing_commands)
425
+ elif "unrelated" in lowered:
426
+ criterion.evidence.append(f"Minimal-diff gate accepted {changes.files_changed} changed file(s)")
427
+ elif changes.paths:
428
+ criterion.evidence.extend(changes.paths)
429
+
430
+
431
+ def _is_test_path(path: str) -> bool:
432
+ lowered = path.lower()
433
+ parts = Path(lowered).parts
434
+ return (
435
+ any(part in {"test", "tests", "spec", "specs", "__tests__"} for part in parts)
436
+ or Path(lowered).name.startswith("test_")
437
+ or ".test." in lowered
438
+ or ".spec." in lowered
439
+ )
440
+
441
+
442
+ def _run_commands(workspace: Workspace, commands: Sequence[tuple[str, ...]], phase: str, *, baseline: bool = False) -> list[VerificationResult]:
443
+ results: list[VerificationResult] = []
444
+ for command in commands:
445
+ try:
446
+ results.append(workspace.run(command, phase=phase, baseline=baseline))
447
+ except (SafetyError, OSError) as exc:
448
+ raise OrchestrationError(f"Cannot run {' '.join(command)}: {exc}") from exc
449
+ return results
450
+
451
+
452
+ def _context_supports_understanding(
453
+ understanding: Understanding, context: dict[str, object], working_root: Path
454
+ ) -> bool:
455
+ if not understanding.implementation_ready(working_root):
456
+ return False
457
+ snippets = context.get("snippets", {})
458
+ supplied_paths = set(snippets) if isinstance(snippets, dict) else set()
459
+ claimed_paths = {
460
+ path
461
+ for evidence in understanding.evidence
462
+ for path in evidence.paths
463
+ }
464
+ existing_affected_paths = {
465
+ path for path in understanding.affected_paths if (working_root / path).is_file()
466
+ }
467
+ return (
468
+ existing_affected_paths.issubset(supplied_paths)
469
+ and bool(claimed_paths)
470
+ and claimed_paths.issubset(supplied_paths)
471
+ and all((working_root / path).is_file() for path in claimed_paths)
472
+ )
473
+
474
+
475
+ def _baseline_test_count_regressed(
476
+ baseline_results: Sequence[VerificationResult],
477
+ final_results: Sequence[VerificationResult],
478
+ ) -> tuple[VerificationResult, VerificationResult] | None:
479
+ for baseline in baseline_results:
480
+ if baseline.tests_run is None:
481
+ continue
482
+ for final in final_results:
483
+ if final.command == baseline.command and final.tests_run is not None:
484
+ if final.tests_run < baseline.tests_run:
485
+ return baseline, final
486
+ break
487
+ return None
488
+
489
+
490
+ class DevAgent:
491
+ """Deterministic orchestration; model reasoning is bounded inside named states."""
492
+
493
+ def __init__(self, provider: ModelProvider, *, max_corrections: int = 2, isolate: bool = True, verbose: bool = False, status: Callable[[str], None] | None = None) -> None:
494
+ self.provider = provider
495
+ self.max_corrections = max(0, min(max_corrections, 5))
496
+ self.isolate = isolate
497
+ self.verbose = verbose
498
+ self.status = status or (lambda message: None)
499
+
500
+ def _announce(self, artifacts: RunArtifacts, lifecycle: Lifecycle) -> None:
501
+ artifacts.record("state", state=lifecycle.state)
502
+ if self.verbose:
503
+ self.status(f"[{lifecycle.state.value}]")
504
+
505
+ def _diagnostic(self, artifacts: RunArtifacts, category: str, message: str) -> None:
506
+ artifacts.record("diagnostic", category=category, message=message)
507
+ if self.verbose:
508
+ self.status(f"[{category}] {message}")
509
+
510
+ def _show_retrieval(self, artifacts: RunArtifacts, context: dict[str, object]) -> None:
511
+ diagnostics = context.get("diagnostics", {})
512
+ if not isinstance(diagnostics, dict):
513
+ return
514
+ self._diagnostic(
515
+ artifacts,
516
+ "RETRIEVAL",
517
+ (
518
+ f"lexical matches: {diagnostics.get('exact_lexical_matches', 0)} exact, "
519
+ f"{diagnostics.get('normalized_lexical_matches', 0)} normalized"
520
+ ),
521
+ )
522
+ self._diagnostic(
523
+ artifacts,
524
+ "RETRIEVAL",
525
+ f"fallback: {diagnostics.get('fallback', 'none')}",
526
+ )
527
+ selected = diagnostics.get("selected", [])
528
+ if isinstance(selected, list):
529
+ for path in selected:
530
+ self._diagnostic(artifacts, "RETRIEVAL", f"selected: {path}")
531
+
532
+ def run(self, repository_root: Path | str, requirement: str) -> RunResult:
533
+ root = Path(repository_root).expanduser().resolve()
534
+ task = compile_task(requirement)
535
+ artifacts = RunArtifacts(root)
536
+ lifecycle = Lifecycle()
537
+ verification: list[VerificationResult] = []
538
+ not_run: list[str] = []
539
+ implementation: list[str] = []
540
+ recommendations: list[str] = []
541
+ understanding = Understanding("", "", [], "", [], [], 0.0)
542
+ review: ReviewDecision | None = None
543
+ source_repository = discover_repository(root, probe_capabilities=False)
544
+ selection = select_worktree(root, artifacts.run_id, enabled=self.isolate, git_head=source_repository.git_head, dirty_files=source_repository.dirty_files)
545
+ working_root = selection.root
546
+ repository = discover_repository(working_root)
547
+ repository.root = str(root)
548
+ repository.git_branch = source_repository.git_branch
549
+ repository.git_head = source_repository.git_head
550
+ repository.dirty_files = source_repository.dirty_files
551
+ workspace = Workspace(working_root, artifacts, source_repository.dirty_files if not selection.isolated else ())
552
+ artifacts.write_json("metadata.json", {"run_id": artifacts.run_id, "task": task, "repository": repository})
553
+ artifacts.record("worktree", source=str(root), working_root=str(working_root), isolated=selection.isolated, reason=selection.reason)
554
+ self._announce(artifacts, lifecycle)
555
+ self._diagnostic(artifacts, "WORKTREE", f"source: {root}")
556
+ self._diagnostic(artifacts, "WORKTREE", f"isolated: {str(selection.isolated).lower()}")
557
+ self._diagnostic(artifacts, "WORKTREE", f"working_root: {working_root}")
558
+
559
+ try:
560
+ if selection.creation_failed:
561
+ raise OrchestrationError(selection.reason)
562
+ lifecycle.transition(AgentState.DISCOVER)
563
+ self._announce(artifacts, lifecycle)
564
+ self._diagnostic(
565
+ artifacts,
566
+ "DISCOVER",
567
+ f"repository files: {repository.inventory_file_count}",
568
+ )
569
+ self._diagnostic(
570
+ artifacts,
571
+ "DISCOVER",
572
+ f"components: {len(repository.components)}",
573
+ )
574
+ for diagnostic in repository.capability_diagnostics:
575
+ self._diagnostic(artifacts, "VERIFICATION_DISCOVERY", diagnostic)
576
+ memory = RepositoryMemory(root)
577
+ memory.load_facts()
578
+ memory.store_facts(repository.facts)
579
+
580
+ lifecycle.transition(AgentState.UNDERSTAND)
581
+ self._announce(artifacts, lifecycle)
582
+ context = retrieve_context(
583
+ workspace,
584
+ repository,
585
+ task.goal,
586
+ requires_tests=task.requires_tests,
587
+ )
588
+ self._show_retrieval(artifacts, context)
589
+ for context_attempt in range(2):
590
+ response = self.provider.request(role="understand", payload={"task": jsonable(task), "repository": context}, schema=UNDERSTANDING_SCHEMA)
591
+ understanding = _understanding(response)
592
+ if _context_supports_understanding(understanding, context, working_root):
593
+ break
594
+ if context_attempt == 0:
595
+ lifecycle.transition(AgentState.GATHER_CONTEXT)
596
+ self._announce(artifacts, lifecycle)
597
+ context = retrieve_context(
598
+ workspace,
599
+ repository,
600
+ task.goal + " " + " ".join(understanding.affected_paths),
601
+ max_chars=32_000,
602
+ requires_tests=task.requires_tests,
603
+ )
604
+ self._show_retrieval(artifacts, context)
605
+ lifecycle.transition(AgentState.UNDERSTAND)
606
+ self._announce(artifacts, lifecycle)
607
+ else:
608
+ raise OrchestrationError("Evidence gate rejected implementation: root cause or source evidence is insufficient")
609
+
610
+ lifecycle.transition(AgentState.TASK_SPEC)
611
+ self._announce(artifacts, lifecycle)
612
+ targeted, broad = _default_commands(repository)
613
+ lifecycle.transition(AgentState.BASELINE)
614
+ self._announce(artifacts, lifecycle)
615
+ if targeted:
616
+ verification.extend(_run_commands(workspace, targeted[:1], "baseline", baseline=True))
617
+ else:
618
+ not_run.append("Baseline: no evidence-backed local test or compile command was discovered")
619
+
620
+ lifecycle.transition(AgentState.PLAN)
621
+ self._announce(artifacts, lifecycle)
622
+ plan_response = self.provider.request(
623
+ role="plan",
624
+ payload={
625
+ "task": jsonable(task),
626
+ "understanding": jsonable(understanding),
627
+ "capabilities": jsonable(repository.capabilities),
628
+ "builtin_verification": jsonable(
629
+ _builtin_verification_commands(repository)
630
+ ),
631
+ },
632
+ schema=PLAN_SCHEMA,
633
+ )
634
+ plan = _plan(plan_response, targeted)
635
+ unsupported = [command for command in plan.verification if _command_kind(command, repository) is None]
636
+ if unsupported:
637
+ raise OrchestrationError(
638
+ "Plan requested verification not supported by repository evidence "
639
+ f"or DevAgent built-ins: {' '.join(unsupported[0])}"
640
+ )
641
+ if task.requires_tests and not any(_command_kind(command, repository) in {"test", "integration"} for command in plan.verification):
642
+ raise OrchestrationError("A task requiring tests must use an evidence-backed test command")
643
+ allowed_paths = set(plan.files_to_inspect)
644
+ if not set(understanding.affected_paths).issubset(allowed_paths):
645
+ raise OrchestrationError("Plan does not include all evidence-backed affected paths")
646
+ for path in allowed_paths:
647
+ if (working_root / path).exists():
648
+ workspace.read_file(path)
649
+
650
+ lifecycle.transition(AgentState.GATHER_CONTEXT)
651
+ self._announce(artifacts, lifecycle)
652
+ lifecycle.transition(AgentState.REPRODUCE)
653
+ self._announce(artifacts, lifecycle)
654
+ baseline_failed = any(result.baseline and not result.passed for result in verification)
655
+ if task.task_type in {TaskType.BUG_FIX, TaskType.RUNTIME_ERROR, TaskType.TEST_FAILURE} and not baseline_failed:
656
+ not_run.append("Pre-change failure was not independently reproduced; implementation is based on source evidence and final regression verification")
657
+
658
+ lifecycle.transition(AgentState.IMPLEMENT)
659
+ self._announce(artifacts, lifecycle)
660
+ implement_response = self.provider.request(
661
+ role="implement",
662
+ payload={"task": jsonable(task), "understanding": jsonable(understanding), "plan": jsonable(plan), "files": {path: workspace.read_file(path) for path in allowed_paths if (working_root / path).is_file()}},
663
+ schema=IMPLEMENT_SCHEMA,
664
+ )
665
+ _execute_actions(workspace, implement_response, allowed_paths)
666
+ implementation.extend(_strings(implement_response.get("summary", []), "summary", "implement") if isinstance(implement_response.get("summary"), list) else [str(implement_response.get("summary", "Implemented planned change"))])
667
+
668
+ corrections = 0
669
+ failure_signatures: list[tuple[object, ...]] = []
670
+ while True:
671
+ lifecycle.transition(AgentState.VERIFY_TARGETED)
672
+ self._announce(artifacts, lifecycle)
673
+ current_targeted = plan.verification or targeted
674
+ if not current_targeted:
675
+ raise OrchestrationError("No targeted verification command is available")
676
+ target_results = _run_commands(workspace, current_targeted[:3], "targeted")
677
+ verification.extend(target_results)
678
+ if all(result.passed for result in target_results):
679
+ break
680
+ lifecycle.transition(AgentState.DIAGNOSE)
681
+ self._announce(artifacts, lifecycle)
682
+ if corrections >= self.max_corrections:
683
+ raise OrchestrationError("Targeted verification still fails after the bounded correction budget")
684
+ diagnosis = self.provider.request(
685
+ role="diagnose",
686
+ payload={"task": jsonable(task), "failures": jsonable([result for result in target_results if not result.passed]), "failed_hypotheses": recommendations},
687
+ schema=DIAGNOSE_SCHEMA,
688
+ )
689
+ decision = diagnosis.get("decision")
690
+ signature = tuple((result.command, result.exit_code, result.classification, result.stderr[-1000:]) for result in target_results if not result.passed)
691
+ failure_signatures.append(signature)
692
+ if failure_signatures.count(signature) >= 2:
693
+ decision = "replan"
694
+ hypothesis = str(diagnosis.get("updated_hypothesis", "")).strip()
695
+ if hypothesis:
696
+ recommendations.append(f"Diagnosis {corrections + 1}: {hypothesis}")
697
+ if decision == "block":
698
+ raise OrchestrationError(hypothesis or "Diagnosis could not identify a safe correction")
699
+ if decision == "replan":
700
+ lifecycle.transition(AgentState.PLAN)
701
+ self._announce(artifacts, lifecycle)
702
+ replanned = self.provider.request(
703
+ role="replan",
704
+ payload={
705
+ "task": jsonable(task),
706
+ "understanding": jsonable(understanding),
707
+ "failures": jsonable(target_results),
708
+ "failed_hypotheses": recommendations,
709
+ "capabilities": jsonable(repository.capabilities),
710
+ "builtin_verification": jsonable(
711
+ _builtin_verification_commands(repository)
712
+ ),
713
+ },
714
+ schema=PLAN_SCHEMA,
715
+ )
716
+ plan = _plan(replanned, targeted)
717
+ if any(_command_kind(command, repository) is None for command in plan.verification):
718
+ raise OrchestrationError(
719
+ "Replan selected verification unsupported by repository "
720
+ "evidence or DevAgent built-ins"
721
+ )
722
+ allowed_paths.update(plan.files_to_inspect)
723
+ lifecycle.transition(AgentState.GATHER_CONTEXT)
724
+ self._announce(artifacts, lifecycle)
725
+ for path in plan.files_to_inspect:
726
+ if (working_root / path).is_file():
727
+ workspace.read_file(path)
728
+ lifecycle.transition(AgentState.REPRODUCE)
729
+ self._announce(artifacts, lifecycle)
730
+ lifecycle.transition(AgentState.IMPLEMENT)
731
+ self._announce(artifacts, lifecycle)
732
+ correction_response = self.provider.request(
733
+ role="implement_replan",
734
+ payload={"plan": jsonable(plan), "diagnosis": hypothesis, "files": {path: workspace.read_file(path) for path in allowed_paths if (working_root / path).is_file()}},
735
+ schema=IMPLEMENT_SCHEMA,
736
+ )
737
+ else:
738
+ lifecycle.transition(AgentState.IMPLEMENT)
739
+ self._announce(artifacts, lifecycle)
740
+ correction_response = {"actions": diagnosis.get("actions"), "summary": [hypothesis or "Applied focused correction"]}
741
+ _execute_actions(workspace, correction_response, allowed_paths)
742
+ implementation.append(hypothesis or "Applied focused correction")
743
+ corrections += 1
744
+
745
+ lifecycle.transition(AgentState.VERIFY_BROAD)
746
+ self._announce(artifacts, lifecycle)
747
+ broad_results = _run_commands(workspace, broad, "broad") if broad else []
748
+ verification.extend(broad_results)
749
+ if any(not result.passed for result in broad_results):
750
+ not_run.append("One or more broader checks did not pass in the local environment")
751
+
752
+ lifecycle.transition(AgentState.REVIEW)
753
+ self._announce(artifacts, lifecycle)
754
+ review = _review(self.provider.request(
755
+ role="review",
756
+ payload={
757
+ "task": jsonable(task),
758
+ "acceptance_criteria": jsonable(task.acceptance_criteria),
759
+ "conventions": jsonable(repository.facts),
760
+ "diff": _diff(working_root, workspace),
761
+ "verification": jsonable(verification),
762
+ },
763
+ schema=REVIEW_SCHEMA,
764
+ ))
765
+ if not review.approved:
766
+ if corrections >= self.max_corrections:
767
+ raise OrchestrationError("Independent review rejected the patch after the correction budget")
768
+ lifecycle.transition(AgentState.IMPLEMENT)
769
+ self._announce(artifacts, lifecycle)
770
+ revision_response = self.provider.request(
771
+ role="implement_review_fixes",
772
+ payload={"issues": jsonable(review.issues), "plan": jsonable(plan), "diff": _diff(working_root, workspace)},
773
+ schema=IMPLEMENT_SCHEMA,
774
+ )
775
+ _execute_actions(workspace, revision_response, allowed_paths)
776
+ implementation.append(str(revision_response.get("summary", "Addressed review findings")))
777
+ corrections += 1
778
+ lifecycle.transition(AgentState.VERIFY_TARGETED)
779
+ self._announce(artifacts, lifecycle)
780
+ review_fix_results = _run_commands(workspace, plan.verification or targeted, "targeted_after_review")
781
+ verification.extend(review_fix_results)
782
+ if any(not result.passed for result in review_fix_results):
783
+ raise OrchestrationError("Verification failed after review corrections")
784
+ lifecycle.transition(AgentState.VERIFY_BROAD)
785
+ self._announce(artifacts, lifecycle)
786
+ verification.extend(_run_commands(workspace, broad, "broad_after_review") if broad else [])
787
+ lifecycle.transition(AgentState.REVIEW)
788
+ self._announce(artifacts, lifecycle)
789
+ review = _review(self.provider.request(role="review", payload={"task": jsonable(task), "diff": _diff(working_root, workspace), "verification": jsonable(verification)}, schema=REVIEW_SCHEMA))
790
+ if not review.approved:
791
+ raise OrchestrationError("Independent reviewer still rejects the corrected patch")
792
+
793
+ lifecycle.transition(AgentState.QUALITY_CHECK)
794
+ self._announce(artifacts, lifecycle)
795
+ changes = _metrics(working_root, workspace.modified_paths)
796
+ if changes.files_changed > 8 or changes.lines_added + changes.lines_deleted > 500:
797
+ raise OrchestrationError("Minimal-diff gate rejected unexpectedly broad scope")
798
+ if task.requires_tests and task.task_type not in {TaskType.TEST_FAILURE, TaskType.UNIT_TEST} and not any(
799
+ _is_test_path(path) for path in changes.paths
800
+ ):
801
+ raise OrchestrationError("Required regression/feature coverage was not added or updated")
802
+ quality_commands = _builtin_verification_commands(repository)
803
+ quality_results = _run_commands(workspace, quality_commands, "quality")
804
+ verification.extend(quality_results)
805
+ if any(not result.passed for result in quality_results):
806
+ raise OrchestrationError("git diff --check rejected the patch")
807
+
808
+ lifecycle.transition(AgentState.FINAL_VERIFY)
809
+ self._announce(artifacts, lifecycle)
810
+ final_commands = list(dict.fromkeys([*(plan.verification or targeted), *broad, *quality_commands]))
811
+ final_results = _run_commands(workspace, final_commands, "final")
812
+ verification.extend(final_results)
813
+ regression = _baseline_test_count_regressed(
814
+ [result for result in verification if result.baseline],
815
+ final_results,
816
+ )
817
+ if regression is not None:
818
+ baseline_result, final_result = regression
819
+ raise OrchestrationError(
820
+ "Final verification collected fewer tests than baseline "
821
+ f"for {' '.join(final_result.command)}: "
822
+ f"{baseline_result.tests_run} -> {final_result.tests_run}"
823
+ )
824
+ current_results = [result for result in final_results if result.revision == workspace.revision]
825
+ if review is not None:
826
+ _support_acceptance_criteria(task, changes, current_results, review)
827
+ unsupported_criteria = [criterion.description for criterion in task.acceptance_criteria if criterion.required and not criterion.evidence]
828
+ if not current_results or any(not result.passed for result in current_results):
829
+ lifecycle.transition(AgentState.PARTIALLY_VERIFIED)
830
+ outcome = Outcome.PARTIALLY_VERIFIED
831
+ elif broad_results and any(not result.passed for result in broad_results):
832
+ lifecycle.transition(AgentState.PARTIALLY_VERIFIED)
833
+ outcome = Outcome.PARTIALLY_VERIFIED
834
+ elif task.risk is RiskLevel.HIGH and not broad:
835
+ not_run.append("High-risk change has no discovered broad/static verification capability")
836
+ lifecycle.transition(AgentState.PARTIALLY_VERIFIED)
837
+ outcome = Outcome.PARTIALLY_VERIFIED
838
+ elif unsupported_criteria:
839
+ not_run.append("Acceptance criteria without final evidence: " + "; ".join(unsupported_criteria))
840
+ lifecycle.transition(AgentState.PARTIALLY_VERIFIED)
841
+ outcome = Outcome.PARTIALLY_VERIFIED
842
+ else:
843
+ lifecycle.transition(AgentState.LEARN)
844
+ self._announce(artifacts, lifecycle)
845
+ for capability in repository.capabilities:
846
+ memory.store_strategy(f"Use {' '.join(capability.command)} for {capability.kind}", [capability.source])
847
+ lifecycle.transition(AgentState.REPORT)
848
+ self._announce(artifacts, lifecycle)
849
+ lifecycle.transition(AgentState.SUCCESS)
850
+ outcome = Outcome.VERIFIED
851
+ except (OrchestrationError, ProviderError, SafetyError, OSError, ValueError) as exc:
852
+ artifacts.record("blocked", reason=str(exc), state=lifecycle.state)
853
+ not_run.append(str(exc))
854
+ if lifecycle.state not in {AgentState.BLOCKED, AgentState.SUCCESS, AgentState.PARTIALLY_VERIFIED}:
855
+ try:
856
+ lifecycle.transition(AgentState.BLOCKED)
857
+ except InvalidTransition:
858
+ lifecycle.state = AgentState.BLOCKED
859
+ lifecycle.history.append(AgentState.BLOCKED)
860
+ outcome = Outcome.BLOCKED
861
+ changes = _metrics(working_root, workspace.modified_paths)
862
+
863
+ result = RunResult(
864
+ outcome=outcome,
865
+ task=task,
866
+ repository=repository,
867
+ run_id=artifacts.run_id,
868
+ run_dir=str(artifacts.root),
869
+ root_cause=understanding.root_cause,
870
+ implementation=implementation,
871
+ changes=changes,
872
+ verification=verification,
873
+ review=review,
874
+ not_run=not_run,
875
+ recommendations=recommendations,
876
+ state_history=lifecycle.history,
877
+ working_root=str(working_root),
878
+ )
879
+ artifacts.verification.write_text(json.dumps(jsonable(verification), indent=2) + "\n", encoding="utf-8")
880
+ artifacts.report.write_text(json.dumps(jsonable(result), indent=2) + "\n", encoding="utf-8")
881
+ artifacts.record("report", outcome=outcome)
882
+ return result
883
+
884
+
885
+ def run_devagent(repository_root: Path | str, requirement: str, provider: ModelProvider, *, verbose: bool = False, status: Callable[[str], None] | None = None) -> tuple[RunResult, str]:
886
+ result = DevAgent(provider, verbose=verbose, status=status).run(repository_root, requirement)
887
+ return result, render_report(result)