devagent-ai 0.8.2__tar.gz → 0.8.4__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {devagent_ai-0.8.2/devagent_ai.egg-info → devagent_ai-0.8.4}/PKG-INFO +34 -1
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/README.md +33 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent/__init__.py +1 -1
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent/cli.py +91 -4
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent/tasking.py +225 -42
- {devagent_ai-0.8.2 → devagent_ai-0.8.4/devagent_ai.egg-info}/PKG-INFO +34 -1
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent_ai.egg-info/SOURCES.txt +2 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/pyproject.toml +1 -1
- devagent_ai-0.8.4/tests/test_cli_progress.py +91 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_production_v040.py +1 -1
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_requirement_compiler_v082.py +7 -10
- devagent_ai-0.8.4/tests/test_requirement_intelligence_v083.py +170 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/LICENSE +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/NOTICE +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/agent/__init__.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/agent/llm.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/agent/loop.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/agent/memory.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/agent/prompts.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/agent/tools.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent/__main__.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent/artifacts.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent/automations.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent/autonomy.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent/browser.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent/config.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent/discovery.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent/evaluation.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent/memory.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent/models.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent/orchestrator.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent/provider_benchmark.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent/providers.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent/qualification.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent/realworld.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent/report.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent/retrieval.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent/routing.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent/runtime.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent/safety.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent/skills.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent/source_control.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent/state_machine.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent/technical_review.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent/workspace.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent/worktree.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent_ai.egg-info/dependency_links.txt +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent_ai.egg-info/entry_points.txt +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent_ai.egg-info/requires.txt +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent_ai.egg-info/top_level.txt +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/setup.cfg +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_acceptance_contract.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_benchmark_catalog.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_browser_verification.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_capability_discovery.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_cli.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_developer_review_report.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_discovery_memory.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_e2e_fake_provider.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_evaluation_harness.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_evaluation_matrix.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_evaluation_regression_evidence.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_functional_qualification.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_huge_monorepo_v070.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_migration_e2e_v070.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_model_routing.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_multilang_technical_review.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_multistack_devagent_e2e.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_multistack_qualification.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_packaging_metadata.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_plan_verification_normalization.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_preservation_contradiction.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_production_hardening.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_realworld_benchmark.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_retrieval.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_runtime_sandbox.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_safety_workspace.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_source_control_publish.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_structural_devagent_e2e_v070.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_structural_operations.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_structured_provider_contract.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_tasking_state.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_v070_engineering_breadth.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_v080_autonomy.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_v080_provider_benchmark.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_v080_skills_automations.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_workspace_environment.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_worktree.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: devagent-ai
|
|
3
|
-
Version: 0.8.
|
|
3
|
+
Version: 0.8.4
|
|
4
4
|
Summary: Evidence-driven local autonomous software engineering agent
|
|
5
5
|
Author: Tom Ha
|
|
6
6
|
Maintainer: Tom Ha
|
|
@@ -152,6 +152,39 @@ reviewer → independently review final diff
|
|
|
152
152
|
|
|
153
153
|
The deterministic harness remains responsible for safety, tool execution, verification validity, acceptance adjudication, final status, reporting, and source-control publication regardless of which model handles a role.
|
|
154
154
|
|
|
155
|
+
#### Example: use multiple AI models in one DevAgent run
|
|
156
|
+
|
|
157
|
+
You do not have to use one AI model for every reasoning step. If you believe different models are better suited to different engineering roles, configure a default model plus any role-specific overrides. Roles that are not explicitly configured fall back to the default model.
|
|
158
|
+
|
|
159
|
+
```bash
|
|
160
|
+
# Keep credentials in environment variables; DevAgent does not store the keys.
|
|
161
|
+
export OPENAI_API_KEY=...
|
|
162
|
+
export GEMINI_API_KEY=...
|
|
163
|
+
export ANTHROPIC_API_KEY=...
|
|
164
|
+
export XAI_API_KEY=...
|
|
165
|
+
|
|
166
|
+
# Default/fallback model.
|
|
167
|
+
devagent setup --provider openai --model YOUR_OPENAI_MODEL
|
|
168
|
+
|
|
169
|
+
# Optional per-role models.
|
|
170
|
+
devagent setup --role investigator --provider gemini --model YOUR_GEMINI_MODEL
|
|
171
|
+
devagent setup --role planner --provider anthropic --model YOUR_CLAUDE_MODEL
|
|
172
|
+
devagent setup --role implementer --provider openai --model YOUR_OPENAI_MODEL
|
|
173
|
+
devagent setup --role reviewer --provider xai --model YOUR_GROK_MODEL
|
|
174
|
+
|
|
175
|
+
# Inspect routing and optionally probe every configured cloud model.
|
|
176
|
+
devagent models
|
|
177
|
+
devagent doctor --live
|
|
178
|
+
|
|
179
|
+
# Run normally; saved role routing is applied automatically.
|
|
180
|
+
cd my-repo
|
|
181
|
+
devagent "Fix the checkout race condition and add regression coverage."
|
|
182
|
+
```
|
|
183
|
+
|
|
184
|
+
For example, a user may choose a fast or lower-cost model for repository investigation, a different model for planning, a preferred coding model for implementation, and another provider for independent review. This can be useful for cost, latency, provider diversity, or model-strength preferences, but it does not guarantee a better result. DevAgent still requires the same repository evidence, acceptance gates, deterministic verification, and publication rules.
|
|
185
|
+
|
|
186
|
+
When using saved role routing, run the task without run-level `--provider`, `--model`, or `--base-url` overrides. Supplying those flags explicitly selects one provider/model for that run instead of the saved per-role routing.
|
|
187
|
+
|
|
155
188
|
## Run
|
|
156
189
|
|
|
157
190
|
From the application repository:
|
|
@@ -121,6 +121,39 @@ reviewer → independently review final diff
|
|
|
121
121
|
|
|
122
122
|
The deterministic harness remains responsible for safety, tool execution, verification validity, acceptance adjudication, final status, reporting, and source-control publication regardless of which model handles a role.
|
|
123
123
|
|
|
124
|
+
#### Example: use multiple AI models in one DevAgent run
|
|
125
|
+
|
|
126
|
+
You do not have to use one AI model for every reasoning step. If you believe different models are better suited to different engineering roles, configure a default model plus any role-specific overrides. Roles that are not explicitly configured fall back to the default model.
|
|
127
|
+
|
|
128
|
+
```bash
|
|
129
|
+
# Keep credentials in environment variables; DevAgent does not store the keys.
|
|
130
|
+
export OPENAI_API_KEY=...
|
|
131
|
+
export GEMINI_API_KEY=...
|
|
132
|
+
export ANTHROPIC_API_KEY=...
|
|
133
|
+
export XAI_API_KEY=...
|
|
134
|
+
|
|
135
|
+
# Default/fallback model.
|
|
136
|
+
devagent setup --provider openai --model YOUR_OPENAI_MODEL
|
|
137
|
+
|
|
138
|
+
# Optional per-role models.
|
|
139
|
+
devagent setup --role investigator --provider gemini --model YOUR_GEMINI_MODEL
|
|
140
|
+
devagent setup --role planner --provider anthropic --model YOUR_CLAUDE_MODEL
|
|
141
|
+
devagent setup --role implementer --provider openai --model YOUR_OPENAI_MODEL
|
|
142
|
+
devagent setup --role reviewer --provider xai --model YOUR_GROK_MODEL
|
|
143
|
+
|
|
144
|
+
# Inspect routing and optionally probe every configured cloud model.
|
|
145
|
+
devagent models
|
|
146
|
+
devagent doctor --live
|
|
147
|
+
|
|
148
|
+
# Run normally; saved role routing is applied automatically.
|
|
149
|
+
cd my-repo
|
|
150
|
+
devagent "Fix the checkout race condition and add regression coverage."
|
|
151
|
+
```
|
|
152
|
+
|
|
153
|
+
For example, a user may choose a fast or lower-cost model for repository investigation, a different model for planning, a preferred coding model for implementation, and another provider for independent review. This can be useful for cost, latency, provider diversity, or model-strength preferences, but it does not guarantee a better result. DevAgent still requires the same repository evidence, acceptance gates, deterministic verification, and publication rules.
|
|
154
|
+
|
|
155
|
+
When using saved role routing, run the task without run-level `--provider`, `--model`, or `--base-url` overrides. Supplying those flags explicitly selects one provider/model for that run instead of the saved per-role routing.
|
|
156
|
+
|
|
124
157
|
## Run
|
|
125
158
|
|
|
126
159
|
From the application repository:
|
|
@@ -4,10 +4,11 @@ import argparse
|
|
|
4
4
|
import importlib.util
|
|
5
5
|
import json
|
|
6
6
|
import os
|
|
7
|
+
import re
|
|
7
8
|
import shutil
|
|
8
9
|
import sys
|
|
9
10
|
from pathlib import Path
|
|
10
|
-
from typing import Sequence
|
|
11
|
+
from typing import Callable, Sequence
|
|
11
12
|
|
|
12
13
|
from devagent import __version__
|
|
13
14
|
from devagent.config import (
|
|
@@ -37,6 +38,84 @@ _LIVE_PROBE_SCHEMA = {
|
|
|
37
38
|
"additionalProperties": False,
|
|
38
39
|
}
|
|
39
40
|
|
|
41
|
+
_PROGRESS_STAGES: dict[str, tuple[int, str]] = {
|
|
42
|
+
"DISCOVER": (1, "DISCOVER / UNDERSTAND"),
|
|
43
|
+
"UNDERSTAND": (1, "DISCOVER / UNDERSTAND"),
|
|
44
|
+
"TASK_SPEC": (2, "REQUIREMENTS / PLAN"),
|
|
45
|
+
"BASELINE": (2, "REQUIREMENTS / PLAN"),
|
|
46
|
+
"PLAN": (2, "REQUIREMENTS / PLAN"),
|
|
47
|
+
"GATHER_CONTEXT": (2, "REQUIREMENTS / PLAN"),
|
|
48
|
+
"REPRODUCE": (2, "REQUIREMENTS / PLAN"),
|
|
49
|
+
"IMPLEMENT": (3, "IMPLEMENT"),
|
|
50
|
+
"VERIFY_TARGETED": (4, "VERIFY / REPAIR IF NEEDED"),
|
|
51
|
+
"VERIFY_BROAD": (4, "VERIFY / REPAIR IF NEEDED"),
|
|
52
|
+
"REVIEW": (5, "INDEPENDENT REVIEW"),
|
|
53
|
+
"QUALITY_CHECK": (6, "FINAL VERIFICATION"),
|
|
54
|
+
"FINAL_VERIFY": (6, "FINAL VERIFICATION"),
|
|
55
|
+
}
|
|
56
|
+
_STATE_LINE = re.compile(r"^\[([A-Z_]+)\]$")
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
class _ProgressStatus:
|
|
60
|
+
"""Render stable user-facing milestones while keeping full diagnostics opt-in."""
|
|
61
|
+
|
|
62
|
+
def __init__(
|
|
63
|
+
self,
|
|
64
|
+
sink: Callable[[str], None] = print,
|
|
65
|
+
*,
|
|
66
|
+
verbose: bool = False,
|
|
67
|
+
) -> None:
|
|
68
|
+
self.sink = sink
|
|
69
|
+
self.verbose = verbose
|
|
70
|
+
self._last_stage = 0
|
|
71
|
+
self._plan_seen = False
|
|
72
|
+
self._implement_seen = False
|
|
73
|
+
self._review_seen = False
|
|
74
|
+
|
|
75
|
+
def __call__(self, message: str) -> None:
|
|
76
|
+
if self.verbose:
|
|
77
|
+
self.sink(message)
|
|
78
|
+
return
|
|
79
|
+
|
|
80
|
+
match = _STATE_LINE.fullmatch(message)
|
|
81
|
+
if match is None:
|
|
82
|
+
return
|
|
83
|
+
state = match.group(1)
|
|
84
|
+
|
|
85
|
+
if state == "DIAGNOSE":
|
|
86
|
+
self.sink(" ↳ DIAGNOSE")
|
|
87
|
+
return
|
|
88
|
+
if state == "PLAN":
|
|
89
|
+
if self._plan_seen:
|
|
90
|
+
self.sink(" ↳ REPLAN")
|
|
91
|
+
return
|
|
92
|
+
self._plan_seen = True
|
|
93
|
+
if state == "IMPLEMENT":
|
|
94
|
+
if self._implement_seen:
|
|
95
|
+
label = "APPLY REVIEW FIXES" if self._review_seen else "APPLY CORRECTION"
|
|
96
|
+
self.sink(f" ↳ {label}")
|
|
97
|
+
return
|
|
98
|
+
self._implement_seen = True
|
|
99
|
+
if state == "REVIEW":
|
|
100
|
+
self._review_seen = True
|
|
101
|
+
|
|
102
|
+
stage = _PROGRESS_STAGES.get(state)
|
|
103
|
+
if stage is None:
|
|
104
|
+
return
|
|
105
|
+
number, label = stage
|
|
106
|
+
if number <= self._last_stage:
|
|
107
|
+
return
|
|
108
|
+
self._last_stage = number
|
|
109
|
+
self.sink(f"[{number}/7] {label}")
|
|
110
|
+
|
|
111
|
+
def report(self) -> None:
|
|
112
|
+
if self.verbose:
|
|
113
|
+
self.sink("[ENGINEERING_REPORT]")
|
|
114
|
+
return
|
|
115
|
+
if self._last_stage < 7:
|
|
116
|
+
self._last_stage = 7
|
|
117
|
+
self.sink("[7/7] ENGINEERING REPORT")
|
|
118
|
+
|
|
40
119
|
|
|
41
120
|
def _top_parser() -> argparse.ArgumentParser:
|
|
42
121
|
parser = argparse.ArgumentParser(prog="devagent", description="Evidence-driven local software engineering agent")
|
|
@@ -53,7 +132,11 @@ def _top_parser() -> argparse.ArgumentParser:
|
|
|
53
132
|
parser.add_argument("--provider", choices=_PROVIDER_CHOICES)
|
|
54
133
|
parser.add_argument("--model")
|
|
55
134
|
parser.add_argument("--base-url")
|
|
56
|
-
parser.add_argument(
|
|
135
|
+
parser.add_argument(
|
|
136
|
+
"--verbose",
|
|
137
|
+
action="store_true",
|
|
138
|
+
help="Show internal state transitions and diagnostics instead of concise progress stages",
|
|
139
|
+
)
|
|
57
140
|
parser.add_argument("--no-isolation", action="store_true", help="Work in place instead of creating a local detached worktree")
|
|
58
141
|
parser.add_argument(
|
|
59
142
|
"--publish",
|
|
@@ -337,11 +420,14 @@ def _run(argv: Sequence[str]) -> int:
|
|
|
337
420
|
from devagent.orchestrator import DevAgent
|
|
338
421
|
from devagent.report import recommendations_for, render_report
|
|
339
422
|
|
|
423
|
+
progress = _ProgressStatus(print, verbose=args.verbose)
|
|
340
424
|
result = DevAgent(
|
|
341
425
|
model_provider,
|
|
342
426
|
isolate=not args.no_isolation,
|
|
343
|
-
|
|
344
|
-
|
|
427
|
+
# Internal status events are always emitted. The progress reporter keeps normal
|
|
428
|
+
# CLI output concise and passes the full state/diagnostic stream only in --verbose.
|
|
429
|
+
verbose=True,
|
|
430
|
+
status=progress,
|
|
345
431
|
base_commit=publication_plan.base_commit if publication_plan else None,
|
|
346
432
|
).run(args.repo, requirement)
|
|
347
433
|
|
|
@@ -361,6 +447,7 @@ def _run(argv: Sequence[str]) -> int:
|
|
|
361
447
|
|
|
362
448
|
# The full engineering report is intentionally emitted before any Git commit/push.
|
|
363
449
|
result.recommendations = recommendations_for(result)
|
|
450
|
+
progress.report()
|
|
364
451
|
print(render_report(result))
|
|
365
452
|
|
|
366
453
|
if publish_requested:
|
|
@@ -17,11 +17,11 @@ _CLASSIFIERS: tuple[tuple[TaskType, tuple[str, ...]], ...] = (
|
|
|
17
17
|
(TaskType.TEST_FAILURE, ("test fail", "failing test", "pytest error")),
|
|
18
18
|
(TaskType.RUNTIME_ERROR, ("traceback", "exception", "runtime error", "crash")),
|
|
19
19
|
(TaskType.MIGRATION, ("migration", "migrate ", "schema change", "alembic", "database migration")),
|
|
20
|
-
(TaskType.PERFORMANCE, ("performance", "optimize", "slow", "latency", "n+1")),
|
|
21
|
-
(TaskType.REFACTOR, ("refactor", "restructure", "cleanup")),
|
|
20
|
+
(TaskType.PERFORMANCE, ("performance", "optimize", "slow", "latency", "n+1", "faster", "speed up")),
|
|
21
|
+
(TaskType.REFACTOR, ("refactor", "restructure", "cleanup", "rename ", "move ", "delete obsolete")),
|
|
22
22
|
(TaskType.UNIT_TEST, ("add unit test", "write tests", "test coverage")),
|
|
23
23
|
(TaskType.BUG_FIX, ("fix", "bug", "incorrect", "broken", "regression failure", "regression bug")),
|
|
24
|
-
(TaskType.FEATURE, ("add ", "implement", "support ", "feature")),
|
|
24
|
+
(TaskType.FEATURE, ("add ", "implement", "support ", "feature", "create ")),
|
|
25
25
|
)
|
|
26
26
|
|
|
27
27
|
_HIGH_RISK = {
|
|
@@ -56,24 +56,39 @@ _KNOWN_SECTIONS = _REQUIREMENT_SECTIONS | {
|
|
|
56
56
|
"non-goals",
|
|
57
57
|
"non goals",
|
|
58
58
|
"notes",
|
|
59
|
+
"engineering design",
|
|
60
|
+
"engineering context",
|
|
59
61
|
}
|
|
60
62
|
_DIRECTIVE = re.compile(
|
|
61
|
-
r"^(?:add|implement|support|preserve|keep|ensure|require|must|should|when|do not|don't|"
|
|
62
|
-
r"verify|run|return|raise|allow|prevent|maintain|migrate|refactor|update|fix|handle)\b",
|
|
63
|
+
r"^(?:add|create|implement|support|preserve|keep|ensure|require|must|should|when|do not|don't|"
|
|
64
|
+
r"verify|run|return|raise|allow|prevent|maintain|migrate|refactor|rename|move|delete|update|fix|handle)\b",
|
|
63
65
|
re.IGNORECASE,
|
|
64
66
|
)
|
|
65
67
|
|
|
66
|
-
# Bounded normalization for terse user intent. This
|
|
67
|
-
#
|
|
68
|
-
# grammatical number, and operation wording while preserving identifiers,
|
|
69
|
-
# quoted contracts, values, and explicit constraints. Task policy and repository
|
|
70
|
-
# evidence still provide the verification/safety contract.
|
|
68
|
+
# Bounded normalization for terse user intent. This intentionally fixes common
|
|
69
|
+
# engineering shorthand and spelling without attempting to invent product behavior.
|
|
71
70
|
_OPERATION_ALIASES: tuple[tuple[str, str], ...] = (
|
|
72
71
|
("substraction", "subtraction"),
|
|
73
72
|
("substract", "subtract"),
|
|
74
73
|
("multipy", "multiply"),
|
|
75
74
|
("mutiply", "multiply"),
|
|
75
|
+
("authentification", "authentication"),
|
|
76
|
+
("autorization", "authorization"),
|
|
77
|
+
("loging", "login"),
|
|
76
78
|
)
|
|
79
|
+
_ACRONYMS = {
|
|
80
|
+
"api": "API",
|
|
81
|
+
"csv": "CSV",
|
|
82
|
+
"db": "DB",
|
|
83
|
+
"http": "HTTP",
|
|
84
|
+
"https": "HTTPS",
|
|
85
|
+
"json": "JSON",
|
|
86
|
+
"jwt": "JWT",
|
|
87
|
+
"oauth": "OAuth",
|
|
88
|
+
"sql": "SQL",
|
|
89
|
+
"ui": "UI",
|
|
90
|
+
"url": "URL",
|
|
91
|
+
}
|
|
77
92
|
|
|
78
93
|
|
|
79
94
|
def _classify(text: str) -> TaskType:
|
|
@@ -110,27 +125,66 @@ def _dedupe(items: list[str]) -> list[str]:
|
|
|
110
125
|
return result
|
|
111
126
|
|
|
112
127
|
|
|
113
|
-
def
|
|
114
|
-
|
|
128
|
+
def _section_header(line: str) -> tuple[str, str] | None:
|
|
129
|
+
stripped = line.strip()
|
|
130
|
+
markdown = re.match(r"^#{1,6}\s+(.+?)\s*$", stripped)
|
|
131
|
+
if markdown:
|
|
132
|
+
return markdown.group(1).strip().rstrip(":").lower(), ""
|
|
133
|
+
colon = re.match(r"^([A-Za-z][A-Za-z0-9 _/-]{0,80})\s*:\s*(.*)$", stripped)
|
|
134
|
+
if colon and colon.group(1).strip().lower() in _KNOWN_SECTIONS:
|
|
135
|
+
return colon.group(1).strip().lower(), colon.group(2).strip()
|
|
136
|
+
return None
|
|
115
137
|
|
|
116
|
-
The compiler is intentionally bounded. It may repair shorthand/grammar and
|
|
117
|
-
make an operation explicit, but it must not add product behavior the user did
|
|
118
|
-
not request. Structured/multi-line requirements are left intact.
|
|
119
|
-
"""
|
|
120
138
|
|
|
121
|
-
|
|
122
|
-
|
|
139
|
+
def _extract_goal(requirement: str) -> str:
|
|
140
|
+
"""Prefer an explicit Goal section while preserving ordinary free-form input."""
|
|
141
|
+
|
|
142
|
+
lines = requirement.splitlines()
|
|
143
|
+
for index, raw in enumerate(lines):
|
|
144
|
+
header = _section_header(raw)
|
|
145
|
+
if header is None or header[0] != "goal":
|
|
146
|
+
continue
|
|
147
|
+
_, inline = header
|
|
148
|
+
if inline:
|
|
149
|
+
return _clean_requirement_item(inline)
|
|
150
|
+
collected: list[str] = []
|
|
151
|
+
for candidate in lines[index + 1 :]:
|
|
152
|
+
if _section_header(candidate) is not None:
|
|
153
|
+
break
|
|
154
|
+
if candidate.strip():
|
|
155
|
+
collected.append(_clean_requirement_item(candidate))
|
|
156
|
+
if collected:
|
|
157
|
+
return " ".join(collected)
|
|
158
|
+
return re.sub(r"\s+", " ", requirement).strip()
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
def _polish_plain_goal(value: str) -> str:
|
|
162
|
+
"""Improve readability without changing the requested product semantics."""
|
|
163
|
+
|
|
164
|
+
result = re.sub(r"\s+", " ", value).strip()
|
|
165
|
+
for source, destination in _OPERATION_ALIASES:
|
|
166
|
+
result = re.sub(rf"\b{re.escape(source)}\b", destination, result, flags=re.IGNORECASE)
|
|
167
|
+
result = re.sub(r"^customer\s+(?:need|needs|want|wants)\s+", "Implement ", result, flags=re.IGNORECASE)
|
|
168
|
+
result = re.sub(r"^user\s+(?:need|needs|want|wants)\s+", "Implement ", result, flags=re.IGNORECASE)
|
|
169
|
+
for source, destination in _ACRONYMS.items():
|
|
170
|
+
result = re.sub(rf"\b{source}\b", destination, result, flags=re.IGNORECASE)
|
|
171
|
+
if result:
|
|
172
|
+
result = result[0].upper() + result[1:]
|
|
173
|
+
return result.rstrip(".;")
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
def _normalize_terse_requirement(requirement: str) -> str:
|
|
177
|
+
"""Compile common rough prompts into a clearer bounded engineering request."""
|
|
178
|
+
|
|
179
|
+
raw_value = _extract_goal(requirement)
|
|
180
|
+
value = _polish_plain_goal(raw_value)
|
|
181
|
+
if not value:
|
|
123
182
|
return value
|
|
124
183
|
# An explicit callable name is already a precise user contract; never rename it.
|
|
125
184
|
if re.search(r"\b[A-Za-z_][A-Za-z0-9_]*\s*\(", value):
|
|
126
185
|
return value
|
|
127
186
|
|
|
128
|
-
for source, destination in _OPERATION_ALIASES:
|
|
129
|
-
value = re.sub(rf"\b{re.escape(source)}\b", destination, value, flags=re.IGNORECASE)
|
|
130
|
-
|
|
131
187
|
# Common shorthand from natural prompts such as "addition 2 matrix 2x2".
|
|
132
|
-
# Keep both "matrix" and "matrices" in the normalized contract so
|
|
133
|
-
# deterministic evidence can link either conventional symbol spelling.
|
|
134
188
|
matrix_match = re.search(
|
|
135
189
|
r"\b(add(?:ition)?|sum|subtract(?:ion)?|multiply|multiplication|divide|division)\b"
|
|
136
190
|
r"(?:\s+(?:of|for))?\s+(?:2|two)\s+matrix(?:es)?\s+(\d+x\d+)\b",
|
|
@@ -157,28 +211,19 @@ def _normalize_terse_requirement(requirement: str) -> str:
|
|
|
157
211
|
f"(matrix inputs)"
|
|
158
212
|
)
|
|
159
213
|
|
|
160
|
-
# Repair simple count+noun shorthand without inventing domain behavior.
|
|
161
214
|
value = re.sub(r"\b2\s+matrix\b", "two matrices", value, flags=re.IGNORECASE)
|
|
162
215
|
value = re.sub(r"\b2\s+file\b", "two files", value, flags=re.IGNORECASE)
|
|
163
216
|
value = re.sub(r"\b2\s+test\b", "two tests", value, flags=re.IGNORECASE)
|
|
164
217
|
return value
|
|
165
218
|
|
|
166
219
|
|
|
167
|
-
def _section_header(line: str) -> tuple[str, str] | None:
|
|
168
|
-
stripped = line.strip()
|
|
169
|
-
markdown = re.match(r"^#{1,6}\s+(.+?)\s*$", stripped)
|
|
170
|
-
if markdown:
|
|
171
|
-
return markdown.group(1).strip().rstrip(":").lower(), ""
|
|
172
|
-
colon = re.match(r"^([A-Za-z][A-Za-z0-9 _/-]{0,80})\s*:\s*(.*)$", stripped)
|
|
173
|
-
if colon and colon.group(1).strip().lower() in _KNOWN_SECTIONS:
|
|
174
|
-
return colon.group(1).strip().lower(), colon.group(2).strip()
|
|
175
|
-
return None
|
|
176
|
-
|
|
177
|
-
|
|
178
220
|
def _user_acceptance_items(requirement: str) -> list[str]:
|
|
179
221
|
lines = requirement.splitlines()
|
|
180
222
|
explicit: list[str] = []
|
|
181
223
|
active_section: str | None = None
|
|
224
|
+
recognized_section = False
|
|
225
|
+
nonempty_lines = [line.strip() for line in lines if line.strip()]
|
|
226
|
+
|
|
182
227
|
for raw in lines:
|
|
183
228
|
stripped = raw.strip()
|
|
184
229
|
if not stripped:
|
|
@@ -192,6 +237,7 @@ def _user_acceptance_items(requirement: str) -> list[str]:
|
|
|
192
237
|
|
|
193
238
|
header = _section_header(stripped)
|
|
194
239
|
if header is not None:
|
|
240
|
+
recognized_section = True
|
|
195
241
|
name, inline = header
|
|
196
242
|
active_section = name if name in _REQUIREMENT_SECTIONS else None
|
|
197
243
|
if active_section is not None and inline:
|
|
@@ -221,6 +267,19 @@ def _user_acceptance_items(requirement: str) -> list[str]:
|
|
|
221
267
|
if _DIRECTIVE.match(item) or re.search(r"\b(?:must|should|shall)\b", item, re.IGNORECASE):
|
|
222
268
|
directives.append(item)
|
|
223
269
|
directives = _dedupe(directives)
|
|
270
|
+
|
|
271
|
+
# For a loose multi-line customer note, do not silently discard fragments merely
|
|
272
|
+
# because one line happens to begin with a directive. Preserve the whole intent as
|
|
273
|
+
# one user criterion unless the text is clearly a structured directive list.
|
|
274
|
+
if len(nonempty_lines) > 1 and not recognized_section:
|
|
275
|
+
all_directive_like = all(
|
|
276
|
+
_DIRECTIVE.match(_clean_requirement_item(item))
|
|
277
|
+
or re.search(r"\b(?:must|should|shall)\b", item, re.IGNORECASE)
|
|
278
|
+
for item in nonempty_lines
|
|
279
|
+
)
|
|
280
|
+
if not all_directive_like:
|
|
281
|
+
return [re.sub(r"\s+", " ", requirement).strip()]
|
|
282
|
+
|
|
224
283
|
if directives:
|
|
225
284
|
return directives
|
|
226
285
|
return [re.sub(r"\s+", " ", requirement).strip()]
|
|
@@ -257,8 +316,6 @@ def compile_task(requirement: str) -> TaskSpec:
|
|
|
257
316
|
requires_tests = task_type is not TaskType.BUILD_FAILURE
|
|
258
317
|
|
|
259
318
|
criteria: list[AcceptanceCriterion] = []
|
|
260
|
-
# Structured user requirements remain authoritative. Only an unstructured,
|
|
261
|
-
# terse prompt is compiled into the clearer canonical request.
|
|
262
319
|
user_items = _user_acceptance_items(requirement)
|
|
263
320
|
if len(user_items) == 1 and user_items[0] == raw_goal and goal != raw_goal:
|
|
264
321
|
user_items = [goal]
|
|
@@ -342,14 +399,13 @@ def _matrix_operation_contract(task: TaskSpec, repository: Any) -> None:
|
|
|
342
399
|
"multiplication": "multiply",
|
|
343
400
|
"division": "divide",
|
|
344
401
|
}[operation]
|
|
345
|
-
compact_dimension = dimension.replace("x", "x")
|
|
346
402
|
language = _repository_language(repository)
|
|
347
403
|
if language in {"java", "javascript", "typescript"}:
|
|
348
|
-
symbol = f"{verb}Matrices{
|
|
404
|
+
symbol = f"{verb}Matrices{dimension}"
|
|
349
405
|
elif language in {"csharp", "c#"}:
|
|
350
|
-
symbol = f"{verb.capitalize()}Matrices{
|
|
406
|
+
symbol = f"{verb.capitalize()}Matrices{dimension}"
|
|
351
407
|
else:
|
|
352
|
-
symbol = f"{verb}_matrices_{
|
|
408
|
+
symbol = f"{verb}_matrices_{dimension}"
|
|
353
409
|
|
|
354
410
|
compiled = (
|
|
355
411
|
f"Add {symbol}(a, b) to perform element-wise matrix {operation} "
|
|
@@ -363,8 +419,133 @@ def _matrix_operation_contract(task: TaskSpec, repository: Any) -> None:
|
|
|
363
419
|
user_criteria[0].description = compiled
|
|
364
420
|
|
|
365
421
|
|
|
422
|
+
def _unique_repository_values(repository: Any, field: str) -> list[str]:
|
|
423
|
+
values: list[str] = []
|
|
424
|
+
for component in repository.components:
|
|
425
|
+
for value in getattr(component, field, []):
|
|
426
|
+
if value and value not in values:
|
|
427
|
+
values.append(value)
|
|
428
|
+
return values
|
|
429
|
+
|
|
430
|
+
|
|
431
|
+
def _task_design_defaults(task: TaskSpec) -> list[str]:
|
|
432
|
+
common = [
|
|
433
|
+
"Integrate with the repository's existing architecture and naming conventions instead of creating a parallel pattern.",
|
|
434
|
+
"Keep the implementation bounded to the requested behavior and avoid unrelated refactors.",
|
|
435
|
+
"Preserve behavior outside the explicitly requested scope unless the user states otherwise.",
|
|
436
|
+
]
|
|
437
|
+
if task.requires_tests:
|
|
438
|
+
common.append("Add or update focused regression coverage using the repository's existing test conventions.")
|
|
439
|
+
|
|
440
|
+
if task.task_type in {TaskType.BUG_FIX, TaskType.RUNTIME_ERROR, TaskType.TEST_FAILURE}:
|
|
441
|
+
common.extend(
|
|
442
|
+
[
|
|
443
|
+
"Identify and fix the underlying cause rather than masking the visible symptom.",
|
|
444
|
+
"Prove the failing scenario with regression coverage when the repository supports it.",
|
|
445
|
+
]
|
|
446
|
+
)
|
|
447
|
+
elif task.task_type is TaskType.REFACTOR:
|
|
448
|
+
common.append("Keep externally observable behavior stable while updating references and tests affected by the refactor.")
|
|
449
|
+
elif task.task_type is TaskType.MIGRATION:
|
|
450
|
+
common.extend(
|
|
451
|
+
[
|
|
452
|
+
"Use the repository's existing migration mechanism and preserve compatibility with supported application state.",
|
|
453
|
+
"Provide a forward path plus rollback or an explicitly safe non-reversible strategy; do not invent destructive data policy.",
|
|
454
|
+
]
|
|
455
|
+
)
|
|
456
|
+
elif task.task_type is TaskType.PERFORMANCE:
|
|
457
|
+
common.append("Preserve functional behavior while improving the requested performance concern; do not invent an unrequested numeric target.")
|
|
458
|
+
|
|
459
|
+
lowered = " ".join(
|
|
460
|
+
criterion.description for criterion in task.acceptance_criteria if criterion.source is AcceptanceSource.USER
|
|
461
|
+
).lower()
|
|
462
|
+
if any(term in lowered for term in ("auth", "login", "oauth", "token", "credential", "api key", "secret")):
|
|
463
|
+
common.extend(
|
|
464
|
+
[
|
|
465
|
+
"Use the repository's existing configuration and secret-handling mechanisms; never hardcode credentials.",
|
|
466
|
+
"Do not invent authorization roles, OAuth scopes, account-linking policy, or other security/product decisions absent from the user request.",
|
|
467
|
+
]
|
|
468
|
+
)
|
|
469
|
+
if any(term in lowered for term in ("payment", "billing", "checkout", "subscription")):
|
|
470
|
+
common.append(
|
|
471
|
+
"Do not invent retry counts, fees, cancellation policy, payment state transitions, or other commercial behavior absent from the user request."
|
|
472
|
+
)
|
|
473
|
+
return _dedupe(common)
|
|
474
|
+
|
|
475
|
+
|
|
476
|
+
def _compile_repository_aware_brief(task: TaskSpec, repository: Any) -> None:
|
|
477
|
+
"""Turn user intent into a richer engineering brief without changing user-owned criteria.
|
|
478
|
+
|
|
479
|
+
This brief is supplied to every later DevAgent role through TaskSpec.goal. It may
|
|
480
|
+
add safe engineering defaults and repository facts, but it explicitly does not
|
|
481
|
+
create new user/business requirements. AcceptanceSource.USER criteria remain the
|
|
482
|
+
authoritative statement of what the user asked for.
|
|
483
|
+
"""
|
|
484
|
+
|
|
485
|
+
if "DEVAGENT REQUIREMENT INTELLIGENCE" in task.goal:
|
|
486
|
+
return
|
|
487
|
+
|
|
488
|
+
core_goal = task.goal.strip()
|
|
489
|
+
user_requirements = [
|
|
490
|
+
criterion.description
|
|
491
|
+
for criterion in task.acceptance_criteria
|
|
492
|
+
if criterion.source is AcceptanceSource.USER
|
|
493
|
+
]
|
|
494
|
+
languages = _unique_repository_values(repository, "languages")
|
|
495
|
+
frameworks = _unique_repository_values(repository, "frameworks")
|
|
496
|
+
manifests = _unique_repository_values(repository, "manifests")
|
|
497
|
+
test_locations = _unique_repository_values(repository, "test_locations")
|
|
498
|
+
|
|
499
|
+
trusted_commands: list[str] = []
|
|
500
|
+
for capability in repository.capabilities:
|
|
501
|
+
if capability.trusted:
|
|
502
|
+
command = " ".join(capability.command)
|
|
503
|
+
if command and command not in trusted_commands:
|
|
504
|
+
trusted_commands.append(command)
|
|
505
|
+
|
|
506
|
+
lines = [
|
|
507
|
+
core_goal,
|
|
508
|
+
"",
|
|
509
|
+
"DEVAGENT REQUIREMENT INTELLIGENCE",
|
|
510
|
+
"User intent remains authoritative; the sections below are engineering design guidance, not invented business requirements.",
|
|
511
|
+
"",
|
|
512
|
+
"USER REQUIREMENTS",
|
|
513
|
+
]
|
|
514
|
+
lines.extend(f"- {item}" for item in user_requirements or [core_goal])
|
|
515
|
+
|
|
516
|
+
lines.extend(["", "SAFE ENGINEERING DEFAULTS"])
|
|
517
|
+
lines.extend(f"- {item}" for item in _task_design_defaults(task))
|
|
518
|
+
|
|
519
|
+
repository_lines: list[str] = []
|
|
520
|
+
if languages:
|
|
521
|
+
repository_lines.append("Languages: " + ", ".join(languages[:8]))
|
|
522
|
+
if frameworks:
|
|
523
|
+
repository_lines.append("Frameworks: " + ", ".join(frameworks[:8]))
|
|
524
|
+
if manifests:
|
|
525
|
+
repository_lines.append("Manifests: " + ", ".join(manifests[:10]))
|
|
526
|
+
if test_locations:
|
|
527
|
+
repository_lines.append("Existing test locations: " + ", ".join(test_locations[:10]))
|
|
528
|
+
if trusted_commands:
|
|
529
|
+
repository_lines.append("Evidence-backed verification: " + "; ".join(trusted_commands[:8]))
|
|
530
|
+
if len(repository.components) > 1:
|
|
531
|
+
repository_lines.append(f"Repository structure: {repository.kind} with {len(repository.components)} discovered components")
|
|
532
|
+
|
|
533
|
+
lines.extend(["", "REPOSITORY-DERIVED CONTEXT"])
|
|
534
|
+
lines.extend(f"- {item}" for item in repository_lines or ["Use discovered repository structure and conventions as implementation evidence."])
|
|
535
|
+
|
|
536
|
+
lines.extend(
|
|
537
|
+
[
|
|
538
|
+
"",
|
|
539
|
+
"DESIGN GUARDRAIL",
|
|
540
|
+
"- Do not invent material product, business, security, data-lifecycle, or external-contract behavior that the user did not request.",
|
|
541
|
+
"- If source evidence shows a material ambiguity, prefer a bounded implementation or BLOCKED/PARTIALLY_VERIFIED outcome over silently choosing product policy.",
|
|
542
|
+
]
|
|
543
|
+
)
|
|
544
|
+
task.goal = "\n".join(lines)
|
|
545
|
+
|
|
546
|
+
|
|
366
547
|
def enrich_acceptance_contract(task: TaskSpec, repository: Any) -> TaskSpec:
|
|
367
|
-
"""Compile
|
|
548
|
+
"""Compile repository-aware requirement intelligence and trusted final checks."""
|
|
368
549
|
|
|
369
550
|
_matrix_operation_contract(task, repository)
|
|
370
551
|
seen_commands: set[tuple[str, ...]] = set()
|
|
@@ -382,4 +563,6 @@ def enrich_acceptance_contract(task: TaskSpec, repository: Any) -> TaskSpec:
|
|
|
382
563
|
source=AcceptanceSource.REPOSITORY,
|
|
383
564
|
verification_command=capability.command,
|
|
384
565
|
)
|
|
566
|
+
|
|
567
|
+
_compile_repository_aware_brief(task, repository)
|
|
385
568
|
return task
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: devagent-ai
|
|
3
|
-
Version: 0.8.
|
|
3
|
+
Version: 0.8.4
|
|
4
4
|
Summary: Evidence-driven local autonomous software engineering agent
|
|
5
5
|
Author: Tom Ha
|
|
6
6
|
Maintainer: Tom Ha
|
|
@@ -152,6 +152,39 @@ reviewer → independently review final diff
|
|
|
152
152
|
|
|
153
153
|
The deterministic harness remains responsible for safety, tool execution, verification validity, acceptance adjudication, final status, reporting, and source-control publication regardless of which model handles a role.
|
|
154
154
|
|
|
155
|
+
#### Example: use multiple AI models in one DevAgent run
|
|
156
|
+
|
|
157
|
+
You do not have to use one AI model for every reasoning step. If you believe different models are better suited to different engineering roles, configure a default model plus any role-specific overrides. Roles that are not explicitly configured fall back to the default model.
|
|
158
|
+
|
|
159
|
+
```bash
|
|
160
|
+
# Keep credentials in environment variables; DevAgent does not store the keys.
|
|
161
|
+
export OPENAI_API_KEY=...
|
|
162
|
+
export GEMINI_API_KEY=...
|
|
163
|
+
export ANTHROPIC_API_KEY=...
|
|
164
|
+
export XAI_API_KEY=...
|
|
165
|
+
|
|
166
|
+
# Default/fallback model.
|
|
167
|
+
devagent setup --provider openai --model YOUR_OPENAI_MODEL
|
|
168
|
+
|
|
169
|
+
# Optional per-role models.
|
|
170
|
+
devagent setup --role investigator --provider gemini --model YOUR_GEMINI_MODEL
|
|
171
|
+
devagent setup --role planner --provider anthropic --model YOUR_CLAUDE_MODEL
|
|
172
|
+
devagent setup --role implementer --provider openai --model YOUR_OPENAI_MODEL
|
|
173
|
+
devagent setup --role reviewer --provider xai --model YOUR_GROK_MODEL
|
|
174
|
+
|
|
175
|
+
# Inspect routing and optionally probe every configured cloud model.
|
|
176
|
+
devagent models
|
|
177
|
+
devagent doctor --live
|
|
178
|
+
|
|
179
|
+
# Run normally; saved role routing is applied automatically.
|
|
180
|
+
cd my-repo
|
|
181
|
+
devagent "Fix the checkout race condition and add regression coverage."
|
|
182
|
+
```
|
|
183
|
+
|
|
184
|
+
For example, a user may choose a fast or lower-cost model for repository investigation, a different model for planning, a preferred coding model for implementation, and another provider for independent review. This can be useful for cost, latency, provider diversity, or model-strength preferences, but it does not guarantee a better result. DevAgent still requires the same repository evidence, acceptance gates, deterministic verification, and publication rules.
|
|
185
|
+
|
|
186
|
+
When using saved role routing, run the task without run-level `--provider`, `--model`, or `--base-url` overrides. Supplying those flags explicitly selects one provider/model for that run instead of the saved per-role routing.
|
|
187
|
+
|
|
155
188
|
## Run
|
|
156
189
|
|
|
157
190
|
From the application repository:
|
|
@@ -48,6 +48,7 @@ tests/test_benchmark_catalog.py
|
|
|
48
48
|
tests/test_browser_verification.py
|
|
49
49
|
tests/test_capability_discovery.py
|
|
50
50
|
tests/test_cli.py
|
|
51
|
+
tests/test_cli_progress.py
|
|
51
52
|
tests/test_developer_review_report.py
|
|
52
53
|
tests/test_discovery_memory.py
|
|
53
54
|
tests/test_e2e_fake_provider.py
|
|
@@ -68,6 +69,7 @@ tests/test_production_hardening.py
|
|
|
68
69
|
tests/test_production_v040.py
|
|
69
70
|
tests/test_realworld_benchmark.py
|
|
70
71
|
tests/test_requirement_compiler_v082.py
|
|
72
|
+
tests/test_requirement_intelligence_v083.py
|
|
71
73
|
tests/test_retrieval.py
|
|
72
74
|
tests/test_runtime_sandbox.py
|
|
73
75
|
tests/test_safety_workspace.py
|