devagent-ai 0.8.2__tar.gz → 0.8.4__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (88) hide show
  1. {devagent_ai-0.8.2/devagent_ai.egg-info → devagent_ai-0.8.4}/PKG-INFO +34 -1
  2. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/README.md +33 -0
  3. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent/__init__.py +1 -1
  4. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent/cli.py +91 -4
  5. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent/tasking.py +225 -42
  6. {devagent_ai-0.8.2 → devagent_ai-0.8.4/devagent_ai.egg-info}/PKG-INFO +34 -1
  7. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent_ai.egg-info/SOURCES.txt +2 -0
  8. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/pyproject.toml +1 -1
  9. devagent_ai-0.8.4/tests/test_cli_progress.py +91 -0
  10. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_production_v040.py +1 -1
  11. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_requirement_compiler_v082.py +7 -10
  12. devagent_ai-0.8.4/tests/test_requirement_intelligence_v083.py +170 -0
  13. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/LICENSE +0 -0
  14. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/NOTICE +0 -0
  15. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/agent/__init__.py +0 -0
  16. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/agent/llm.py +0 -0
  17. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/agent/loop.py +0 -0
  18. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/agent/memory.py +0 -0
  19. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/agent/prompts.py +0 -0
  20. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/agent/tools.py +0 -0
  21. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent/__main__.py +0 -0
  22. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent/artifacts.py +0 -0
  23. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent/automations.py +0 -0
  24. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent/autonomy.py +0 -0
  25. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent/browser.py +0 -0
  26. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent/config.py +0 -0
  27. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent/discovery.py +0 -0
  28. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent/evaluation.py +0 -0
  29. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent/memory.py +0 -0
  30. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent/models.py +0 -0
  31. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent/orchestrator.py +0 -0
  32. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent/provider_benchmark.py +0 -0
  33. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent/providers.py +0 -0
  34. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent/qualification.py +0 -0
  35. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent/realworld.py +0 -0
  36. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent/report.py +0 -0
  37. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent/retrieval.py +0 -0
  38. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent/routing.py +0 -0
  39. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent/runtime.py +0 -0
  40. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent/safety.py +0 -0
  41. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent/skills.py +0 -0
  42. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent/source_control.py +0 -0
  43. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent/state_machine.py +0 -0
  44. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent/technical_review.py +0 -0
  45. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent/workspace.py +0 -0
  46. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent/worktree.py +0 -0
  47. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent_ai.egg-info/dependency_links.txt +0 -0
  48. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent_ai.egg-info/entry_points.txt +0 -0
  49. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent_ai.egg-info/requires.txt +0 -0
  50. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/devagent_ai.egg-info/top_level.txt +0 -0
  51. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/setup.cfg +0 -0
  52. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_acceptance_contract.py +0 -0
  53. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_benchmark_catalog.py +0 -0
  54. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_browser_verification.py +0 -0
  55. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_capability_discovery.py +0 -0
  56. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_cli.py +0 -0
  57. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_developer_review_report.py +0 -0
  58. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_discovery_memory.py +0 -0
  59. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_e2e_fake_provider.py +0 -0
  60. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_evaluation_harness.py +0 -0
  61. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_evaluation_matrix.py +0 -0
  62. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_evaluation_regression_evidence.py +0 -0
  63. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_functional_qualification.py +0 -0
  64. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_huge_monorepo_v070.py +0 -0
  65. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_migration_e2e_v070.py +0 -0
  66. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_model_routing.py +0 -0
  67. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_multilang_technical_review.py +0 -0
  68. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_multistack_devagent_e2e.py +0 -0
  69. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_multistack_qualification.py +0 -0
  70. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_packaging_metadata.py +0 -0
  71. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_plan_verification_normalization.py +0 -0
  72. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_preservation_contradiction.py +0 -0
  73. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_production_hardening.py +0 -0
  74. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_realworld_benchmark.py +0 -0
  75. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_retrieval.py +0 -0
  76. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_runtime_sandbox.py +0 -0
  77. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_safety_workspace.py +0 -0
  78. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_source_control_publish.py +0 -0
  79. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_structural_devagent_e2e_v070.py +0 -0
  80. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_structural_operations.py +0 -0
  81. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_structured_provider_contract.py +0 -0
  82. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_tasking_state.py +0 -0
  83. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_v070_engineering_breadth.py +0 -0
  84. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_v080_autonomy.py +0 -0
  85. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_v080_provider_benchmark.py +0 -0
  86. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_v080_skills_automations.py +0 -0
  87. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_workspace_environment.py +0 -0
  88. {devagent_ai-0.8.2 → devagent_ai-0.8.4}/tests/test_worktree.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: devagent-ai
3
- Version: 0.8.2
3
+ Version: 0.8.4
4
4
  Summary: Evidence-driven local autonomous software engineering agent
5
5
  Author: Tom Ha
6
6
  Maintainer: Tom Ha
@@ -152,6 +152,39 @@ reviewer → independently review final diff
152
152
 
153
153
  The deterministic harness remains responsible for safety, tool execution, verification validity, acceptance adjudication, final status, reporting, and source-control publication regardless of which model handles a role.
154
154
 
155
+ #### Example: use multiple AI models in one DevAgent run
156
+
157
+ You do not have to use one AI model for every reasoning step. If you believe different models are better suited to different engineering roles, configure a default model plus any role-specific overrides. Roles that are not explicitly configured fall back to the default model.
158
+
159
+ ```bash
160
+ # Keep credentials in environment variables; DevAgent does not store the keys.
161
+ export OPENAI_API_KEY=...
162
+ export GEMINI_API_KEY=...
163
+ export ANTHROPIC_API_KEY=...
164
+ export XAI_API_KEY=...
165
+
166
+ # Default/fallback model.
167
+ devagent setup --provider openai --model YOUR_OPENAI_MODEL
168
+
169
+ # Optional per-role models.
170
+ devagent setup --role investigator --provider gemini --model YOUR_GEMINI_MODEL
171
+ devagent setup --role planner --provider anthropic --model YOUR_CLAUDE_MODEL
172
+ devagent setup --role implementer --provider openai --model YOUR_OPENAI_MODEL
173
+ devagent setup --role reviewer --provider xai --model YOUR_GROK_MODEL
174
+
175
+ # Inspect routing and optionally probe every configured cloud model.
176
+ devagent models
177
+ devagent doctor --live
178
+
179
+ # Run normally; saved role routing is applied automatically.
180
+ cd my-repo
181
+ devagent "Fix the checkout race condition and add regression coverage."
182
+ ```
183
+
184
+ For example, a user may choose a fast or lower-cost model for repository investigation, a different model for planning, a preferred coding model for implementation, and another provider for independent review. This can be useful for cost, latency, provider diversity, or model-strength preferences, but it does not guarantee a better result. DevAgent still requires the same repository evidence, acceptance gates, deterministic verification, and publication rules.
185
+
186
+ When using saved role routing, run the task without run-level `--provider`, `--model`, or `--base-url` overrides. Supplying those flags explicitly selects one provider/model for that run instead of the saved per-role routing.
187
+
155
188
  ## Run
156
189
 
157
190
  From the application repository:
@@ -121,6 +121,39 @@ reviewer → independently review final diff
121
121
 
122
122
  The deterministic harness remains responsible for safety, tool execution, verification validity, acceptance adjudication, final status, reporting, and source-control publication regardless of which model handles a role.
123
123
 
124
+ #### Example: use multiple AI models in one DevAgent run
125
+
126
+ You do not have to use one AI model for every reasoning step. If you believe different models are better suited to different engineering roles, configure a default model plus any role-specific overrides. Roles that are not explicitly configured fall back to the default model.
127
+
128
+ ```bash
129
+ # Keep credentials in environment variables; DevAgent does not store the keys.
130
+ export OPENAI_API_KEY=...
131
+ export GEMINI_API_KEY=...
132
+ export ANTHROPIC_API_KEY=...
133
+ export XAI_API_KEY=...
134
+
135
+ # Default/fallback model.
136
+ devagent setup --provider openai --model YOUR_OPENAI_MODEL
137
+
138
+ # Optional per-role models.
139
+ devagent setup --role investigator --provider gemini --model YOUR_GEMINI_MODEL
140
+ devagent setup --role planner --provider anthropic --model YOUR_CLAUDE_MODEL
141
+ devagent setup --role implementer --provider openai --model YOUR_OPENAI_MODEL
142
+ devagent setup --role reviewer --provider xai --model YOUR_GROK_MODEL
143
+
144
+ # Inspect routing and optionally probe every configured cloud model.
145
+ devagent models
146
+ devagent doctor --live
147
+
148
+ # Run normally; saved role routing is applied automatically.
149
+ cd my-repo
150
+ devagent "Fix the checkout race condition and add regression coverage."
151
+ ```
152
+
153
+ For example, a user may choose a fast or lower-cost model for repository investigation, a different model for planning, a preferred coding model for implementation, and another provider for independent review. This can be useful for cost, latency, provider diversity, or model-strength preferences, but it does not guarantee a better result. DevAgent still requires the same repository evidence, acceptance gates, deterministic verification, and publication rules.
154
+
155
+ When using saved role routing, run the task without run-level `--provider`, `--model`, or `--base-url` overrides. Supplying those flags explicitly selects one provider/model for that run instead of the saved per-role routing.
156
+
124
157
  ## Run
125
158
 
126
159
  From the application repository:
@@ -1,3 +1,3 @@
1
1
  """DevAgent: evidence-driven local software engineering automation."""
2
2
 
3
- __version__ = "0.8.2"
3
+ __version__ = "0.8.4"
@@ -4,10 +4,11 @@ import argparse
4
4
  import importlib.util
5
5
  import json
6
6
  import os
7
+ import re
7
8
  import shutil
8
9
  import sys
9
10
  from pathlib import Path
10
- from typing import Sequence
11
+ from typing import Callable, Sequence
11
12
 
12
13
  from devagent import __version__
13
14
  from devagent.config import (
@@ -37,6 +38,84 @@ _LIVE_PROBE_SCHEMA = {
37
38
  "additionalProperties": False,
38
39
  }
39
40
 
41
+ _PROGRESS_STAGES: dict[str, tuple[int, str]] = {
42
+ "DISCOVER": (1, "DISCOVER / UNDERSTAND"),
43
+ "UNDERSTAND": (1, "DISCOVER / UNDERSTAND"),
44
+ "TASK_SPEC": (2, "REQUIREMENTS / PLAN"),
45
+ "BASELINE": (2, "REQUIREMENTS / PLAN"),
46
+ "PLAN": (2, "REQUIREMENTS / PLAN"),
47
+ "GATHER_CONTEXT": (2, "REQUIREMENTS / PLAN"),
48
+ "REPRODUCE": (2, "REQUIREMENTS / PLAN"),
49
+ "IMPLEMENT": (3, "IMPLEMENT"),
50
+ "VERIFY_TARGETED": (4, "VERIFY / REPAIR IF NEEDED"),
51
+ "VERIFY_BROAD": (4, "VERIFY / REPAIR IF NEEDED"),
52
+ "REVIEW": (5, "INDEPENDENT REVIEW"),
53
+ "QUALITY_CHECK": (6, "FINAL VERIFICATION"),
54
+ "FINAL_VERIFY": (6, "FINAL VERIFICATION"),
55
+ }
56
+ _STATE_LINE = re.compile(r"^\[([A-Z_]+)\]$")
57
+
58
+
59
+ class _ProgressStatus:
60
+ """Render stable user-facing milestones while keeping full diagnostics opt-in."""
61
+
62
+ def __init__(
63
+ self,
64
+ sink: Callable[[str], None] = print,
65
+ *,
66
+ verbose: bool = False,
67
+ ) -> None:
68
+ self.sink = sink
69
+ self.verbose = verbose
70
+ self._last_stage = 0
71
+ self._plan_seen = False
72
+ self._implement_seen = False
73
+ self._review_seen = False
74
+
75
+ def __call__(self, message: str) -> None:
76
+ if self.verbose:
77
+ self.sink(message)
78
+ return
79
+
80
+ match = _STATE_LINE.fullmatch(message)
81
+ if match is None:
82
+ return
83
+ state = match.group(1)
84
+
85
+ if state == "DIAGNOSE":
86
+ self.sink(" ↳ DIAGNOSE")
87
+ return
88
+ if state == "PLAN":
89
+ if self._plan_seen:
90
+ self.sink(" ↳ REPLAN")
91
+ return
92
+ self._plan_seen = True
93
+ if state == "IMPLEMENT":
94
+ if self._implement_seen:
95
+ label = "APPLY REVIEW FIXES" if self._review_seen else "APPLY CORRECTION"
96
+ self.sink(f" ↳ {label}")
97
+ return
98
+ self._implement_seen = True
99
+ if state == "REVIEW":
100
+ self._review_seen = True
101
+
102
+ stage = _PROGRESS_STAGES.get(state)
103
+ if stage is None:
104
+ return
105
+ number, label = stage
106
+ if number <= self._last_stage:
107
+ return
108
+ self._last_stage = number
109
+ self.sink(f"[{number}/7] {label}")
110
+
111
+ def report(self) -> None:
112
+ if self.verbose:
113
+ self.sink("[ENGINEERING_REPORT]")
114
+ return
115
+ if self._last_stage < 7:
116
+ self._last_stage = 7
117
+ self.sink("[7/7] ENGINEERING REPORT")
118
+
40
119
 
41
120
  def _top_parser() -> argparse.ArgumentParser:
42
121
  parser = argparse.ArgumentParser(prog="devagent", description="Evidence-driven local software engineering agent")
@@ -53,7 +132,11 @@ def _top_parser() -> argparse.ArgumentParser:
53
132
  parser.add_argument("--provider", choices=_PROVIDER_CHOICES)
54
133
  parser.add_argument("--model")
55
134
  parser.add_argument("--base-url")
56
- parser.add_argument("--verbose", action="store_true", help="Show state transitions")
135
+ parser.add_argument(
136
+ "--verbose",
137
+ action="store_true",
138
+ help="Show internal state transitions and diagnostics instead of concise progress stages",
139
+ )
57
140
  parser.add_argument("--no-isolation", action="store_true", help="Work in place instead of creating a local detached worktree")
58
141
  parser.add_argument(
59
142
  "--publish",
@@ -337,11 +420,14 @@ def _run(argv: Sequence[str]) -> int:
337
420
  from devagent.orchestrator import DevAgent
338
421
  from devagent.report import recommendations_for, render_report
339
422
 
423
+ progress = _ProgressStatus(print, verbose=args.verbose)
340
424
  result = DevAgent(
341
425
  model_provider,
342
426
  isolate=not args.no_isolation,
343
- verbose=args.verbose,
344
- status=print,
427
+ # Internal status events are always emitted. The progress reporter keeps normal
428
+ # CLI output concise and passes the full state/diagnostic stream only in --verbose.
429
+ verbose=True,
430
+ status=progress,
345
431
  base_commit=publication_plan.base_commit if publication_plan else None,
346
432
  ).run(args.repo, requirement)
347
433
 
@@ -361,6 +447,7 @@ def _run(argv: Sequence[str]) -> int:
361
447
 
362
448
  # The full engineering report is intentionally emitted before any Git commit/push.
363
449
  result.recommendations = recommendations_for(result)
450
+ progress.report()
364
451
  print(render_report(result))
365
452
 
366
453
  if publish_requested:
@@ -17,11 +17,11 @@ _CLASSIFIERS: tuple[tuple[TaskType, tuple[str, ...]], ...] = (
17
17
  (TaskType.TEST_FAILURE, ("test fail", "failing test", "pytest error")),
18
18
  (TaskType.RUNTIME_ERROR, ("traceback", "exception", "runtime error", "crash")),
19
19
  (TaskType.MIGRATION, ("migration", "migrate ", "schema change", "alembic", "database migration")),
20
- (TaskType.PERFORMANCE, ("performance", "optimize", "slow", "latency", "n+1")),
21
- (TaskType.REFACTOR, ("refactor", "restructure", "cleanup")),
20
+ (TaskType.PERFORMANCE, ("performance", "optimize", "slow", "latency", "n+1", "faster", "speed up")),
21
+ (TaskType.REFACTOR, ("refactor", "restructure", "cleanup", "rename ", "move ", "delete obsolete")),
22
22
  (TaskType.UNIT_TEST, ("add unit test", "write tests", "test coverage")),
23
23
  (TaskType.BUG_FIX, ("fix", "bug", "incorrect", "broken", "regression failure", "regression bug")),
24
- (TaskType.FEATURE, ("add ", "implement", "support ", "feature")),
24
+ (TaskType.FEATURE, ("add ", "implement", "support ", "feature", "create ")),
25
25
  )
26
26
 
27
27
  _HIGH_RISK = {
@@ -56,24 +56,39 @@ _KNOWN_SECTIONS = _REQUIREMENT_SECTIONS | {
56
56
  "non-goals",
57
57
  "non goals",
58
58
  "notes",
59
+ "engineering design",
60
+ "engineering context",
59
61
  }
60
62
  _DIRECTIVE = re.compile(
61
- r"^(?:add|implement|support|preserve|keep|ensure|require|must|should|when|do not|don't|"
62
- r"verify|run|return|raise|allow|prevent|maintain|migrate|refactor|update|fix|handle)\b",
63
+ r"^(?:add|create|implement|support|preserve|keep|ensure|require|must|should|when|do not|don't|"
64
+ r"verify|run|return|raise|allow|prevent|maintain|migrate|refactor|rename|move|delete|update|fix|handle)\b",
63
65
  re.IGNORECASE,
64
66
  )
65
67
 
66
- # Bounded normalization for terse user intent. This is deliberately not a fuzzy
67
- # "guess what the user meant" layer: it corrects common engineering shorthand,
68
- # grammatical number, and operation wording while preserving identifiers,
69
- # quoted contracts, values, and explicit constraints. Task policy and repository
70
- # evidence still provide the verification/safety contract.
68
+ # Bounded normalization for terse user intent. This intentionally fixes common
69
+ # engineering shorthand and spelling without attempting to invent product behavior.
71
70
  _OPERATION_ALIASES: tuple[tuple[str, str], ...] = (
72
71
  ("substraction", "subtraction"),
73
72
  ("substract", "subtract"),
74
73
  ("multipy", "multiply"),
75
74
  ("mutiply", "multiply"),
75
+ ("authentification", "authentication"),
76
+ ("autorization", "authorization"),
77
+ ("loging", "login"),
76
78
  )
79
+ _ACRONYMS = {
80
+ "api": "API",
81
+ "csv": "CSV",
82
+ "db": "DB",
83
+ "http": "HTTP",
84
+ "https": "HTTPS",
85
+ "json": "JSON",
86
+ "jwt": "JWT",
87
+ "oauth": "OAuth",
88
+ "sql": "SQL",
89
+ "ui": "UI",
90
+ "url": "URL",
91
+ }
77
92
 
78
93
 
79
94
  def _classify(text: str) -> TaskType:
@@ -110,27 +125,66 @@ def _dedupe(items: list[str]) -> list[str]:
110
125
  return result
111
126
 
112
127
 
113
- def _normalize_terse_requirement(requirement: str) -> str:
114
- """Compile common rough one-line prompts into a clearer engineering request.
128
+ def _section_header(line: str) -> tuple[str, str] | None:
129
+ stripped = line.strip()
130
+ markdown = re.match(r"^#{1,6}\s+(.+?)\s*$", stripped)
131
+ if markdown:
132
+ return markdown.group(1).strip().rstrip(":").lower(), ""
133
+ colon = re.match(r"^([A-Za-z][A-Za-z0-9 _/-]{0,80})\s*:\s*(.*)$", stripped)
134
+ if colon and colon.group(1).strip().lower() in _KNOWN_SECTIONS:
135
+ return colon.group(1).strip().lower(), colon.group(2).strip()
136
+ return None
115
137
 
116
- The compiler is intentionally bounded. It may repair shorthand/grammar and
117
- make an operation explicit, but it must not add product behavior the user did
118
- not request. Structured/multi-line requirements are left intact.
119
- """
120
138
 
121
- value = re.sub(r"\s+", " ", requirement).strip()
122
- if not value or "\n" in requirement or _section_header(value) is not None:
139
+ def _extract_goal(requirement: str) -> str:
140
+ """Prefer an explicit Goal section while preserving ordinary free-form input."""
141
+
142
+ lines = requirement.splitlines()
143
+ for index, raw in enumerate(lines):
144
+ header = _section_header(raw)
145
+ if header is None or header[0] != "goal":
146
+ continue
147
+ _, inline = header
148
+ if inline:
149
+ return _clean_requirement_item(inline)
150
+ collected: list[str] = []
151
+ for candidate in lines[index + 1 :]:
152
+ if _section_header(candidate) is not None:
153
+ break
154
+ if candidate.strip():
155
+ collected.append(_clean_requirement_item(candidate))
156
+ if collected:
157
+ return " ".join(collected)
158
+ return re.sub(r"\s+", " ", requirement).strip()
159
+
160
+
161
+ def _polish_plain_goal(value: str) -> str:
162
+ """Improve readability without changing the requested product semantics."""
163
+
164
+ result = re.sub(r"\s+", " ", value).strip()
165
+ for source, destination in _OPERATION_ALIASES:
166
+ result = re.sub(rf"\b{re.escape(source)}\b", destination, result, flags=re.IGNORECASE)
167
+ result = re.sub(r"^customer\s+(?:need|needs|want|wants)\s+", "Implement ", result, flags=re.IGNORECASE)
168
+ result = re.sub(r"^user\s+(?:need|needs|want|wants)\s+", "Implement ", result, flags=re.IGNORECASE)
169
+ for source, destination in _ACRONYMS.items():
170
+ result = re.sub(rf"\b{source}\b", destination, result, flags=re.IGNORECASE)
171
+ if result:
172
+ result = result[0].upper() + result[1:]
173
+ return result.rstrip(".;")
174
+
175
+
176
+ def _normalize_terse_requirement(requirement: str) -> str:
177
+ """Compile common rough prompts into a clearer bounded engineering request."""
178
+
179
+ raw_value = _extract_goal(requirement)
180
+ value = _polish_plain_goal(raw_value)
181
+ if not value:
123
182
  return value
124
183
  # An explicit callable name is already a precise user contract; never rename it.
125
184
  if re.search(r"\b[A-Za-z_][A-Za-z0-9_]*\s*\(", value):
126
185
  return value
127
186
 
128
- for source, destination in _OPERATION_ALIASES:
129
- value = re.sub(rf"\b{re.escape(source)}\b", destination, value, flags=re.IGNORECASE)
130
-
131
187
  # Common shorthand from natural prompts such as "addition 2 matrix 2x2".
132
- # Keep both "matrix" and "matrices" in the normalized contract so
133
- # deterministic evidence can link either conventional symbol spelling.
134
188
  matrix_match = re.search(
135
189
  r"\b(add(?:ition)?|sum|subtract(?:ion)?|multiply|multiplication|divide|division)\b"
136
190
  r"(?:\s+(?:of|for))?\s+(?:2|two)\s+matrix(?:es)?\s+(\d+x\d+)\b",
@@ -157,28 +211,19 @@ def _normalize_terse_requirement(requirement: str) -> str:
157
211
  f"(matrix inputs)"
158
212
  )
159
213
 
160
- # Repair simple count+noun shorthand without inventing domain behavior.
161
214
  value = re.sub(r"\b2\s+matrix\b", "two matrices", value, flags=re.IGNORECASE)
162
215
  value = re.sub(r"\b2\s+file\b", "two files", value, flags=re.IGNORECASE)
163
216
  value = re.sub(r"\b2\s+test\b", "two tests", value, flags=re.IGNORECASE)
164
217
  return value
165
218
 
166
219
 
167
- def _section_header(line: str) -> tuple[str, str] | None:
168
- stripped = line.strip()
169
- markdown = re.match(r"^#{1,6}\s+(.+?)\s*$", stripped)
170
- if markdown:
171
- return markdown.group(1).strip().rstrip(":").lower(), ""
172
- colon = re.match(r"^([A-Za-z][A-Za-z0-9 _/-]{0,80})\s*:\s*(.*)$", stripped)
173
- if colon and colon.group(1).strip().lower() in _KNOWN_SECTIONS:
174
- return colon.group(1).strip().lower(), colon.group(2).strip()
175
- return None
176
-
177
-
178
220
  def _user_acceptance_items(requirement: str) -> list[str]:
179
221
  lines = requirement.splitlines()
180
222
  explicit: list[str] = []
181
223
  active_section: str | None = None
224
+ recognized_section = False
225
+ nonempty_lines = [line.strip() for line in lines if line.strip()]
226
+
182
227
  for raw in lines:
183
228
  stripped = raw.strip()
184
229
  if not stripped:
@@ -192,6 +237,7 @@ def _user_acceptance_items(requirement: str) -> list[str]:
192
237
 
193
238
  header = _section_header(stripped)
194
239
  if header is not None:
240
+ recognized_section = True
195
241
  name, inline = header
196
242
  active_section = name if name in _REQUIREMENT_SECTIONS else None
197
243
  if active_section is not None and inline:
@@ -221,6 +267,19 @@ def _user_acceptance_items(requirement: str) -> list[str]:
221
267
  if _DIRECTIVE.match(item) or re.search(r"\b(?:must|should|shall)\b", item, re.IGNORECASE):
222
268
  directives.append(item)
223
269
  directives = _dedupe(directives)
270
+
271
+ # For a loose multi-line customer note, do not silently discard fragments merely
272
+ # because one line happens to begin with a directive. Preserve the whole intent as
273
+ # one user criterion unless the text is clearly a structured directive list.
274
+ if len(nonempty_lines) > 1 and not recognized_section:
275
+ all_directive_like = all(
276
+ _DIRECTIVE.match(_clean_requirement_item(item))
277
+ or re.search(r"\b(?:must|should|shall)\b", item, re.IGNORECASE)
278
+ for item in nonempty_lines
279
+ )
280
+ if not all_directive_like:
281
+ return [re.sub(r"\s+", " ", requirement).strip()]
282
+
224
283
  if directives:
225
284
  return directives
226
285
  return [re.sub(r"\s+", " ", requirement).strip()]
@@ -257,8 +316,6 @@ def compile_task(requirement: str) -> TaskSpec:
257
316
  requires_tests = task_type is not TaskType.BUILD_FAILURE
258
317
 
259
318
  criteria: list[AcceptanceCriterion] = []
260
- # Structured user requirements remain authoritative. Only an unstructured,
261
- # terse prompt is compiled into the clearer canonical request.
262
319
  user_items = _user_acceptance_items(requirement)
263
320
  if len(user_items) == 1 and user_items[0] == raw_goal and goal != raw_goal:
264
321
  user_items = [goal]
@@ -342,14 +399,13 @@ def _matrix_operation_contract(task: TaskSpec, repository: Any) -> None:
342
399
  "multiplication": "multiply",
343
400
  "division": "divide",
344
401
  }[operation]
345
- compact_dimension = dimension.replace("x", "x")
346
402
  language = _repository_language(repository)
347
403
  if language in {"java", "javascript", "typescript"}:
348
- symbol = f"{verb}Matrices{compact_dimension}"
404
+ symbol = f"{verb}Matrices{dimension}"
349
405
  elif language in {"csharp", "c#"}:
350
- symbol = f"{verb.capitalize()}Matrices{compact_dimension}"
406
+ symbol = f"{verb.capitalize()}Matrices{dimension}"
351
407
  else:
352
- symbol = f"{verb}_matrices_{compact_dimension}"
408
+ symbol = f"{verb}_matrices_{dimension}"
353
409
 
354
410
  compiled = (
355
411
  f"Add {symbol}(a, b) to perform element-wise matrix {operation} "
@@ -363,8 +419,133 @@ def _matrix_operation_contract(task: TaskSpec, repository: Any) -> None:
363
419
  user_criteria[0].description = compiled
364
420
 
365
421
 
422
+ def _unique_repository_values(repository: Any, field: str) -> list[str]:
423
+ values: list[str] = []
424
+ for component in repository.components:
425
+ for value in getattr(component, field, []):
426
+ if value and value not in values:
427
+ values.append(value)
428
+ return values
429
+
430
+
431
+ def _task_design_defaults(task: TaskSpec) -> list[str]:
432
+ common = [
433
+ "Integrate with the repository's existing architecture and naming conventions instead of creating a parallel pattern.",
434
+ "Keep the implementation bounded to the requested behavior and avoid unrelated refactors.",
435
+ "Preserve behavior outside the explicitly requested scope unless the user states otherwise.",
436
+ ]
437
+ if task.requires_tests:
438
+ common.append("Add or update focused regression coverage using the repository's existing test conventions.")
439
+
440
+ if task.task_type in {TaskType.BUG_FIX, TaskType.RUNTIME_ERROR, TaskType.TEST_FAILURE}:
441
+ common.extend(
442
+ [
443
+ "Identify and fix the underlying cause rather than masking the visible symptom.",
444
+ "Prove the failing scenario with regression coverage when the repository supports it.",
445
+ ]
446
+ )
447
+ elif task.task_type is TaskType.REFACTOR:
448
+ common.append("Keep externally observable behavior stable while updating references and tests affected by the refactor.")
449
+ elif task.task_type is TaskType.MIGRATION:
450
+ common.extend(
451
+ [
452
+ "Use the repository's existing migration mechanism and preserve compatibility with supported application state.",
453
+ "Provide a forward path plus rollback or an explicitly safe non-reversible strategy; do not invent destructive data policy.",
454
+ ]
455
+ )
456
+ elif task.task_type is TaskType.PERFORMANCE:
457
+ common.append("Preserve functional behavior while improving the requested performance concern; do not invent an unrequested numeric target.")
458
+
459
+ lowered = " ".join(
460
+ criterion.description for criterion in task.acceptance_criteria if criterion.source is AcceptanceSource.USER
461
+ ).lower()
462
+ if any(term in lowered for term in ("auth", "login", "oauth", "token", "credential", "api key", "secret")):
463
+ common.extend(
464
+ [
465
+ "Use the repository's existing configuration and secret-handling mechanisms; never hardcode credentials.",
466
+ "Do not invent authorization roles, OAuth scopes, account-linking policy, or other security/product decisions absent from the user request.",
467
+ ]
468
+ )
469
+ if any(term in lowered for term in ("payment", "billing", "checkout", "subscription")):
470
+ common.append(
471
+ "Do not invent retry counts, fees, cancellation policy, payment state transitions, or other commercial behavior absent from the user request."
472
+ )
473
+ return _dedupe(common)
474
+
475
+
476
+ def _compile_repository_aware_brief(task: TaskSpec, repository: Any) -> None:
477
+ """Turn user intent into a richer engineering brief without changing user-owned criteria.
478
+
479
+ This brief is supplied to every later DevAgent role through TaskSpec.goal. It may
480
+ add safe engineering defaults and repository facts, but it explicitly does not
481
+ create new user/business requirements. AcceptanceSource.USER criteria remain the
482
+ authoritative statement of what the user asked for.
483
+ """
484
+
485
+ if "DEVAGENT REQUIREMENT INTELLIGENCE" in task.goal:
486
+ return
487
+
488
+ core_goal = task.goal.strip()
489
+ user_requirements = [
490
+ criterion.description
491
+ for criterion in task.acceptance_criteria
492
+ if criterion.source is AcceptanceSource.USER
493
+ ]
494
+ languages = _unique_repository_values(repository, "languages")
495
+ frameworks = _unique_repository_values(repository, "frameworks")
496
+ manifests = _unique_repository_values(repository, "manifests")
497
+ test_locations = _unique_repository_values(repository, "test_locations")
498
+
499
+ trusted_commands: list[str] = []
500
+ for capability in repository.capabilities:
501
+ if capability.trusted:
502
+ command = " ".join(capability.command)
503
+ if command and command not in trusted_commands:
504
+ trusted_commands.append(command)
505
+
506
+ lines = [
507
+ core_goal,
508
+ "",
509
+ "DEVAGENT REQUIREMENT INTELLIGENCE",
510
+ "User intent remains authoritative; the sections below are engineering design guidance, not invented business requirements.",
511
+ "",
512
+ "USER REQUIREMENTS",
513
+ ]
514
+ lines.extend(f"- {item}" for item in user_requirements or [core_goal])
515
+
516
+ lines.extend(["", "SAFE ENGINEERING DEFAULTS"])
517
+ lines.extend(f"- {item}" for item in _task_design_defaults(task))
518
+
519
+ repository_lines: list[str] = []
520
+ if languages:
521
+ repository_lines.append("Languages: " + ", ".join(languages[:8]))
522
+ if frameworks:
523
+ repository_lines.append("Frameworks: " + ", ".join(frameworks[:8]))
524
+ if manifests:
525
+ repository_lines.append("Manifests: " + ", ".join(manifests[:10]))
526
+ if test_locations:
527
+ repository_lines.append("Existing test locations: " + ", ".join(test_locations[:10]))
528
+ if trusted_commands:
529
+ repository_lines.append("Evidence-backed verification: " + "; ".join(trusted_commands[:8]))
530
+ if len(repository.components) > 1:
531
+ repository_lines.append(f"Repository structure: {repository.kind} with {len(repository.components)} discovered components")
532
+
533
+ lines.extend(["", "REPOSITORY-DERIVED CONTEXT"])
534
+ lines.extend(f"- {item}" for item in repository_lines or ["Use discovered repository structure and conventions as implementation evidence."])
535
+
536
+ lines.extend(
537
+ [
538
+ "",
539
+ "DESIGN GUARDRAIL",
540
+ "- Do not invent material product, business, security, data-lifecycle, or external-contract behavior that the user did not request.",
541
+ "- If source evidence shows a material ambiguity, prefer a bounded implementation or BLOCKED/PARTIALLY_VERIFIED outcome over silently choosing product policy.",
542
+ ]
543
+ )
544
+ task.goal = "\n".join(lines)
545
+
546
+
366
547
  def enrich_acceptance_contract(task: TaskSpec, repository: Any) -> TaskSpec:
367
- """Compile safe repository-aware defaults, then add trusted repository checks."""
548
+ """Compile repository-aware requirement intelligence and trusted final checks."""
368
549
 
369
550
  _matrix_operation_contract(task, repository)
370
551
  seen_commands: set[tuple[str, ...]] = set()
@@ -382,4 +563,6 @@ def enrich_acceptance_contract(task: TaskSpec, repository: Any) -> TaskSpec:
382
563
  source=AcceptanceSource.REPOSITORY,
383
564
  verification_command=capability.command,
384
565
  )
566
+
567
+ _compile_repository_aware_brief(task, repository)
385
568
  return task
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: devagent-ai
3
- Version: 0.8.2
3
+ Version: 0.8.4
4
4
  Summary: Evidence-driven local autonomous software engineering agent
5
5
  Author: Tom Ha
6
6
  Maintainer: Tom Ha
@@ -152,6 +152,39 @@ reviewer → independently review final diff
152
152
 
153
153
  The deterministic harness remains responsible for safety, tool execution, verification validity, acceptance adjudication, final status, reporting, and source-control publication regardless of which model handles a role.
154
154
 
155
+ #### Example: use multiple AI models in one DevAgent run
156
+
157
+ You do not have to use one AI model for every reasoning step. If you believe different models are better suited to different engineering roles, configure a default model plus any role-specific overrides. Roles that are not explicitly configured fall back to the default model.
158
+
159
+ ```bash
160
+ # Keep credentials in environment variables; DevAgent does not store the keys.
161
+ export OPENAI_API_KEY=...
162
+ export GEMINI_API_KEY=...
163
+ export ANTHROPIC_API_KEY=...
164
+ export XAI_API_KEY=...
165
+
166
+ # Default/fallback model.
167
+ devagent setup --provider openai --model YOUR_OPENAI_MODEL
168
+
169
+ # Optional per-role models.
170
+ devagent setup --role investigator --provider gemini --model YOUR_GEMINI_MODEL
171
+ devagent setup --role planner --provider anthropic --model YOUR_CLAUDE_MODEL
172
+ devagent setup --role implementer --provider openai --model YOUR_OPENAI_MODEL
173
+ devagent setup --role reviewer --provider xai --model YOUR_GROK_MODEL
174
+
175
+ # Inspect routing and optionally probe every configured cloud model.
176
+ devagent models
177
+ devagent doctor --live
178
+
179
+ # Run normally; saved role routing is applied automatically.
180
+ cd my-repo
181
+ devagent "Fix the checkout race condition and add regression coverage."
182
+ ```
183
+
184
+ For example, a user may choose a fast or lower-cost model for repository investigation, a different model for planning, a preferred coding model for implementation, and another provider for independent review. This can be useful for cost, latency, provider diversity, or model-strength preferences, but it does not guarantee a better result. DevAgent still requires the same repository evidence, acceptance gates, deterministic verification, and publication rules.
185
+
186
+ When using saved role routing, run the task without run-level `--provider`, `--model`, or `--base-url` overrides. Supplying those flags explicitly selects one provider/model for that run instead of the saved per-role routing.
187
+
155
188
  ## Run
156
189
 
157
190
  From the application repository:
@@ -48,6 +48,7 @@ tests/test_benchmark_catalog.py
48
48
  tests/test_browser_verification.py
49
49
  tests/test_capability_discovery.py
50
50
  tests/test_cli.py
51
+ tests/test_cli_progress.py
51
52
  tests/test_developer_review_report.py
52
53
  tests/test_discovery_memory.py
53
54
  tests/test_e2e_fake_provider.py
@@ -68,6 +69,7 @@ tests/test_production_hardening.py
68
69
  tests/test_production_v040.py
69
70
  tests/test_realworld_benchmark.py
70
71
  tests/test_requirement_compiler_v082.py
72
+ tests/test_requirement_intelligence_v083.py
71
73
  tests/test_retrieval.py
72
74
  tests/test_runtime_sandbox.py
73
75
  tests/test_safety_workspace.py
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "devagent-ai"
7
- version = "0.8.2"
7
+ version = "0.8.4"
8
8
  description = "Evidence-driven local autonomous software engineering agent"
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.10"