devagent-ai 0.8.2__tar.gz → 0.8.3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {devagent_ai-0.8.2/devagent_ai.egg-info → devagent_ai-0.8.3}/PKG-INFO +1 -1
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/devagent/__init__.py +1 -1
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/devagent/tasking.py +225 -42
- {devagent_ai-0.8.2 → devagent_ai-0.8.3/devagent_ai.egg-info}/PKG-INFO +1 -1
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/devagent_ai.egg-info/SOURCES.txt +1 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/pyproject.toml +1 -1
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/tests/test_production_v040.py +1 -1
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/tests/test_requirement_compiler_v082.py +7 -10
- devagent_ai-0.8.3/tests/test_requirement_intelligence_v083.py +170 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/LICENSE +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/NOTICE +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/README.md +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/agent/__init__.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/agent/llm.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/agent/loop.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/agent/memory.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/agent/prompts.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/agent/tools.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/devagent/__main__.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/devagent/artifacts.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/devagent/automations.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/devagent/autonomy.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/devagent/browser.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/devagent/cli.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/devagent/config.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/devagent/discovery.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/devagent/evaluation.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/devagent/memory.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/devagent/models.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/devagent/orchestrator.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/devagent/provider_benchmark.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/devagent/providers.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/devagent/qualification.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/devagent/realworld.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/devagent/report.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/devagent/retrieval.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/devagent/routing.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/devagent/runtime.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/devagent/safety.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/devagent/skills.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/devagent/source_control.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/devagent/state_machine.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/devagent/technical_review.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/devagent/workspace.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/devagent/worktree.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/devagent_ai.egg-info/dependency_links.txt +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/devagent_ai.egg-info/entry_points.txt +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/devagent_ai.egg-info/requires.txt +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/devagent_ai.egg-info/top_level.txt +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/setup.cfg +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/tests/test_acceptance_contract.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/tests/test_benchmark_catalog.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/tests/test_browser_verification.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/tests/test_capability_discovery.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/tests/test_cli.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/tests/test_developer_review_report.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/tests/test_discovery_memory.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/tests/test_e2e_fake_provider.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/tests/test_evaluation_harness.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/tests/test_evaluation_matrix.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/tests/test_evaluation_regression_evidence.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/tests/test_functional_qualification.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/tests/test_huge_monorepo_v070.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/tests/test_migration_e2e_v070.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/tests/test_model_routing.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/tests/test_multilang_technical_review.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/tests/test_multistack_devagent_e2e.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/tests/test_multistack_qualification.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/tests/test_packaging_metadata.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/tests/test_plan_verification_normalization.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/tests/test_preservation_contradiction.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/tests/test_production_hardening.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/tests/test_realworld_benchmark.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/tests/test_retrieval.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/tests/test_runtime_sandbox.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/tests/test_safety_workspace.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/tests/test_source_control_publish.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/tests/test_structural_devagent_e2e_v070.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/tests/test_structural_operations.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/tests/test_structured_provider_contract.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/tests/test_tasking_state.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/tests/test_v070_engineering_breadth.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/tests/test_v080_autonomy.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/tests/test_v080_provider_benchmark.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/tests/test_v080_skills_automations.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/tests/test_workspace_environment.py +0 -0
- {devagent_ai-0.8.2 → devagent_ai-0.8.3}/tests/test_worktree.py +0 -0
|
@@ -17,11 +17,11 @@ _CLASSIFIERS: tuple[tuple[TaskType, tuple[str, ...]], ...] = (
|
|
|
17
17
|
(TaskType.TEST_FAILURE, ("test fail", "failing test", "pytest error")),
|
|
18
18
|
(TaskType.RUNTIME_ERROR, ("traceback", "exception", "runtime error", "crash")),
|
|
19
19
|
(TaskType.MIGRATION, ("migration", "migrate ", "schema change", "alembic", "database migration")),
|
|
20
|
-
(TaskType.PERFORMANCE, ("performance", "optimize", "slow", "latency", "n+1")),
|
|
21
|
-
(TaskType.REFACTOR, ("refactor", "restructure", "cleanup")),
|
|
20
|
+
(TaskType.PERFORMANCE, ("performance", "optimize", "slow", "latency", "n+1", "faster", "speed up")),
|
|
21
|
+
(TaskType.REFACTOR, ("refactor", "restructure", "cleanup", "rename ", "move ", "delete obsolete")),
|
|
22
22
|
(TaskType.UNIT_TEST, ("add unit test", "write tests", "test coverage")),
|
|
23
23
|
(TaskType.BUG_FIX, ("fix", "bug", "incorrect", "broken", "regression failure", "regression bug")),
|
|
24
|
-
(TaskType.FEATURE, ("add ", "implement", "support ", "feature")),
|
|
24
|
+
(TaskType.FEATURE, ("add ", "implement", "support ", "feature", "create ")),
|
|
25
25
|
)
|
|
26
26
|
|
|
27
27
|
_HIGH_RISK = {
|
|
@@ -56,24 +56,39 @@ _KNOWN_SECTIONS = _REQUIREMENT_SECTIONS | {
|
|
|
56
56
|
"non-goals",
|
|
57
57
|
"non goals",
|
|
58
58
|
"notes",
|
|
59
|
+
"engineering design",
|
|
60
|
+
"engineering context",
|
|
59
61
|
}
|
|
60
62
|
_DIRECTIVE = re.compile(
|
|
61
|
-
r"^(?:add|implement|support|preserve|keep|ensure|require|must|should|when|do not|don't|"
|
|
62
|
-
r"verify|run|return|raise|allow|prevent|maintain|migrate|refactor|update|fix|handle)\b",
|
|
63
|
+
r"^(?:add|create|implement|support|preserve|keep|ensure|require|must|should|when|do not|don't|"
|
|
64
|
+
r"verify|run|return|raise|allow|prevent|maintain|migrate|refactor|rename|move|delete|update|fix|handle)\b",
|
|
63
65
|
re.IGNORECASE,
|
|
64
66
|
)
|
|
65
67
|
|
|
66
|
-
# Bounded normalization for terse user intent. This
|
|
67
|
-
#
|
|
68
|
-
# grammatical number, and operation wording while preserving identifiers,
|
|
69
|
-
# quoted contracts, values, and explicit constraints. Task policy and repository
|
|
70
|
-
# evidence still provide the verification/safety contract.
|
|
68
|
+
# Bounded normalization for terse user intent. This intentionally fixes common
|
|
69
|
+
# engineering shorthand and spelling without attempting to invent product behavior.
|
|
71
70
|
_OPERATION_ALIASES: tuple[tuple[str, str], ...] = (
|
|
72
71
|
("substraction", "subtraction"),
|
|
73
72
|
("substract", "subtract"),
|
|
74
73
|
("multipy", "multiply"),
|
|
75
74
|
("mutiply", "multiply"),
|
|
75
|
+
("authentification", "authentication"),
|
|
76
|
+
("autorization", "authorization"),
|
|
77
|
+
("loging", "login"),
|
|
76
78
|
)
|
|
79
|
+
_ACRONYMS = {
|
|
80
|
+
"api": "API",
|
|
81
|
+
"csv": "CSV",
|
|
82
|
+
"db": "DB",
|
|
83
|
+
"http": "HTTP",
|
|
84
|
+
"https": "HTTPS",
|
|
85
|
+
"json": "JSON",
|
|
86
|
+
"jwt": "JWT",
|
|
87
|
+
"oauth": "OAuth",
|
|
88
|
+
"sql": "SQL",
|
|
89
|
+
"ui": "UI",
|
|
90
|
+
"url": "URL",
|
|
91
|
+
}
|
|
77
92
|
|
|
78
93
|
|
|
79
94
|
def _classify(text: str) -> TaskType:
|
|
@@ -110,27 +125,66 @@ def _dedupe(items: list[str]) -> list[str]:
|
|
|
110
125
|
return result
|
|
111
126
|
|
|
112
127
|
|
|
113
|
-
def
|
|
114
|
-
|
|
128
|
+
def _section_header(line: str) -> tuple[str, str] | None:
|
|
129
|
+
stripped = line.strip()
|
|
130
|
+
markdown = re.match(r"^#{1,6}\s+(.+?)\s*$", stripped)
|
|
131
|
+
if markdown:
|
|
132
|
+
return markdown.group(1).strip().rstrip(":").lower(), ""
|
|
133
|
+
colon = re.match(r"^([A-Za-z][A-Za-z0-9 _/-]{0,80})\s*:\s*(.*)$", stripped)
|
|
134
|
+
if colon and colon.group(1).strip().lower() in _KNOWN_SECTIONS:
|
|
135
|
+
return colon.group(1).strip().lower(), colon.group(2).strip()
|
|
136
|
+
return None
|
|
115
137
|
|
|
116
|
-
The compiler is intentionally bounded. It may repair shorthand/grammar and
|
|
117
|
-
make an operation explicit, but it must not add product behavior the user did
|
|
118
|
-
not request. Structured/multi-line requirements are left intact.
|
|
119
|
-
"""
|
|
120
138
|
|
|
121
|
-
|
|
122
|
-
|
|
139
|
+
def _extract_goal(requirement: str) -> str:
|
|
140
|
+
"""Prefer an explicit Goal section while preserving ordinary free-form input."""
|
|
141
|
+
|
|
142
|
+
lines = requirement.splitlines()
|
|
143
|
+
for index, raw in enumerate(lines):
|
|
144
|
+
header = _section_header(raw)
|
|
145
|
+
if header is None or header[0] != "goal":
|
|
146
|
+
continue
|
|
147
|
+
_, inline = header
|
|
148
|
+
if inline:
|
|
149
|
+
return _clean_requirement_item(inline)
|
|
150
|
+
collected: list[str] = []
|
|
151
|
+
for candidate in lines[index + 1 :]:
|
|
152
|
+
if _section_header(candidate) is not None:
|
|
153
|
+
break
|
|
154
|
+
if candidate.strip():
|
|
155
|
+
collected.append(_clean_requirement_item(candidate))
|
|
156
|
+
if collected:
|
|
157
|
+
return " ".join(collected)
|
|
158
|
+
return re.sub(r"\s+", " ", requirement).strip()
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
def _polish_plain_goal(value: str) -> str:
|
|
162
|
+
"""Improve readability without changing the requested product semantics."""
|
|
163
|
+
|
|
164
|
+
result = re.sub(r"\s+", " ", value).strip()
|
|
165
|
+
for source, destination in _OPERATION_ALIASES:
|
|
166
|
+
result = re.sub(rf"\b{re.escape(source)}\b", destination, result, flags=re.IGNORECASE)
|
|
167
|
+
result = re.sub(r"^customer\s+(?:need|needs|want|wants)\s+", "Implement ", result, flags=re.IGNORECASE)
|
|
168
|
+
result = re.sub(r"^user\s+(?:need|needs|want|wants)\s+", "Implement ", result, flags=re.IGNORECASE)
|
|
169
|
+
for source, destination in _ACRONYMS.items():
|
|
170
|
+
result = re.sub(rf"\b{source}\b", destination, result, flags=re.IGNORECASE)
|
|
171
|
+
if result:
|
|
172
|
+
result = result[0].upper() + result[1:]
|
|
173
|
+
return result.rstrip(".;")
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
def _normalize_terse_requirement(requirement: str) -> str:
|
|
177
|
+
"""Compile common rough prompts into a clearer bounded engineering request."""
|
|
178
|
+
|
|
179
|
+
raw_value = _extract_goal(requirement)
|
|
180
|
+
value = _polish_plain_goal(raw_value)
|
|
181
|
+
if not value:
|
|
123
182
|
return value
|
|
124
183
|
# An explicit callable name is already a precise user contract; never rename it.
|
|
125
184
|
if re.search(r"\b[A-Za-z_][A-Za-z0-9_]*\s*\(", value):
|
|
126
185
|
return value
|
|
127
186
|
|
|
128
|
-
for source, destination in _OPERATION_ALIASES:
|
|
129
|
-
value = re.sub(rf"\b{re.escape(source)}\b", destination, value, flags=re.IGNORECASE)
|
|
130
|
-
|
|
131
187
|
# Common shorthand from natural prompts such as "addition 2 matrix 2x2".
|
|
132
|
-
# Keep both "matrix" and "matrices" in the normalized contract so
|
|
133
|
-
# deterministic evidence can link either conventional symbol spelling.
|
|
134
188
|
matrix_match = re.search(
|
|
135
189
|
r"\b(add(?:ition)?|sum|subtract(?:ion)?|multiply|multiplication|divide|division)\b"
|
|
136
190
|
r"(?:\s+(?:of|for))?\s+(?:2|two)\s+matrix(?:es)?\s+(\d+x\d+)\b",
|
|
@@ -157,28 +211,19 @@ def _normalize_terse_requirement(requirement: str) -> str:
|
|
|
157
211
|
f"(matrix inputs)"
|
|
158
212
|
)
|
|
159
213
|
|
|
160
|
-
# Repair simple count+noun shorthand without inventing domain behavior.
|
|
161
214
|
value = re.sub(r"\b2\s+matrix\b", "two matrices", value, flags=re.IGNORECASE)
|
|
162
215
|
value = re.sub(r"\b2\s+file\b", "two files", value, flags=re.IGNORECASE)
|
|
163
216
|
value = re.sub(r"\b2\s+test\b", "two tests", value, flags=re.IGNORECASE)
|
|
164
217
|
return value
|
|
165
218
|
|
|
166
219
|
|
|
167
|
-
def _section_header(line: str) -> tuple[str, str] | None:
|
|
168
|
-
stripped = line.strip()
|
|
169
|
-
markdown = re.match(r"^#{1,6}\s+(.+?)\s*$", stripped)
|
|
170
|
-
if markdown:
|
|
171
|
-
return markdown.group(1).strip().rstrip(":").lower(), ""
|
|
172
|
-
colon = re.match(r"^([A-Za-z][A-Za-z0-9 _/-]{0,80})\s*:\s*(.*)$", stripped)
|
|
173
|
-
if colon and colon.group(1).strip().lower() in _KNOWN_SECTIONS:
|
|
174
|
-
return colon.group(1).strip().lower(), colon.group(2).strip()
|
|
175
|
-
return None
|
|
176
|
-
|
|
177
|
-
|
|
178
220
|
def _user_acceptance_items(requirement: str) -> list[str]:
|
|
179
221
|
lines = requirement.splitlines()
|
|
180
222
|
explicit: list[str] = []
|
|
181
223
|
active_section: str | None = None
|
|
224
|
+
recognized_section = False
|
|
225
|
+
nonempty_lines = [line.strip() for line in lines if line.strip()]
|
|
226
|
+
|
|
182
227
|
for raw in lines:
|
|
183
228
|
stripped = raw.strip()
|
|
184
229
|
if not stripped:
|
|
@@ -192,6 +237,7 @@ def _user_acceptance_items(requirement: str) -> list[str]:
|
|
|
192
237
|
|
|
193
238
|
header = _section_header(stripped)
|
|
194
239
|
if header is not None:
|
|
240
|
+
recognized_section = True
|
|
195
241
|
name, inline = header
|
|
196
242
|
active_section = name if name in _REQUIREMENT_SECTIONS else None
|
|
197
243
|
if active_section is not None and inline:
|
|
@@ -221,6 +267,19 @@ def _user_acceptance_items(requirement: str) -> list[str]:
|
|
|
221
267
|
if _DIRECTIVE.match(item) or re.search(r"\b(?:must|should|shall)\b", item, re.IGNORECASE):
|
|
222
268
|
directives.append(item)
|
|
223
269
|
directives = _dedupe(directives)
|
|
270
|
+
|
|
271
|
+
# For a loose multi-line customer note, do not silently discard fragments merely
|
|
272
|
+
# because one line happens to begin with a directive. Preserve the whole intent as
|
|
273
|
+
# one user criterion unless the text is clearly a structured directive list.
|
|
274
|
+
if len(nonempty_lines) > 1 and not recognized_section:
|
|
275
|
+
all_directive_like = all(
|
|
276
|
+
_DIRECTIVE.match(_clean_requirement_item(item))
|
|
277
|
+
or re.search(r"\b(?:must|should|shall)\b", item, re.IGNORECASE)
|
|
278
|
+
for item in nonempty_lines
|
|
279
|
+
)
|
|
280
|
+
if not all_directive_like:
|
|
281
|
+
return [re.sub(r"\s+", " ", requirement).strip()]
|
|
282
|
+
|
|
224
283
|
if directives:
|
|
225
284
|
return directives
|
|
226
285
|
return [re.sub(r"\s+", " ", requirement).strip()]
|
|
@@ -257,8 +316,6 @@ def compile_task(requirement: str) -> TaskSpec:
|
|
|
257
316
|
requires_tests = task_type is not TaskType.BUILD_FAILURE
|
|
258
317
|
|
|
259
318
|
criteria: list[AcceptanceCriterion] = []
|
|
260
|
-
# Structured user requirements remain authoritative. Only an unstructured,
|
|
261
|
-
# terse prompt is compiled into the clearer canonical request.
|
|
262
319
|
user_items = _user_acceptance_items(requirement)
|
|
263
320
|
if len(user_items) == 1 and user_items[0] == raw_goal and goal != raw_goal:
|
|
264
321
|
user_items = [goal]
|
|
@@ -342,14 +399,13 @@ def _matrix_operation_contract(task: TaskSpec, repository: Any) -> None:
|
|
|
342
399
|
"multiplication": "multiply",
|
|
343
400
|
"division": "divide",
|
|
344
401
|
}[operation]
|
|
345
|
-
compact_dimension = dimension.replace("x", "x")
|
|
346
402
|
language = _repository_language(repository)
|
|
347
403
|
if language in {"java", "javascript", "typescript"}:
|
|
348
|
-
symbol = f"{verb}Matrices{
|
|
404
|
+
symbol = f"{verb}Matrices{dimension}"
|
|
349
405
|
elif language in {"csharp", "c#"}:
|
|
350
|
-
symbol = f"{verb.capitalize()}Matrices{
|
|
406
|
+
symbol = f"{verb.capitalize()}Matrices{dimension}"
|
|
351
407
|
else:
|
|
352
|
-
symbol = f"{verb}_matrices_{
|
|
408
|
+
symbol = f"{verb}_matrices_{dimension}"
|
|
353
409
|
|
|
354
410
|
compiled = (
|
|
355
411
|
f"Add {symbol}(a, b) to perform element-wise matrix {operation} "
|
|
@@ -363,8 +419,133 @@ def _matrix_operation_contract(task: TaskSpec, repository: Any) -> None:
|
|
|
363
419
|
user_criteria[0].description = compiled
|
|
364
420
|
|
|
365
421
|
|
|
422
|
+
def _unique_repository_values(repository: Any, field: str) -> list[str]:
|
|
423
|
+
values: list[str] = []
|
|
424
|
+
for component in repository.components:
|
|
425
|
+
for value in getattr(component, field, []):
|
|
426
|
+
if value and value not in values:
|
|
427
|
+
values.append(value)
|
|
428
|
+
return values
|
|
429
|
+
|
|
430
|
+
|
|
431
|
+
def _task_design_defaults(task: TaskSpec) -> list[str]:
|
|
432
|
+
common = [
|
|
433
|
+
"Integrate with the repository's existing architecture and naming conventions instead of creating a parallel pattern.",
|
|
434
|
+
"Keep the implementation bounded to the requested behavior and avoid unrelated refactors.",
|
|
435
|
+
"Preserve behavior outside the explicitly requested scope unless the user states otherwise.",
|
|
436
|
+
]
|
|
437
|
+
if task.requires_tests:
|
|
438
|
+
common.append("Add or update focused regression coverage using the repository's existing test conventions.")
|
|
439
|
+
|
|
440
|
+
if task.task_type in {TaskType.BUG_FIX, TaskType.RUNTIME_ERROR, TaskType.TEST_FAILURE}:
|
|
441
|
+
common.extend(
|
|
442
|
+
[
|
|
443
|
+
"Identify and fix the underlying cause rather than masking the visible symptom.",
|
|
444
|
+
"Prove the failing scenario with regression coverage when the repository supports it.",
|
|
445
|
+
]
|
|
446
|
+
)
|
|
447
|
+
elif task.task_type is TaskType.REFACTOR:
|
|
448
|
+
common.append("Keep externally observable behavior stable while updating references and tests affected by the refactor.")
|
|
449
|
+
elif task.task_type is TaskType.MIGRATION:
|
|
450
|
+
common.extend(
|
|
451
|
+
[
|
|
452
|
+
"Use the repository's existing migration mechanism and preserve compatibility with supported application state.",
|
|
453
|
+
"Provide a forward path plus rollback or an explicitly safe non-reversible strategy; do not invent destructive data policy.",
|
|
454
|
+
]
|
|
455
|
+
)
|
|
456
|
+
elif task.task_type is TaskType.PERFORMANCE:
|
|
457
|
+
common.append("Preserve functional behavior while improving the requested performance concern; do not invent an unrequested numeric target.")
|
|
458
|
+
|
|
459
|
+
lowered = " ".join(
|
|
460
|
+
criterion.description for criterion in task.acceptance_criteria if criterion.source is AcceptanceSource.USER
|
|
461
|
+
).lower()
|
|
462
|
+
if any(term in lowered for term in ("auth", "login", "oauth", "token", "credential", "api key", "secret")):
|
|
463
|
+
common.extend(
|
|
464
|
+
[
|
|
465
|
+
"Use the repository's existing configuration and secret-handling mechanisms; never hardcode credentials.",
|
|
466
|
+
"Do not invent authorization roles, OAuth scopes, account-linking policy, or other security/product decisions absent from the user request.",
|
|
467
|
+
]
|
|
468
|
+
)
|
|
469
|
+
if any(term in lowered for term in ("payment", "billing", "checkout", "subscription")):
|
|
470
|
+
common.append(
|
|
471
|
+
"Do not invent retry counts, fees, cancellation policy, payment state transitions, or other commercial behavior absent from the user request."
|
|
472
|
+
)
|
|
473
|
+
return _dedupe(common)
|
|
474
|
+
|
|
475
|
+
|
|
476
|
+
def _compile_repository_aware_brief(task: TaskSpec, repository: Any) -> None:
|
|
477
|
+
"""Turn user intent into a richer engineering brief without changing user-owned criteria.
|
|
478
|
+
|
|
479
|
+
This brief is supplied to every later DevAgent role through TaskSpec.goal. It may
|
|
480
|
+
add safe engineering defaults and repository facts, but it explicitly does not
|
|
481
|
+
create new user/business requirements. AcceptanceSource.USER criteria remain the
|
|
482
|
+
authoritative statement of what the user asked for.
|
|
483
|
+
"""
|
|
484
|
+
|
|
485
|
+
if "DEVAGENT REQUIREMENT INTELLIGENCE" in task.goal:
|
|
486
|
+
return
|
|
487
|
+
|
|
488
|
+
core_goal = task.goal.strip()
|
|
489
|
+
user_requirements = [
|
|
490
|
+
criterion.description
|
|
491
|
+
for criterion in task.acceptance_criteria
|
|
492
|
+
if criterion.source is AcceptanceSource.USER
|
|
493
|
+
]
|
|
494
|
+
languages = _unique_repository_values(repository, "languages")
|
|
495
|
+
frameworks = _unique_repository_values(repository, "frameworks")
|
|
496
|
+
manifests = _unique_repository_values(repository, "manifests")
|
|
497
|
+
test_locations = _unique_repository_values(repository, "test_locations")
|
|
498
|
+
|
|
499
|
+
trusted_commands: list[str] = []
|
|
500
|
+
for capability in repository.capabilities:
|
|
501
|
+
if capability.trusted:
|
|
502
|
+
command = " ".join(capability.command)
|
|
503
|
+
if command and command not in trusted_commands:
|
|
504
|
+
trusted_commands.append(command)
|
|
505
|
+
|
|
506
|
+
lines = [
|
|
507
|
+
core_goal,
|
|
508
|
+
"",
|
|
509
|
+
"DEVAGENT REQUIREMENT INTELLIGENCE",
|
|
510
|
+
"User intent remains authoritative; the sections below are engineering design guidance, not invented business requirements.",
|
|
511
|
+
"",
|
|
512
|
+
"USER REQUIREMENTS",
|
|
513
|
+
]
|
|
514
|
+
lines.extend(f"- {item}" for item in user_requirements or [core_goal])
|
|
515
|
+
|
|
516
|
+
lines.extend(["", "SAFE ENGINEERING DEFAULTS"])
|
|
517
|
+
lines.extend(f"- {item}" for item in _task_design_defaults(task))
|
|
518
|
+
|
|
519
|
+
repository_lines: list[str] = []
|
|
520
|
+
if languages:
|
|
521
|
+
repository_lines.append("Languages: " + ", ".join(languages[:8]))
|
|
522
|
+
if frameworks:
|
|
523
|
+
repository_lines.append("Frameworks: " + ", ".join(frameworks[:8]))
|
|
524
|
+
if manifests:
|
|
525
|
+
repository_lines.append("Manifests: " + ", ".join(manifests[:10]))
|
|
526
|
+
if test_locations:
|
|
527
|
+
repository_lines.append("Existing test locations: " + ", ".join(test_locations[:10]))
|
|
528
|
+
if trusted_commands:
|
|
529
|
+
repository_lines.append("Evidence-backed verification: " + "; ".join(trusted_commands[:8]))
|
|
530
|
+
if len(repository.components) > 1:
|
|
531
|
+
repository_lines.append(f"Repository structure: {repository.kind} with {len(repository.components)} discovered components")
|
|
532
|
+
|
|
533
|
+
lines.extend(["", "REPOSITORY-DERIVED CONTEXT"])
|
|
534
|
+
lines.extend(f"- {item}" for item in repository_lines or ["Use discovered repository structure and conventions as implementation evidence."])
|
|
535
|
+
|
|
536
|
+
lines.extend(
|
|
537
|
+
[
|
|
538
|
+
"",
|
|
539
|
+
"DESIGN GUARDRAIL",
|
|
540
|
+
"- Do not invent material product, business, security, data-lifecycle, or external-contract behavior that the user did not request.",
|
|
541
|
+
"- If source evidence shows a material ambiguity, prefer a bounded implementation or BLOCKED/PARTIALLY_VERIFIED outcome over silently choosing product policy.",
|
|
542
|
+
]
|
|
543
|
+
)
|
|
544
|
+
task.goal = "\n".join(lines)
|
|
545
|
+
|
|
546
|
+
|
|
366
547
|
def enrich_acceptance_contract(task: TaskSpec, repository: Any) -> TaskSpec:
|
|
367
|
-
"""Compile
|
|
548
|
+
"""Compile repository-aware requirement intelligence and trusted final checks."""
|
|
368
549
|
|
|
369
550
|
_matrix_operation_contract(task, repository)
|
|
370
551
|
seen_commands: set[tuple[str, ...]] = set()
|
|
@@ -382,4 +563,6 @@ def enrich_acceptance_contract(task: TaskSpec, repository: Any) -> TaskSpec:
|
|
|
382
563
|
source=AcceptanceSource.REPOSITORY,
|
|
383
564
|
verification_command=capability.command,
|
|
384
565
|
)
|
|
566
|
+
|
|
567
|
+
_compile_repository_aware_brief(task, repository)
|
|
385
568
|
return task
|
|
@@ -68,6 +68,7 @@ tests/test_production_hardening.py
|
|
|
68
68
|
tests/test_production_v040.py
|
|
69
69
|
tests/test_realworld_benchmark.py
|
|
70
70
|
tests/test_requirement_compiler_v082.py
|
|
71
|
+
tests/test_requirement_intelligence_v083.py
|
|
71
72
|
tests/test_retrieval.py
|
|
72
73
|
tests/test_runtime_sandbox.py
|
|
73
74
|
tests/test_safety_workspace.py
|
|
@@ -62,7 +62,7 @@ def test_release_version_is_semver_and_consistent() -> None:
|
|
|
62
62
|
version = match.group(1)
|
|
63
63
|
assert re.fullmatch(r"\d+\.\d+\.\d+", version)
|
|
64
64
|
assert version == __version__
|
|
65
|
-
assert version == "0.8.
|
|
65
|
+
assert version == "0.8.3"
|
|
66
66
|
|
|
67
67
|
|
|
68
68
|
def test_release_workflow_requires_green_exact_main_revision() -> None:
|
|
@@ -64,19 +64,17 @@ def test_rough_matrix_prompt_becomes_precise_python_contract() -> None:
|
|
|
64
64
|
assert task.goal == "Add a matrix addition function for two 2x2 matrices (matrix inputs)"
|
|
65
65
|
|
|
66
66
|
enrich_acceptance_contract(task, _repo())
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
"for two 2x2 matrices"
|
|
70
|
-
)
|
|
67
|
+
core = "Add add_matrices_2x2(a, b) to perform element-wise matrix addition for two 2x2 matrices"
|
|
68
|
+
assert task.goal.startswith(core + "\n\nDEVAGENT REQUIREMENT INTELLIGENCE")
|
|
71
69
|
user = [item for item in task.acceptance_criteria if item.source is AcceptanceSource.USER]
|
|
72
|
-
assert [item.description for item in user] == [
|
|
70
|
+
assert [item.description for item in user] == [core]
|
|
73
71
|
|
|
74
72
|
|
|
75
73
|
def test_requirement_compiler_preserves_explicit_user_callable() -> None:
|
|
76
74
|
requirement = "Add matrix_sum(a, b) for two 2x2 matrices"
|
|
77
75
|
task = compile_task(requirement)
|
|
78
76
|
enrich_acceptance_contract(task, _repo())
|
|
79
|
-
assert task.goal
|
|
77
|
+
assert task.goal.startswith(requirement + "\n\nDEVAGENT REQUIREMENT INTELLIGENCE")
|
|
80
78
|
user = [item for item in task.acceptance_criteria if item.source is AcceptanceSource.USER]
|
|
81
79
|
assert [item.description for item in user] == [requirement]
|
|
82
80
|
|
|
@@ -84,12 +82,11 @@ def test_requirement_compiler_preserves_explicit_user_callable() -> None:
|
|
|
84
82
|
def test_common_typo_is_compiled_without_inventing_behavior() -> None:
|
|
85
83
|
task = compile_task("add new function substraction 2 matrix 2x2")
|
|
86
84
|
enrich_acceptance_contract(task, _repo())
|
|
87
|
-
assert task.goal
|
|
88
|
-
"Add subtract_matrices_2x2(a, b) to perform element-wise matrix subtraction "
|
|
89
|
-
"for two 2x2 matrices"
|
|
85
|
+
assert task.goal.startswith(
|
|
86
|
+
"Add subtract_matrices_2x2(a, b) to perform element-wise matrix subtraction for two 2x2 matrices"
|
|
90
87
|
)
|
|
91
88
|
assert "mutation" not in task.goal.lower()
|
|
92
|
-
assert "invalid" not in task.goal.lower()
|
|
89
|
+
assert "invalid-input" not in task.goal.lower()
|
|
93
90
|
|
|
94
91
|
|
|
95
92
|
def test_repository_language_selects_conventional_java_callable() -> None:
|
|
@@ -0,0 +1,170 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from pathlib import Path
|
|
4
|
+
|
|
5
|
+
from devagent.cli import _read_requirement_file
|
|
6
|
+
from devagent.models import (
|
|
7
|
+
AcceptanceSource,
|
|
8
|
+
Capability,
|
|
9
|
+
CapabilityProvenance,
|
|
10
|
+
Component,
|
|
11
|
+
RepositoryModel,
|
|
12
|
+
TaskType,
|
|
13
|
+
)
|
|
14
|
+
from devagent.tasking import compile_task, enrich_acceptance_contract
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def _repo(*, language: str = "python", framework: str = "FastAPI") -> RepositoryModel:
|
|
18
|
+
return RepositoryModel(
|
|
19
|
+
root="/repo",
|
|
20
|
+
kind="single-component",
|
|
21
|
+
components=[
|
|
22
|
+
Component(
|
|
23
|
+
path=".",
|
|
24
|
+
languages=[language],
|
|
25
|
+
frameworks=[framework] if framework else [],
|
|
26
|
+
manifests=["pyproject.toml" if language == "python" else "pom.xml"],
|
|
27
|
+
test_locations=["tests"],
|
|
28
|
+
capabilities=[
|
|
29
|
+
Capability(
|
|
30
|
+
kind="test",
|
|
31
|
+
command=("python", "-m", "pytest", "-q"),
|
|
32
|
+
source="pyproject.toml",
|
|
33
|
+
provenance=CapabilityProvenance.EXPLICIT,
|
|
34
|
+
)
|
|
35
|
+
],
|
|
36
|
+
)
|
|
37
|
+
],
|
|
38
|
+
facts=[],
|
|
39
|
+
git_head="abc",
|
|
40
|
+
)
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _user_criteria(task) -> list[str]:
|
|
44
|
+
return [
|
|
45
|
+
item.description
|
|
46
|
+
for item in task.acceptance_criteria
|
|
47
|
+
if item.source is AcceptanceSource.USER
|
|
48
|
+
]
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def test_rough_terminal_prompt_becomes_repository_aware_engineering_brief() -> None:
|
|
52
|
+
task = compile_task("add login google")
|
|
53
|
+
assert task.task_type is TaskType.FEATURE
|
|
54
|
+
|
|
55
|
+
enrich_acceptance_contract(task, _repo())
|
|
56
|
+
|
|
57
|
+
assert task.goal.startswith("Add login google\n\nDEVAGENT REQUIREMENT INTELLIGENCE")
|
|
58
|
+
assert "USER REQUIREMENTS\n- Add login google" in task.goal
|
|
59
|
+
assert "SAFE ENGINEERING DEFAULTS" in task.goal
|
|
60
|
+
assert "existing architecture and naming conventions" in task.goal
|
|
61
|
+
assert "never hardcode credentials" in task.goal
|
|
62
|
+
assert "Frameworks: FastAPI" in task.goal
|
|
63
|
+
assert "Languages: python" in task.goal
|
|
64
|
+
assert "Evidence-backed verification: python -m pytest -q" in task.goal
|
|
65
|
+
assert "Do not invent material product, business, security" in task.goal
|
|
66
|
+
assert _user_criteria(task) == ["Add login google"]
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def test_terminal_and_file_path_feed_identical_requirement_intelligence(tmp_path: Path) -> None:
|
|
70
|
+
text = "add CSV export for filtered reports and preserve JSON export"
|
|
71
|
+
requirement = tmp_path / "customer-request.anything"
|
|
72
|
+
requirement.write_text(text + "\n", encoding="utf-8")
|
|
73
|
+
|
|
74
|
+
direct = compile_task(text)
|
|
75
|
+
from_file = compile_task(_read_requirement_file(requirement))
|
|
76
|
+
enrich_acceptance_contract(direct, _repo())
|
|
77
|
+
enrich_acceptance_contract(from_file, _repo())
|
|
78
|
+
|
|
79
|
+
assert direct.goal == from_file.goal
|
|
80
|
+
assert _user_criteria(direct) == _user_criteria(from_file)
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def test_unstructured_multiline_customer_note_does_not_lose_fragments() -> None:
|
|
84
|
+
requirement = """customer need export
|
|
85
|
+
csv maybe
|
|
86
|
+
filtered data
|
|
87
|
+
keep json old one
|
|
88
|
+
button ui
|
|
89
|
+
"""
|
|
90
|
+
task = compile_task(requirement)
|
|
91
|
+
user = " ".join(_user_criteria(task)).lower()
|
|
92
|
+
|
|
93
|
+
assert "export" in user
|
|
94
|
+
assert "csv" in user
|
|
95
|
+
assert "filtered data" in user
|
|
96
|
+
assert "json" in user
|
|
97
|
+
assert "button ui" in user
|
|
98
|
+
|
|
99
|
+
enrich_acceptance_contract(task, _repo())
|
|
100
|
+
assert "CSV" in task.goal
|
|
101
|
+
assert "JSON" in task.goal
|
|
102
|
+
assert "UI" in task.goal
|
|
103
|
+
assert "not invented business requirements" in task.goal
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def test_structured_file_goal_and_explicit_requirements_remain_authoritative() -> None:
|
|
107
|
+
requirement = """Goal: Add CSV export for filtered reports
|
|
108
|
+
|
|
109
|
+
Requirements:
|
|
110
|
+
- Preserve existing JSON export behavior
|
|
111
|
+
- Export the currently filtered result set
|
|
112
|
+
|
|
113
|
+
Constraints:
|
|
114
|
+
- Do not change the existing JSON API
|
|
115
|
+
"""
|
|
116
|
+
task = compile_task(requirement)
|
|
117
|
+
assert task.goal == "Add CSV export for filtered reports"
|
|
118
|
+
assert _user_criteria(task) == [
|
|
119
|
+
"Preserve existing JSON export behavior",
|
|
120
|
+
"Export the currently filtered result set",
|
|
121
|
+
"Do not change the existing JSON API",
|
|
122
|
+
]
|
|
123
|
+
|
|
124
|
+
enrich_acceptance_contract(task, _repo())
|
|
125
|
+
assert task.goal.startswith("Add CSV export for filtered reports\n\nDEVAGENT REQUIREMENT INTELLIGENCE")
|
|
126
|
+
for item in _user_criteria(task):
|
|
127
|
+
assert f"- {item}" in task.goal
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def test_requirement_intelligence_does_not_invent_payment_policy() -> None:
|
|
131
|
+
task = compile_task("handle payment failure")
|
|
132
|
+
enrich_acceptance_contract(task, _repo())
|
|
133
|
+
lowered = task.goal.lower()
|
|
134
|
+
|
|
135
|
+
# Payment-specific safety guidance is added only because the request is about
|
|
136
|
+
# payment. It guides implementation without becoming a fabricated USER criterion.
|
|
137
|
+
assert "do not invent retry counts, fees, cancellation policy" in lowered
|
|
138
|
+
assert _user_criteria(task) == ["Handle payment failure"]
|
|
139
|
+
assert not any("three retries" in item.lower() for item in _user_criteria(task))
|
|
140
|
+
assert not any("cancel" in item.lower() for item in _user_criteria(task))
|
|
141
|
+
assert not any("fee" in item.lower() for item in _user_criteria(task))
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
def test_generic_prompt_does_not_gain_unrelated_domain_retrieval_terms() -> None:
|
|
145
|
+
task = compile_task("add report search")
|
|
146
|
+
enrich_acceptance_contract(task, _repo())
|
|
147
|
+
lowered = task.goal.lower()
|
|
148
|
+
|
|
149
|
+
assert "retry counts" not in lowered
|
|
150
|
+
assert "oauth scopes" not in lowered
|
|
151
|
+
assert "payment state transitions" not in lowered
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
def test_performance_shorthand_gets_safe_design_without_fake_target() -> None:
|
|
155
|
+
task = compile_task("make checkout faster")
|
|
156
|
+
assert task.task_type is TaskType.PERFORMANCE
|
|
157
|
+
enrich_acceptance_contract(task, _repo())
|
|
158
|
+
|
|
159
|
+
assert "Preserve functional behavior" in task.goal
|
|
160
|
+
assert "do not invent an unrequested numeric target" in task.goal
|
|
161
|
+
assert _user_criteria(task) == ["Make checkout faster"]
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
def test_detailed_callable_contract_is_preserved_inside_new_design_brief() -> None:
|
|
165
|
+
requirement = "Add calculate_total(items) and preserve calculate_tax behavior"
|
|
166
|
+
task = compile_task(requirement)
|
|
167
|
+
enrich_acceptance_contract(task, _repo())
|
|
168
|
+
|
|
169
|
+
assert task.goal.startswith(requirement + "\n\nDEVAGENT REQUIREMENT INTELLIGENCE")
|
|
170
|
+
assert _user_criteria(task) == [requirement]
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|