devagent-ai 0.8.0__tar.gz → 0.8.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {devagent_ai-0.8.0/devagent_ai.egg-info → devagent_ai-0.8.2}/PKG-INFO +139 -13
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/README.md +139 -13
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent/__init__.py +1 -1
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent/models.py +13 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent/tasking.py +128 -4
- {devagent_ai-0.8.0 → devagent_ai-0.8.2/devagent_ai.egg-info}/PKG-INFO +139 -13
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent_ai.egg-info/SOURCES.txt +2 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/pyproject.toml +1 -1
- devagent_ai-0.8.2/tests/test_plan_verification_normalization.py +176 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_production_v040.py +1 -1
- devagent_ai-0.8.2/tests/test_requirement_compiler_v082.py +185 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/LICENSE +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/NOTICE +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/agent/__init__.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/agent/llm.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/agent/loop.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/agent/memory.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/agent/prompts.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/agent/tools.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent/__main__.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent/artifacts.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent/automations.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent/autonomy.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent/browser.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent/cli.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent/config.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent/discovery.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent/evaluation.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent/memory.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent/orchestrator.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent/provider_benchmark.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent/providers.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent/qualification.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent/realworld.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent/report.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent/retrieval.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent/routing.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent/runtime.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent/safety.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent/skills.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent/source_control.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent/state_machine.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent/technical_review.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent/workspace.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent/worktree.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent_ai.egg-info/dependency_links.txt +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent_ai.egg-info/entry_points.txt +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent_ai.egg-info/requires.txt +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent_ai.egg-info/top_level.txt +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/setup.cfg +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_acceptance_contract.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_benchmark_catalog.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_browser_verification.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_capability_discovery.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_cli.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_developer_review_report.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_discovery_memory.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_e2e_fake_provider.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_evaluation_harness.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_evaluation_matrix.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_evaluation_regression_evidence.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_functional_qualification.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_huge_monorepo_v070.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_migration_e2e_v070.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_model_routing.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_multilang_technical_review.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_multistack_devagent_e2e.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_multistack_qualification.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_packaging_metadata.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_preservation_contradiction.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_production_hardening.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_realworld_benchmark.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_retrieval.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_runtime_sandbox.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_safety_workspace.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_source_control_publish.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_structural_devagent_e2e_v070.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_structural_operations.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_structured_provider_contract.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_tasking_state.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_v070_engineering_breadth.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_v080_autonomy.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_v080_provider_benchmark.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_v080_skills_automations.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_workspace_environment.py +0 -0
- {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_worktree.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: devagent-ai
|
|
3
|
-
Version: 0.8.
|
|
3
|
+
Version: 0.8.2
|
|
4
4
|
Summary: Evidence-driven local autonomous software engineering agent
|
|
5
5
|
Author: Tom Ha
|
|
6
6
|
Maintainer: Tom Ha
|
|
@@ -216,6 +216,112 @@ devagent --input ../specs/release.requirement
|
|
|
216
216
|
|
|
217
217
|
Binary data, invalid UTF-8, secret-like paths, and files above the input-size bound are refused.
|
|
218
218
|
|
|
219
|
+
## Practical examples
|
|
220
|
+
|
|
221
|
+
DevAgent is intended for real repository work, not only one-line code generation. Run it from the repository you want to change and describe the engineering outcome you need.
|
|
222
|
+
|
|
223
|
+
### 1. Fix a bug and prove the regression is covered
|
|
224
|
+
|
|
225
|
+
```bash
|
|
226
|
+
cd my-service
|
|
227
|
+
devagent "Fix the websocket reconnect bug that duplicates subscriptions after a network drop. Add a regression test and keep the public API unchanged."
|
|
228
|
+
```
|
|
229
|
+
|
|
230
|
+
**Benefit:** DevAgent first discovers the relevant implementation and tests, turns the request into explicit acceptance criteria, makes a bounded patch, runs repository-supported verification, independently reviews the final diff, and only reports `VERIFIED` when the required evidence supports it.
|
|
231
|
+
|
|
232
|
+
### 2. Add a feature on a dedicated branch
|
|
233
|
+
|
|
234
|
+
```bash
|
|
235
|
+
cd my-app
|
|
236
|
+
devagent \
|
|
237
|
+
--publish-branch feature/csv-export \
|
|
238
|
+
"Add CSV export for filtered reports. Preserve the existing JSON export behavior and add tests."
|
|
239
|
+
```
|
|
240
|
+
|
|
241
|
+
**Benefit:** a verified change can be committed and pushed to the requested feature branch while DevAgent stops before PR creation or merge, leaving integration control with the developer or repository owner.
|
|
242
|
+
|
|
243
|
+
### 3. Give DevAgent a longer product or customer requirement
|
|
244
|
+
|
|
245
|
+
```bash
|
|
246
|
+
cd my-repo
|
|
247
|
+
devagent --input requirements/customer-billing-retry.md
|
|
248
|
+
```
|
|
249
|
+
|
|
250
|
+
The input can be any bounded UTF-8 text file; it does not need a special extension or DevAgent-specific format.
|
|
251
|
+
|
|
252
|
+
**Benefit:** long requirements stay in a reviewable file instead of being compressed into a short prompt, while DevAgent still derives bounded implementation and verification work from repository evidence.
|
|
253
|
+
|
|
254
|
+
### 4. Perform a refactor that includes rename/move/delete operations
|
|
255
|
+
|
|
256
|
+
```bash
|
|
257
|
+
cd my-repo
|
|
258
|
+
devagent "Rename LegacyOrderService to OrderService, move it into the services package, update all references, remove the obsolete module, and preserve behavior."
|
|
259
|
+
```
|
|
260
|
+
|
|
261
|
+
**Benefit:** structural changes go through backup-first workspace operations, path/scope checks, repository verification, and final-diff review instead of uncontrolled file manipulation.
|
|
262
|
+
|
|
263
|
+
### 5. Change a database schema with forward/rollback verification
|
|
264
|
+
|
|
265
|
+
```bash
|
|
266
|
+
cd my-python-service
|
|
267
|
+
devagent "Add a nullable status column to the SQLite orders table, provide a forward and rollback migration, update the data-access layer, and verify both migration directions."
|
|
268
|
+
```
|
|
269
|
+
|
|
270
|
+
**Benefit:** migration work can be treated as high-risk engineering work with explicit acceptance evidence instead of assuming that a generated migration is correct because it looks plausible. The current qualified production fixture covers SQLite forward/rollback migration behavior; broader PostgreSQL/MySQL coverage remains an external-validation area.
|
|
271
|
+
|
|
272
|
+
### 6. Work in Java or .NET repositories
|
|
273
|
+
|
|
274
|
+
```bash
|
|
275
|
+
cd my-java-service
|
|
276
|
+
devagent "Add validation for duplicate customer IDs in this Maven service and add the appropriate JUnit regression test."
|
|
277
|
+
```
|
|
278
|
+
|
|
279
|
+
```bash
|
|
280
|
+
cd my-dotnet-service
|
|
281
|
+
devagent "Fix the null-handling bug in the order import path and verify the .NET project still builds successfully."
|
|
282
|
+
```
|
|
283
|
+
|
|
284
|
+
**Benefit:** DevAgent discovers repository-native Maven/Gradle and .NET project evidence instead of forcing every repository through a Python-centric workflow.
|
|
285
|
+
|
|
286
|
+
### 7. Keep all changes local for inspection
|
|
287
|
+
|
|
288
|
+
```bash
|
|
289
|
+
cd my-repo
|
|
290
|
+
devagent --no-publish "Refactor retry handling to remove duplicate logic and keep behavior unchanged."
|
|
291
|
+
```
|
|
292
|
+
|
|
293
|
+
**Benefit:** you still get implementation, verification, independent review, and the engineering report, but DevAgent does not commit or push the result.
|
|
294
|
+
|
|
295
|
+
### 8. Use the model/provider you prefer
|
|
296
|
+
|
|
297
|
+
```bash
|
|
298
|
+
# Configure once
|
|
299
|
+
devagent setup --provider anthropic --model YOUR_MODEL
|
|
300
|
+
export ANTHROPIC_API_KEY=...
|
|
301
|
+
|
|
302
|
+
# Then use the same DevAgent engineering workflow
|
|
303
|
+
devagent "Fix the failing checkout integration test without weakening the assertion."
|
|
304
|
+
```
|
|
305
|
+
|
|
306
|
+
You can similarly configure OpenAI, Gemini, Grok/xAI, or an OpenAI-compatible endpoint.
|
|
307
|
+
|
|
308
|
+
**Benefit:** the model supplies reasoning, while DevAgent keeps the same deterministic acceptance, safety, verification, reporting, and publication rules around it.
|
|
309
|
+
|
|
310
|
+
## What DevAgent adds around an AI coding model
|
|
311
|
+
|
|
312
|
+
| Common engineering risk | DevAgent behavior |
|
|
313
|
+
| --- | --- |
|
|
314
|
+
| The model says “done” without enough proof | Required acceptance criteria remain `UNPROVEN` or the run becomes `PARTIALLY_VERIFIED` / `BLOCKED` instead of falsely claiming success. |
|
|
315
|
+
| A patch touches unrelated code | Evidence gathering, explicit scope, minimal-change planning, and independent diff review constrain the change. |
|
|
316
|
+
| Existing developer work is damaged | Clean repositories use isolated worktrees by default; dirty tracked/untracked developer work is protected; files are backed up before first modification. |
|
|
317
|
+
| Tests passed before a later edit | Verification is revision-aware, so older successful evidence does not prove a newer tree. |
|
|
318
|
+
| A generated change breaks the build or tests | DevAgent runs repository-supported targeted/broad checks and can diagnose/replan before final verification. |
|
|
319
|
+
| An agent pushes directly to a protected primary branch | Starting from `main`, `master`, or `trunk` causes DevAgent to work on a safe branch; runtime DevAgent does not merge or deploy. |
|
|
320
|
+
| You are locked to one model vendor | OpenAI, Claude, Gemini, Grok/xAI, and compatible endpoints can use the same engineering harness. |
|
|
321
|
+
| It is hard to audit what the agent actually did | DevAgent emits an engineering report with decisions, changed symbols, tests, acceptance evidence, verification, failures, gaps, and source-control status. |
|
|
322
|
+
|
|
323
|
+
The goal is not to replace developer judgment. The goal is to make autonomous engineering work **bounded, reviewable, reproducible, and harder to falsely declare complete**.
|
|
324
|
+
|
|
219
325
|
Useful commands:
|
|
220
326
|
|
|
221
327
|
```bash
|
|
@@ -230,7 +336,7 @@ devagent benchmark --help
|
|
|
230
336
|
|
|
231
337
|
### Pinned real-world benchmark
|
|
232
338
|
|
|
233
|
-
DevAgent
|
|
339
|
+
DevAgent includes an opt-in benchmark runner for pinned GitHub repositories. A benchmark case injects a deterministic defect into an exact commit and uses an **external oracle** before and after DevAgent. This avoids treating DevAgent's own report as the benchmark oracle.
|
|
234
340
|
|
|
235
341
|
```bash
|
|
236
342
|
devagent benchmark \
|
|
@@ -278,7 +384,7 @@ DevAgent uses defense-in-depth controls around repository modification, command
|
|
|
278
384
|
- protected targets are refused and force push is never used;
|
|
279
385
|
- no runtime PR, merge, rebase, force-push, or deployment automation.
|
|
280
386
|
|
|
281
|
-
DevAgent
|
|
387
|
+
On Linux, DevAgent can execute engineering commands inside a bubblewrap-based operating-system sandbox. Production qualification exercises required sandbox mode with network access denied. Required mode fails closed when isolation cannot be established rather than silently falling back. Review the report and pushed branch before integrating customer or production code: sandboxing reduces execution risk, but it does not make arbitrary generated changes universally safe.
|
|
282
388
|
|
|
283
389
|
## Repository intelligence and verification
|
|
284
390
|
|
|
@@ -298,13 +404,20 @@ Verification can include baseline tests, targeted tests, component/broad checks,
|
|
|
298
404
|
|
|
299
405
|
## Production qualification
|
|
300
406
|
|
|
301
|
-
DevAgent 0.
|
|
407
|
+
DevAgent 0.8.0 uses cumulative production qualification rather than replacing older evidence with a smaller new suite.
|
|
408
|
+
|
|
409
|
+
- **v4 — 70 required cases** covering end-to-end engineering behavior, acceptance truthfulness, task/risk scope, provider contracts and parity, model routing, worktree and Git publication safety, CLI input, review/repair loops, report/evaluation integrity, release integrity, large-repository behavior, structural refactors, Java/.NET discovery and execution, SQLite migration forward/rollback, and real repository-native stacks.
|
|
410
|
+
- **v5 — 9 required autonomy cases** covering bounded parallel coordination, dirty-source refusal, real isolated parallel DevAgent runs, bounded/relevant skills and provider injection, automation overlap claim/recovery, and provider-benchmark deduplication, live structured-contract behavior, and secret redaction.
|
|
411
|
+
|
|
412
|
+
The v0.8 merge commit on `main` passed both catalogs in required Linux sandbox mode:
|
|
302
413
|
|
|
303
414
|
```text
|
|
304
|
-
|
|
415
|
+
v4: 70/70 passed
|
|
416
|
+
v5: 9/9 passed
|
|
417
|
+
combined: 79/79 passed
|
|
305
418
|
```
|
|
306
419
|
|
|
307
|
-
|
|
420
|
+
The qualification environment exercises real local toolchains for:
|
|
308
421
|
|
|
309
422
|
```text
|
|
310
423
|
Python / pytest
|
|
@@ -312,21 +425,30 @@ Node + TypeScript repository discovery
|
|
|
312
425
|
Go
|
|
313
426
|
Rust / Cargo
|
|
314
427
|
C++ / Make
|
|
428
|
+
Java / Maven
|
|
429
|
+
.NET build
|
|
430
|
+
SQLite migration forward + rollback
|
|
315
431
|
```
|
|
316
432
|
|
|
317
|
-
Run the release qualification locally on a machine with
|
|
433
|
+
Run the same release qualification catalogs locally on a machine with the required toolchains:
|
|
318
434
|
|
|
319
435
|
```bash
|
|
436
|
+
DEVAGENT_SANDBOX=required DEVAGENT_NETWORK=deny \
|
|
320
437
|
python -m devagent.qualification \
|
|
321
438
|
--catalog evaluation/benchmark_v4.json \
|
|
322
439
|
--report .devagent/production-qualification-v4.json
|
|
440
|
+
|
|
441
|
+
DEVAGENT_SANDBOX=required DEVAGENT_NETWORK=deny \
|
|
442
|
+
python -m devagent.qualification \
|
|
443
|
+
--catalog evaluation/benchmark_v5.json \
|
|
444
|
+
--report .devagent/production-qualification-v5.json
|
|
323
445
|
```
|
|
324
446
|
|
|
325
|
-
Production CI runs Python 3.10/3.11/3.12, a clean wheel install, and
|
|
447
|
+
Production CI also runs Python 3.10/3.11/3.12, a clean wheel build/install, real bubblewrap sandbox smoke, and both qualification catalogs. Qualification JSON is retained as CI evidence.
|
|
326
448
|
|
|
327
|
-
**100% qualified means 100% of
|
|
449
|
+
**100% qualified means 100% of these explicit catalogs passed on that revision and environment.** It does not mean mathematical correctness for every unseen repository, environment, model response, language, provider, or engineering task, and it is not a claim that DevAgent is universally superior to every hosted coding platform.
|
|
328
450
|
|
|
329
|
-
See [docs/production-readiness.md](docs/production-readiness.md) for the
|
|
451
|
+
See [docs/production-readiness.md](docs/production-readiness.md) for the project's earlier readiness assessment and its explicit limitations.
|
|
330
452
|
|
|
331
453
|
## Provider architecture
|
|
332
454
|
|
|
@@ -380,11 +502,15 @@ Automated provider tests normally use deterministic or mocked clients and do not
|
|
|
380
502
|
|
|
381
503
|
## Project status
|
|
382
504
|
|
|
383
|
-
DevAgent 0.
|
|
505
|
+
DevAgent 0.8.0 is **beta software with a verified core release baseline**. The exact v0.8 merge revision on `main` passed Production CI across Python 3.10/3.11/3.12, clean wheel installation, real Linux bubblewrap sandbox execution, production qualification v4 (**70/70**), and autonomy qualification v5 (**9/9**), for **79/79 cumulative required qualification cases**.
|
|
506
|
+
|
|
507
|
+
The current core includes evidence-backed `VERIFIED` / `PARTIALLY_VERIFIED` / `BLOCKED` outcomes, backup-first editing, isolated worktrees, bounded structural file operations, repository-native verification, independent review, safe branch publication, provider/model choice, Java and .NET engineering discovery/execution, SQLite migration forward/rollback verification, large-monorepo deep-manifest discovery, bounded parallel agents, repository-local skills, foreground automations, Linux OS sandboxing, bounded browser/local-UI verification, and real-provider structured-contract benchmarking.
|
|
508
|
+
|
|
509
|
+
These results are **bounded engineering claims**, not universal-correctness or market-superiority claims. They are tied to explicit qualification cases, pinned revisions, deterministic fixtures/external oracles where applicable, and the environments actually exercised by CI.
|
|
384
510
|
|
|
385
|
-
Remaining
|
|
511
|
+
Remaining work is primarily **breadth and external validation**, not missing core architecture: a larger public corpus of pinned upstream repositories and tasks; broader browser/UI coverage across dynamic applications and multiple browser environments; a wider Java/Gradle, .NET test-framework, and PostgreSQL/MySQL migration matrix beyond the current qualified fixtures; larger and more diverse monorepo stress cases beyond the current >12,000-file deep-manifest case; more real-world multi-agent workload studies; and continuous paid real-provider benchmarking across a broader set of model/provider combinations. GitHub branch protection/rulesets are external repository settings and must be configured separately; DevAgent does not claim to configure them itself.
|
|
386
512
|
|
|
387
|
-
The project intentionally prioritizes trustworthy outcomes over feature count.
|
|
513
|
+
The project intentionally prioritizes trustworthy outcomes, reproducible evidence, and safe engineering behavior over feature count or unsupported "best agent" claims.
|
|
388
514
|
|
|
389
515
|
## Contributing
|
|
390
516
|
|
|
@@ -185,6 +185,112 @@ devagent --input ../specs/release.requirement
|
|
|
185
185
|
|
|
186
186
|
Binary data, invalid UTF-8, secret-like paths, and files above the input-size bound are refused.
|
|
187
187
|
|
|
188
|
+
## Practical examples
|
|
189
|
+
|
|
190
|
+
DevAgent is intended for real repository work, not only one-line code generation. Run it from the repository you want to change and describe the engineering outcome you need.
|
|
191
|
+
|
|
192
|
+
### 1. Fix a bug and prove the regression is covered
|
|
193
|
+
|
|
194
|
+
```bash
|
|
195
|
+
cd my-service
|
|
196
|
+
devagent "Fix the websocket reconnect bug that duplicates subscriptions after a network drop. Add a regression test and keep the public API unchanged."
|
|
197
|
+
```
|
|
198
|
+
|
|
199
|
+
**Benefit:** DevAgent first discovers the relevant implementation and tests, turns the request into explicit acceptance criteria, makes a bounded patch, runs repository-supported verification, independently reviews the final diff, and only reports `VERIFIED` when the required evidence supports it.
|
|
200
|
+
|
|
201
|
+
### 2. Add a feature on a dedicated branch
|
|
202
|
+
|
|
203
|
+
```bash
|
|
204
|
+
cd my-app
|
|
205
|
+
devagent \
|
|
206
|
+
--publish-branch feature/csv-export \
|
|
207
|
+
"Add CSV export for filtered reports. Preserve the existing JSON export behavior and add tests."
|
|
208
|
+
```
|
|
209
|
+
|
|
210
|
+
**Benefit:** a verified change can be committed and pushed to the requested feature branch while DevAgent stops before PR creation or merge, leaving integration control with the developer or repository owner.
|
|
211
|
+
|
|
212
|
+
### 3. Give DevAgent a longer product or customer requirement
|
|
213
|
+
|
|
214
|
+
```bash
|
|
215
|
+
cd my-repo
|
|
216
|
+
devagent --input requirements/customer-billing-retry.md
|
|
217
|
+
```
|
|
218
|
+
|
|
219
|
+
The input can be any bounded UTF-8 text file; it does not need a special extension or DevAgent-specific format.
|
|
220
|
+
|
|
221
|
+
**Benefit:** long requirements stay in a reviewable file instead of being compressed into a short prompt, while DevAgent still derives bounded implementation and verification work from repository evidence.
|
|
222
|
+
|
|
223
|
+
### 4. Perform a refactor that includes rename/move/delete operations
|
|
224
|
+
|
|
225
|
+
```bash
|
|
226
|
+
cd my-repo
|
|
227
|
+
devagent "Rename LegacyOrderService to OrderService, move it into the services package, update all references, remove the obsolete module, and preserve behavior."
|
|
228
|
+
```
|
|
229
|
+
|
|
230
|
+
**Benefit:** structural changes go through backup-first workspace operations, path/scope checks, repository verification, and final-diff review instead of uncontrolled file manipulation.
|
|
231
|
+
|
|
232
|
+
### 5. Change a database schema with forward/rollback verification
|
|
233
|
+
|
|
234
|
+
```bash
|
|
235
|
+
cd my-python-service
|
|
236
|
+
devagent "Add a nullable status column to the SQLite orders table, provide a forward and rollback migration, update the data-access layer, and verify both migration directions."
|
|
237
|
+
```
|
|
238
|
+
|
|
239
|
+
**Benefit:** migration work can be treated as high-risk engineering work with explicit acceptance evidence instead of assuming that a generated migration is correct because it looks plausible. The current qualified production fixture covers SQLite forward/rollback migration behavior; broader PostgreSQL/MySQL coverage remains an external-validation area.
|
|
240
|
+
|
|
241
|
+
### 6. Work in Java or .NET repositories
|
|
242
|
+
|
|
243
|
+
```bash
|
|
244
|
+
cd my-java-service
|
|
245
|
+
devagent "Add validation for duplicate customer IDs in this Maven service and add the appropriate JUnit regression test."
|
|
246
|
+
```
|
|
247
|
+
|
|
248
|
+
```bash
|
|
249
|
+
cd my-dotnet-service
|
|
250
|
+
devagent "Fix the null-handling bug in the order import path and verify the .NET project still builds successfully."
|
|
251
|
+
```
|
|
252
|
+
|
|
253
|
+
**Benefit:** DevAgent discovers repository-native Maven/Gradle and .NET project evidence instead of forcing every repository through a Python-centric workflow.
|
|
254
|
+
|
|
255
|
+
### 7. Keep all changes local for inspection
|
|
256
|
+
|
|
257
|
+
```bash
|
|
258
|
+
cd my-repo
|
|
259
|
+
devagent --no-publish "Refactor retry handling to remove duplicate logic and keep behavior unchanged."
|
|
260
|
+
```
|
|
261
|
+
|
|
262
|
+
**Benefit:** you still get implementation, verification, independent review, and the engineering report, but DevAgent does not commit or push the result.
|
|
263
|
+
|
|
264
|
+
### 8. Use the model/provider you prefer
|
|
265
|
+
|
|
266
|
+
```bash
|
|
267
|
+
# Configure once
|
|
268
|
+
devagent setup --provider anthropic --model YOUR_MODEL
|
|
269
|
+
export ANTHROPIC_API_KEY=...
|
|
270
|
+
|
|
271
|
+
# Then use the same DevAgent engineering workflow
|
|
272
|
+
devagent "Fix the failing checkout integration test without weakening the assertion."
|
|
273
|
+
```
|
|
274
|
+
|
|
275
|
+
You can similarly configure OpenAI, Gemini, Grok/xAI, or an OpenAI-compatible endpoint.
|
|
276
|
+
|
|
277
|
+
**Benefit:** the model supplies reasoning, while DevAgent keeps the same deterministic acceptance, safety, verification, reporting, and publication rules around it.
|
|
278
|
+
|
|
279
|
+
## What DevAgent adds around an AI coding model
|
|
280
|
+
|
|
281
|
+
| Common engineering risk | DevAgent behavior |
|
|
282
|
+
| --- | --- |
|
|
283
|
+
| The model says “done” without enough proof | Required acceptance criteria remain `UNPROVEN` or the run becomes `PARTIALLY_VERIFIED` / `BLOCKED` instead of falsely claiming success. |
|
|
284
|
+
| A patch touches unrelated code | Evidence gathering, explicit scope, minimal-change planning, and independent diff review constrain the change. |
|
|
285
|
+
| Existing developer work is damaged | Clean repositories use isolated worktrees by default; dirty tracked/untracked developer work is protected; files are backed up before first modification. |
|
|
286
|
+
| Tests passed before a later edit | Verification is revision-aware, so older successful evidence does not prove a newer tree. |
|
|
287
|
+
| A generated change breaks the build or tests | DevAgent runs repository-supported targeted/broad checks and can diagnose/replan before final verification. |
|
|
288
|
+
| An agent pushes directly to a protected primary branch | Starting from `main`, `master`, or `trunk` causes DevAgent to work on a safe branch; runtime DevAgent does not merge or deploy. |
|
|
289
|
+
| You are locked to one model vendor | OpenAI, Claude, Gemini, Grok/xAI, and compatible endpoints can use the same engineering harness. |
|
|
290
|
+
| It is hard to audit what the agent actually did | DevAgent emits an engineering report with decisions, changed symbols, tests, acceptance evidence, verification, failures, gaps, and source-control status. |
|
|
291
|
+
|
|
292
|
+
The goal is not to replace developer judgment. The goal is to make autonomous engineering work **bounded, reviewable, reproducible, and harder to falsely declare complete**.
|
|
293
|
+
|
|
188
294
|
Useful commands:
|
|
189
295
|
|
|
190
296
|
```bash
|
|
@@ -199,7 +305,7 @@ devagent benchmark --help
|
|
|
199
305
|
|
|
200
306
|
### Pinned real-world benchmark
|
|
201
307
|
|
|
202
|
-
DevAgent
|
|
308
|
+
DevAgent includes an opt-in benchmark runner for pinned GitHub repositories. A benchmark case injects a deterministic defect into an exact commit and uses an **external oracle** before and after DevAgent. This avoids treating DevAgent's own report as the benchmark oracle.
|
|
203
309
|
|
|
204
310
|
```bash
|
|
205
311
|
devagent benchmark \
|
|
@@ -247,7 +353,7 @@ DevAgent uses defense-in-depth controls around repository modification, command
|
|
|
247
353
|
- protected targets are refused and force push is never used;
|
|
248
354
|
- no runtime PR, merge, rebase, force-push, or deployment automation.
|
|
249
355
|
|
|
250
|
-
DevAgent
|
|
356
|
+
On Linux, DevAgent can execute engineering commands inside a bubblewrap-based operating-system sandbox. Production qualification exercises required sandbox mode with network access denied. Required mode fails closed when isolation cannot be established rather than silently falling back. Review the report and pushed branch before integrating customer or production code: sandboxing reduces execution risk, but it does not make arbitrary generated changes universally safe.
|
|
251
357
|
|
|
252
358
|
## Repository intelligence and verification
|
|
253
359
|
|
|
@@ -267,13 +373,20 @@ Verification can include baseline tests, targeted tests, component/broad checks,
|
|
|
267
373
|
|
|
268
374
|
## Production qualification
|
|
269
375
|
|
|
270
|
-
DevAgent 0.
|
|
376
|
+
DevAgent 0.8.0 uses cumulative production qualification rather than replacing older evidence with a smaller new suite.
|
|
377
|
+
|
|
378
|
+
- **v4 — 70 required cases** covering end-to-end engineering behavior, acceptance truthfulness, task/risk scope, provider contracts and parity, model routing, worktree and Git publication safety, CLI input, review/repair loops, report/evaluation integrity, release integrity, large-repository behavior, structural refactors, Java/.NET discovery and execution, SQLite migration forward/rollback, and real repository-native stacks.
|
|
379
|
+
- **v5 — 9 required autonomy cases** covering bounded parallel coordination, dirty-source refusal, real isolated parallel DevAgent runs, bounded/relevant skills and provider injection, automation overlap claim/recovery, and provider-benchmark deduplication, live structured-contract behavior, and secret redaction.
|
|
380
|
+
|
|
381
|
+
The v0.8 merge commit on `main` passed both catalogs in required Linux sandbox mode:
|
|
271
382
|
|
|
272
383
|
```text
|
|
273
|
-
|
|
384
|
+
v4: 70/70 passed
|
|
385
|
+
v5: 9/9 passed
|
|
386
|
+
combined: 79/79 passed
|
|
274
387
|
```
|
|
275
388
|
|
|
276
|
-
|
|
389
|
+
The qualification environment exercises real local toolchains for:
|
|
277
390
|
|
|
278
391
|
```text
|
|
279
392
|
Python / pytest
|
|
@@ -281,21 +394,30 @@ Node + TypeScript repository discovery
|
|
|
281
394
|
Go
|
|
282
395
|
Rust / Cargo
|
|
283
396
|
C++ / Make
|
|
397
|
+
Java / Maven
|
|
398
|
+
.NET build
|
|
399
|
+
SQLite migration forward + rollback
|
|
284
400
|
```
|
|
285
401
|
|
|
286
|
-
Run the release qualification locally on a machine with
|
|
402
|
+
Run the same release qualification catalogs locally on a machine with the required toolchains:
|
|
287
403
|
|
|
288
404
|
```bash
|
|
405
|
+
DEVAGENT_SANDBOX=required DEVAGENT_NETWORK=deny \
|
|
289
406
|
python -m devagent.qualification \
|
|
290
407
|
--catalog evaluation/benchmark_v4.json \
|
|
291
408
|
--report .devagent/production-qualification-v4.json
|
|
409
|
+
|
|
410
|
+
DEVAGENT_SANDBOX=required DEVAGENT_NETWORK=deny \
|
|
411
|
+
python -m devagent.qualification \
|
|
412
|
+
--catalog evaluation/benchmark_v5.json \
|
|
413
|
+
--report .devagent/production-qualification-v5.json
|
|
292
414
|
```
|
|
293
415
|
|
|
294
|
-
Production CI runs Python 3.10/3.11/3.12, a clean wheel install, and
|
|
416
|
+
Production CI also runs Python 3.10/3.11/3.12, a clean wheel build/install, real bubblewrap sandbox smoke, and both qualification catalogs. Qualification JSON is retained as CI evidence.
|
|
295
417
|
|
|
296
|
-
**100% qualified means 100% of
|
|
418
|
+
**100% qualified means 100% of these explicit catalogs passed on that revision and environment.** It does not mean mathematical correctness for every unseen repository, environment, model response, language, provider, or engineering task, and it is not a claim that DevAgent is universally superior to every hosted coding platform.
|
|
297
419
|
|
|
298
|
-
See [docs/production-readiness.md](docs/production-readiness.md) for the
|
|
420
|
+
See [docs/production-readiness.md](docs/production-readiness.md) for the project's earlier readiness assessment and its explicit limitations.
|
|
299
421
|
|
|
300
422
|
## Provider architecture
|
|
301
423
|
|
|
@@ -349,11 +471,15 @@ Automated provider tests normally use deterministic or mocked clients and do not
|
|
|
349
471
|
|
|
350
472
|
## Project status
|
|
351
473
|
|
|
352
|
-
DevAgent 0.
|
|
474
|
+
DevAgent 0.8.0 is **beta software with a verified core release baseline**. The exact v0.8 merge revision on `main` passed Production CI across Python 3.10/3.11/3.12, clean wheel installation, real Linux bubblewrap sandbox execution, production qualification v4 (**70/70**), and autonomy qualification v5 (**9/9**), for **79/79 cumulative required qualification cases**.
|
|
475
|
+
|
|
476
|
+
The current core includes evidence-backed `VERIFIED` / `PARTIALLY_VERIFIED` / `BLOCKED` outcomes, backup-first editing, isolated worktrees, bounded structural file operations, repository-native verification, independent review, safe branch publication, provider/model choice, Java and .NET engineering discovery/execution, SQLite migration forward/rollback verification, large-monorepo deep-manifest discovery, bounded parallel agents, repository-local skills, foreground automations, Linux OS sandboxing, bounded browser/local-UI verification, and real-provider structured-contract benchmarking.
|
|
477
|
+
|
|
478
|
+
These results are **bounded engineering claims**, not universal-correctness or market-superiority claims. They are tied to explicit qualification cases, pinned revisions, deterministic fixtures/external oracles where applicable, and the environments actually exercised by CI.
|
|
353
479
|
|
|
354
|
-
Remaining
|
|
480
|
+
Remaining work is primarily **breadth and external validation**, not missing core architecture: a larger public corpus of pinned upstream repositories and tasks; broader browser/UI coverage across dynamic applications and multiple browser environments; a wider Java/Gradle, .NET test-framework, and PostgreSQL/MySQL migration matrix beyond the current qualified fixtures; larger and more diverse monorepo stress cases beyond the current >12,000-file deep-manifest case; more real-world multi-agent workload studies; and continuous paid real-provider benchmarking across a broader set of model/provider combinations. GitHub branch protection/rulesets are external repository settings and must be configured separately; DevAgent does not claim to configure them itself.
|
|
355
481
|
|
|
356
|
-
The project intentionally prioritizes trustworthy outcomes over feature count.
|
|
482
|
+
The project intentionally prioritizes trustworthy outcomes, reproducible evidence, and safe engineering behavior over feature count or unsupported "best agent" claims.
|
|
357
483
|
|
|
358
484
|
## Contributing
|
|
359
485
|
|
|
@@ -371,4 +497,4 @@ DevAgent is open source under the [MIT License](LICENSE).
|
|
|
371
497
|
Copyright (c) 2026 Tom Ha
|
|
372
498
|
```
|
|
373
499
|
|
|
374
|
-
DevAgent was created by **Tom Ha**. Original repository: **https://github.com/tomha85/devagent**. See [NOTICE](NOTICE) and [COPYRIGHT](COPYRIGHT) for project attribution.
|
|
500
|
+
DevAgent was created by **Tom Ha**. Original repository: **https://github.com/tomha85/devagent**. See [NOTICE](NOTICE) and [COPYRIGHT](COPYRIGHT) for project attribution.
|
|
@@ -186,6 +186,19 @@ class EngineeringPlan:
|
|
|
186
186
|
verification: list[tuple[str, ...]]
|
|
187
187
|
rationale: str
|
|
188
188
|
|
|
189
|
+
def __post_init__(self) -> None:
|
|
190
|
+
# A path-scoped `git diff -- <paths...>` is planner inspection, not a
|
|
191
|
+
# repository verification capability. DevAgent already captures and
|
|
192
|
+
# independently reviews the final diff. Keeping this command in the
|
|
193
|
+
# executable verification plan makes an otherwise valid plan fail the
|
|
194
|
+
# evidence-backed command allowlist before implementation can begin.
|
|
195
|
+
# Preserve real Git verification such as `git diff --check`.
|
|
196
|
+
self.verification = [
|
|
197
|
+
command
|
|
198
|
+
for command in self.verification
|
|
199
|
+
if not (len(command) > 3 and command[:3] == ("git", "diff", "--"))
|
|
200
|
+
]
|
|
201
|
+
|
|
189
202
|
|
|
190
203
|
_PRESERVATION_PATTERNS = (
|
|
191
204
|
re.compile(r"\bpreserv(?:e|es|ed|ing)\s+(?:the\s+)?existing\s+([^.;\n]+)", re.IGNORECASE),
|
|
@@ -63,6 +63,18 @@ _DIRECTIVE = re.compile(
|
|
|
63
63
|
re.IGNORECASE,
|
|
64
64
|
)
|
|
65
65
|
|
|
66
|
+
# Bounded normalization for terse user intent. This is deliberately not a fuzzy
|
|
67
|
+
# "guess what the user meant" layer: it corrects common engineering shorthand,
|
|
68
|
+
# grammatical number, and operation wording while preserving identifiers,
|
|
69
|
+
# quoted contracts, values, and explicit constraints. Task policy and repository
|
|
70
|
+
# evidence still provide the verification/safety contract.
|
|
71
|
+
_OPERATION_ALIASES: tuple[tuple[str, str], ...] = (
|
|
72
|
+
("substraction", "subtraction"),
|
|
73
|
+
("substract", "subtract"),
|
|
74
|
+
("multipy", "multiply"),
|
|
75
|
+
("mutiply", "multiply"),
|
|
76
|
+
)
|
|
77
|
+
|
|
66
78
|
|
|
67
79
|
def _classify(text: str) -> TaskType:
|
|
68
80
|
lowered = text.lower()
|
|
@@ -98,6 +110,60 @@ def _dedupe(items: list[str]) -> list[str]:
|
|
|
98
110
|
return result
|
|
99
111
|
|
|
100
112
|
|
|
113
|
+
def _normalize_terse_requirement(requirement: str) -> str:
|
|
114
|
+
"""Compile common rough one-line prompts into a clearer engineering request.
|
|
115
|
+
|
|
116
|
+
The compiler is intentionally bounded. It may repair shorthand/grammar and
|
|
117
|
+
make an operation explicit, but it must not add product behavior the user did
|
|
118
|
+
not request. Structured/multi-line requirements are left intact.
|
|
119
|
+
"""
|
|
120
|
+
|
|
121
|
+
value = re.sub(r"\s+", " ", requirement).strip()
|
|
122
|
+
if not value or "\n" in requirement or _section_header(value) is not None:
|
|
123
|
+
return value
|
|
124
|
+
# An explicit callable name is already a precise user contract; never rename it.
|
|
125
|
+
if re.search(r"\b[A-Za-z_][A-Za-z0-9_]*\s*\(", value):
|
|
126
|
+
return value
|
|
127
|
+
|
|
128
|
+
for source, destination in _OPERATION_ALIASES:
|
|
129
|
+
value = re.sub(rf"\b{re.escape(source)}\b", destination, value, flags=re.IGNORECASE)
|
|
130
|
+
|
|
131
|
+
# Common shorthand from natural prompts such as "addition 2 matrix 2x2".
|
|
132
|
+
# Keep both "matrix" and "matrices" in the normalized contract so
|
|
133
|
+
# deterministic evidence can link either conventional symbol spelling.
|
|
134
|
+
matrix_match = re.search(
|
|
135
|
+
r"\b(add(?:ition)?|sum|subtract(?:ion)?|multiply|multiplication|divide|division)\b"
|
|
136
|
+
r"(?:\s+(?:of|for))?\s+(?:2|two)\s+matrix(?:es)?\s+(\d+x\d+)\b",
|
|
137
|
+
value,
|
|
138
|
+
flags=re.IGNORECASE,
|
|
139
|
+
)
|
|
140
|
+
if matrix_match:
|
|
141
|
+
operation = matrix_match.group(1).lower()
|
|
142
|
+
dimension = matrix_match.group(2).lower()
|
|
143
|
+
canonical_operation = {
|
|
144
|
+
"add": "addition",
|
|
145
|
+
"addition": "addition",
|
|
146
|
+
"sum": "addition",
|
|
147
|
+
"subtract": "subtraction",
|
|
148
|
+
"subtraction": "subtraction",
|
|
149
|
+
"multiply": "multiplication",
|
|
150
|
+
"multiplication": "multiplication",
|
|
151
|
+
"divide": "division",
|
|
152
|
+
"division": "division",
|
|
153
|
+
}[operation]
|
|
154
|
+
prefix = "Add" if re.search(r"\b(add|new|function|implement)\b", value, re.IGNORECASE) else "Implement"
|
|
155
|
+
return (
|
|
156
|
+
f"{prefix} a matrix {canonical_operation} function for two {dimension} matrices "
|
|
157
|
+
f"(matrix inputs)"
|
|
158
|
+
)
|
|
159
|
+
|
|
160
|
+
# Repair simple count+noun shorthand without inventing domain behavior.
|
|
161
|
+
value = re.sub(r"\b2\s+matrix\b", "two matrices", value, flags=re.IGNORECASE)
|
|
162
|
+
value = re.sub(r"\b2\s+file\b", "two files", value, flags=re.IGNORECASE)
|
|
163
|
+
value = re.sub(r"\b2\s+test\b", "two tests", value, flags=re.IGNORECASE)
|
|
164
|
+
return value
|
|
165
|
+
|
|
166
|
+
|
|
101
167
|
def _section_header(line: str) -> tuple[str, str] | None:
|
|
102
168
|
stripped = line.strip()
|
|
103
169
|
markdown = re.match(r"^#{1,6}\s+(.+?)\s*$", stripped)
|
|
@@ -182,15 +248,21 @@ def _append_criterion(
|
|
|
182
248
|
|
|
183
249
|
|
|
184
250
|
def compile_task(requirement: str) -> TaskSpec:
|
|
185
|
-
|
|
186
|
-
if not
|
|
251
|
+
raw_goal = re.sub(r"\s+", " ", requirement).strip()
|
|
252
|
+
if not raw_goal:
|
|
187
253
|
raise ValueError("Engineering requirement cannot be empty")
|
|
254
|
+
goal = _normalize_terse_requirement(requirement)
|
|
188
255
|
task_type = _classify(goal)
|
|
189
256
|
code_change = task_type is not TaskType.UNIT_TEST or "only" not in goal.lower()
|
|
190
257
|
requires_tests = task_type is not TaskType.BUILD_FAILURE
|
|
191
258
|
|
|
192
259
|
criteria: list[AcceptanceCriterion] = []
|
|
193
|
-
|
|
260
|
+
# Structured user requirements remain authoritative. Only an unstructured,
|
|
261
|
+
# terse prompt is compiled into the clearer canonical request.
|
|
262
|
+
user_items = _user_acceptance_items(requirement)
|
|
263
|
+
if len(user_items) == 1 and user_items[0] == raw_goal and goal != raw_goal:
|
|
264
|
+
user_items = [goal]
|
|
265
|
+
for item in user_items:
|
|
194
266
|
_append_criterion(criteria, item, source=AcceptanceSource.USER)
|
|
195
267
|
|
|
196
268
|
if task_type in {TaskType.BUG_FIX, TaskType.RUNTIME_ERROR, TaskType.TEST_FAILURE}:
|
|
@@ -240,9 +312,61 @@ def compile_task(requirement: str) -> TaskSpec:
|
|
|
240
312
|
)
|
|
241
313
|
|
|
242
314
|
|
|
315
|
+
def _repository_language(repository: Any) -> str | None:
|
|
316
|
+
languages = [
|
|
317
|
+
language.lower()
|
|
318
|
+
for component in repository.components
|
|
319
|
+
for language in component.languages
|
|
320
|
+
]
|
|
321
|
+
for preferred in ("python", "java", "csharp", "c#", "javascript", "typescript", "go", "rust"):
|
|
322
|
+
if preferred in languages:
|
|
323
|
+
return preferred
|
|
324
|
+
return languages[0] if languages else None
|
|
325
|
+
|
|
326
|
+
|
|
327
|
+
def _matrix_operation_contract(task: TaskSpec, repository: Any) -> None:
|
|
328
|
+
"""Turn the bounded matrix shorthand compiler output into a repo-style callable contract."""
|
|
329
|
+
|
|
330
|
+
match = re.fullmatch(
|
|
331
|
+
r"(?:Add|Implement) a matrix (addition|subtraction|multiplication|division) "
|
|
332
|
+
r"function for two (\d+x\d+) matrices \(matrix inputs\)",
|
|
333
|
+
task.goal,
|
|
334
|
+
)
|
|
335
|
+
if match is None:
|
|
336
|
+
return
|
|
337
|
+
|
|
338
|
+
operation, dimension = match.groups()
|
|
339
|
+
verb = {
|
|
340
|
+
"addition": "add",
|
|
341
|
+
"subtraction": "subtract",
|
|
342
|
+
"multiplication": "multiply",
|
|
343
|
+
"division": "divide",
|
|
344
|
+
}[operation]
|
|
345
|
+
compact_dimension = dimension.replace("x", "x")
|
|
346
|
+
language = _repository_language(repository)
|
|
347
|
+
if language in {"java", "javascript", "typescript"}:
|
|
348
|
+
symbol = f"{verb}Matrices{compact_dimension}"
|
|
349
|
+
elif language in {"csharp", "c#"}:
|
|
350
|
+
symbol = f"{verb.capitalize()}Matrices{compact_dimension}"
|
|
351
|
+
else:
|
|
352
|
+
symbol = f"{verb}_matrices_{compact_dimension}"
|
|
353
|
+
|
|
354
|
+
compiled = (
|
|
355
|
+
f"Add {symbol}(a, b) to perform element-wise matrix {operation} "
|
|
356
|
+
f"for two {dimension} matrices"
|
|
357
|
+
)
|
|
358
|
+
task.goal = compiled
|
|
359
|
+
user_criteria = [
|
|
360
|
+
criterion for criterion in task.acceptance_criteria if criterion.source is AcceptanceSource.USER
|
|
361
|
+
]
|
|
362
|
+
if len(user_criteria) == 1:
|
|
363
|
+
user_criteria[0].description = compiled
|
|
364
|
+
|
|
365
|
+
|
|
243
366
|
def enrich_acceptance_contract(task: TaskSpec, repository: Any) -> TaskSpec:
|
|
244
|
-
"""
|
|
367
|
+
"""Compile safe repository-aware defaults, then add trusted repository checks."""
|
|
245
368
|
|
|
369
|
+
_matrix_operation_contract(task, repository)
|
|
246
370
|
seen_commands: set[tuple[str, ...]] = set()
|
|
247
371
|
for capability in repository.capabilities:
|
|
248
372
|
if not capability.trusted:
|