devagent-ai 0.8.0__tar.gz → 0.8.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (86) hide show
  1. {devagent_ai-0.8.0/devagent_ai.egg-info → devagent_ai-0.8.2}/PKG-INFO +139 -13
  2. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/README.md +139 -13
  3. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent/__init__.py +1 -1
  4. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent/models.py +13 -0
  5. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent/tasking.py +128 -4
  6. {devagent_ai-0.8.0 → devagent_ai-0.8.2/devagent_ai.egg-info}/PKG-INFO +139 -13
  7. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent_ai.egg-info/SOURCES.txt +2 -0
  8. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/pyproject.toml +1 -1
  9. devagent_ai-0.8.2/tests/test_plan_verification_normalization.py +176 -0
  10. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_production_v040.py +1 -1
  11. devagent_ai-0.8.2/tests/test_requirement_compiler_v082.py +185 -0
  12. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/LICENSE +0 -0
  13. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/NOTICE +0 -0
  14. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/agent/__init__.py +0 -0
  15. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/agent/llm.py +0 -0
  16. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/agent/loop.py +0 -0
  17. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/agent/memory.py +0 -0
  18. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/agent/prompts.py +0 -0
  19. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/agent/tools.py +0 -0
  20. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent/__main__.py +0 -0
  21. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent/artifacts.py +0 -0
  22. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent/automations.py +0 -0
  23. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent/autonomy.py +0 -0
  24. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent/browser.py +0 -0
  25. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent/cli.py +0 -0
  26. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent/config.py +0 -0
  27. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent/discovery.py +0 -0
  28. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent/evaluation.py +0 -0
  29. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent/memory.py +0 -0
  30. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent/orchestrator.py +0 -0
  31. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent/provider_benchmark.py +0 -0
  32. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent/providers.py +0 -0
  33. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent/qualification.py +0 -0
  34. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent/realworld.py +0 -0
  35. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent/report.py +0 -0
  36. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent/retrieval.py +0 -0
  37. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent/routing.py +0 -0
  38. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent/runtime.py +0 -0
  39. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent/safety.py +0 -0
  40. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent/skills.py +0 -0
  41. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent/source_control.py +0 -0
  42. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent/state_machine.py +0 -0
  43. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent/technical_review.py +0 -0
  44. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent/workspace.py +0 -0
  45. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent/worktree.py +0 -0
  46. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent_ai.egg-info/dependency_links.txt +0 -0
  47. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent_ai.egg-info/entry_points.txt +0 -0
  48. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent_ai.egg-info/requires.txt +0 -0
  49. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/devagent_ai.egg-info/top_level.txt +0 -0
  50. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/setup.cfg +0 -0
  51. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_acceptance_contract.py +0 -0
  52. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_benchmark_catalog.py +0 -0
  53. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_browser_verification.py +0 -0
  54. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_capability_discovery.py +0 -0
  55. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_cli.py +0 -0
  56. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_developer_review_report.py +0 -0
  57. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_discovery_memory.py +0 -0
  58. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_e2e_fake_provider.py +0 -0
  59. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_evaluation_harness.py +0 -0
  60. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_evaluation_matrix.py +0 -0
  61. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_evaluation_regression_evidence.py +0 -0
  62. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_functional_qualification.py +0 -0
  63. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_huge_monorepo_v070.py +0 -0
  64. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_migration_e2e_v070.py +0 -0
  65. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_model_routing.py +0 -0
  66. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_multilang_technical_review.py +0 -0
  67. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_multistack_devagent_e2e.py +0 -0
  68. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_multistack_qualification.py +0 -0
  69. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_packaging_metadata.py +0 -0
  70. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_preservation_contradiction.py +0 -0
  71. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_production_hardening.py +0 -0
  72. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_realworld_benchmark.py +0 -0
  73. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_retrieval.py +0 -0
  74. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_runtime_sandbox.py +0 -0
  75. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_safety_workspace.py +0 -0
  76. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_source_control_publish.py +0 -0
  77. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_structural_devagent_e2e_v070.py +0 -0
  78. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_structural_operations.py +0 -0
  79. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_structured_provider_contract.py +0 -0
  80. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_tasking_state.py +0 -0
  81. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_v070_engineering_breadth.py +0 -0
  82. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_v080_autonomy.py +0 -0
  83. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_v080_provider_benchmark.py +0 -0
  84. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_v080_skills_automations.py +0 -0
  85. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_workspace_environment.py +0 -0
  86. {devagent_ai-0.8.0 → devagent_ai-0.8.2}/tests/test_worktree.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: devagent-ai
3
- Version: 0.8.0
3
+ Version: 0.8.2
4
4
  Summary: Evidence-driven local autonomous software engineering agent
5
5
  Author: Tom Ha
6
6
  Maintainer: Tom Ha
@@ -216,6 +216,112 @@ devagent --input ../specs/release.requirement
216
216
 
217
217
  Binary data, invalid UTF-8, secret-like paths, and files above the input-size bound are refused.
218
218
 
219
+ ## Practical examples
220
+
221
+ DevAgent is intended for real repository work, not only one-line code generation. Run it from the repository you want to change and describe the engineering outcome you need.
222
+
223
+ ### 1. Fix a bug and prove the regression is covered
224
+
225
+ ```bash
226
+ cd my-service
227
+ devagent "Fix the websocket reconnect bug that duplicates subscriptions after a network drop. Add a regression test and keep the public API unchanged."
228
+ ```
229
+
230
+ **Benefit:** DevAgent first discovers the relevant implementation and tests, turns the request into explicit acceptance criteria, makes a bounded patch, runs repository-supported verification, independently reviews the final diff, and only reports `VERIFIED` when the required evidence supports it.
231
+
232
+ ### 2. Add a feature on a dedicated branch
233
+
234
+ ```bash
235
+ cd my-app
236
+ devagent \
237
+ --publish-branch feature/csv-export \
238
+ "Add CSV export for filtered reports. Preserve the existing JSON export behavior and add tests."
239
+ ```
240
+
241
+ **Benefit:** a verified change can be committed and pushed to the requested feature branch while DevAgent stops before PR creation or merge, leaving integration control with the developer or repository owner.
242
+
243
+ ### 3. Give DevAgent a longer product or customer requirement
244
+
245
+ ```bash
246
+ cd my-repo
247
+ devagent --input requirements/customer-billing-retry.md
248
+ ```
249
+
250
+ The input can be any bounded UTF-8 text file; it does not need a special extension or DevAgent-specific format.
251
+
252
+ **Benefit:** long requirements stay in a reviewable file instead of being compressed into a short prompt, while DevAgent still derives bounded implementation and verification work from repository evidence.
253
+
254
+ ### 4. Perform a refactor that includes rename/move/delete operations
255
+
256
+ ```bash
257
+ cd my-repo
258
+ devagent "Rename LegacyOrderService to OrderService, move it into the services package, update all references, remove the obsolete module, and preserve behavior."
259
+ ```
260
+
261
+ **Benefit:** structural changes go through backup-first workspace operations, path/scope checks, repository verification, and final-diff review instead of uncontrolled file manipulation.
262
+
263
+ ### 5. Change a database schema with forward/rollback verification
264
+
265
+ ```bash
266
+ cd my-python-service
267
+ devagent "Add a nullable status column to the SQLite orders table, provide a forward and rollback migration, update the data-access layer, and verify both migration directions."
268
+ ```
269
+
270
+ **Benefit:** migration work can be treated as high-risk engineering work with explicit acceptance evidence instead of assuming that a generated migration is correct because it looks plausible. The current qualified production fixture covers SQLite forward/rollback migration behavior; broader PostgreSQL/MySQL coverage remains an external-validation area.
271
+
272
+ ### 6. Work in Java or .NET repositories
273
+
274
+ ```bash
275
+ cd my-java-service
276
+ devagent "Add validation for duplicate customer IDs in this Maven service and add the appropriate JUnit regression test."
277
+ ```
278
+
279
+ ```bash
280
+ cd my-dotnet-service
281
+ devagent "Fix the null-handling bug in the order import path and verify the .NET project still builds successfully."
282
+ ```
283
+
284
+ **Benefit:** DevAgent discovers repository-native Maven/Gradle and .NET project evidence instead of forcing every repository through a Python-centric workflow.
285
+
286
+ ### 7. Keep all changes local for inspection
287
+
288
+ ```bash
289
+ cd my-repo
290
+ devagent --no-publish "Refactor retry handling to remove duplicate logic and keep behavior unchanged."
291
+ ```
292
+
293
+ **Benefit:** you still get implementation, verification, independent review, and the engineering report, but DevAgent does not commit or push the result.
294
+
295
+ ### 8. Use the model/provider you prefer
296
+
297
+ ```bash
298
+ # Configure once
299
+ devagent setup --provider anthropic --model YOUR_MODEL
300
+ export ANTHROPIC_API_KEY=...
301
+
302
+ # Then use the same DevAgent engineering workflow
303
+ devagent "Fix the failing checkout integration test without weakening the assertion."
304
+ ```
305
+
306
+ You can similarly configure OpenAI, Gemini, Grok/xAI, or an OpenAI-compatible endpoint.
307
+
308
+ **Benefit:** the model supplies reasoning, while DevAgent keeps the same deterministic acceptance, safety, verification, reporting, and publication rules around it.
309
+
310
+ ## What DevAgent adds around an AI coding model
311
+
312
+ | Common engineering risk | DevAgent behavior |
313
+ | --- | --- |
314
+ | The model says “done” without enough proof | Required acceptance criteria remain `UNPROVEN` or the run becomes `PARTIALLY_VERIFIED` / `BLOCKED` instead of falsely claiming success. |
315
+ | A patch touches unrelated code | Evidence gathering, explicit scope, minimal-change planning, and independent diff review constrain the change. |
316
+ | Existing developer work is damaged | Clean repositories use isolated worktrees by default; dirty tracked/untracked developer work is protected; files are backed up before first modification. |
317
+ | Tests passed before a later edit | Verification is revision-aware, so older successful evidence does not prove a newer tree. |
318
+ | A generated change breaks the build or tests | DevAgent runs repository-supported targeted/broad checks and can diagnose/replan before final verification. |
319
+ | An agent pushes directly to a protected primary branch | Starting from `main`, `master`, or `trunk` causes DevAgent to work on a safe branch; runtime DevAgent does not merge or deploy. |
320
+ | You are locked to one model vendor | OpenAI, Claude, Gemini, Grok/xAI, and compatible endpoints can use the same engineering harness. |
321
+ | It is hard to audit what the agent actually did | DevAgent emits an engineering report with decisions, changed symbols, tests, acceptance evidence, verification, failures, gaps, and source-control status. |
322
+
323
+ The goal is not to replace developer judgment. The goal is to make autonomous engineering work **bounded, reviewable, reproducible, and harder to falsely declare complete**.
324
+
219
325
  Useful commands:
220
326
 
221
327
  ```bash
@@ -230,7 +336,7 @@ devagent benchmark --help
230
336
 
231
337
  ### Pinned real-world benchmark
232
338
 
233
- DevAgent 0.5 adds an opt-in benchmark runner for pinned GitHub repositories. A benchmark case injects a deterministic defect into an exact commit and uses an **external oracle** before and after DevAgent. This avoids treating DevAgent's own report as the benchmark oracle.
339
+ DevAgent includes an opt-in benchmark runner for pinned GitHub repositories. A benchmark case injects a deterministic defect into an exact commit and uses an **external oracle** before and after DevAgent. This avoids treating DevAgent's own report as the benchmark oracle.
234
340
 
235
341
  ```bash
236
342
  devagent benchmark \
@@ -278,7 +384,7 @@ DevAgent uses defense-in-depth controls around repository modification, command
278
384
  - protected targets are refused and force push is never used;
279
385
  - no runtime PR, merge, rebase, force-push, or deployment automation.
280
386
 
281
- DevAgent is **not an operating-system sandbox**. Review the report and pushed branch before integrating customer or production code.
387
+ On Linux, DevAgent can execute engineering commands inside a bubblewrap-based operating-system sandbox. Production qualification exercises required sandbox mode with network access denied. Required mode fails closed when isolation cannot be established rather than silently falling back. Review the report and pushed branch before integrating customer or production code: sandboxing reduces execution risk, but it does not make arbitrary generated changes universally safe.
282
388
 
283
389
  ## Repository intelligence and verification
284
390
 
@@ -298,13 +404,20 @@ Verification can include baseline tests, targeted tests, component/broad checks,
298
404
 
299
405
  ## Production qualification
300
406
 
301
- DevAgent 0.5.0 uses **production qualification v4**. It extends the v3 release gate with large-repository bounded-retrieval and real-world benchmark truthfulness contracts while preserving the primary invariant:
407
+ DevAgent 0.8.0 uses cumulative production qualification rather than replacing older evidence with a smaller new suite.
408
+
409
+ - **v4 — 70 required cases** covering end-to-end engineering behavior, acceptance truthfulness, task/risk scope, provider contracts and parity, model routing, worktree and Git publication safety, CLI input, review/repair loops, report/evaluation integrity, release integrity, large-repository behavior, structural refactors, Java/.NET discovery and execution, SQLite migration forward/rollback, and real repository-native stacks.
410
+ - **v5 — 9 required autonomy cases** covering bounded parallel coordination, dirty-source refusal, real isolated parallel DevAgent runs, bounded/relevant skills and provider injection, automation overlap claim/recovery, and provider-benchmark deduplication, live structured-contract behavior, and secret redaction.
411
+
412
+ The v0.8 merge commit on `main` passed both catalogs in required Linux sandbox mode:
302
413
 
303
414
  ```text
304
- false_verified == 0
415
+ v4: 70/70 passed
416
+ v5: 9/9 passed
417
+ combined: 79/79 passed
305
418
  ```
306
419
 
307
- It covers end-to-end engineering behavior, acceptance truthfulness, task scope, provider contracts/parity, model routing, worktree and Git publication safety, CLI input, review/repair loops, report/evaluation integrity, release integrity, and actual repository-native toolchain execution for:
420
+ The qualification environment exercises real local toolchains for:
308
421
 
309
422
  ```text
310
423
  Python / pytest
@@ -312,21 +425,30 @@ Node + TypeScript repository discovery
312
425
  Go
313
426
  Rust / Cargo
314
427
  C++ / Make
428
+ Java / Maven
429
+ .NET build
430
+ SQLite migration forward + rollback
315
431
  ```
316
432
 
317
- Run the release qualification locally on a machine with those toolchains:
433
+ Run the same release qualification catalogs locally on a machine with the required toolchains:
318
434
 
319
435
  ```bash
436
+ DEVAGENT_SANDBOX=required DEVAGENT_NETWORK=deny \
320
437
  python -m devagent.qualification \
321
438
  --catalog evaluation/benchmark_v4.json \
322
439
  --report .devagent/production-qualification-v4.json
440
+
441
+ DEVAGENT_SANDBOX=required DEVAGENT_NETWORK=deny \
442
+ python -m devagent.qualification \
443
+ --catalog evaluation/benchmark_v5.json \
444
+ --report .devagent/production-qualification-v5.json
323
445
  ```
324
446
 
325
- Production CI runs Python 3.10/3.11/3.12, a clean wheel install, and this production qualification gate. The qualification JSON is retained as CI evidence.
447
+ Production CI also runs Python 3.10/3.11/3.12, a clean wheel build/install, real bubblewrap sandbox smoke, and both qualification catalogs. Qualification JSON is retained as CI evidence.
326
448
 
327
- **100% qualified means 100% of this explicit catalog passed.** It does not mean mathematical correctness for every unseen repository, environment, model response, language, or engineering task.
449
+ **100% qualified means 100% of these explicit catalogs passed on that revision and environment.** It does not mean mathematical correctness for every unseen repository, environment, model response, language, provider, or engineering task, and it is not a claim that DevAgent is universally superior to every hosted coding platform.
328
450
 
329
- See [docs/production-readiness.md](docs/production-readiness.md) for the evidence behind the 0.4.0 production-readiness assessment and its explicit limitations.
451
+ See [docs/production-readiness.md](docs/production-readiness.md) for the project's earlier readiness assessment and its explicit limitations.
330
452
 
331
453
  ## Provider architecture
332
454
 
@@ -380,11 +502,15 @@ Automated provider tests normally use deterministic or mocked clients and do not
380
502
 
381
503
  ## Project status
382
504
 
383
- DevAgent 0.5 is **beta software**. Its qualification and benchmark results are bounded claims tied to explicit cases, pinned revisions, and external oracles; they are not a claim of universal correctness or parity with every hosted coding platform.
505
+ DevAgent 0.8.0 is **beta software with a verified core release baseline**. The exact v0.8 merge revision on `main` passed Production CI across Python 3.10/3.11/3.12, clean wheel installation, real Linux bubblewrap sandbox execution, production qualification v4 (**70/70**), and autonomy qualification v5 (**9/9**), for **79/79 cumulative required qualification cases**.
506
+
507
+ The current core includes evidence-backed `VERIFIED` / `PARTIALLY_VERIFIED` / `BLOCKED` outcomes, backup-first editing, isolated worktrees, bounded structural file operations, repository-native verification, independent review, safe branch publication, provider/model choice, Java and .NET engineering discovery/execution, SQLite migration forward/rollback verification, large-monorepo deep-manifest discovery, bounded parallel agents, repository-local skills, foreground automations, Linux OS sandboxing, bounded browser/local-UI verification, and real-provider structured-contract benchmarking.
508
+
509
+ These results are **bounded engineering claims**, not universal-correctness or market-superiority claims. They are tied to explicit qualification cases, pinned revisions, deterministic fixtures/external oracles where applicable, and the environments actually exercised by CI.
384
510
 
385
- Remaining gaps include a larger published corpus of pinned upstream benchmark cases, browser/UI runtime qualification, a broader Java/.NET/database-migration matrix, very large monorepo stress runs above the current bounded inventory, parallel multi-agent orchestration, operating-system sandboxing, and continuous paid real-provider testing across every model/provider combination.
511
+ Remaining work is primarily **breadth and external validation**, not missing core architecture: a larger public corpus of pinned upstream repositories and tasks; broader browser/UI coverage across dynamic applications and multiple browser environments; a wider Java/Gradle, .NET test-framework, and PostgreSQL/MySQL migration matrix beyond the current qualified fixtures; larger and more diverse monorepo stress cases beyond the current >12,000-file deep-manifest case; more real-world multi-agent workload studies; and continuous paid real-provider benchmarking across a broader set of model/provider combinations. GitHub branch protection/rulesets are external repository settings and must be configured separately; DevAgent does not claim to configure them itself.
386
512
 
387
- The project intentionally prioritizes trustworthy outcomes over feature count.
513
+ The project intentionally prioritizes trustworthy outcomes, reproducible evidence, and safe engineering behavior over feature count or unsupported "best agent" claims.
388
514
 
389
515
  ## Contributing
390
516
 
@@ -185,6 +185,112 @@ devagent --input ../specs/release.requirement
185
185
 
186
186
  Binary data, invalid UTF-8, secret-like paths, and files above the input-size bound are refused.
187
187
 
188
+ ## Practical examples
189
+
190
+ DevAgent is intended for real repository work, not only one-line code generation. Run it from the repository you want to change and describe the engineering outcome you need.
191
+
192
+ ### 1. Fix a bug and prove the regression is covered
193
+
194
+ ```bash
195
+ cd my-service
196
+ devagent "Fix the websocket reconnect bug that duplicates subscriptions after a network drop. Add a regression test and keep the public API unchanged."
197
+ ```
198
+
199
+ **Benefit:** DevAgent first discovers the relevant implementation and tests, turns the request into explicit acceptance criteria, makes a bounded patch, runs repository-supported verification, independently reviews the final diff, and only reports `VERIFIED` when the required evidence supports it.
200
+
201
+ ### 2. Add a feature on a dedicated branch
202
+
203
+ ```bash
204
+ cd my-app
205
+ devagent \
206
+ --publish-branch feature/csv-export \
207
+ "Add CSV export for filtered reports. Preserve the existing JSON export behavior and add tests."
208
+ ```
209
+
210
+ **Benefit:** a verified change can be committed and pushed to the requested feature branch while DevAgent stops before PR creation or merge, leaving integration control with the developer or repository owner.
211
+
212
+ ### 3. Give DevAgent a longer product or customer requirement
213
+
214
+ ```bash
215
+ cd my-repo
216
+ devagent --input requirements/customer-billing-retry.md
217
+ ```
218
+
219
+ The input can be any bounded UTF-8 text file; it does not need a special extension or DevAgent-specific format.
220
+
221
+ **Benefit:** long requirements stay in a reviewable file instead of being compressed into a short prompt, while DevAgent still derives bounded implementation and verification work from repository evidence.
222
+
223
+ ### 4. Perform a refactor that includes rename/move/delete operations
224
+
225
+ ```bash
226
+ cd my-repo
227
+ devagent "Rename LegacyOrderService to OrderService, move it into the services package, update all references, remove the obsolete module, and preserve behavior."
228
+ ```
229
+
230
+ **Benefit:** structural changes go through backup-first workspace operations, path/scope checks, repository verification, and final-diff review instead of uncontrolled file manipulation.
231
+
232
+ ### 5. Change a database schema with forward/rollback verification
233
+
234
+ ```bash
235
+ cd my-python-service
236
+ devagent "Add a nullable status column to the SQLite orders table, provide a forward and rollback migration, update the data-access layer, and verify both migration directions."
237
+ ```
238
+
239
+ **Benefit:** migration work can be treated as high-risk engineering work with explicit acceptance evidence instead of assuming that a generated migration is correct because it looks plausible. The current qualified production fixture covers SQLite forward/rollback migration behavior; broader PostgreSQL/MySQL coverage remains an external-validation area.
240
+
241
+ ### 6. Work in Java or .NET repositories
242
+
243
+ ```bash
244
+ cd my-java-service
245
+ devagent "Add validation for duplicate customer IDs in this Maven service and add the appropriate JUnit regression test."
246
+ ```
247
+
248
+ ```bash
249
+ cd my-dotnet-service
250
+ devagent "Fix the null-handling bug in the order import path and verify the .NET project still builds successfully."
251
+ ```
252
+
253
+ **Benefit:** DevAgent discovers repository-native Maven/Gradle and .NET project evidence instead of forcing every repository through a Python-centric workflow.
254
+
255
+ ### 7. Keep all changes local for inspection
256
+
257
+ ```bash
258
+ cd my-repo
259
+ devagent --no-publish "Refactor retry handling to remove duplicate logic and keep behavior unchanged."
260
+ ```
261
+
262
+ **Benefit:** you still get implementation, verification, independent review, and the engineering report, but DevAgent does not commit or push the result.
263
+
264
+ ### 8. Use the model/provider you prefer
265
+
266
+ ```bash
267
+ # Configure once
268
+ devagent setup --provider anthropic --model YOUR_MODEL
269
+ export ANTHROPIC_API_KEY=...
270
+
271
+ # Then use the same DevAgent engineering workflow
272
+ devagent "Fix the failing checkout integration test without weakening the assertion."
273
+ ```
274
+
275
+ You can similarly configure OpenAI, Gemini, Grok/xAI, or an OpenAI-compatible endpoint.
276
+
277
+ **Benefit:** the model supplies reasoning, while DevAgent keeps the same deterministic acceptance, safety, verification, reporting, and publication rules around it.
278
+
279
+ ## What DevAgent adds around an AI coding model
280
+
281
+ | Common engineering risk | DevAgent behavior |
282
+ | --- | --- |
283
+ | The model says “done” without enough proof | Required acceptance criteria remain `UNPROVEN` or the run becomes `PARTIALLY_VERIFIED` / `BLOCKED` instead of falsely claiming success. |
284
+ | A patch touches unrelated code | Evidence gathering, explicit scope, minimal-change planning, and independent diff review constrain the change. |
285
+ | Existing developer work is damaged | Clean repositories use isolated worktrees by default; dirty tracked/untracked developer work is protected; files are backed up before first modification. |
286
+ | Tests passed before a later edit | Verification is revision-aware, so older successful evidence does not prove a newer tree. |
287
+ | A generated change breaks the build or tests | DevAgent runs repository-supported targeted/broad checks and can diagnose/replan before final verification. |
288
+ | An agent pushes directly to a protected primary branch | Starting from `main`, `master`, or `trunk` causes DevAgent to work on a safe branch; runtime DevAgent does not merge or deploy. |
289
+ | You are locked to one model vendor | OpenAI, Claude, Gemini, Grok/xAI, and compatible endpoints can use the same engineering harness. |
290
+ | It is hard to audit what the agent actually did | DevAgent emits an engineering report with decisions, changed symbols, tests, acceptance evidence, verification, failures, gaps, and source-control status. |
291
+
292
+ The goal is not to replace developer judgment. The goal is to make autonomous engineering work **bounded, reviewable, reproducible, and harder to falsely declare complete**.
293
+
188
294
  Useful commands:
189
295
 
190
296
  ```bash
@@ -199,7 +305,7 @@ devagent benchmark --help
199
305
 
200
306
  ### Pinned real-world benchmark
201
307
 
202
- DevAgent 0.5 adds an opt-in benchmark runner for pinned GitHub repositories. A benchmark case injects a deterministic defect into an exact commit and uses an **external oracle** before and after DevAgent. This avoids treating DevAgent's own report as the benchmark oracle.
308
+ DevAgent includes an opt-in benchmark runner for pinned GitHub repositories. A benchmark case injects a deterministic defect into an exact commit and uses an **external oracle** before and after DevAgent. This avoids treating DevAgent's own report as the benchmark oracle.
203
309
 
204
310
  ```bash
205
311
  devagent benchmark \
@@ -247,7 +353,7 @@ DevAgent uses defense-in-depth controls around repository modification, command
247
353
  - protected targets are refused and force push is never used;
248
354
  - no runtime PR, merge, rebase, force-push, or deployment automation.
249
355
 
250
- DevAgent is **not an operating-system sandbox**. Review the report and pushed branch before integrating customer or production code.
356
+ On Linux, DevAgent can execute engineering commands inside a bubblewrap-based operating-system sandbox. Production qualification exercises required sandbox mode with network access denied. Required mode fails closed when isolation cannot be established rather than silently falling back. Review the report and pushed branch before integrating customer or production code: sandboxing reduces execution risk, but it does not make arbitrary generated changes universally safe.
251
357
 
252
358
  ## Repository intelligence and verification
253
359
 
@@ -267,13 +373,20 @@ Verification can include baseline tests, targeted tests, component/broad checks,
267
373
 
268
374
  ## Production qualification
269
375
 
270
- DevAgent 0.5.0 uses **production qualification v4**. It extends the v3 release gate with large-repository bounded-retrieval and real-world benchmark truthfulness contracts while preserving the primary invariant:
376
+ DevAgent 0.8.0 uses cumulative production qualification rather than replacing older evidence with a smaller new suite.
377
+
378
+ - **v4 — 70 required cases** covering end-to-end engineering behavior, acceptance truthfulness, task/risk scope, provider contracts and parity, model routing, worktree and Git publication safety, CLI input, review/repair loops, report/evaluation integrity, release integrity, large-repository behavior, structural refactors, Java/.NET discovery and execution, SQLite migration forward/rollback, and real repository-native stacks.
379
+ - **v5 — 9 required autonomy cases** covering bounded parallel coordination, dirty-source refusal, real isolated parallel DevAgent runs, bounded/relevant skills and provider injection, automation overlap claim/recovery, and provider-benchmark deduplication, live structured-contract behavior, and secret redaction.
380
+
381
+ The v0.8 merge commit on `main` passed both catalogs in required Linux sandbox mode:
271
382
 
272
383
  ```text
273
- false_verified == 0
384
+ v4: 70/70 passed
385
+ v5: 9/9 passed
386
+ combined: 79/79 passed
274
387
  ```
275
388
 
276
- It covers end-to-end engineering behavior, acceptance truthfulness, task scope, provider contracts/parity, model routing, worktree and Git publication safety, CLI input, review/repair loops, report/evaluation integrity, release integrity, and actual repository-native toolchain execution for:
389
+ The qualification environment exercises real local toolchains for:
277
390
 
278
391
  ```text
279
392
  Python / pytest
@@ -281,21 +394,30 @@ Node + TypeScript repository discovery
281
394
  Go
282
395
  Rust / Cargo
283
396
  C++ / Make
397
+ Java / Maven
398
+ .NET build
399
+ SQLite migration forward + rollback
284
400
  ```
285
401
 
286
- Run the release qualification locally on a machine with those toolchains:
402
+ Run the same release qualification catalogs locally on a machine with the required toolchains:
287
403
 
288
404
  ```bash
405
+ DEVAGENT_SANDBOX=required DEVAGENT_NETWORK=deny \
289
406
  python -m devagent.qualification \
290
407
  --catalog evaluation/benchmark_v4.json \
291
408
  --report .devagent/production-qualification-v4.json
409
+
410
+ DEVAGENT_SANDBOX=required DEVAGENT_NETWORK=deny \
411
+ python -m devagent.qualification \
412
+ --catalog evaluation/benchmark_v5.json \
413
+ --report .devagent/production-qualification-v5.json
292
414
  ```
293
415
 
294
- Production CI runs Python 3.10/3.11/3.12, a clean wheel install, and this production qualification gate. The qualification JSON is retained as CI evidence.
416
+ Production CI also runs Python 3.10/3.11/3.12, a clean wheel build/install, real bubblewrap sandbox smoke, and both qualification catalogs. Qualification JSON is retained as CI evidence.
295
417
 
296
- **100% qualified means 100% of this explicit catalog passed.** It does not mean mathematical correctness for every unseen repository, environment, model response, language, or engineering task.
418
+ **100% qualified means 100% of these explicit catalogs passed on that revision and environment.** It does not mean mathematical correctness for every unseen repository, environment, model response, language, provider, or engineering task, and it is not a claim that DevAgent is universally superior to every hosted coding platform.
297
419
 
298
- See [docs/production-readiness.md](docs/production-readiness.md) for the evidence behind the 0.4.0 production-readiness assessment and its explicit limitations.
420
+ See [docs/production-readiness.md](docs/production-readiness.md) for the project's earlier readiness assessment and its explicit limitations.
299
421
 
300
422
  ## Provider architecture
301
423
 
@@ -349,11 +471,15 @@ Automated provider tests normally use deterministic or mocked clients and do not
349
471
 
350
472
  ## Project status
351
473
 
352
- DevAgent 0.5 is **beta software**. Its qualification and benchmark results are bounded claims tied to explicit cases, pinned revisions, and external oracles; they are not a claim of universal correctness or parity with every hosted coding platform.
474
+ DevAgent 0.8.0 is **beta software with a verified core release baseline**. The exact v0.8 merge revision on `main` passed Production CI across Python 3.10/3.11/3.12, clean wheel installation, real Linux bubblewrap sandbox execution, production qualification v4 (**70/70**), and autonomy qualification v5 (**9/9**), for **79/79 cumulative required qualification cases**.
475
+
476
+ The current core includes evidence-backed `VERIFIED` / `PARTIALLY_VERIFIED` / `BLOCKED` outcomes, backup-first editing, isolated worktrees, bounded structural file operations, repository-native verification, independent review, safe branch publication, provider/model choice, Java and .NET engineering discovery/execution, SQLite migration forward/rollback verification, large-monorepo deep-manifest discovery, bounded parallel agents, repository-local skills, foreground automations, Linux OS sandboxing, bounded browser/local-UI verification, and real-provider structured-contract benchmarking.
477
+
478
+ These results are **bounded engineering claims**, not universal-correctness or market-superiority claims. They are tied to explicit qualification cases, pinned revisions, deterministic fixtures/external oracles where applicable, and the environments actually exercised by CI.
353
479
 
354
- Remaining gaps include a larger published corpus of pinned upstream benchmark cases, browser/UI runtime qualification, a broader Java/.NET/database-migration matrix, very large monorepo stress runs above the current bounded inventory, parallel multi-agent orchestration, operating-system sandboxing, and continuous paid real-provider testing across every model/provider combination.
480
+ Remaining work is primarily **breadth and external validation**, not missing core architecture: a larger public corpus of pinned upstream repositories and tasks; broader browser/UI coverage across dynamic applications and multiple browser environments; a wider Java/Gradle, .NET test-framework, and PostgreSQL/MySQL migration matrix beyond the current qualified fixtures; larger and more diverse monorepo stress cases beyond the current >12,000-file deep-manifest case; more real-world multi-agent workload studies; and continuous paid real-provider benchmarking across a broader set of model/provider combinations. GitHub branch protection/rulesets are external repository settings and must be configured separately; DevAgent does not claim to configure them itself.
355
481
 
356
- The project intentionally prioritizes trustworthy outcomes over feature count.
482
+ The project intentionally prioritizes trustworthy outcomes, reproducible evidence, and safe engineering behavior over feature count or unsupported "best agent" claims.
357
483
 
358
484
  ## Contributing
359
485
 
@@ -371,4 +497,4 @@ DevAgent is open source under the [MIT License](LICENSE).
371
497
  Copyright (c) 2026 Tom Ha
372
498
  ```
373
499
 
374
- DevAgent was created by **Tom Ha**. Original repository: **https://github.com/tomha85/devagent**. See [NOTICE](NOTICE) and [COPYRIGHT](COPYRIGHT) for project attribution.
500
+ DevAgent was created by **Tom Ha**. Original repository: **https://github.com/tomha85/devagent**. See [NOTICE](NOTICE) and [COPYRIGHT](COPYRIGHT) for project attribution.
@@ -1,3 +1,3 @@
1
1
  """DevAgent: evidence-driven local software engineering automation."""
2
2
 
3
- __version__ = "0.8.0"
3
+ __version__ = "0.8.2"
@@ -186,6 +186,19 @@ class EngineeringPlan:
186
186
  verification: list[tuple[str, ...]]
187
187
  rationale: str
188
188
 
189
+ def __post_init__(self) -> None:
190
+ # A path-scoped `git diff -- <paths...>` is planner inspection, not a
191
+ # repository verification capability. DevAgent already captures and
192
+ # independently reviews the final diff. Keeping this command in the
193
+ # executable verification plan makes an otherwise valid plan fail the
194
+ # evidence-backed command allowlist before implementation can begin.
195
+ # Preserve real Git verification such as `git diff --check`.
196
+ self.verification = [
197
+ command
198
+ for command in self.verification
199
+ if not (len(command) > 3 and command[:3] == ("git", "diff", "--"))
200
+ ]
201
+
189
202
 
190
203
  _PRESERVATION_PATTERNS = (
191
204
  re.compile(r"\bpreserv(?:e|es|ed|ing)\s+(?:the\s+)?existing\s+([^.;\n]+)", re.IGNORECASE),
@@ -63,6 +63,18 @@ _DIRECTIVE = re.compile(
63
63
  re.IGNORECASE,
64
64
  )
65
65
 
66
+ # Bounded normalization for terse user intent. This is deliberately not a fuzzy
67
+ # "guess what the user meant" layer: it corrects common engineering shorthand,
68
+ # grammatical number, and operation wording while preserving identifiers,
69
+ # quoted contracts, values, and explicit constraints. Task policy and repository
70
+ # evidence still provide the verification/safety contract.
71
+ _OPERATION_ALIASES: tuple[tuple[str, str], ...] = (
72
+ ("substraction", "subtraction"),
73
+ ("substract", "subtract"),
74
+ ("multipy", "multiply"),
75
+ ("mutiply", "multiply"),
76
+ )
77
+
66
78
 
67
79
  def _classify(text: str) -> TaskType:
68
80
  lowered = text.lower()
@@ -98,6 +110,60 @@ def _dedupe(items: list[str]) -> list[str]:
98
110
  return result
99
111
 
100
112
 
113
+ def _normalize_terse_requirement(requirement: str) -> str:
114
+ """Compile common rough one-line prompts into a clearer engineering request.
115
+
116
+ The compiler is intentionally bounded. It may repair shorthand/grammar and
117
+ make an operation explicit, but it must not add product behavior the user did
118
+ not request. Structured/multi-line requirements are left intact.
119
+ """
120
+
121
+ value = re.sub(r"\s+", " ", requirement).strip()
122
+ if not value or "\n" in requirement or _section_header(value) is not None:
123
+ return value
124
+ # An explicit callable name is already a precise user contract; never rename it.
125
+ if re.search(r"\b[A-Za-z_][A-Za-z0-9_]*\s*\(", value):
126
+ return value
127
+
128
+ for source, destination in _OPERATION_ALIASES:
129
+ value = re.sub(rf"\b{re.escape(source)}\b", destination, value, flags=re.IGNORECASE)
130
+
131
+ # Common shorthand from natural prompts such as "addition 2 matrix 2x2".
132
+ # Keep both "matrix" and "matrices" in the normalized contract so
133
+ # deterministic evidence can link either conventional symbol spelling.
134
+ matrix_match = re.search(
135
+ r"\b(add(?:ition)?|sum|subtract(?:ion)?|multiply|multiplication|divide|division)\b"
136
+ r"(?:\s+(?:of|for))?\s+(?:2|two)\s+matrix(?:es)?\s+(\d+x\d+)\b",
137
+ value,
138
+ flags=re.IGNORECASE,
139
+ )
140
+ if matrix_match:
141
+ operation = matrix_match.group(1).lower()
142
+ dimension = matrix_match.group(2).lower()
143
+ canonical_operation = {
144
+ "add": "addition",
145
+ "addition": "addition",
146
+ "sum": "addition",
147
+ "subtract": "subtraction",
148
+ "subtraction": "subtraction",
149
+ "multiply": "multiplication",
150
+ "multiplication": "multiplication",
151
+ "divide": "division",
152
+ "division": "division",
153
+ }[operation]
154
+ prefix = "Add" if re.search(r"\b(add|new|function|implement)\b", value, re.IGNORECASE) else "Implement"
155
+ return (
156
+ f"{prefix} a matrix {canonical_operation} function for two {dimension} matrices "
157
+ f"(matrix inputs)"
158
+ )
159
+
160
+ # Repair simple count+noun shorthand without inventing domain behavior.
161
+ value = re.sub(r"\b2\s+matrix\b", "two matrices", value, flags=re.IGNORECASE)
162
+ value = re.sub(r"\b2\s+file\b", "two files", value, flags=re.IGNORECASE)
163
+ value = re.sub(r"\b2\s+test\b", "two tests", value, flags=re.IGNORECASE)
164
+ return value
165
+
166
+
101
167
  def _section_header(line: str) -> tuple[str, str] | None:
102
168
  stripped = line.strip()
103
169
  markdown = re.match(r"^#{1,6}\s+(.+?)\s*$", stripped)
@@ -182,15 +248,21 @@ def _append_criterion(
182
248
 
183
249
 
184
250
  def compile_task(requirement: str) -> TaskSpec:
185
- goal = re.sub(r"\s+", " ", requirement).strip()
186
- if not goal:
251
+ raw_goal = re.sub(r"\s+", " ", requirement).strip()
252
+ if not raw_goal:
187
253
  raise ValueError("Engineering requirement cannot be empty")
254
+ goal = _normalize_terse_requirement(requirement)
188
255
  task_type = _classify(goal)
189
256
  code_change = task_type is not TaskType.UNIT_TEST or "only" not in goal.lower()
190
257
  requires_tests = task_type is not TaskType.BUILD_FAILURE
191
258
 
192
259
  criteria: list[AcceptanceCriterion] = []
193
- for item in _user_acceptance_items(requirement):
260
+ # Structured user requirements remain authoritative. Only an unstructured,
261
+ # terse prompt is compiled into the clearer canonical request.
262
+ user_items = _user_acceptance_items(requirement)
263
+ if len(user_items) == 1 and user_items[0] == raw_goal and goal != raw_goal:
264
+ user_items = [goal]
265
+ for item in user_items:
194
266
  _append_criterion(criteria, item, source=AcceptanceSource.USER)
195
267
 
196
268
  if task_type in {TaskType.BUG_FIX, TaskType.RUNTIME_ERROR, TaskType.TEST_FAILURE}:
@@ -240,9 +312,61 @@ def compile_task(requirement: str) -> TaskSpec:
240
312
  )
241
313
 
242
314
 
315
+ def _repository_language(repository: Any) -> str | None:
316
+ languages = [
317
+ language.lower()
318
+ for component in repository.components
319
+ for language in component.languages
320
+ ]
321
+ for preferred in ("python", "java", "csharp", "c#", "javascript", "typescript", "go", "rust"):
322
+ if preferred in languages:
323
+ return preferred
324
+ return languages[0] if languages else None
325
+
326
+
327
+ def _matrix_operation_contract(task: TaskSpec, repository: Any) -> None:
328
+ """Turn the bounded matrix shorthand compiler output into a repo-style callable contract."""
329
+
330
+ match = re.fullmatch(
331
+ r"(?:Add|Implement) a matrix (addition|subtraction|multiplication|division) "
332
+ r"function for two (\d+x\d+) matrices \(matrix inputs\)",
333
+ task.goal,
334
+ )
335
+ if match is None:
336
+ return
337
+
338
+ operation, dimension = match.groups()
339
+ verb = {
340
+ "addition": "add",
341
+ "subtraction": "subtract",
342
+ "multiplication": "multiply",
343
+ "division": "divide",
344
+ }[operation]
345
+ compact_dimension = dimension.replace("x", "x")
346
+ language = _repository_language(repository)
347
+ if language in {"java", "javascript", "typescript"}:
348
+ symbol = f"{verb}Matrices{compact_dimension}"
349
+ elif language in {"csharp", "c#"}:
350
+ symbol = f"{verb.capitalize()}Matrices{compact_dimension}"
351
+ else:
352
+ symbol = f"{verb}_matrices_{compact_dimension}"
353
+
354
+ compiled = (
355
+ f"Add {symbol}(a, b) to perform element-wise matrix {operation} "
356
+ f"for two {dimension} matrices"
357
+ )
358
+ task.goal = compiled
359
+ user_criteria = [
360
+ criterion for criterion in task.acceptance_criteria if criterion.source is AcceptanceSource.USER
361
+ ]
362
+ if len(user_criteria) == 1:
363
+ user_criteria[0].description = compiled
364
+
365
+
243
366
  def enrich_acceptance_contract(task: TaskSpec, repository: Any) -> TaskSpec:
244
- """Add checks derived from trusted repository capabilities without replacing user intent."""
367
+ """Compile safe repository-aware defaults, then add trusted repository checks."""
245
368
 
369
+ _matrix_operation_contract(task, repository)
246
370
  seen_commands: set[tuple[str, ...]] = set()
247
371
  for capability in repository.capabilities:
248
372
  if not capability.trusted: