qaas-python 0.2.2__tar.gz → 0.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (166) hide show
  1. {qaas_python-0.2.2 → qaas_python-0.3.0}/CLAUDE.md +5 -3
  2. {qaas_python-0.2.2 → qaas_python-0.3.0}/PKG-INFO +28 -3
  3. {qaas_python-0.2.2 → qaas_python-0.3.0}/README.md +27 -2
  4. {qaas_python-0.2.2 → qaas_python-0.3.0}/pyproject.toml +1 -1
  5. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/cli.py +16 -23
  6. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/conductor.py +37 -7
  7. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/config.py +4 -2
  8. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/defaults/config/agents/arbiter.yaml +0 -1
  9. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/defaults/config/agents/cartographer.yaml +0 -1
  10. qaas_python-0.3.0/src/qaas/defaults/config/agents/chronicle.yaml +19 -0
  11. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/defaults/config/agents/clerk.yaml +0 -1
  12. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/defaults/config/agents/conduit.yaml +0 -1
  13. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/defaults/config/agents/forge.yaml +0 -1
  14. qaas_python-0.3.0/src/qaas/defaults/config/agents/gauge.yaml +26 -0
  15. qaas_python-0.3.0/src/qaas/defaults/config/agents/keystone.yaml +21 -0
  16. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/defaults/config/agents/mender.yaml +0 -1
  17. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/defaults/config/agents/proof.yaml +0 -1
  18. qaas_python-0.3.0/src/qaas/defaults/config/agents/pulse.yaml +23 -0
  19. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/defaults/config/agents/surface.yaml +0 -1
  20. qaas_python-0.3.0/src/qaas/defaults/config/agents/usher.yaml +23 -0
  21. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/defaults/config/agents/vault.yaml +0 -1
  22. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/defaults/config/agents/warden.yaml +0 -4
  23. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/defaults/config/system.yaml +12 -17
  24. qaas_python-0.3.0/src/qaas/prompts/CHRONICLE.md +61 -0
  25. qaas_python-0.3.0/src/qaas/prompts/GAUGE.md +109 -0
  26. qaas_python-0.3.0/src/qaas/prompts/KEYSTONE.md +80 -0
  27. qaas_python-0.3.0/src/qaas/prompts/PULSE.md +100 -0
  28. qaas_python-0.3.0/src/qaas/prompts/USHER.md +94 -0
  29. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/target.py +8 -1
  30. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/tasks.py +34 -0
  31. {qaas_python-0.2.2 → qaas_python-0.3.0}/tests/test_conductor.py +55 -8
  32. {qaas_python-0.2.2 → qaas_python-0.3.0}/tests/test_config.py +32 -17
  33. {qaas_python-0.2.2 → qaas_python-0.3.0}/tests/test_paths.py +3 -3
  34. {qaas_python-0.2.2 → qaas_python-0.3.0}/tests/test_trace.py +19 -5
  35. {qaas_python-0.2.2 → qaas_python-0.3.0}/.gitignore +0 -0
  36. {qaas_python-0.2.2 → qaas_python-0.3.0}/ARCHITECTURE.md +0 -0
  37. {qaas_python-0.2.2 → qaas_python-0.3.0}/BUILD_PLAN.md +0 -0
  38. {qaas_python-0.2.2 → qaas_python-0.3.0}/LICENSE +0 -0
  39. {qaas_python-0.2.2 → qaas_python-0.3.0}/config/targets/corvid.yaml +0 -0
  40. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/adapters/__init__.py +0 -0
  41. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/adapters/tracker.py +0 -0
  42. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/adapters/vcs.py +0 -0
  43. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/discover.py +0 -0
  44. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/envelope.py +0 -0
  45. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/guardrails.py +0 -0
  46. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/mcp/__init__.py +0 -0
  47. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/mcp/context.py +0 -0
  48. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/mcp/contract_diff.py +0 -0
  49. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/mcp/defect_memory.py +0 -0
  50. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/mcp/env_control.py +0 -0
  51. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/mcp/envelope_server.py +0 -0
  52. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/mcp/test_runner.py +0 -0
  53. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/mcp/tracker.py +0 -0
  54. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/mcp/vcs.py +0 -0
  55. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/paths.py +0 -0
  56. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/plugin/.claude-plugin/plugin.json +0 -0
  57. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/plugin/skills/a11y-audit/SKILL.md +0 -0
  58. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/plugin/skills/adversarial-review/SKILL.md +0 -0
  59. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/plugin/skills/api-surface-extraction/SKILL.md +0 -0
  60. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/plugin/skills/authz-matrix-check/SKILL.md +0 -0
  61. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/plugin/skills/console-error-triage/SKILL.md +0 -0
  62. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/plugin/skills/contract-test-generation/SKILL.md +0 -0
  63. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/plugin/skills/dedupe-strategy/SKILL.md +0 -0
  64. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/plugin/skills/environment-pinning/SKILL.md +0 -0
  65. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/plugin/skills/error-taxonomy/SKILL.md +0 -0
  66. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/plugin/skills/exploratory-ui-walk/SKILL.md +0 -0
  67. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/plugin/skills/failing-test-authoring/SKILL.md +0 -0
  68. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/plugin/skills/flake-detection/SKILL.md +0 -0
  69. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/plugin/skills/form-state-probe/SKILL.md +0 -0
  70. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/plugin/skills/minimal-diff-discipline/SKILL.md +0 -0
  71. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/plugin/skills/openapi-diff/SKILL.md +0 -0
  72. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/plugin/skills/ownership-resolution/SKILL.md +0 -0
  73. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/plugin/skills/product-task-graph/SKILL.md +0 -0
  74. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/plugin/skills/regression-risk-scoring/SKILL.md +0 -0
  75. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/plugin/skills/regression-suite-selection/SKILL.md +0 -0
  76. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/plugin/skills/repo-cartography/SKILL.md +0 -0
  77. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/plugin/skills/repro-minimisation/SKILL.md +0 -0
  78. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/plugin/skills/rollback-plan-authoring/SKILL.md +0 -0
  79. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/plugin/skills/root-cause-vs-symptom/SKILL.md +0 -0
  80. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/plugin/skills/routing-rules/SKILL.md +0 -0
  81. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/plugin/skills/severity-rubric/SKILL.md +0 -0
  82. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/plugin/skills/test-first-fix/SKILL.md +0 -0
  83. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/plugin/skills/test-quality-audit/SKILL.md +0 -0
  84. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/plugin/skills/ticket-writer/SKILL.md +0 -0
  85. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/plugin/skills/verdict-reporting/SKILL.md +0 -0
  86. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/plugin/skills/verification-protocol/SKILL.md +0 -0
  87. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/prompts/ARBITER.md +0 -0
  88. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/prompts/CARTOGRAPHER.md +0 -0
  89. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/prompts/CLERK.md +0 -0
  90. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/prompts/CONDUIT.md +0 -0
  91. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/prompts/FORGE.md +0 -0
  92. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/prompts/MENDER.md +0 -0
  93. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/prompts/PROOF.md +0 -0
  94. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/prompts/SURFACE.md +0 -0
  95. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/prompts/VAULT.md +0 -0
  96. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/prompts/WARDEN.md +0 -0
  97. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/prompts/_shared.md +0 -0
  98. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/registry.py +0 -0
  99. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/runner.py +0 -0
  100. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/scorecard.py +0 -0
  101. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/sdk_compat.py +0 -0
  102. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/store.py +0 -0
  103. {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/trace.py +0 -0
  104. {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/CODEOWNERS +0 -0
  105. {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/api/Dockerfile +0 -0
  106. {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/api/app/__init__.py +0 -0
  107. {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/api/app/auth.py +0 -0
  108. {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/api/app/config.py +0 -0
  109. {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/api/app/db.py +0 -0
  110. {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/api/app/errors.py +0 -0
  111. {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/api/app/main.py +0 -0
  112. {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/api/app/models.py +0 -0
  113. {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/api/app/routes/__init__.py +0 -0
  114. {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/api/app/routes/auth.py +0 -0
  115. {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/api/app/routes/invoices.py +0 -0
  116. {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/api/app/routes/orders.py +0 -0
  117. {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/api/app/routes/stream.py +0 -0
  118. {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/api/app/schemas.py +0 -0
  119. {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/api/migrations/001_init.sql +0 -0
  120. {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/api/pyproject.toml +0 -0
  121. {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/api/seed/fixtures.sql +0 -0
  122. {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/defects.yaml +0 -0
  123. {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/docker-compose.yml +0 -0
  124. {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/openapi.yaml +0 -0
  125. {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/web/.gitignore +0 -0
  126. {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/web/Dockerfile +0 -0
  127. {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/web/index.html +0 -0
  128. {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/web/package-lock.json +0 -0
  129. {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/web/package.json +0 -0
  130. {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/web/src/api.ts +0 -0
  131. {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/web/src/components/Button.tsx +0 -0
  132. {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/web/src/components/Layout.tsx +0 -0
  133. {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/web/src/components/SearchInput.tsx +0 -0
  134. {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/web/src/main.tsx +0 -0
  135. {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/web/src/routes/CheckoutReview.tsx +0 -0
  136. {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/web/src/routes/Login.tsx +0 -0
  137. {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/web/src/routes/NewOrder.tsx +0 -0
  138. {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/web/src/routes/OrderDetail.tsx +0 -0
  139. {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/web/src/routes/OrdersList.tsx +0 -0
  140. {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/web/src/styles.css +0 -0
  141. {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/web/tsconfig.json +0 -0
  142. {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/web/vite.config.ts +0 -0
  143. {qaas_python-0.2.2 → qaas_python-0.3.0}/tests/adapters/test_github_vcs.py +0 -0
  144. {qaas_python-0.2.2 → qaas_python-0.3.0}/tests/adapters/test_jira_tracker.py +0 -0
  145. {qaas_python-0.2.2 → qaas_python-0.3.0}/tests/conftest.py +0 -0
  146. {qaas_python-0.2.2 → qaas_python-0.3.0}/tests/mcp/conftest.py +0 -0
  147. {qaas_python-0.2.2 → qaas_python-0.3.0}/tests/mcp/test_contract_diff.py +0 -0
  148. {qaas_python-0.2.2 → qaas_python-0.3.0}/tests/mcp/test_defect_memory.py +0 -0
  149. {qaas_python-0.2.2 → qaas_python-0.3.0}/tests/mcp/test_env_control.py +0 -0
  150. {qaas_python-0.2.2 → qaas_python-0.3.0}/tests/mcp/test_test_runner.py +0 -0
  151. {qaas_python-0.2.2 → qaas_python-0.3.0}/tests/mcp/test_tracker.py +0 -0
  152. {qaas_python-0.2.2 → qaas_python-0.3.0}/tests/mcp/test_vcs.py +0 -0
  153. {qaas_python-0.2.2 → qaas_python-0.3.0}/tests/support.py +0 -0
  154. {qaas_python-0.2.2 → qaas_python-0.3.0}/tests/target_app/test_seeded_defects.py +0 -0
  155. {qaas_python-0.2.2 → qaas_python-0.3.0}/tests/test_cli.py +0 -0
  156. {qaas_python-0.2.2 → qaas_python-0.3.0}/tests/test_discover.py +0 -0
  157. {qaas_python-0.2.2 → qaas_python-0.3.0}/tests/test_envelope.py +0 -0
  158. {qaas_python-0.2.2 → qaas_python-0.3.0}/tests/test_guardrails.py +0 -0
  159. {qaas_python-0.2.2 → qaas_python-0.3.0}/tests/test_hooks_and_skills.py +0 -0
  160. {qaas_python-0.2.2 → qaas_python-0.3.0}/tests/test_prompt_overrides.py +0 -0
  161. {qaas_python-0.2.2 → qaas_python-0.3.0}/tests/test_registry.py +0 -0
  162. {qaas_python-0.2.2 → qaas_python-0.3.0}/tests/test_scorecard.py +0 -0
  163. {qaas_python-0.2.2 → qaas_python-0.3.0}/tests/test_skills_actually_load.py +0 -0
  164. {qaas_python-0.2.2 → qaas_python-0.3.0}/tests/test_store.py +0 -0
  165. {qaas_python-0.2.2 → qaas_python-0.3.0}/tests/test_target_profile_is_honoured.py +0 -0
  166. {qaas_python-0.2.2 → qaas_python-0.3.0}/tests/test_user_mcp_servers.py +0 -0
@@ -52,8 +52,8 @@ default `pytest` run is offline and free, and must stay that way.
52
52
  ### Phase pipeline (`conductor.py`)
53
53
 
54
54
  ```
55
- map -> discover -> reproduce -> file -> verify
56
- CARTOGRAPHER CONDUIT/SURFACE FORGE CLERK PROOF
55
+ map -> discover -> reproduce -> file -> verify -> report
56
+ CARTOGRAPHER CONDUIT/SURFACE/… FORGE CLERK PROOF CHRONICLE
57
57
  ```
58
58
 
59
59
  Discovery agents run concurrently up to the mode's cap; FORGE runs **once per
@@ -80,7 +80,9 @@ adding an agent as a sign something is wrong. Constraints enforced in
80
80
  `config.py`: at most 6 MCP servers per agent (§5.3, tool-selection accuracy),
81
81
  and every `must_call` tool must name a server the agent actually has.
82
82
 
83
- Adding a discovery agent is a prompt file plus a YAML file and no Python --
83
+ Discovery AND reporting agents dispatch **by layer**; every other phase
84
+ dispatches by name. So adding a discovery or reporting agent is a prompt file
85
+ plus a YAML file and no Python --
84
86
  `_phase_discover` falls back to `tasks.discovery` for anything without a
85
87
  bespoke builder. VAULT and WARDEN were added that way and found that it was
86
88
  not true before them.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: qaas-python
3
- Version: 0.2.2
3
+ Version: 0.3.0
4
4
  Summary: A multi-agent QA system: finds real defects, reproduces them, files tickets, fixes them, and proves the fix
5
5
  Project-URL: Homepage, https://github.com/allaabdella2-us/qa-multi-agent-system
6
6
  Project-URL: Repository, https://github.com/allaabdella2-us/qa-multi-agent-system
@@ -84,6 +84,31 @@ Nothing crosses between the loops except a ticket — which is also the audit tr
84
84
 
85
85
  ---
86
86
 
87
+ ## 🤖 The roster
88
+
89
+ | agent | layer | what it does |
90
+ |---|---|---|
91
+ | 🗺️ CARTOGRAPHER | map | services, routes, schema, ownership → the system map everything reads |
92
+ | 🏛️ KEYSTONE | discovery | circular deps, layering violations, god modules, dead code |
93
+ | 🔌 CONDUIT | discovery | API contract drift, authz gaps, error-shape inconsistency |
94
+ | 🖱️ SURFACE | discovery | drives the UI through real journeys |
95
+ | 🗄️ VAULT | discovery | schema constraints the code assumes and the database does not enforce |
96
+ | 🔒 WARDEN | discovery | missing authorization, secrets, vulnerable dependencies, leaks |
97
+ | 📡 PULSE | discovery | WebSocket auth, reconnect, ordering, backpressure |
98
+ | 🧭 USHER | discovery | whether a person can *find* a feature, not just whether it works |
99
+ | ⏱️ GAUGE | discovery | N+1 queries, unindexed hot paths, unbounded results, bundle outliers |
100
+ | 🔨 FORGE | triage | reproduces, minimises, measures flake, commits a failing test |
101
+ | 📝 CLERK | triage | dedupes, scores severity, routes, files — the only tracker writer |
102
+ | 🔧 MENDER | remediation | the minimal fix, on a branch |
103
+ | ⚖️ ARBITER | remediation | adversarial review: APPROVE / REQUEST_CHANGES / ESCALATE |
104
+ | ✅ PROOF | verify | re-runs the original test → VERIFIED / NOT_FIXED / REGRESSED |
105
+ | 📊 CHRONICLE | reporting | what the run found, what recurred, and what it could not reach |
106
+
107
+ **CONDUCTOR** is the sixteenth. It is the Python state machine rather than an
108
+ agent — a model cannot enforce a budget it is itself spending.
109
+
110
+ ---
111
+
87
112
  ## 🚀 Quickstart in 60 seconds
88
113
 
89
114
  ```bash
@@ -318,8 +343,8 @@ precision is measured rather than assumed.
318
343
 
319
344
  Honest about what exists:
320
345
 
321
- - ✅ **10 of the 16 agents** in the design are built — CARTOGRAPHER, CONDUIT, SURFACE, VAULT, WARDEN, FORGE, CLERK, MENDER, ARBITER, PROOF. CONDUCTOR is the Python state machine rather than an agent. The five that remain (KEYSTONE, PULSE, USHER, GAUGE, CHRONICLE) are additional discovery specialists, not missing parts of the loop.
322
- - ✅ **Adding an agent needs a prompt file and a YAML file — no Python.** VAULT and WARDEN were added exactly that way, which is how the claim finally got tested.
346
+ - ✅ **All 16 agents in the design are built.**
347
+ - ✅ **Adding one needs a prompt file and a YAML file — no Python.** Six were added that way, which is how the claim got tested.
323
348
  - ✅ The fix loop has closed end to end on a real defect: `NOT_FIXED → MENDER → ARBITER APPROVE → VERIFIED`.
324
349
  - ✅ 30 skills, 7 in-process MCP servers, 649 offline tests.
325
350
  - ⚠️ Running the bundled demo needs `export CORVID_PASSWORD=password123` — credentials come from the environment, including the demo's.
@@ -54,6 +54,31 @@ Nothing crosses between the loops except a ticket — which is also the audit tr
54
54
 
55
55
  ---
56
56
 
57
+ ## 🤖 The roster
58
+
59
+ | agent | layer | what it does |
60
+ |---|---|---|
61
+ | 🗺️ CARTOGRAPHER | map | services, routes, schema, ownership → the system map everything reads |
62
+ | 🏛️ KEYSTONE | discovery | circular deps, layering violations, god modules, dead code |
63
+ | 🔌 CONDUIT | discovery | API contract drift, authz gaps, error-shape inconsistency |
64
+ | 🖱️ SURFACE | discovery | drives the UI through real journeys |
65
+ | 🗄️ VAULT | discovery | schema constraints the code assumes and the database does not enforce |
66
+ | 🔒 WARDEN | discovery | missing authorization, secrets, vulnerable dependencies, leaks |
67
+ | 📡 PULSE | discovery | WebSocket auth, reconnect, ordering, backpressure |
68
+ | 🧭 USHER | discovery | whether a person can *find* a feature, not just whether it works |
69
+ | ⏱️ GAUGE | discovery | N+1 queries, unindexed hot paths, unbounded results, bundle outliers |
70
+ | 🔨 FORGE | triage | reproduces, minimises, measures flake, commits a failing test |
71
+ | 📝 CLERK | triage | dedupes, scores severity, routes, files — the only tracker writer |
72
+ | 🔧 MENDER | remediation | the minimal fix, on a branch |
73
+ | ⚖️ ARBITER | remediation | adversarial review: APPROVE / REQUEST_CHANGES / ESCALATE |
74
+ | ✅ PROOF | verify | re-runs the original test → VERIFIED / NOT_FIXED / REGRESSED |
75
+ | 📊 CHRONICLE | reporting | what the run found, what recurred, and what it could not reach |
76
+
77
+ **CONDUCTOR** is the sixteenth. It is the Python state machine rather than an
78
+ agent — a model cannot enforce a budget it is itself spending.
79
+
80
+ ---
81
+
57
82
  ## 🚀 Quickstart in 60 seconds
58
83
 
59
84
  ```bash
@@ -288,8 +313,8 @@ precision is measured rather than assumed.
288
313
 
289
314
  Honest about what exists:
290
315
 
291
- - ✅ **10 of the 16 agents** in the design are built — CARTOGRAPHER, CONDUIT, SURFACE, VAULT, WARDEN, FORGE, CLERK, MENDER, ARBITER, PROOF. CONDUCTOR is the Python state machine rather than an agent. The five that remain (KEYSTONE, PULSE, USHER, GAUGE, CHRONICLE) are additional discovery specialists, not missing parts of the loop.
292
- - ✅ **Adding an agent needs a prompt file and a YAML file — no Python.** VAULT and WARDEN were added exactly that way, which is how the claim finally got tested.
316
+ - ✅ **All 16 agents in the design are built.**
317
+ - ✅ **Adding one needs a prompt file and a YAML file — no Python.** Six were added that way, which is how the claim got tested.
293
318
  - ✅ The fix loop has closed end to end on a real defect: `NOT_FIXED → MENDER → ARBITER APPROVE → VERIFIED`.
294
319
  - ✅ 30 skills, 7 in-process MCP servers, 649 offline tests.
295
320
  - ⚠️ Running the bundled demo needs `export CORVID_PASSWORD=password123` — credentials come from the environment, including the demo's.
@@ -2,7 +2,7 @@
2
2
  # Distribution name is `qaas-python` (`qaas` was taken); the import package and
3
3
  # the CLI are both `qaas`.
4
4
  name = "qaas-python"
5
- version = "0.2.2"
5
+ version = "0.3.0"
6
6
  description = "A multi-agent QA system: finds real defects, reproduces them, files tickets, fixes them, and proves the fix"
7
7
  readme = "README.md"
8
8
  license = "MIT"
@@ -500,14 +500,18 @@ def validate(config_dir: Path | None = ConfigDir) -> None:
500
500
  # never dispatched. The mode meant for every pull request could not file a
501
501
  # ticket. It took a live run to notice; this check makes it free.
502
502
  for mode_name, mode in sorted(cfg.run_modes.items()):
503
- needed = sum(
504
- cfg.agents[a].max_budget_usd for a in mode.agents if a in cfg.agents
505
- )
506
- if needed > mode.max_budget_usd:
503
+ # Only meaningful when both sides declare a cap. The shipped config
504
+ # declares none, so this check simply does not fire there.
505
+ agent_caps = [
506
+ cfg.agents[a].max_budget_usd for a in mode.agents
507
+ if a in cfg.agents and cfg.agents[a].max_budget_usd is not None
508
+ ]
509
+ needed = sum(agent_caps)
510
+ if mode.max_budget_usd is not None and agent_caps and needed > mode.max_budget_usd:
507
511
  missing = [a for a in mode.agents if a in cfg.agents][-1]
508
512
  problems.append(
509
- f"mode '{mode_name}': agents can spend ${needed:.2f} but the cap is "
510
- f"${mode.max_budget_usd:.2f}, so the run stops before it reaches "
513
+ f"mode '{mode_name}': the agents' caps exceed the mode's cap, so "
514
+ f"its cap, so the run stops before it reaches "
511
515
  f"{missing} and files nothing. Raise max_budget_usd or drop an agent"
512
516
  )
513
517
 
@@ -567,7 +571,7 @@ def validate(config_dir: Path | None = ConfigDir) -> None:
567
571
  filing = "" if rm.files_tickets else " [dim](no filing)[/dim]"
568
572
  console.print(
569
573
  f"[bold]{mode}[/bold]: {', '.join(rm.agents)} "
570
- f"[dim]budget ${rm.max_budget_usd:.2f}, {rm.max_wall_clock_s}s[/dim]{filing}"
574
+ f"[dim]{rm.max_wall_clock_s}s[/dim]{filing}"
571
575
  )
572
576
 
573
577
  if notes:
@@ -834,7 +838,7 @@ def runs(root: Path = Root, limit: int = 10) -> None:
834
838
  console.print("[dim]no runs yet[/dim]")
835
839
  return
836
840
  table = Table(header_style="bold")
837
- for col in ("run", "envelopes", "agents", "cost"):
841
+ for col in ("run", "envelopes", "agents"):
838
842
  table.add_column(col)
839
843
  for run_id in ids:
840
844
  store = RunStore(run_id, root)
@@ -843,7 +847,6 @@ def runs(root: Path = Root, limit: int = 10) -> None:
843
847
  run_id,
844
848
  str(len(store.envelopes())),
845
849
  str(len(results)),
846
- f"${store.total_cost_usd():.2f}",
847
850
  )
848
851
  console.print(table)
849
852
 
@@ -872,9 +875,6 @@ def show(run_id: str, root: Path = Root) -> None:
872
875
  header.append(f"started {summary.started:%Y-%m-%d %H:%M:%S}Z")
873
876
  if summary.duration_s is not None:
874
877
  header.append(f"duration {summary.duration_s:.0f}s")
875
- header.append(f"cost ${summary.cost_usd:.2f}")
876
- if summary.budget_usd:
877
- header.append(f"of ${summary.budget_usd:.2f} budget")
878
878
  console.print(" " + " ".join(header))
879
879
  if summary.target_sha:
880
880
  dirty = " [yellow](dirty tree)[/yellow]" if summary.target_dirty else ""
@@ -960,7 +960,6 @@ def trace(
960
960
  table.add_column("agent", style="cyan")
961
961
  table.add_column("kind")
962
962
  table.add_column("detail", overflow="fold")
963
- table.add_column("cost", justify="right", style="dim")
964
963
  for row in trace_mod.timeline(entries):
965
964
  label = f"{row.kind} ×{row.count}" if row.count > 1 else row.kind
966
965
  table.add_row(
@@ -968,7 +967,6 @@ def trace(
968
967
  row.agent,
969
968
  f"[{KIND_STYLE.get(row.kind, 'white')}]{label}[/]",
970
969
  row.detail,
971
- f"${row.cost_usd:.2f}" if row.cost_usd is not None else "",
972
970
  )
973
971
  console.print(table)
974
972
  console.print(f"\n[dim]{len(entries)} entries[/dim]")
@@ -1082,7 +1080,7 @@ def run(
1082
1080
  rm = cfg.run_modes[mode]
1083
1081
  console.print(
1084
1082
  f"[bold]{mode}[/bold] — {len(specs)} agents, "
1085
- f"budget ${rm.max_budget_usd:.2f}, concurrency {rm.max_concurrency}"
1083
+ f"concurrency {rm.max_concurrency}"
1086
1084
  )
1087
1085
 
1088
1086
  if dry_run:
@@ -1093,7 +1091,7 @@ def run(
1093
1091
  d = describe(spec, prompt_dirs)
1094
1092
  console.print(
1095
1093
  f" [bold]{spec.name:14s}[/bold] {spec.model:18s} effort={spec.effort:7s} "
1096
- f"turns<={spec.max_turns:<3d} ${spec.max_budget_usd:.2f}"
1094
+ f"turns<={spec.max_turns}"
1097
1095
  )
1098
1096
  console.print(f" tools: {', '.join(d['allowed_tools'])}")
1099
1097
  console.print(f" prompt: {d['prompt_chars']} chars")
@@ -1101,11 +1099,11 @@ def run(
1101
1099
 
1102
1100
  def on_event(kind: str, detail: dict) -> None:
1103
1101
  if kind == "agent_started":
1104
- console.print(f"[dim]->[/dim] {detail.get('agent')} [dim](${detail.get('budget', 0):.2f})[/dim]")
1102
+ console.print(f"[dim]->[/dim] {detail.get('agent')}")
1105
1103
  elif kind == "finished":
1106
1104
  console.print(
1107
1105
  f"[dim]<-[/dim] {detail.get('agent')} "
1108
- f"[dim]${detail.get('cost', 0):.3f}, {detail.get('envelopes', 0)} findings[/dim]"
1106
+ f"[dim]{detail.get('envelopes', 0)} findings[/dim]"
1109
1107
  )
1110
1108
  elif kind == "stopped":
1111
1109
  console.print(f"[yellow]stopped: {detail.get('reason')}[/yellow]")
@@ -1172,11 +1170,6 @@ def score(
1172
1170
  table.add_row("false positives", f"{s['false_positives']} ({s['false_positive_rate']:.0%})")
1173
1171
  table.add_row("duplicates", f"{s['duplicates']} ({s['duplicate_rate']:.0%})")
1174
1172
  table.add_row("severity agreement", f"{s['severity_agreement']:.0%}")
1175
- table.add_row("cost", f"${s['cost_usd']:.2f}")
1176
- table.add_row(
1177
- "cost per accepted",
1178
- f"${s['cost_per_accepted']:.2f}" if s["cost_per_accepted"] is not None else "-",
1179
- )
1180
1173
  console.print(table)
1181
1174
 
1182
1175
  if card.matches:
@@ -110,7 +110,7 @@ class RunReport:
110
110
  class Budget:
111
111
  """The spend and wall-clock governor. Checked before every dispatch."""
112
112
 
113
- def __init__(self, max_usd: float, max_seconds: int, *, already_spent: float = 0.0):
113
+ def __init__(self, max_usd: float | None, max_seconds: int, *, already_spent: float = 0.0):
114
114
  self.max_usd = max_usd
115
115
  self.max_seconds = max_seconds
116
116
  #: What this run has already cost, including earlier invocations.
@@ -130,23 +130,28 @@ class Budget:
130
130
  return time.monotonic() - self.started
131
131
 
132
132
  @property
133
- def remaining_usd(self) -> float:
134
- return max(0.0, self.max_usd - self.spent)
133
+ def remaining_usd(self) -> float | None:
134
+ return None if self.max_usd is None else max(0.0, self.max_usd - self.spent)
135
135
 
136
136
  def spend(self, amount: float) -> None:
137
137
  self.spent += amount
138
138
 
139
139
  def check(self) -> None:
140
- if self.spent >= self.max_usd:
140
+ # `max_usd is None` means no spend ceiling -- the shipped config sets
141
+ # none, because a dollar figure bakes one vendor's pricing into a tool
142
+ # meant to run against local models too. The wall-clock cap and each
143
+ # agent's `max_turns` still bound a run; those are model-agnostic.
144
+ if self.max_usd is not None and self.spent >= self.max_usd:
141
145
  raise BudgetExceeded(f"spend cap reached: ${self.spent:.2f} of ${self.max_usd:.2f}")
142
146
  if self.elapsed >= self.max_seconds:
143
147
  raise BudgetExceeded(
144
148
  f"wall-clock cap reached: {self.elapsed:.0f}s of {self.max_seconds}s"
145
149
  )
146
150
 
147
- def allowance(self, spec: AgentSpec) -> float:
148
- """What this agent may spend: its own cap, or what the run has left."""
149
- return max(0.01, min(spec.max_budget_usd, self.remaining_usd))
151
+ def allowance(self, spec: AgentSpec) -> float | None:
152
+ """What this agent may spend, or None when neither it nor the run caps it."""
153
+ caps = [c for c in (spec.max_budget_usd, self.remaining_usd) if c is not None]
154
+ return max(0.01, min(caps)) if caps else None
150
155
 
151
156
 
152
157
  class Conductor:
@@ -234,6 +239,7 @@ class Conductor:
234
239
  else:
235
240
  store.log("skipped", reason="mode does not file tickets", mode=mode)
236
241
  await self._phase_verify(specs, store, budget, report, map_version)
242
+ await self._phase_report(specs, store, budget, report, mode, map_version)
237
243
  except BudgetExceeded as exc:
238
244
  report.stopped_early = str(exc)
239
245
  report.escalations.append(str(exc))
@@ -311,6 +317,30 @@ class Conductor:
311
317
 
312
318
  await self._gather(jobs, store, budget, report, map_version, self.config.run_modes[mode].max_concurrency)
313
319
 
320
+ async def _phase_report(self, specs, store, budget, report, mode, map_version) -> None:
321
+ """Reporting agents run last, over what the run itself produced.
322
+
323
+ Dispatched by LAYER, like discovery, and deliberately not by name. Every
324
+ other phase looks up a specific agent (`specs.get("FORGE")`), which is
325
+ why CHRONICLE could be configured, validated, assembled and shown in
326
+ `--dry-run` while never running: no phase asked for it. That is the same
327
+ silent skip VAULT and WARDEN exposed for discovery, and it is worth
328
+ fixing the shape rather than the instance -- a second reporting agent
329
+ now needs no Python either.
330
+
331
+ A run with no reporting agent is the ordinary case and not worth a log
332
+ line; most modes have none.
333
+ """
334
+ reporting = [s for s in specs.values() if s.layer == "reporting"]
335
+ if not reporting:
336
+ return
337
+
338
+ for spec in reporting:
339
+ budget.check()
340
+ await self._dispatch(
341
+ spec, store, budget, report, tasks.report(self.config, mode), map_version
342
+ )
343
+
314
344
  async def _phase_reproduce(self, specs, store, budget, report, map_version) -> None:
315
345
  """One FORGE invocation per finding.
316
346
 
@@ -74,7 +74,8 @@ class AgentSpec(BaseModel):
74
74
  model: str = "claude-opus-5"
75
75
  effort: Literal["low", "medium", "high", "xhigh", "max"] = "high"
76
76
  max_turns: int = 40
77
- max_budget_usd: float = 2.0
77
+ max_budget_usd: float | None = None
78
+
78
79
 
79
80
  mcp_servers: list[str] = Field(default_factory=list)
80
81
  builtin_tools: list[str] = Field(default_factory=list)
@@ -165,7 +166,8 @@ class RunMode(BaseModel):
165
166
 
166
167
  trigger: str
167
168
  agents: list[str]
168
- max_budget_usd: float = 10.0
169
+ max_budget_usd: float | None = None
170
+
169
171
  max_wall_clock_s: int = 3600
170
172
  max_concurrency: int = 3
171
173
  files_tickets: bool = True
@@ -9,7 +9,6 @@ prompt: ARBITER.md
9
9
  model: claude-opus-5
10
10
  effort: high
11
11
  max_turns: 50
12
- max_budget_usd: 3.0
13
12
  mcp_servers: [envelope, vcs, contract_diff, test_runner]
14
13
  builtin_tools: [Read, Grep, Glob]
15
14
  skills: [adversarial-review, root-cause-vs-symptom, regression-risk-scoring, test-quality-audit]
@@ -7,7 +7,6 @@ prompt: CARTOGRAPHER.md
7
7
  model: claude-sonnet-5 # extraction, not judgment
8
8
  effort: medium
9
9
  max_turns: 60
10
- max_budget_usd: 2.0
11
10
  mcp_servers: [envelope]
12
11
  builtin_tools: [Read, Grep, Glob]
13
12
  policy: {} # read-only
@@ -0,0 +1,19 @@
1
+ name: CHRONICLE
2
+ layer: reporting
3
+ role: >
4
+ Reporting analyst. Reads what a run produced -- findings, verdicts, denials,
5
+ escalations, recurrence -- and reports the pattern across them, including what
6
+ the run could not reach. Audits the run, never the application.
7
+ prompt: CHRONICLE.md
8
+ model: claude-sonnet-5
9
+ effort: medium
10
+ max_turns: 40
11
+ mcp_servers: [envelope, defect_memory, tracker]
12
+ builtin_tools: [Read]
13
+ policy: {}
14
+
15
+ skills: [severity-rubric, dedupe-strategy, verdict-reporting]
16
+
17
+ # No must_call: a run that found nothing still deserves a report saying so, and
18
+ # a report is not an envelope. Requiring an emission would turn "nothing to say"
19
+ # into an invented finding.
@@ -8,7 +8,6 @@ prompt: CLERK.md
8
8
  model: claude-sonnet-5 # composition against a fixed rubric and house format
9
9
  effort: high
10
10
  max_turns: 50
11
- max_budget_usd: 2.0
12
11
  mcp_servers: [envelope, tracker, defect_memory]
13
12
  builtin_tools: [Read]
14
13
  policy:
@@ -8,7 +8,6 @@ prompt: CONDUIT.md
8
8
  model: claude-opus-5
9
9
  effort: high
10
10
  max_turns: 60
11
- max_budget_usd: 3.0
12
11
  mcp_servers: [envelope, contract_diff, env_control, defect_memory]
13
12
  builtin_tools: [Read, Grep, Glob]
14
13
  policy: {}
@@ -8,7 +8,6 @@ prompt: FORGE.md
8
8
  model: claude-opus-5
9
9
  effort: high
10
10
  max_turns: 80
11
- max_budget_usd: 4.0
12
11
  mcp_servers: [envelope, test_runner, env_control, vcs]
13
12
  builtin_tools: [Read, Grep, Glob, Write, Edit, Bash]
14
13
  policy:
@@ -0,0 +1,26 @@
1
+ name: GAUGE
2
+ layer: discovery
3
+ role: >
4
+ Performance analyst. Finds N+1 query patterns, unbounded result sets, queries
5
+ filtering on unindexed columns on request paths, endpoint latency outliers,
6
+ bundle-size outliers, unbounded caches and leaked connections. Evidences them
7
+ from the code and from timed requests, because this deployment has no load
8
+ runner, no metrics backend and no profiler -- so it never claims behaviour
9
+ under load that it did not observe.
10
+ prompt: GAUGE.md
11
+ model: claude-opus-5
12
+ effort: high
13
+ max_turns: 60
14
+ mcp_servers: [envelope, env_control, defect_memory]
15
+ builtin_tools: [Read, Grep, Glob]
16
+ policy: {}
17
+
18
+ skills: [environment-pinning, repro-minimisation, root-cause-vs-symptom, severity-rubric]
19
+
20
+ # No must_call: finding nothing is a valid outcome for a discovery agent, and
21
+ # requiring an emission would manufacture findings to satisfy it -- which for a
22
+ # performance agent means speculation about load it cannot apply.
23
+
24
+ # Nightly and pre-release only (§4.10): too slow and too noisy per-PR. The
25
+ # run_modes rosters in system.yaml decide that; this file only declares the
26
+ # agent.
@@ -0,0 +1,21 @@
1
+ name: KEYSTONE
2
+ layer: discovery
3
+ role: >
4
+ Architecture analyst. Finds dependency cycles, layering violations, god modules
5
+ and fan-in outliers, domain logic duplicated across services, drift between the
6
+ architecture documents and the code, wrong service boundaries, and dead or
7
+ orphaned code. Pure static analysis -- it needs no running application, so it
8
+ is the one discovery agent that works against a target with no environment.
9
+ prompt: KEYSTONE.md
10
+ model: claude-opus-5
11
+ effort: high
12
+ max_turns: 60
13
+ mcp_servers: [envelope, defect_memory]
14
+ builtin_tools: [Read, Grep, Glob]
15
+ policy: {} # read-only; it never touches the app it reads
16
+
17
+ skills: [repo-cartography, api-surface-extraction, ownership-resolution, severity-rubric]
18
+
19
+ # No must_call: see VAULT. Also, an architecture agent that must emit something
20
+ # will emit taste, and "this could be cleaner" filed as a defect is the fastest
21
+ # way for a team to stop reading structural findings at all.
@@ -10,7 +10,6 @@ prompt: MENDER.md
10
10
  model: claude-opus-5
11
11
  effort: high
12
12
  max_turns: 80
13
- max_budget_usd: 5.0
14
13
  mcp_servers: [envelope, test_runner, env_control, vcs, tracker, contract_diff]
15
14
  builtin_tools: [Read, Grep, Glob, Write, Edit, Bash]
16
15
  skills: [test-first-fix, minimal-diff-discipline, root-cause-vs-symptom, rollback-plan-authoring]
@@ -8,7 +8,6 @@ prompt: PROOF.md
8
8
  model: claude-opus-5
9
9
  effort: high
10
10
  max_turns: 60
11
- max_budget_usd: 4.0
12
11
  mcp_servers: [envelope, test_runner, env_control, tracker, vcs]
13
12
  builtin_tools: [Read, Grep, Glob, Bash]
14
13
  policy:
@@ -0,0 +1,23 @@
1
+ name: PULSE
2
+ layer: discovery
3
+ role: >
4
+ Realtime and WebSocket analyst. Finds auth bypass on the upgrade handshake,
5
+ reconnect without jittered backoff, message loss with no resume token,
6
+ order-dependent consumers with nothing carrying order, missing heartbeats,
7
+ absent backpressure, and channel authorization never re-checked after
8
+ subscribe. The WebSocket harness the design gives this role does not exist
9
+ here, so PULSE reads the connection code and reports at the confidence of a
10
+ source reading, not of a measurement.
11
+ prompt: PULSE.md
12
+ model: claude-opus-5
13
+ effort: high
14
+ max_turns: 60
15
+ mcp_servers: [envelope, env_control, defect_memory]
16
+ builtin_tools: [Read, Grep, Glob]
17
+ policy: {} # read-only
18
+
19
+ skills: [api-surface-extraction, authz-matrix-check, environment-pinning, severity-rubric]
20
+
21
+ # No must_call: see VAULT. It matters more here than anywhere -- a target with no
22
+ # realtime surface should produce zero envelopes, and a required emission would
23
+ # turn "there are no websockets" into a manufactured finding about one.
@@ -8,7 +8,6 @@ prompt: SURFACE.md
8
8
  model: claude-opus-5
9
9
  effort: high
10
10
  max_turns: 80
11
- max_budget_usd: 4.0
12
11
  mcp_servers: [envelope, env_control, playwright]
13
12
  builtin_tools: [Read, Grep, Glob]
14
13
  policy: {}
@@ -0,0 +1,23 @@
1
+ name: USHER
2
+ layer: discovery
3
+ role: >
4
+ Product navigation and UX guide. Navigates the live product to answer "how do
5
+ I do X here?", and reports every place that navigation struggled: tasks
6
+ reachable only by typing a URL, dead ends, unlabelled paths, step counts out
7
+ of proportion to the task, and product vocabulary that does not match the
8
+ user's. SURFACE owns whether a feature works; USHER owns whether anyone can
9
+ find it.
10
+ prompt: USHER.md
11
+ model: claude-opus-5
12
+ effort: high
13
+ max_turns: 80
14
+ mcp_servers: [envelope, env_control, playwright, defect_memory]
15
+ builtin_tools: [Read, Grep, Glob]
16
+ policy: {}
17
+
18
+ skills: [product-task-graph, exploratory-ui-walk, environment-pinning, severity-rubric, repro-minimisation]
19
+
20
+ # No must_call: finding nothing is a valid outcome for a discovery agent, and
21
+ # requiring an emission would manufacture findings to satisfy it. That matters
22
+ # more here than elsewhere -- a product that is easy to navigate produces no
23
+ # friction envelopes, which is the result, not a failure of the run.
@@ -10,7 +10,6 @@ prompt: VAULT.md
10
10
  model: claude-opus-5
11
11
  effort: high
12
12
  max_turns: 60
13
- max_budget_usd: 3.0
14
13
  mcp_servers: [envelope, env_control, defect_memory]
15
14
  builtin_tools: [Read, Grep, Glob]
16
15
  policy: {}
@@ -9,10 +9,6 @@ prompt: WARDEN.md
9
9
  model: claude-opus-5
10
10
  effort: high
11
11
  max_turns: 60
12
- # Measured: WARDEN exhausted $3.00 on its first real run against the demo app
13
- # and was killed mid-audit. Building the endpoint-by-role matrix and actually
14
- # impersonating each role costs more than reading a spec does.
15
- max_budget_usd: 5.0
16
12
  mcp_servers: [envelope, env_control, defect_memory]
17
13
  builtin_tools: [Read, Grep, Glob]
18
14
  policy: {}
@@ -9,6 +9,11 @@ target:
9
9
  tracker: local # local | jira -- swap to file against real Jira
10
10
  vcs: local # local | github
11
11
 
12
+ # No spend cap by default. A dollar figure bakes one vendor's pricing into
13
+ # the config, and this is meant to run against local models too. `max_turns`
14
+ # is the model-agnostic bound. Set `max_budget_usd` here if you want a
15
+ # ceiling; the governor enforces one whenever it is present.
16
+
12
17
  thresholds:
13
18
  min_confidence_to_file: 0.6 # §7 confidence gate
14
19
  max_findings_per_agent_run: 25 # §8.3 loop breaker: pause and escalate, don't file
@@ -20,31 +25,23 @@ thresholds:
20
25
  run_modes:
21
26
  pr-check:
22
27
  trigger: pull_request
23
- agents: [CARTOGRAPHER, CONDUIT, SURFACE, FORGE, CLERK]
24
- # Was 6.0, which this roster could not complete. Measured on a real run:
25
- # CARTOGRAPHER $0.69 + CONDUIT $2.25 + SURFACE $3.16 = $6.09, so the governor
26
- # stopped the run before FORGE or CLERK ever dispatched. The mode meant to
27
- # run on every pull request could not file a ticket, and it failed silently:
28
- # agents ran, findings landed in the ledger, nothing errored. `qaas validate`
29
- # now refuses a mode whose agents cannot fit inside its cap.
30
- max_budget_usd: 16.0
28
+ # KEYSTONE joins the PR sweep because it is pure static analysis and needs
29
+ # no running app. USHER, GAUGE and CHRONICLE do not: the design says GAUGE is
30
+ # "nightly and pre-release only -- too expensive and too noisy per-PR", and
31
+ # the same argument holds for a UX walk and a report about the week.
32
+ agents: [CARTOGRAPHER, KEYSTONE, CONDUIT, SURFACE, VAULT, WARDEN, FORGE, CLERK]
31
33
  max_wall_clock_s: 900
32
34
  max_concurrency: 2
33
35
 
34
36
  nightly:
35
37
  trigger: cron
36
- agents: [CARTOGRAPHER, CONDUIT, SURFACE, VAULT, WARDEN, FORGE, CLERK]
37
- # FORGE runs once per finding, so the deep sweep's budget scales with how
38
- # much discovery found, not with the number of agents. Measured: discovery
39
- # ~$6, then roughly $1-2 per finding reproduced.
40
- max_budget_usd: 50.0
38
+ agents: [CARTOGRAPHER, KEYSTONE, CONDUIT, SURFACE, VAULT, WARDEN, PULSE, USHER, GAUGE, FORGE, CLERK, CHRONICLE]
41
39
  max_wall_clock_s: 7200
42
40
  max_concurrency: 3
43
41
 
44
42
  incident:
45
43
  trigger: alert
46
44
  agents: [CONDUIT]
47
- max_budget_usd: 4.0
48
45
  max_wall_clock_s: 600
49
46
  max_concurrency: 2
50
47
  files_tickets: false # §9: diagnostic only, read-only, no filing
@@ -55,7 +52,6 @@ run_modes:
55
52
  fix-cycle:
56
53
  trigger: agent_ready_ticket
57
54
  agents: [PROOF, MENDER, ARBITER]
58
- max_budget_usd: 20.0
59
55
  max_wall_clock_s: 3600
60
56
  max_concurrency: 1
61
57
 
@@ -63,7 +59,6 @@ run_modes:
63
59
  # expensive mode in the system and the only one that closes the loop.
64
60
  full-loop:
65
61
  trigger: on_demand
66
- agents: [CARTOGRAPHER, CONDUIT, SURFACE, VAULT, WARDEN, FORGE, CLERK, MENDER, ARBITER, PROOF]
67
- max_budget_usd: 70.0
62
+ agents: [CARTOGRAPHER, KEYSTONE, CONDUIT, SURFACE, VAULT, WARDEN, PULSE, USHER, GAUGE, FORGE, CLERK, MENDER, ARBITER, PROOF, CHRONICLE]
68
63
  max_wall_clock_s: 10800
69
64
  max_concurrency: 3