qaas-python 0.2.2__tar.gz → 0.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {qaas_python-0.2.2 → qaas_python-0.3.0}/CLAUDE.md +5 -3
- {qaas_python-0.2.2 → qaas_python-0.3.0}/PKG-INFO +28 -3
- {qaas_python-0.2.2 → qaas_python-0.3.0}/README.md +27 -2
- {qaas_python-0.2.2 → qaas_python-0.3.0}/pyproject.toml +1 -1
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/cli.py +16 -23
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/conductor.py +37 -7
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/config.py +4 -2
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/defaults/config/agents/arbiter.yaml +0 -1
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/defaults/config/agents/cartographer.yaml +0 -1
- qaas_python-0.3.0/src/qaas/defaults/config/agents/chronicle.yaml +19 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/defaults/config/agents/clerk.yaml +0 -1
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/defaults/config/agents/conduit.yaml +0 -1
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/defaults/config/agents/forge.yaml +0 -1
- qaas_python-0.3.0/src/qaas/defaults/config/agents/gauge.yaml +26 -0
- qaas_python-0.3.0/src/qaas/defaults/config/agents/keystone.yaml +21 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/defaults/config/agents/mender.yaml +0 -1
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/defaults/config/agents/proof.yaml +0 -1
- qaas_python-0.3.0/src/qaas/defaults/config/agents/pulse.yaml +23 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/defaults/config/agents/surface.yaml +0 -1
- qaas_python-0.3.0/src/qaas/defaults/config/agents/usher.yaml +23 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/defaults/config/agents/vault.yaml +0 -1
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/defaults/config/agents/warden.yaml +0 -4
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/defaults/config/system.yaml +12 -17
- qaas_python-0.3.0/src/qaas/prompts/CHRONICLE.md +61 -0
- qaas_python-0.3.0/src/qaas/prompts/GAUGE.md +109 -0
- qaas_python-0.3.0/src/qaas/prompts/KEYSTONE.md +80 -0
- qaas_python-0.3.0/src/qaas/prompts/PULSE.md +100 -0
- qaas_python-0.3.0/src/qaas/prompts/USHER.md +94 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/target.py +8 -1
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/tasks.py +34 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/tests/test_conductor.py +55 -8
- {qaas_python-0.2.2 → qaas_python-0.3.0}/tests/test_config.py +32 -17
- {qaas_python-0.2.2 → qaas_python-0.3.0}/tests/test_paths.py +3 -3
- {qaas_python-0.2.2 → qaas_python-0.3.0}/tests/test_trace.py +19 -5
- {qaas_python-0.2.2 → qaas_python-0.3.0}/.gitignore +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/ARCHITECTURE.md +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/BUILD_PLAN.md +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/LICENSE +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/config/targets/corvid.yaml +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/adapters/__init__.py +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/adapters/tracker.py +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/adapters/vcs.py +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/discover.py +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/envelope.py +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/guardrails.py +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/mcp/__init__.py +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/mcp/context.py +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/mcp/contract_diff.py +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/mcp/defect_memory.py +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/mcp/env_control.py +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/mcp/envelope_server.py +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/mcp/test_runner.py +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/mcp/tracker.py +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/mcp/vcs.py +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/paths.py +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/plugin/.claude-plugin/plugin.json +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/plugin/skills/a11y-audit/SKILL.md +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/plugin/skills/adversarial-review/SKILL.md +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/plugin/skills/api-surface-extraction/SKILL.md +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/plugin/skills/authz-matrix-check/SKILL.md +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/plugin/skills/console-error-triage/SKILL.md +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/plugin/skills/contract-test-generation/SKILL.md +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/plugin/skills/dedupe-strategy/SKILL.md +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/plugin/skills/environment-pinning/SKILL.md +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/plugin/skills/error-taxonomy/SKILL.md +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/plugin/skills/exploratory-ui-walk/SKILL.md +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/plugin/skills/failing-test-authoring/SKILL.md +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/plugin/skills/flake-detection/SKILL.md +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/plugin/skills/form-state-probe/SKILL.md +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/plugin/skills/minimal-diff-discipline/SKILL.md +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/plugin/skills/openapi-diff/SKILL.md +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/plugin/skills/ownership-resolution/SKILL.md +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/plugin/skills/product-task-graph/SKILL.md +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/plugin/skills/regression-risk-scoring/SKILL.md +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/plugin/skills/regression-suite-selection/SKILL.md +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/plugin/skills/repo-cartography/SKILL.md +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/plugin/skills/repro-minimisation/SKILL.md +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/plugin/skills/rollback-plan-authoring/SKILL.md +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/plugin/skills/root-cause-vs-symptom/SKILL.md +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/plugin/skills/routing-rules/SKILL.md +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/plugin/skills/severity-rubric/SKILL.md +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/plugin/skills/test-first-fix/SKILL.md +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/plugin/skills/test-quality-audit/SKILL.md +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/plugin/skills/ticket-writer/SKILL.md +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/plugin/skills/verdict-reporting/SKILL.md +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/plugin/skills/verification-protocol/SKILL.md +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/prompts/ARBITER.md +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/prompts/CARTOGRAPHER.md +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/prompts/CLERK.md +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/prompts/CONDUIT.md +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/prompts/FORGE.md +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/prompts/MENDER.md +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/prompts/PROOF.md +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/prompts/SURFACE.md +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/prompts/VAULT.md +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/prompts/WARDEN.md +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/prompts/_shared.md +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/registry.py +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/runner.py +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/scorecard.py +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/sdk_compat.py +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/store.py +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/src/qaas/trace.py +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/CODEOWNERS +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/api/Dockerfile +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/api/app/__init__.py +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/api/app/auth.py +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/api/app/config.py +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/api/app/db.py +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/api/app/errors.py +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/api/app/main.py +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/api/app/models.py +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/api/app/routes/__init__.py +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/api/app/routes/auth.py +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/api/app/routes/invoices.py +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/api/app/routes/orders.py +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/api/app/routes/stream.py +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/api/app/schemas.py +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/api/migrations/001_init.sql +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/api/pyproject.toml +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/api/seed/fixtures.sql +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/defects.yaml +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/docker-compose.yml +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/openapi.yaml +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/web/.gitignore +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/web/Dockerfile +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/web/index.html +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/web/package-lock.json +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/web/package.json +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/web/src/api.ts +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/web/src/components/Button.tsx +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/web/src/components/Layout.tsx +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/web/src/components/SearchInput.tsx +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/web/src/main.tsx +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/web/src/routes/CheckoutReview.tsx +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/web/src/routes/Login.tsx +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/web/src/routes/NewOrder.tsx +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/web/src/routes/OrderDetail.tsx +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/web/src/routes/OrdersList.tsx +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/web/src/styles.css +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/web/tsconfig.json +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/target-app/web/vite.config.ts +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/tests/adapters/test_github_vcs.py +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/tests/adapters/test_jira_tracker.py +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/tests/conftest.py +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/tests/mcp/conftest.py +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/tests/mcp/test_contract_diff.py +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/tests/mcp/test_defect_memory.py +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/tests/mcp/test_env_control.py +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/tests/mcp/test_test_runner.py +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/tests/mcp/test_tracker.py +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/tests/mcp/test_vcs.py +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/tests/support.py +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/tests/target_app/test_seeded_defects.py +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/tests/test_cli.py +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/tests/test_discover.py +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/tests/test_envelope.py +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/tests/test_guardrails.py +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/tests/test_hooks_and_skills.py +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/tests/test_prompt_overrides.py +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/tests/test_registry.py +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/tests/test_scorecard.py +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/tests/test_skills_actually_load.py +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/tests/test_store.py +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/tests/test_target_profile_is_honoured.py +0 -0
- {qaas_python-0.2.2 → qaas_python-0.3.0}/tests/test_user_mcp_servers.py +0 -0
|
@@ -52,8 +52,8 @@ default `pytest` run is offline and free, and must stay that way.
|
|
|
52
52
|
### Phase pipeline (`conductor.py`)
|
|
53
53
|
|
|
54
54
|
```
|
|
55
|
-
map -> discover -> reproduce -> file -> verify
|
|
56
|
-
CARTOGRAPHER CONDUIT/SURFACE FORGE CLERK PROOF
|
|
55
|
+
map -> discover -> reproduce -> file -> verify -> report
|
|
56
|
+
CARTOGRAPHER CONDUIT/SURFACE/… FORGE CLERK PROOF CHRONICLE
|
|
57
57
|
```
|
|
58
58
|
|
|
59
59
|
Discovery agents run concurrently up to the mode's cap; FORGE runs **once per
|
|
@@ -80,7 +80,9 @@ adding an agent as a sign something is wrong. Constraints enforced in
|
|
|
80
80
|
`config.py`: at most 6 MCP servers per agent (§5.3, tool-selection accuracy),
|
|
81
81
|
and every `must_call` tool must name a server the agent actually has.
|
|
82
82
|
|
|
83
|
-
|
|
83
|
+
Discovery AND reporting agents dispatch **by layer**; every other phase
|
|
84
|
+
dispatches by name. So adding a discovery or reporting agent is a prompt file
|
|
85
|
+
plus a YAML file and no Python --
|
|
84
86
|
`_phase_discover` falls back to `tasks.discovery` for anything without a
|
|
85
87
|
bespoke builder. VAULT and WARDEN were added that way and found that it was
|
|
86
88
|
not true before them.
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: qaas-python
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.3.0
|
|
4
4
|
Summary: A multi-agent QA system: finds real defects, reproduces them, files tickets, fixes them, and proves the fix
|
|
5
5
|
Project-URL: Homepage, https://github.com/allaabdella2-us/qa-multi-agent-system
|
|
6
6
|
Project-URL: Repository, https://github.com/allaabdella2-us/qa-multi-agent-system
|
|
@@ -84,6 +84,31 @@ Nothing crosses between the loops except a ticket — which is also the audit tr
|
|
|
84
84
|
|
|
85
85
|
---
|
|
86
86
|
|
|
87
|
+
## 🤖 The roster
|
|
88
|
+
|
|
89
|
+
| agent | layer | what it does |
|
|
90
|
+
|---|---|---|
|
|
91
|
+
| 🗺️ CARTOGRAPHER | map | services, routes, schema, ownership → the system map everything reads |
|
|
92
|
+
| 🏛️ KEYSTONE | discovery | circular deps, layering violations, god modules, dead code |
|
|
93
|
+
| 🔌 CONDUIT | discovery | API contract drift, authz gaps, error-shape inconsistency |
|
|
94
|
+
| 🖱️ SURFACE | discovery | drives the UI through real journeys |
|
|
95
|
+
| 🗄️ VAULT | discovery | schema constraints the code assumes and the database does not enforce |
|
|
96
|
+
| 🔒 WARDEN | discovery | missing authorization, secrets, vulnerable dependencies, leaks |
|
|
97
|
+
| 📡 PULSE | discovery | WebSocket auth, reconnect, ordering, backpressure |
|
|
98
|
+
| 🧭 USHER | discovery | whether a person can *find* a feature, not just whether it works |
|
|
99
|
+
| ⏱️ GAUGE | discovery | N+1 queries, unindexed hot paths, unbounded results, bundle outliers |
|
|
100
|
+
| 🔨 FORGE | triage | reproduces, minimises, measures flake, commits a failing test |
|
|
101
|
+
| 📝 CLERK | triage | dedupes, scores severity, routes, files — the only tracker writer |
|
|
102
|
+
| 🔧 MENDER | remediation | the minimal fix, on a branch |
|
|
103
|
+
| ⚖️ ARBITER | remediation | adversarial review: APPROVE / REQUEST_CHANGES / ESCALATE |
|
|
104
|
+
| ✅ PROOF | verify | re-runs the original test → VERIFIED / NOT_FIXED / REGRESSED |
|
|
105
|
+
| 📊 CHRONICLE | reporting | what the run found, what recurred, and what it could not reach |
|
|
106
|
+
|
|
107
|
+
**CONDUCTOR** is the sixteenth. It is the Python state machine rather than an
|
|
108
|
+
agent — a model cannot enforce a budget it is itself spending.
|
|
109
|
+
|
|
110
|
+
---
|
|
111
|
+
|
|
87
112
|
## 🚀 Quickstart in 60 seconds
|
|
88
113
|
|
|
89
114
|
```bash
|
|
@@ -318,8 +343,8 @@ precision is measured rather than assumed.
|
|
|
318
343
|
|
|
319
344
|
Honest about what exists:
|
|
320
345
|
|
|
321
|
-
- ✅ **
|
|
322
|
-
- ✅ **Adding
|
|
346
|
+
- ✅ **All 16 agents in the design are built.**
|
|
347
|
+
- ✅ **Adding one needs a prompt file and a YAML file — no Python.** Six were added that way, which is how the claim got tested.
|
|
323
348
|
- ✅ The fix loop has closed end to end on a real defect: `NOT_FIXED → MENDER → ARBITER APPROVE → VERIFIED`.
|
|
324
349
|
- ✅ 30 skills, 7 in-process MCP servers, 649 offline tests.
|
|
325
350
|
- ⚠️ Running the bundled demo needs `export CORVID_PASSWORD=password123` — credentials come from the environment, including the demo's.
|
|
@@ -54,6 +54,31 @@ Nothing crosses between the loops except a ticket — which is also the audit tr
|
|
|
54
54
|
|
|
55
55
|
---
|
|
56
56
|
|
|
57
|
+
## 🤖 The roster
|
|
58
|
+
|
|
59
|
+
| agent | layer | what it does |
|
|
60
|
+
|---|---|---|
|
|
61
|
+
| 🗺️ CARTOGRAPHER | map | services, routes, schema, ownership → the system map everything reads |
|
|
62
|
+
| 🏛️ KEYSTONE | discovery | circular deps, layering violations, god modules, dead code |
|
|
63
|
+
| 🔌 CONDUIT | discovery | API contract drift, authz gaps, error-shape inconsistency |
|
|
64
|
+
| 🖱️ SURFACE | discovery | drives the UI through real journeys |
|
|
65
|
+
| 🗄️ VAULT | discovery | schema constraints the code assumes and the database does not enforce |
|
|
66
|
+
| 🔒 WARDEN | discovery | missing authorization, secrets, vulnerable dependencies, leaks |
|
|
67
|
+
| 📡 PULSE | discovery | WebSocket auth, reconnect, ordering, backpressure |
|
|
68
|
+
| 🧭 USHER | discovery | whether a person can *find* a feature, not just whether it works |
|
|
69
|
+
| ⏱️ GAUGE | discovery | N+1 queries, unindexed hot paths, unbounded results, bundle outliers |
|
|
70
|
+
| 🔨 FORGE | triage | reproduces, minimises, measures flake, commits a failing test |
|
|
71
|
+
| 📝 CLERK | triage | dedupes, scores severity, routes, files — the only tracker writer |
|
|
72
|
+
| 🔧 MENDER | remediation | the minimal fix, on a branch |
|
|
73
|
+
| ⚖️ ARBITER | remediation | adversarial review: APPROVE / REQUEST_CHANGES / ESCALATE |
|
|
74
|
+
| ✅ PROOF | verify | re-runs the original test → VERIFIED / NOT_FIXED / REGRESSED |
|
|
75
|
+
| 📊 CHRONICLE | reporting | what the run found, what recurred, and what it could not reach |
|
|
76
|
+
|
|
77
|
+
**CONDUCTOR** is the sixteenth. It is the Python state machine rather than an
|
|
78
|
+
agent — a model cannot enforce a budget it is itself spending.
|
|
79
|
+
|
|
80
|
+
---
|
|
81
|
+
|
|
57
82
|
## 🚀 Quickstart in 60 seconds
|
|
58
83
|
|
|
59
84
|
```bash
|
|
@@ -288,8 +313,8 @@ precision is measured rather than assumed.
|
|
|
288
313
|
|
|
289
314
|
Honest about what exists:
|
|
290
315
|
|
|
291
|
-
- ✅ **
|
|
292
|
-
- ✅ **Adding
|
|
316
|
+
- ✅ **All 16 agents in the design are built.**
|
|
317
|
+
- ✅ **Adding one needs a prompt file and a YAML file — no Python.** Six were added that way, which is how the claim got tested.
|
|
293
318
|
- ✅ The fix loop has closed end to end on a real defect: `NOT_FIXED → MENDER → ARBITER APPROVE → VERIFIED`.
|
|
294
319
|
- ✅ 30 skills, 7 in-process MCP servers, 649 offline tests.
|
|
295
320
|
- ⚠️ Running the bundled demo needs `export CORVID_PASSWORD=password123` — credentials come from the environment, including the demo's.
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
# Distribution name is `qaas-python` (`qaas` was taken); the import package and
|
|
3
3
|
# the CLI are both `qaas`.
|
|
4
4
|
name = "qaas-python"
|
|
5
|
-
version = "0.
|
|
5
|
+
version = "0.3.0"
|
|
6
6
|
description = "A multi-agent QA system: finds real defects, reproduces them, files tickets, fixes them, and proves the fix"
|
|
7
7
|
readme = "README.md"
|
|
8
8
|
license = "MIT"
|
|
@@ -500,14 +500,18 @@ def validate(config_dir: Path | None = ConfigDir) -> None:
|
|
|
500
500
|
# never dispatched. The mode meant for every pull request could not file a
|
|
501
501
|
# ticket. It took a live run to notice; this check makes it free.
|
|
502
502
|
for mode_name, mode in sorted(cfg.run_modes.items()):
|
|
503
|
-
|
|
504
|
-
|
|
505
|
-
|
|
506
|
-
|
|
503
|
+
# Only meaningful when both sides declare a cap. The shipped config
|
|
504
|
+
# declares none, so this check simply does not fire there.
|
|
505
|
+
agent_caps = [
|
|
506
|
+
cfg.agents[a].max_budget_usd for a in mode.agents
|
|
507
|
+
if a in cfg.agents and cfg.agents[a].max_budget_usd is not None
|
|
508
|
+
]
|
|
509
|
+
needed = sum(agent_caps)
|
|
510
|
+
if mode.max_budget_usd is not None and agent_caps and needed > mode.max_budget_usd:
|
|
507
511
|
missing = [a for a in mode.agents if a in cfg.agents][-1]
|
|
508
512
|
problems.append(
|
|
509
|
-
f"mode '{mode_name}': agents
|
|
510
|
-
f"
|
|
513
|
+
f"mode '{mode_name}': the agents' caps exceed the mode's cap, so "
|
|
514
|
+
f"its cap, so the run stops before it reaches "
|
|
511
515
|
f"{missing} and files nothing. Raise max_budget_usd or drop an agent"
|
|
512
516
|
)
|
|
513
517
|
|
|
@@ -567,7 +571,7 @@ def validate(config_dir: Path | None = ConfigDir) -> None:
|
|
|
567
571
|
filing = "" if rm.files_tickets else " [dim](no filing)[/dim]"
|
|
568
572
|
console.print(
|
|
569
573
|
f"[bold]{mode}[/bold]: {', '.join(rm.agents)} "
|
|
570
|
-
f"[dim]
|
|
574
|
+
f"[dim]{rm.max_wall_clock_s}s[/dim]{filing}"
|
|
571
575
|
)
|
|
572
576
|
|
|
573
577
|
if notes:
|
|
@@ -834,7 +838,7 @@ def runs(root: Path = Root, limit: int = 10) -> None:
|
|
|
834
838
|
console.print("[dim]no runs yet[/dim]")
|
|
835
839
|
return
|
|
836
840
|
table = Table(header_style="bold")
|
|
837
|
-
for col in ("run", "envelopes", "agents"
|
|
841
|
+
for col in ("run", "envelopes", "agents"):
|
|
838
842
|
table.add_column(col)
|
|
839
843
|
for run_id in ids:
|
|
840
844
|
store = RunStore(run_id, root)
|
|
@@ -843,7 +847,6 @@ def runs(root: Path = Root, limit: int = 10) -> None:
|
|
|
843
847
|
run_id,
|
|
844
848
|
str(len(store.envelopes())),
|
|
845
849
|
str(len(results)),
|
|
846
|
-
f"${store.total_cost_usd():.2f}",
|
|
847
850
|
)
|
|
848
851
|
console.print(table)
|
|
849
852
|
|
|
@@ -872,9 +875,6 @@ def show(run_id: str, root: Path = Root) -> None:
|
|
|
872
875
|
header.append(f"started {summary.started:%Y-%m-%d %H:%M:%S}Z")
|
|
873
876
|
if summary.duration_s is not None:
|
|
874
877
|
header.append(f"duration {summary.duration_s:.0f}s")
|
|
875
|
-
header.append(f"cost ${summary.cost_usd:.2f}")
|
|
876
|
-
if summary.budget_usd:
|
|
877
|
-
header.append(f"of ${summary.budget_usd:.2f} budget")
|
|
878
878
|
console.print(" " + " ".join(header))
|
|
879
879
|
if summary.target_sha:
|
|
880
880
|
dirty = " [yellow](dirty tree)[/yellow]" if summary.target_dirty else ""
|
|
@@ -960,7 +960,6 @@ def trace(
|
|
|
960
960
|
table.add_column("agent", style="cyan")
|
|
961
961
|
table.add_column("kind")
|
|
962
962
|
table.add_column("detail", overflow="fold")
|
|
963
|
-
table.add_column("cost", justify="right", style="dim")
|
|
964
963
|
for row in trace_mod.timeline(entries):
|
|
965
964
|
label = f"{row.kind} ×{row.count}" if row.count > 1 else row.kind
|
|
966
965
|
table.add_row(
|
|
@@ -968,7 +967,6 @@ def trace(
|
|
|
968
967
|
row.agent,
|
|
969
968
|
f"[{KIND_STYLE.get(row.kind, 'white')}]{label}[/]",
|
|
970
969
|
row.detail,
|
|
971
|
-
f"${row.cost_usd:.2f}" if row.cost_usd is not None else "",
|
|
972
970
|
)
|
|
973
971
|
console.print(table)
|
|
974
972
|
console.print(f"\n[dim]{len(entries)} entries[/dim]")
|
|
@@ -1082,7 +1080,7 @@ def run(
|
|
|
1082
1080
|
rm = cfg.run_modes[mode]
|
|
1083
1081
|
console.print(
|
|
1084
1082
|
f"[bold]{mode}[/bold] — {len(specs)} agents, "
|
|
1085
|
-
f"
|
|
1083
|
+
f"concurrency {rm.max_concurrency}"
|
|
1086
1084
|
)
|
|
1087
1085
|
|
|
1088
1086
|
if dry_run:
|
|
@@ -1093,7 +1091,7 @@ def run(
|
|
|
1093
1091
|
d = describe(spec, prompt_dirs)
|
|
1094
1092
|
console.print(
|
|
1095
1093
|
f" [bold]{spec.name:14s}[/bold] {spec.model:18s} effort={spec.effort:7s} "
|
|
1096
|
-
f"turns<={spec.max_turns
|
|
1094
|
+
f"turns<={spec.max_turns}"
|
|
1097
1095
|
)
|
|
1098
1096
|
console.print(f" tools: {', '.join(d['allowed_tools'])}")
|
|
1099
1097
|
console.print(f" prompt: {d['prompt_chars']} chars")
|
|
@@ -1101,11 +1099,11 @@ def run(
|
|
|
1101
1099
|
|
|
1102
1100
|
def on_event(kind: str, detail: dict) -> None:
|
|
1103
1101
|
if kind == "agent_started":
|
|
1104
|
-
console.print(f"[dim]->[/dim] {detail.get('agent')}
|
|
1102
|
+
console.print(f"[dim]->[/dim] {detail.get('agent')}")
|
|
1105
1103
|
elif kind == "finished":
|
|
1106
1104
|
console.print(
|
|
1107
1105
|
f"[dim]<-[/dim] {detail.get('agent')} "
|
|
1108
|
-
f"[dim]
|
|
1106
|
+
f"[dim]{detail.get('envelopes', 0)} findings[/dim]"
|
|
1109
1107
|
)
|
|
1110
1108
|
elif kind == "stopped":
|
|
1111
1109
|
console.print(f"[yellow]stopped: {detail.get('reason')}[/yellow]")
|
|
@@ -1172,11 +1170,6 @@ def score(
|
|
|
1172
1170
|
table.add_row("false positives", f"{s['false_positives']} ({s['false_positive_rate']:.0%})")
|
|
1173
1171
|
table.add_row("duplicates", f"{s['duplicates']} ({s['duplicate_rate']:.0%})")
|
|
1174
1172
|
table.add_row("severity agreement", f"{s['severity_agreement']:.0%}")
|
|
1175
|
-
table.add_row("cost", f"${s['cost_usd']:.2f}")
|
|
1176
|
-
table.add_row(
|
|
1177
|
-
"cost per accepted",
|
|
1178
|
-
f"${s['cost_per_accepted']:.2f}" if s["cost_per_accepted"] is not None else "-",
|
|
1179
|
-
)
|
|
1180
1173
|
console.print(table)
|
|
1181
1174
|
|
|
1182
1175
|
if card.matches:
|
|
@@ -110,7 +110,7 @@ class RunReport:
|
|
|
110
110
|
class Budget:
|
|
111
111
|
"""The spend and wall-clock governor. Checked before every dispatch."""
|
|
112
112
|
|
|
113
|
-
def __init__(self, max_usd: float, max_seconds: int, *, already_spent: float = 0.0):
|
|
113
|
+
def __init__(self, max_usd: float | None, max_seconds: int, *, already_spent: float = 0.0):
|
|
114
114
|
self.max_usd = max_usd
|
|
115
115
|
self.max_seconds = max_seconds
|
|
116
116
|
#: What this run has already cost, including earlier invocations.
|
|
@@ -130,23 +130,28 @@ class Budget:
|
|
|
130
130
|
return time.monotonic() - self.started
|
|
131
131
|
|
|
132
132
|
@property
|
|
133
|
-
def remaining_usd(self) -> float:
|
|
134
|
-
return max(0.0, self.max_usd - self.spent)
|
|
133
|
+
def remaining_usd(self) -> float | None:
|
|
134
|
+
return None if self.max_usd is None else max(0.0, self.max_usd - self.spent)
|
|
135
135
|
|
|
136
136
|
def spend(self, amount: float) -> None:
|
|
137
137
|
self.spent += amount
|
|
138
138
|
|
|
139
139
|
def check(self) -> None:
|
|
140
|
-
|
|
140
|
+
# `max_usd is None` means no spend ceiling -- the shipped config sets
|
|
141
|
+
# none, because a dollar figure bakes one vendor's pricing into a tool
|
|
142
|
+
# meant to run against local models too. The wall-clock cap and each
|
|
143
|
+
# agent's `max_turns` still bound a run; those are model-agnostic.
|
|
144
|
+
if self.max_usd is not None and self.spent >= self.max_usd:
|
|
141
145
|
raise BudgetExceeded(f"spend cap reached: ${self.spent:.2f} of ${self.max_usd:.2f}")
|
|
142
146
|
if self.elapsed >= self.max_seconds:
|
|
143
147
|
raise BudgetExceeded(
|
|
144
148
|
f"wall-clock cap reached: {self.elapsed:.0f}s of {self.max_seconds}s"
|
|
145
149
|
)
|
|
146
150
|
|
|
147
|
-
def allowance(self, spec: AgentSpec) -> float:
|
|
148
|
-
"""What this agent may spend
|
|
149
|
-
|
|
151
|
+
def allowance(self, spec: AgentSpec) -> float | None:
|
|
152
|
+
"""What this agent may spend, or None when neither it nor the run caps it."""
|
|
153
|
+
caps = [c for c in (spec.max_budget_usd, self.remaining_usd) if c is not None]
|
|
154
|
+
return max(0.01, min(caps)) if caps else None
|
|
150
155
|
|
|
151
156
|
|
|
152
157
|
class Conductor:
|
|
@@ -234,6 +239,7 @@ class Conductor:
|
|
|
234
239
|
else:
|
|
235
240
|
store.log("skipped", reason="mode does not file tickets", mode=mode)
|
|
236
241
|
await self._phase_verify(specs, store, budget, report, map_version)
|
|
242
|
+
await self._phase_report(specs, store, budget, report, mode, map_version)
|
|
237
243
|
except BudgetExceeded as exc:
|
|
238
244
|
report.stopped_early = str(exc)
|
|
239
245
|
report.escalations.append(str(exc))
|
|
@@ -311,6 +317,30 @@ class Conductor:
|
|
|
311
317
|
|
|
312
318
|
await self._gather(jobs, store, budget, report, map_version, self.config.run_modes[mode].max_concurrency)
|
|
313
319
|
|
|
320
|
+
async def _phase_report(self, specs, store, budget, report, mode, map_version) -> None:
|
|
321
|
+
"""Reporting agents run last, over what the run itself produced.
|
|
322
|
+
|
|
323
|
+
Dispatched by LAYER, like discovery, and deliberately not by name. Every
|
|
324
|
+
other phase looks up a specific agent (`specs.get("FORGE")`), which is
|
|
325
|
+
why CHRONICLE could be configured, validated, assembled and shown in
|
|
326
|
+
`--dry-run` while never running: no phase asked for it. That is the same
|
|
327
|
+
silent skip VAULT and WARDEN exposed for discovery, and it is worth
|
|
328
|
+
fixing the shape rather than the instance -- a second reporting agent
|
|
329
|
+
now needs no Python either.
|
|
330
|
+
|
|
331
|
+
A run with no reporting agent is the ordinary case and not worth a log
|
|
332
|
+
line; most modes have none.
|
|
333
|
+
"""
|
|
334
|
+
reporting = [s for s in specs.values() if s.layer == "reporting"]
|
|
335
|
+
if not reporting:
|
|
336
|
+
return
|
|
337
|
+
|
|
338
|
+
for spec in reporting:
|
|
339
|
+
budget.check()
|
|
340
|
+
await self._dispatch(
|
|
341
|
+
spec, store, budget, report, tasks.report(self.config, mode), map_version
|
|
342
|
+
)
|
|
343
|
+
|
|
314
344
|
async def _phase_reproduce(self, specs, store, budget, report, map_version) -> None:
|
|
315
345
|
"""One FORGE invocation per finding.
|
|
316
346
|
|
|
@@ -74,7 +74,8 @@ class AgentSpec(BaseModel):
|
|
|
74
74
|
model: str = "claude-opus-5"
|
|
75
75
|
effort: Literal["low", "medium", "high", "xhigh", "max"] = "high"
|
|
76
76
|
max_turns: int = 40
|
|
77
|
-
max_budget_usd: float =
|
|
77
|
+
max_budget_usd: float | None = None
|
|
78
|
+
|
|
78
79
|
|
|
79
80
|
mcp_servers: list[str] = Field(default_factory=list)
|
|
80
81
|
builtin_tools: list[str] = Field(default_factory=list)
|
|
@@ -165,7 +166,8 @@ class RunMode(BaseModel):
|
|
|
165
166
|
|
|
166
167
|
trigger: str
|
|
167
168
|
agents: list[str]
|
|
168
|
-
max_budget_usd: float =
|
|
169
|
+
max_budget_usd: float | None = None
|
|
170
|
+
|
|
169
171
|
max_wall_clock_s: int = 3600
|
|
170
172
|
max_concurrency: int = 3
|
|
171
173
|
files_tickets: bool = True
|
|
@@ -9,7 +9,6 @@ prompt: ARBITER.md
|
|
|
9
9
|
model: claude-opus-5
|
|
10
10
|
effort: high
|
|
11
11
|
max_turns: 50
|
|
12
|
-
max_budget_usd: 3.0
|
|
13
12
|
mcp_servers: [envelope, vcs, contract_diff, test_runner]
|
|
14
13
|
builtin_tools: [Read, Grep, Glob]
|
|
15
14
|
skills: [adversarial-review, root-cause-vs-symptom, regression-risk-scoring, test-quality-audit]
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
name: CHRONICLE
|
|
2
|
+
layer: reporting
|
|
3
|
+
role: >
|
|
4
|
+
Reporting analyst. Reads what a run produced -- findings, verdicts, denials,
|
|
5
|
+
escalations, recurrence -- and reports the pattern across them, including what
|
|
6
|
+
the run could not reach. Audits the run, never the application.
|
|
7
|
+
prompt: CHRONICLE.md
|
|
8
|
+
model: claude-sonnet-5
|
|
9
|
+
effort: medium
|
|
10
|
+
max_turns: 40
|
|
11
|
+
mcp_servers: [envelope, defect_memory, tracker]
|
|
12
|
+
builtin_tools: [Read]
|
|
13
|
+
policy: {}
|
|
14
|
+
|
|
15
|
+
skills: [severity-rubric, dedupe-strategy, verdict-reporting]
|
|
16
|
+
|
|
17
|
+
# No must_call: a run that found nothing still deserves a report saying so, and
|
|
18
|
+
# a report is not an envelope. Requiring an emission would turn "nothing to say"
|
|
19
|
+
# into an invented finding.
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
name: GAUGE
|
|
2
|
+
layer: discovery
|
|
3
|
+
role: >
|
|
4
|
+
Performance analyst. Finds N+1 query patterns, unbounded result sets, queries
|
|
5
|
+
filtering on unindexed columns on request paths, endpoint latency outliers,
|
|
6
|
+
bundle-size outliers, unbounded caches and leaked connections. Evidences them
|
|
7
|
+
from the code and from timed requests, because this deployment has no load
|
|
8
|
+
runner, no metrics backend and no profiler -- so it never claims behaviour
|
|
9
|
+
under load that it did not observe.
|
|
10
|
+
prompt: GAUGE.md
|
|
11
|
+
model: claude-opus-5
|
|
12
|
+
effort: high
|
|
13
|
+
max_turns: 60
|
|
14
|
+
mcp_servers: [envelope, env_control, defect_memory]
|
|
15
|
+
builtin_tools: [Read, Grep, Glob]
|
|
16
|
+
policy: {}
|
|
17
|
+
|
|
18
|
+
skills: [environment-pinning, repro-minimisation, root-cause-vs-symptom, severity-rubric]
|
|
19
|
+
|
|
20
|
+
# No must_call: finding nothing is a valid outcome for a discovery agent, and
|
|
21
|
+
# requiring an emission would manufacture findings to satisfy it -- which for a
|
|
22
|
+
# performance agent means speculation about load it cannot apply.
|
|
23
|
+
|
|
24
|
+
# Nightly and pre-release only (§4.10): too slow and too noisy per-PR. The
|
|
25
|
+
# run_modes rosters in system.yaml decide that; this file only declares the
|
|
26
|
+
# agent.
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
name: KEYSTONE
|
|
2
|
+
layer: discovery
|
|
3
|
+
role: >
|
|
4
|
+
Architecture analyst. Finds dependency cycles, layering violations, god modules
|
|
5
|
+
and fan-in outliers, domain logic duplicated across services, drift between the
|
|
6
|
+
architecture documents and the code, wrong service boundaries, and dead or
|
|
7
|
+
orphaned code. Pure static analysis -- it needs no running application, so it
|
|
8
|
+
is the one discovery agent that works against a target with no environment.
|
|
9
|
+
prompt: KEYSTONE.md
|
|
10
|
+
model: claude-opus-5
|
|
11
|
+
effort: high
|
|
12
|
+
max_turns: 60
|
|
13
|
+
mcp_servers: [envelope, defect_memory]
|
|
14
|
+
builtin_tools: [Read, Grep, Glob]
|
|
15
|
+
policy: {} # read-only; it never touches the app it reads
|
|
16
|
+
|
|
17
|
+
skills: [repo-cartography, api-surface-extraction, ownership-resolution, severity-rubric]
|
|
18
|
+
|
|
19
|
+
# No must_call: see VAULT. Also, an architecture agent that must emit something
|
|
20
|
+
# will emit taste, and "this could be cleaner" filed as a defect is the fastest
|
|
21
|
+
# way for a team to stop reading structural findings at all.
|
|
@@ -10,7 +10,6 @@ prompt: MENDER.md
|
|
|
10
10
|
model: claude-opus-5
|
|
11
11
|
effort: high
|
|
12
12
|
max_turns: 80
|
|
13
|
-
max_budget_usd: 5.0
|
|
14
13
|
mcp_servers: [envelope, test_runner, env_control, vcs, tracker, contract_diff]
|
|
15
14
|
builtin_tools: [Read, Grep, Glob, Write, Edit, Bash]
|
|
16
15
|
skills: [test-first-fix, minimal-diff-discipline, root-cause-vs-symptom, rollback-plan-authoring]
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
name: PULSE
|
|
2
|
+
layer: discovery
|
|
3
|
+
role: >
|
|
4
|
+
Realtime and WebSocket analyst. Finds auth bypass on the upgrade handshake,
|
|
5
|
+
reconnect without jittered backoff, message loss with no resume token,
|
|
6
|
+
order-dependent consumers with nothing carrying order, missing heartbeats,
|
|
7
|
+
absent backpressure, and channel authorization never re-checked after
|
|
8
|
+
subscribe. The WebSocket harness the design gives this role does not exist
|
|
9
|
+
here, so PULSE reads the connection code and reports at the confidence of a
|
|
10
|
+
source reading, not of a measurement.
|
|
11
|
+
prompt: PULSE.md
|
|
12
|
+
model: claude-opus-5
|
|
13
|
+
effort: high
|
|
14
|
+
max_turns: 60
|
|
15
|
+
mcp_servers: [envelope, env_control, defect_memory]
|
|
16
|
+
builtin_tools: [Read, Grep, Glob]
|
|
17
|
+
policy: {} # read-only
|
|
18
|
+
|
|
19
|
+
skills: [api-surface-extraction, authz-matrix-check, environment-pinning, severity-rubric]
|
|
20
|
+
|
|
21
|
+
# No must_call: see VAULT. It matters more here than anywhere -- a target with no
|
|
22
|
+
# realtime surface should produce zero envelopes, and a required emission would
|
|
23
|
+
# turn "there are no websockets" into a manufactured finding about one.
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
name: USHER
|
|
2
|
+
layer: discovery
|
|
3
|
+
role: >
|
|
4
|
+
Product navigation and UX guide. Navigates the live product to answer "how do
|
|
5
|
+
I do X here?", and reports every place that navigation struggled: tasks
|
|
6
|
+
reachable only by typing a URL, dead ends, unlabelled paths, step counts out
|
|
7
|
+
of proportion to the task, and product vocabulary that does not match the
|
|
8
|
+
user's. SURFACE owns whether a feature works; USHER owns whether anyone can
|
|
9
|
+
find it.
|
|
10
|
+
prompt: USHER.md
|
|
11
|
+
model: claude-opus-5
|
|
12
|
+
effort: high
|
|
13
|
+
max_turns: 80
|
|
14
|
+
mcp_servers: [envelope, env_control, playwright, defect_memory]
|
|
15
|
+
builtin_tools: [Read, Grep, Glob]
|
|
16
|
+
policy: {}
|
|
17
|
+
|
|
18
|
+
skills: [product-task-graph, exploratory-ui-walk, environment-pinning, severity-rubric, repro-minimisation]
|
|
19
|
+
|
|
20
|
+
# No must_call: finding nothing is a valid outcome for a discovery agent, and
|
|
21
|
+
# requiring an emission would manufacture findings to satisfy it. That matters
|
|
22
|
+
# more here than elsewhere -- a product that is easy to navigate produces no
|
|
23
|
+
# friction envelopes, which is the result, not a failure of the run.
|
|
@@ -9,10 +9,6 @@ prompt: WARDEN.md
|
|
|
9
9
|
model: claude-opus-5
|
|
10
10
|
effort: high
|
|
11
11
|
max_turns: 60
|
|
12
|
-
# Measured: WARDEN exhausted $3.00 on its first real run against the demo app
|
|
13
|
-
# and was killed mid-audit. Building the endpoint-by-role matrix and actually
|
|
14
|
-
# impersonating each role costs more than reading a spec does.
|
|
15
|
-
max_budget_usd: 5.0
|
|
16
12
|
mcp_servers: [envelope, env_control, defect_memory]
|
|
17
13
|
builtin_tools: [Read, Grep, Glob]
|
|
18
14
|
policy: {}
|
|
@@ -9,6 +9,11 @@ target:
|
|
|
9
9
|
tracker: local # local | jira -- swap to file against real Jira
|
|
10
10
|
vcs: local # local | github
|
|
11
11
|
|
|
12
|
+
# No spend cap by default. A dollar figure bakes one vendor's pricing into
|
|
13
|
+
# the config, and this is meant to run against local models too. `max_turns`
|
|
14
|
+
# is the model-agnostic bound. Set `max_budget_usd` here if you want a
|
|
15
|
+
# ceiling; the governor enforces one whenever it is present.
|
|
16
|
+
|
|
12
17
|
thresholds:
|
|
13
18
|
min_confidence_to_file: 0.6 # §7 confidence gate
|
|
14
19
|
max_findings_per_agent_run: 25 # §8.3 loop breaker: pause and escalate, don't file
|
|
@@ -20,31 +25,23 @@ thresholds:
|
|
|
20
25
|
run_modes:
|
|
21
26
|
pr-check:
|
|
22
27
|
trigger: pull_request
|
|
23
|
-
|
|
24
|
-
#
|
|
25
|
-
#
|
|
26
|
-
#
|
|
27
|
-
|
|
28
|
-
# agents ran, findings landed in the ledger, nothing errored. `qaas validate`
|
|
29
|
-
# now refuses a mode whose agents cannot fit inside its cap.
|
|
30
|
-
max_budget_usd: 16.0
|
|
28
|
+
# KEYSTONE joins the PR sweep because it is pure static analysis and needs
|
|
29
|
+
# no running app. USHER, GAUGE and CHRONICLE do not: the design says GAUGE is
|
|
30
|
+
# "nightly and pre-release only -- too expensive and too noisy per-PR", and
|
|
31
|
+
# the same argument holds for a UX walk and a report about the week.
|
|
32
|
+
agents: [CARTOGRAPHER, KEYSTONE, CONDUIT, SURFACE, VAULT, WARDEN, FORGE, CLERK]
|
|
31
33
|
max_wall_clock_s: 900
|
|
32
34
|
max_concurrency: 2
|
|
33
35
|
|
|
34
36
|
nightly:
|
|
35
37
|
trigger: cron
|
|
36
|
-
agents: [CARTOGRAPHER, CONDUIT, SURFACE, VAULT, WARDEN, FORGE, CLERK]
|
|
37
|
-
# FORGE runs once per finding, so the deep sweep's budget scales with how
|
|
38
|
-
# much discovery found, not with the number of agents. Measured: discovery
|
|
39
|
-
# ~$6, then roughly $1-2 per finding reproduced.
|
|
40
|
-
max_budget_usd: 50.0
|
|
38
|
+
agents: [CARTOGRAPHER, KEYSTONE, CONDUIT, SURFACE, VAULT, WARDEN, PULSE, USHER, GAUGE, FORGE, CLERK, CHRONICLE]
|
|
41
39
|
max_wall_clock_s: 7200
|
|
42
40
|
max_concurrency: 3
|
|
43
41
|
|
|
44
42
|
incident:
|
|
45
43
|
trigger: alert
|
|
46
44
|
agents: [CONDUIT]
|
|
47
|
-
max_budget_usd: 4.0
|
|
48
45
|
max_wall_clock_s: 600
|
|
49
46
|
max_concurrency: 2
|
|
50
47
|
files_tickets: false # §9: diagnostic only, read-only, no filing
|
|
@@ -55,7 +52,6 @@ run_modes:
|
|
|
55
52
|
fix-cycle:
|
|
56
53
|
trigger: agent_ready_ticket
|
|
57
54
|
agents: [PROOF, MENDER, ARBITER]
|
|
58
|
-
max_budget_usd: 20.0
|
|
59
55
|
max_wall_clock_s: 3600
|
|
60
56
|
max_concurrency: 1
|
|
61
57
|
|
|
@@ -63,7 +59,6 @@ run_modes:
|
|
|
63
59
|
# expensive mode in the system and the only one that closes the loop.
|
|
64
60
|
full-loop:
|
|
65
61
|
trigger: on_demand
|
|
66
|
-
agents: [CARTOGRAPHER, CONDUIT, SURFACE, VAULT, WARDEN, FORGE, CLERK, MENDER, ARBITER, PROOF]
|
|
67
|
-
max_budget_usd: 70.0
|
|
62
|
+
agents: [CARTOGRAPHER, KEYSTONE, CONDUIT, SURFACE, VAULT, WARDEN, PULSE, USHER, GAUGE, FORGE, CLERK, MENDER, ARBITER, PROOF, CHRONICLE]
|
|
68
63
|
max_wall_clock_s: 10800
|
|
69
64
|
max_concurrency: 3
|