qaas-python 0.1.0__py3-none-any.whl → 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- qaas/conductor.py +9 -7
- qaas/defaults/config/agents/vault.yaml +21 -0
- qaas/defaults/config/agents/warden.yaml +23 -0
- qaas/defaults/config/system.yaml +4 -4
- qaas/prompts/VAULT.md +59 -0
- qaas/prompts/WARDEN.md +62 -0
- qaas/tasks.py +44 -1
- {qaas_python-0.1.0.dist-info → qaas_python-0.2.0.dist-info}/METADATA +10 -41
- {qaas_python-0.1.0.dist-info → qaas_python-0.2.0.dist-info}/RECORD +12 -8
- {qaas_python-0.1.0.dist-info → qaas_python-0.2.0.dist-info}/WHEEL +0 -0
- {qaas_python-0.1.0.dist-info → qaas_python-0.2.0.dist-info}/entry_points.txt +0 -0
- {qaas_python-0.1.0.dist-info → qaas_python-0.2.0.dist-info}/licenses/LICENSE +0 -0
qaas/conductor.py
CHANGED
|
@@ -272,18 +272,20 @@ class Conductor:
|
|
|
272
272
|
if not discovery:
|
|
273
273
|
return
|
|
274
274
|
|
|
275
|
+
# CONDUIT and SURFACE keep bespoke tasks because they name tools only
|
|
276
|
+
# they have. Everything else gets the generic discovery task, which is
|
|
277
|
+
# what makes "a new agent is a prompt plus a YAML" true: this used to be
|
|
278
|
+
# a closed dict, so a new discovery agent was skipped with `no task
|
|
279
|
+
# builder` -- it validated, it assembled, it showed up in `--dry-run`,
|
|
280
|
+
# and then it silently did nothing.
|
|
275
281
|
builders = {
|
|
276
|
-
"CONDUIT": lambda: tasks.conduit(self.config, mode),
|
|
277
|
-
"SURFACE": lambda: tasks.surface(self.config, mode),
|
|
282
|
+
"CONDUIT": lambda spec: tasks.conduit(self.config, mode),
|
|
283
|
+
"SURFACE": lambda spec: tasks.surface(self.config, mode),
|
|
278
284
|
}
|
|
279
285
|
jobs = [
|
|
280
|
-
(spec, builders
|
|
286
|
+
(spec, builders.get(spec.name, lambda sp: tasks.discovery(self.config, mode, sp))(spec))
|
|
281
287
|
for spec in discovery
|
|
282
|
-
if spec.name in builders
|
|
283
288
|
]
|
|
284
|
-
unknown = [s.name for s in discovery if s.name not in builders]
|
|
285
|
-
if unknown:
|
|
286
|
-
store.log("skipped", reason="no task builder", agents=unknown)
|
|
287
289
|
|
|
288
290
|
await self._gather(jobs, store, budget, report, map_version, self.config.run_modes[mode].max_concurrency)
|
|
289
291
|
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
name: VAULT
|
|
2
|
+
layer: discovery
|
|
3
|
+
role: >
|
|
4
|
+
Database and data-integrity analyst. Finds schema constraints the application
|
|
5
|
+
assumes but the database does not enforce, migrations that lose or corrupt
|
|
6
|
+
data, missing indexes on paths the code queries, and cross-tenant reads that
|
|
7
|
+
the ORM makes easy to write. Reports what the schema actually says, never what
|
|
8
|
+
the model layer claims.
|
|
9
|
+
prompt: VAULT.md
|
|
10
|
+
model: claude-opus-5
|
|
11
|
+
effort: high
|
|
12
|
+
max_turns: 60
|
|
13
|
+
max_budget_usd: 3.0
|
|
14
|
+
mcp_servers: [envelope, env_control, defect_memory]
|
|
15
|
+
builtin_tools: [Read, Grep, Glob]
|
|
16
|
+
policy: {}
|
|
17
|
+
|
|
18
|
+
skills: [authz-matrix-check, environment-pinning, severity-rubric, repro-minimisation]
|
|
19
|
+
|
|
20
|
+
# No must_call: finding nothing is a valid outcome for a discovery agent, and
|
|
21
|
+
# requiring an emission would manufacture findings to satisfy it.
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
name: WARDEN
|
|
2
|
+
layer: discovery
|
|
3
|
+
role: >
|
|
4
|
+
Security and dependency auditor. Finds missing authorization, secrets committed
|
|
5
|
+
to the repository, dependencies with known advisories, and error paths that
|
|
6
|
+
leak internals to a caller. Reports a concrete exploit path or lowers its
|
|
7
|
+
confidence -- a security finding without one is a guess wearing a severity.
|
|
8
|
+
prompt: WARDEN.md
|
|
9
|
+
model: claude-opus-5
|
|
10
|
+
effort: high
|
|
11
|
+
max_turns: 60
|
|
12
|
+
# Measured: WARDEN exhausted $3.00 on its first real run against the demo app
|
|
13
|
+
# and was killed mid-audit. Building the endpoint-by-role matrix and actually
|
|
14
|
+
# impersonating each role costs more than reading a spec does.
|
|
15
|
+
max_budget_usd: 5.0
|
|
16
|
+
mcp_servers: [envelope, env_control, defect_memory]
|
|
17
|
+
builtin_tools: [Read, Grep, Glob]
|
|
18
|
+
policy: {}
|
|
19
|
+
|
|
20
|
+
skills: [authz-matrix-check, error-taxonomy, severity-rubric, routing-rules, repro-minimisation]
|
|
21
|
+
|
|
22
|
+
# No must_call: see VAULT. Also: an auditor that must report something will
|
|
23
|
+
# report something, and security noise is the fastest way to be ignored.
|
qaas/defaults/config/system.yaml
CHANGED
|
@@ -33,11 +33,11 @@ run_modes:
|
|
|
33
33
|
|
|
34
34
|
nightly:
|
|
35
35
|
trigger: cron
|
|
36
|
-
agents: [CARTOGRAPHER, CONDUIT, SURFACE, FORGE, CLERK]
|
|
36
|
+
agents: [CARTOGRAPHER, CONDUIT, SURFACE, VAULT, WARDEN, FORGE, CLERK]
|
|
37
37
|
# FORGE runs once per finding, so the deep sweep's budget scales with how
|
|
38
38
|
# much discovery found, not with the number of agents. Measured: discovery
|
|
39
39
|
# ~$6, then roughly $1-2 per finding reproduced.
|
|
40
|
-
max_budget_usd:
|
|
40
|
+
max_budget_usd: 50.0
|
|
41
41
|
max_wall_clock_s: 7200
|
|
42
42
|
max_concurrency: 3
|
|
43
43
|
|
|
@@ -63,7 +63,7 @@ run_modes:
|
|
|
63
63
|
# expensive mode in the system and the only one that closes the loop.
|
|
64
64
|
full-loop:
|
|
65
65
|
trigger: on_demand
|
|
66
|
-
agents: [CARTOGRAPHER, CONDUIT, SURFACE, FORGE, CLERK, MENDER, ARBITER, PROOF]
|
|
67
|
-
max_budget_usd:
|
|
66
|
+
agents: [CARTOGRAPHER, CONDUIT, SURFACE, VAULT, WARDEN, FORGE, CLERK, MENDER, ARBITER, PROOF]
|
|
67
|
+
max_budget_usd: 70.0
|
|
68
68
|
max_wall_clock_s: 10800
|
|
69
69
|
max_concurrency: 3
|
qaas/prompts/VAULT.md
ADDED
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
You are VAULT, the database and data-integrity analyst.
|
|
2
|
+
|
|
3
|
+
## Your domain
|
|
4
|
+
|
|
5
|
+
The schema, and the distance between what it enforces and what the application
|
|
6
|
+
assumes. Application code is full of invariants nobody wrote down; your job is to
|
|
7
|
+
find the ones the database will not hold up.
|
|
8
|
+
|
|
9
|
+
Detect:
|
|
10
|
+
|
|
11
|
+
- **Constraints the code assumes and the schema does not enforce** — a field the
|
|
12
|
+
application treats as required with no `NOT NULL`, a relationship it treats as
|
|
13
|
+
unique with no unique index, an enum validated only in the model layer.
|
|
14
|
+
- **Missing foreign keys**, or ones declared without a delete rule, so a parent
|
|
15
|
+
row can leave orphans behind.
|
|
16
|
+
- **Cross-tenant reads** — a query filtered by id but not by the owning
|
|
17
|
+
organisation, on a table that has an owner column. The ORM makes this easy to
|
|
18
|
+
write and hard to see.
|
|
19
|
+
- **Migrations that lose or corrupt data** — a column dropped and re-added, a type
|
|
20
|
+
narrowed without a backfill, a `NOT NULL` added without a default over existing
|
|
21
|
+
rows.
|
|
22
|
+
- **Indexes the query patterns need and the schema lacks** — a column filtered or
|
|
23
|
+
joined on in application code with no index behind it. Say which query, not
|
|
24
|
+
just which column.
|
|
25
|
+
- **Seed and fixture drift** — fixtures that no longer satisfy the constraints the
|
|
26
|
+
migrations now declare.
|
|
27
|
+
|
|
28
|
+
## How you work
|
|
29
|
+
|
|
30
|
+
1. Read the system map for the schema snapshot and the route inventory. Do not
|
|
31
|
+
rediscover them.
|
|
32
|
+
2. Read the migrations in order. The current schema is the sum of them, and a
|
|
33
|
+
defect is often visible only in the sequence — a constraint added, then
|
|
34
|
+
dropped two migrations later to make a deploy pass.
|
|
35
|
+
3. Read the model and query layer and compare its assumptions against what the
|
|
36
|
+
schema actually declares. The gap between the two is your finding.
|
|
37
|
+
4. Where an environment is available, confirm the behaviour rather than inferring
|
|
38
|
+
it: insert the row the code believes is impossible, and see whether the
|
|
39
|
+
database refuses it.
|
|
40
|
+
5. Pin the environment for anything you reproduce, so it runs the same way later.
|
|
41
|
+
|
|
42
|
+
## What counts as evidence
|
|
43
|
+
|
|
44
|
+
The schema text, the migration, and the query. A finding that says "this column
|
|
45
|
+
should be indexed" without naming the query that scans it is an opinion. A
|
|
46
|
+
finding that says "this insert succeeds and the model layer says it cannot" with
|
|
47
|
+
the statement and the response is a defect.
|
|
48
|
+
|
|
49
|
+
Where you could not observe the behaviour — no reachable database, no fixture
|
|
50
|
+
that reaches the path — say so plainly and lower your confidence. An honest
|
|
51
|
+
`unattempted` reproduction is worth more than a confident guess, because the next
|
|
52
|
+
agent will treat your confidence as real.
|
|
53
|
+
|
|
54
|
+
## What is not yours
|
|
55
|
+
|
|
56
|
+
The HTTP surface is CONDUIT's, the UI is SURFACE's, and dependency advisories are
|
|
57
|
+
WARDEN's. A cross-tenant read is yours when the defect is in the query, and
|
|
58
|
+
CONDUIT's when the defect is in the missing authorization check. If both are true,
|
|
59
|
+
report the one you can evidence.
|
qaas/prompts/WARDEN.md
ADDED
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
You are WARDEN, the security and dependency auditor.
|
|
2
|
+
|
|
3
|
+
## Your domain
|
|
4
|
+
|
|
5
|
+
The things that let someone do what they should not be able to do. You are the
|
|
6
|
+
agent whose findings carry the most weight and therefore cost the most when they
|
|
7
|
+
are wrong.
|
|
8
|
+
|
|
9
|
+
Detect:
|
|
10
|
+
|
|
11
|
+
- **Missing or wrong authorization** — an endpoint that mutates or reads data
|
|
12
|
+
without checking the caller's role, or that checks authentication and calls it
|
|
13
|
+
authorization. The presence of an auth dependency is not evidence that access
|
|
14
|
+
is checked.
|
|
15
|
+
- **Cross-tenant access** — one organisation's data reachable by another's user.
|
|
16
|
+
- **Secrets in the repository** — keys, tokens, passwords and connection strings
|
|
17
|
+
in source, fixtures, CI config or committed environment files.
|
|
18
|
+
- **Dependencies with known advisories**, and dependencies pinned to a version
|
|
19
|
+
behind a security release.
|
|
20
|
+
- **Internal detail leaking to a caller** — stack traces, SQL, file paths, library
|
|
21
|
+
versions in an error response.
|
|
22
|
+
- **Mass assignment** — a handler that accepts fields the client should not
|
|
23
|
+
control, such as a role, a price, or a status.
|
|
24
|
+
- **Weak or absent rate limiting** on authentication and password-reset paths.
|
|
25
|
+
|
|
26
|
+
## How you work
|
|
27
|
+
|
|
28
|
+
1. Read the system map for the route inventory and the role matrix. Do not
|
|
29
|
+
rediscover them.
|
|
30
|
+
2. Build the endpoint-by-role matrix and look for the holes, rather than reading
|
|
31
|
+
handlers in file order and hoping to notice.
|
|
32
|
+
3. Where an environment is available, **demonstrate the access** — impersonate the
|
|
33
|
+
lower-privilege role and make the call. A refusal you predicted and a refusal
|
|
34
|
+
you observed are different findings.
|
|
35
|
+
4. For dependencies, name the advisory and the version that fixes it.
|
|
36
|
+
|
|
37
|
+
## The bar for a security finding
|
|
38
|
+
|
|
39
|
+
**A concrete exploit path, or lower your confidence.** Say which role, which
|
|
40
|
+
endpoint, which field, and what they get. "This endpoint may be missing an
|
|
41
|
+
authorization check" is a note to yourself; "a viewer can POST
|
|
42
|
+
/v1/orders/3/refund and it succeeds" is a finding.
|
|
43
|
+
|
|
44
|
+
This matters more here than anywhere else in the system. A security finding is
|
|
45
|
+
routed to a restricted project, wakes people up, and is read as urgent. A false
|
|
46
|
+
one spends that credibility, and the next real finding is read more slowly. If
|
|
47
|
+
you cannot evidence it, report it with the confidence it actually deserves and
|
|
48
|
+
say what you could not test.
|
|
49
|
+
|
|
50
|
+
## Routing
|
|
51
|
+
|
|
52
|
+
Security findings are routed to a restricted project, and the tracker will
|
|
53
|
+
**refuse** to file one if no restricted project is configured rather than filing
|
|
54
|
+
it somewhere the whole company can read. That refusal is correct; do not work
|
|
55
|
+
around it by relabelling the finding as something else.
|
|
56
|
+
|
|
57
|
+
## What is not yours
|
|
58
|
+
|
|
59
|
+
Spec drift and error-shape inconsistency are CONDUIT's unless the leak has a
|
|
60
|
+
security consequence. Schema constraints are VAULT's. A missing index is nobody's
|
|
61
|
+
security problem. When a finding is genuinely both, report the security
|
|
62
|
+
consequence and say which other surface it also touches.
|
qaas/tasks.py
CHANGED
|
@@ -11,7 +11,7 @@ one repository's directory layout or one app's seeded users works exactly once.
|
|
|
11
11
|
|
|
12
12
|
from __future__ import annotations
|
|
13
13
|
|
|
14
|
-
from qaas.config import SystemConfig
|
|
14
|
+
from qaas.config import AgentSpec, SystemConfig
|
|
15
15
|
from qaas.envelope import DefectEnvelope
|
|
16
16
|
from qaas.target import TargetProfile
|
|
17
17
|
|
|
@@ -218,6 +218,49 @@ fails, misleads, blocks or excludes someone. Do not report what you would have
|
|
|
218
218
|
designed differently."""
|
|
219
219
|
|
|
220
220
|
|
|
221
|
+
def discovery(config: SystemConfig, mode: str, spec: "AgentSpec") -> str:
|
|
222
|
+
"""The task for a discovery agent with no hand-written builder.
|
|
223
|
+
|
|
224
|
+
The architecture's claim is that adding an agent needs a prompt file and a
|
|
225
|
+
YAML file and no Python. That was not true: `_phase_discover` dispatched
|
|
226
|
+
from a hardcoded dict of builders, so a new discovery agent was silently
|
|
227
|
+
skipped with `no task builder` -- it validated, it assembled, it appeared in
|
|
228
|
+
`--dry-run`, and then it did nothing. VAULT and WARDEN were added exactly
|
|
229
|
+
that way and this is the bug they found.
|
|
230
|
+
|
|
231
|
+
What an agent should be told is: which application, what it can reach, and
|
|
232
|
+
what its own prompt says its domain is. Everything specific to a domain
|
|
233
|
+
belongs in that agent's prompt, not here -- CONDUIT and SURFACE keep their
|
|
234
|
+
bespoke builders because they name tools (`diff_openapi`, the browser) that
|
|
235
|
+
only they have.
|
|
236
|
+
"""
|
|
237
|
+
p = _profile(config)
|
|
238
|
+
reach = (
|
|
239
|
+
"The application is reachable, so prove what you report: observe the "
|
|
240
|
+
"behaviour and capture the evidence. A finding you have not observed is a "
|
|
241
|
+
"hypothesis, and its confidence should say so."
|
|
242
|
+
if p.environment.is_reachable
|
|
243
|
+
else "There is no reachable instance, so every finding is a reading of the "
|
|
244
|
+
"code. Quote the lines that support it and keep your confidence honest "
|
|
245
|
+
"about not having observed the behaviour."
|
|
246
|
+
)
|
|
247
|
+
return f"""Audit {p.name} for defects in your domain.
|
|
248
|
+
|
|
249
|
+
Layout — {p.layout.described()}
|
|
250
|
+
|
|
251
|
+
Your own instructions define what your domain is and what counts as evidence in
|
|
252
|
+
it. Work within it and leave the other surfaces to the agents that own them.
|
|
253
|
+
|
|
254
|
+
{reach}
|
|
255
|
+
|
|
256
|
+
Emit one envelope per distinct defect with `emit_envelope`. Finding nothing is a
|
|
257
|
+
valid outcome; inventing something to report is not. Deduplicate against
|
|
258
|
+
`search_similar` before you emit, so a defect this system already knows about
|
|
259
|
+
comes back as an occurrence rather than a new finding.
|
|
260
|
+
|
|
261
|
+
Mode: {mode}."""
|
|
262
|
+
|
|
263
|
+
|
|
221
264
|
def forge(envelope: DefectEnvelope, config: SystemConfig, flake_runs: int) -> str:
|
|
222
265
|
p = _profile(config)
|
|
223
266
|
evidence = "\n".join(f" - {e.type.value}: {e.uri} {e.note}".rstrip() for e in envelope.evidence)
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: qaas-python
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.2.0
|
|
4
4
|
Summary: A multi-agent QA system: finds real defects, reproduces them, files tickets, fixes them, and proves the fix
|
|
5
5
|
Project-URL: Homepage, https://github.com/allaabdella2-us/qa-multi-agent-system
|
|
6
6
|
Project-URL: Repository, https://github.com/allaabdella2-us/qa-multi-agent-system
|
|
@@ -22,7 +22,7 @@ Requires-Dist: claude-agent-sdk>=0.2.127
|
|
|
22
22
|
Requires-Dist: pydantic>=2.9
|
|
23
23
|
Requires-Dist: pyyaml>=6.0
|
|
24
24
|
Requires-Dist: rich>=13.9
|
|
25
|
-
Requires-Dist: typer>=0.
|
|
25
|
+
Requires-Dist: typer>=0.16
|
|
26
26
|
Provides-Extra: dev
|
|
27
27
|
Requires-Dist: pytest-asyncio>=0.24; extra == 'dev'
|
|
28
28
|
Requires-Dist: pytest>=8.3; extra == 'dev'
|
|
@@ -43,7 +43,7 @@ Description-Content-Type: text/markdown
|
|
|
43
43
|
[](#-contributing)
|
|
44
44
|
[](https://docs.claude.com/en/api/agent-sdk/overview)
|
|
45
45
|
|
|
46
|
-
[Quickstart](#-quickstart-in-60-seconds) · [
|
|
46
|
+
[Quickstart](#-quickstart-in-60-seconds) · [Your repo](#-point-it-at-your-repository) · [Jira](#-file-into-jira) · [Architecture](ARCHITECTURE.md)
|
|
47
47
|
|
|
48
48
|
</div>
|
|
49
49
|
|
|
@@ -51,7 +51,7 @@ Description-Content-Type: text/markdown
|
|
|
51
51
|
|
|
52
52
|
Most "AI QA" tools generate tests. **This one behaves like a QA team.**
|
|
53
53
|
|
|
54
|
-
|
|
54
|
+
Ten agents, each with its own context, tool allowlist and budget, coordinated by
|
|
55
55
|
a state machine that is ordinary Python — because a model cannot enforce a budget
|
|
56
56
|
it is itself spending.
|
|
57
57
|
|
|
@@ -111,37 +111,6 @@ qaas run --repo https://github.com/you/your-app --dry-run
|
|
|
111
111
|
|
|
112
112
|
---
|
|
113
113
|
|
|
114
|
-
## 💰 What it costs
|
|
115
|
-
|
|
116
|
-
> [!IMPORTANT]
|
|
117
|
-
> **Real runs spend real money.** Read this before your first one.
|
|
118
|
-
|
|
119
|
-
Median cost per dispatch, **measured** across real runs — not estimated:
|
|
120
|
-
|
|
121
|
-
| agent | median | what you get |
|
|
122
|
-
|---|--:|---|
|
|
123
|
-
| 🖱️ `SURFACE` | **$3.34** | broken flows, console errors, a11y, forms |
|
|
124
|
-
| 🔧 `MENDER` | **$2.03** | the minimal fix, on a branch |
|
|
125
|
-
| 🔌 `CONDUIT` | **$1.97** | API contract, authz and error-shape defects |
|
|
126
|
-
| 🔨 `FORGE` | **$1.61** | a minimal repro + failing test — **per finding** |
|
|
127
|
-
| ✅ `PROOF` / ⚖️ `ARBITER` | ~$1.00 | verification and adversarial review |
|
|
128
|
-
| 📝 `CLERK` / 🗺️ `CARTOGRAPHER` | ~$0.70 | filing, and the map everything reads |
|
|
129
|
-
|
|
130
|
-
A full discovery run over the demo app found **13 of 16** seeded defects for
|
|
131
|
-
about **$15**. A `fix-cycle` pass costs **$5–7**.
|
|
132
|
-
|
|
133
|
-
> `FORGE` runs **once per finding** in a fresh context, so cost scales with what
|
|
134
|
-
> was found, not with how many agents exist.
|
|
135
|
-
|
|
136
|
-
**The controls are real, not advisory:**
|
|
137
|
-
|
|
138
|
-
- `max_budget_usd` per agent *and* per run mode; the governor checks before every dispatch and **stops the run** rather than overspending.
|
|
139
|
-
- The cap survives a resume — `qaas run --run-id <existing>` carries forward what that run already spent.
|
|
140
|
-
- `qaas validate` refuses a run mode whose agents could outspend its cap.
|
|
141
|
-
- `--dry-run` on everything.
|
|
142
|
-
|
|
143
|
-
---
|
|
144
|
-
|
|
145
114
|
## 🎯 Point it at your repository
|
|
146
115
|
|
|
147
116
|
```bash
|
|
@@ -276,20 +245,20 @@ Every tool call, denial, verdict and escalation is on the record.
|
|
|
276
245
|
|
|
277
246
|
```console
|
|
278
247
|
$ qaas trace run-20260908T182034-c6ed26
|
|
279
|
-
t+ agent kind detail
|
|
280
|
-
0s - run_started mode=nightly agents=[
|
|
248
|
+
t+ agent kind detail
|
|
249
|
+
0s - run_started mode=nightly agents=[7]
|
|
281
250
|
0s CARTOGRAPHER agent_started model=claude-sonnet-5
|
|
282
251
|
4s CARTOGRAPHER tool_call ×34 Read×25, Glob×6, ToolSearch×2
|
|
283
252
|
6s CARTOGRAPHER denial tool=Bash reason=Bash is not in CARTOGRAPHER's
|
|
284
253
|
tool allowlist (Read, Grep, Glob).
|
|
285
254
|
146s CARTOGRAPHER system_map version=20260907T233530 sections=[12]
|
|
286
|
-
156s CARTOGRAPHER agent_finished subtype=success num_turns=45
|
|
255
|
+
156s CARTOGRAPHER agent_finished subtype=success num_turns=45
|
|
287
256
|
```
|
|
288
257
|
|
|
289
258
|
```bash
|
|
290
259
|
qaas trace <run-id> --agent proof --kind verdict # filter
|
|
291
260
|
qaas trace <run-id> --json # export
|
|
292
|
-
qaas show <run-id> # mode, commit,
|
|
261
|
+
qaas show <run-id> # mode, commit, tickets, escalations
|
|
293
262
|
qaas runs # everything that ever ran
|
|
294
263
|
```
|
|
295
264
|
|
|
@@ -327,7 +296,6 @@ Two runs against the demo app, scored automatically:
|
|
|
327
296
|
| 🎯 recall | **81%** — 13 of 16 | **69%** — 11 of 16 |
|
|
328
297
|
| 🔇 precision | **100%** — 0 FP | **92%** — 1 FP |
|
|
329
298
|
| 🏷️ severity agreement | **100%** | **100%** |
|
|
330
|
-
| 💵 cost per accepted finding | $1.12 | $0.64 |
|
|
331
299
|
|
|
332
300
|
**Both numbers are shown on purpose.** A single figure would be the flattering
|
|
333
301
|
one, and it would not survive contact with a second run. These are stochastic
|
|
@@ -350,7 +318,8 @@ precision is measured rather than assumed.
|
|
|
350
318
|
|
|
351
319
|
Honest about what exists:
|
|
352
320
|
|
|
353
|
-
- ✅ **
|
|
321
|
+
- ✅ **10 of the 16 agents** in the design are built — CARTOGRAPHER, CONDUIT, SURFACE, VAULT, WARDEN, FORGE, CLERK, MENDER, ARBITER, PROOF. CONDUCTOR is the Python state machine rather than an agent. The five that remain (KEYSTONE, PULSE, USHER, GAUGE, CHRONICLE) are additional discovery specialists, not missing parts of the loop.
|
|
322
|
+
- ✅ **Adding an agent needs a prompt file and a YAML file — no Python.** VAULT and WARDEN were added exactly that way, which is how the claim finally got tested.
|
|
354
323
|
- ✅ The fix loop has closed end to end on a real defect: `NOT_FIXED → MENDER → ARBITER APPROVE → VERIFIED`.
|
|
355
324
|
- ✅ 30 skills, 7 in-process MCP servers, 649 offline tests.
|
|
356
325
|
- ⚠️ Running the bundled demo needs `export CORVID_PASSWORD=password123` — credentials come from the environment, including the demo's.
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
qaas/cli.py,sha256=L12m7SGyCYNcraUXzu0bCp9A1UtiSF0L6CLzS7xVnVc,65477
|
|
2
|
-
qaas/conductor.py,sha256=
|
|
2
|
+
qaas/conductor.py,sha256=pURmyPm825s3xg1y-ypFRToeB_OoO89OQxI0oy-Gx8E,23075
|
|
3
3
|
qaas/config.py,sha256=ZzgK0KUy7MR31w0o7fe1D2tYl--jkZCN_VaWR2kK1Xs,17528
|
|
4
4
|
qaas/discover.py,sha256=L5ejBsYs_s41OWAL-Dhc8VlQ5EWw2hGayWTDzfj05R8,8524
|
|
5
5
|
qaas/envelope.py,sha256=IiqyOy96CHZw2A0pSDWBNIg6yQ2MzfcwkpQWmgMNvzY,9095
|
|
@@ -11,12 +11,12 @@ qaas/scorecard.py,sha256=aUU7g5OtXIW4752A4NxpIZH-OdC8bpNtzJUvCi41JYg,15809
|
|
|
11
11
|
qaas/sdk_compat.py,sha256=ftE6PK0jZY85zkYqS7UJTGYwRkYDlH4NspiKbKVfVXQ,1678
|
|
12
12
|
qaas/store.py,sha256=aDJmeCog17zo_Dw2Pw_JHTcWnDQlDW3QKzNnUttVvg8,11038
|
|
13
13
|
qaas/target.py,sha256=-wgFDOPm8tCYzideEJV4hdVb1aakLqAM-KJT4Egfgzw,10378
|
|
14
|
-
qaas/tasks.py,sha256=
|
|
14
|
+
qaas/tasks.py,sha256=vQh3leld5OE6zK3gfxZxbCM5ZixOQOzheJ_8ZomgHgE,17146
|
|
15
15
|
qaas/trace.py,sha256=V-uFqh1iCYRqgWL3VzDbRdHgAkwLsr1uRFfwwycWZ94,11461
|
|
16
16
|
qaas/adapters/__init__.py,sha256=bw2pqtDqhZGP730gwV28BJ-8TF-rhanwCEjdIizjP6A,882
|
|
17
17
|
qaas/adapters/tracker.py,sha256=K1U7weiA_K2MM58yji3WQn3PEATVqDD9_qcRbrcZwik,54348
|
|
18
18
|
qaas/adapters/vcs.py,sha256=9su-4QLLxR6yTyLZWCJAky9BROJci80M5jV6nGo6pjg,19430
|
|
19
|
-
qaas/defaults/config/system.yaml,sha256=
|
|
19
|
+
qaas/defaults/config/system.yaml,sha256=io7NRqtiU9HLvV5dUkQn7YUNxCCzi5TOdnnQUOK5bc0,2837
|
|
20
20
|
qaas/defaults/config/agents/arbiter.yaml,sha256=HuPv4E9-p2LW3-bzt3ard9QTJSw98HZYKf-ae6Skd1o,739
|
|
21
21
|
qaas/defaults/config/agents/cartographer.yaml,sha256=9FWySvbwD1bHuMm8CkyaJhmrtgIJrRuRGrli_ZZa_pw,772
|
|
22
22
|
qaas/defaults/config/agents/clerk.yaml,sha256=VgjdqROYr7khipxmou-Ol3Bv-FWj5CZENspUQDA1UB4,768
|
|
@@ -25,6 +25,8 @@ qaas/defaults/config/agents/forge.yaml,sha256=WGwkueX0HvaKep-5m2bWR7v6Hx29bs-nuo
|
|
|
25
25
|
qaas/defaults/config/agents/mender.yaml,sha256=-HoSJtm5GQSlozgr0O5BOvyibJanYSuJuUYhSs7MRQE,2182
|
|
26
26
|
qaas/defaults/config/agents/proof.yaml,sha256=JbXl2yKUHRv7q-A6MEcz88E3gJn78vNxw6fdmsDbXus,853
|
|
27
27
|
qaas/defaults/config/agents/surface.yaml,sha256=IqmWhQQYArEsNiiRKyw2uTue04CfYreHF9yxJ6z7p_Y,532
|
|
28
|
+
qaas/defaults/config/agents/vault.yaml,sha256=Ib8gqKxW7LGSlxQ4UGWSZgJB7t78FYPQrphxKrgef1g,795
|
|
29
|
+
qaas/defaults/config/agents/warden.yaml,sha256=iu3FHRRxsRraN68PtlfrlUqbBu4x9OIjc4F0XAjQsYQ,1002
|
|
28
30
|
qaas/mcp/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
29
31
|
qaas/mcp/context.py,sha256=Z45wQuP0nIZOmM5GTfxs6KHxFckLQ61H4zbsVqiVqr8,2551
|
|
30
32
|
qaas/mcp/contract_diff.py,sha256=efuma0WQSnxHC0NE7adF_8ctCZoovdB4k080QdiEJK8,43942
|
|
@@ -73,9 +75,11 @@ qaas/prompts/FORGE.md,sha256=7KvnMrzS1UtucIoE8IbWtAd6BC9J-0YO-e2fSdyptds,2269
|
|
|
73
75
|
qaas/prompts/MENDER.md,sha256=eot0WsyilfpWkJ9OsEmzeimikHXVSzd3jFKdslSn5GI,2921
|
|
74
76
|
qaas/prompts/PROOF.md,sha256=xJ5pi4O0N1AFq_xZwAeUawJrP6kknRhnkfpDLqgyo2Q,2026
|
|
75
77
|
qaas/prompts/SURFACE.md,sha256=cZ9df27bXXRh39yEkwByyxvHskpcNmh6HvC9C3Jt83Y,2220
|
|
78
|
+
qaas/prompts/VAULT.md,sha256=Ansowimx-wMbinQkn9_gGgZFmvmMFMc9v_N8s2Ahws4,2915
|
|
79
|
+
qaas/prompts/WARDEN.md,sha256=39KsbwWRwe7mtjYEik1owBIDgu4qyD52UHG_T_7gTMw,2950
|
|
76
80
|
qaas/prompts/_shared.md,sha256=lN_s_rAmakhGuyRuWItC--iyy-Er6ohT_dzGeE_g-fo,2620
|
|
77
|
-
qaas_python-0.
|
|
78
|
-
qaas_python-0.
|
|
79
|
-
qaas_python-0.
|
|
80
|
-
qaas_python-0.
|
|
81
|
-
qaas_python-0.
|
|
81
|
+
qaas_python-0.2.0.dist-info/METADATA,sha256=HZx6pPuXaxXYhL2HwjThoiTpbB9o3LUGnh4ffziVdcE,15528
|
|
82
|
+
qaas_python-0.2.0.dist-info/WHEEL,sha256=zOwg4jB6zX2kU910N-cMawjivD6tO8NEWvE12je1bVk,87
|
|
83
|
+
qaas_python-0.2.0.dist-info/entry_points.txt,sha256=6UScfruyhP9N_xGx3tXJGkaoAiB36dkINuKyOH6OkK4,38
|
|
84
|
+
qaas_python-0.2.0.dist-info/licenses/LICENSE,sha256=pHWke5oMtv7PLjIQbN6hRa31J0AKj51VCd5TCTUbbX0,1069
|
|
85
|
+
qaas_python-0.2.0.dist-info/RECORD,,
|
|
File without changes
|
|
File without changes
|
|
File without changes
|