qaas-python 0.1.1__py3-none-any.whl → 0.2.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- qaas/cli.py +6 -8
- qaas/conductor.py +31 -7
- qaas/defaults/config/agents/vault.yaml +21 -0
- qaas/defaults/config/agents/warden.yaml +23 -0
- qaas/defaults/config/system.yaml +4 -4
- qaas/prompts/VAULT.md +59 -0
- qaas/prompts/WARDEN.md +62 -0
- qaas/target.py +18 -0
- qaas/tasks.py +44 -1
- {qaas_python-0.1.1.dist-info → qaas_python-0.2.1.dist-info}/METADATA +9 -9
- {qaas_python-0.1.1.dist-info → qaas_python-0.2.1.dist-info}/RECORD +14 -10
- {qaas_python-0.1.1.dist-info → qaas_python-0.2.1.dist-info}/WHEEL +0 -0
- {qaas_python-0.1.1.dist-info → qaas_python-0.2.1.dist-info}/entry_points.txt +0 -0
- {qaas_python-0.1.1.dist-info → qaas_python-0.2.1.dist-info}/licenses/LICENSE +0 -0
qaas/cli.py
CHANGED
|
@@ -449,16 +449,14 @@ def doctor(
|
|
|
449
449
|
|
|
450
450
|
|
|
451
451
|
def _agent_usable(spec, caps: dict[str, bool]) -> bool:
|
|
452
|
-
"""
|
|
452
|
+
"""Delegates to `target.agent_usable`, which the conductor also uses.
|
|
453
453
|
|
|
454
|
-
|
|
455
|
-
|
|
454
|
+
Two copies of this rule meant `qaas doctor` could report an agent unusable
|
|
455
|
+
while a run dispatched it anyway.
|
|
456
456
|
"""
|
|
457
|
-
|
|
458
|
-
|
|
459
|
-
|
|
460
|
-
return caps["live_api"] or caps["static_analysis"]
|
|
461
|
-
return True
|
|
457
|
+
from qaas.target import agent_usable
|
|
458
|
+
|
|
459
|
+
return agent_usable(spec.name, caps)
|
|
462
460
|
|
|
463
461
|
|
|
464
462
|
@app.command()
|
qaas/conductor.py
CHANGED
|
@@ -25,6 +25,7 @@ from typing import Any, Callable
|
|
|
25
25
|
|
|
26
26
|
from qaas.config import AgentSpec, SystemConfig, load_config
|
|
27
27
|
from qaas.envelope import DefectEnvelope
|
|
28
|
+
from qaas.target import agent_usable
|
|
28
29
|
from qaas.mcp.context import ToolContext
|
|
29
30
|
from qaas.runner import RunOutcome, run_agent
|
|
30
31
|
from qaas.store import RunStore, SystemMapStore
|
|
@@ -269,21 +270,44 @@ class Conductor:
|
|
|
269
270
|
async def _phase_discover(self, specs, store, budget, report, mode, map_version) -> None:
|
|
270
271
|
"""Discovery agents are independent. Run them concurrently, bounded."""
|
|
271
272
|
discovery = [s for name, s in specs.items() if s.layer == "discovery"]
|
|
273
|
+
|
|
274
|
+
# Skip agents this target cannot support. `qaas doctor` has always
|
|
275
|
+
# reported these ("agents that cannot: SURFACE"), but nothing acted on
|
|
276
|
+
# it, so a run against a target with no reachable UI would still
|
|
277
|
+
# dispatch SURFACE and spend its entire budget hunting a browser that
|
|
278
|
+
# was never there. Being told an agent cannot work and then watching it
|
|
279
|
+
# run is worse than not being told.
|
|
280
|
+
profile = getattr(self.config, "profile", None)
|
|
281
|
+
if profile is not None:
|
|
282
|
+
caps = profile.capabilities()
|
|
283
|
+
unusable = [s for s in discovery if not agent_usable(s.name, caps)]
|
|
284
|
+
if unusable:
|
|
285
|
+
discovery = [s for s in discovery if s not in unusable]
|
|
286
|
+
store.log(
|
|
287
|
+
"skipped",
|
|
288
|
+
reason="target cannot support these agents",
|
|
289
|
+
agents=[s.name for s in unusable],
|
|
290
|
+
)
|
|
291
|
+
for spec in unusable:
|
|
292
|
+
self._emit("skipped", agent=spec.name, reason="target lacks the capability")
|
|
293
|
+
|
|
272
294
|
if not discovery:
|
|
273
295
|
return
|
|
274
296
|
|
|
297
|
+
# CONDUIT and SURFACE keep bespoke tasks because they name tools only
|
|
298
|
+
# they have. Everything else gets the generic discovery task, which is
|
|
299
|
+
# what makes "a new agent is a prompt plus a YAML" true: this used to be
|
|
300
|
+
# a closed dict, so a new discovery agent was skipped with `no task
|
|
301
|
+
# builder` -- it validated, it assembled, it showed up in `--dry-run`,
|
|
302
|
+
# and then it silently did nothing.
|
|
275
303
|
builders = {
|
|
276
|
-
"CONDUIT": lambda: tasks.conduit(self.config, mode),
|
|
277
|
-
"SURFACE": lambda: tasks.surface(self.config, mode),
|
|
304
|
+
"CONDUIT": lambda spec: tasks.conduit(self.config, mode),
|
|
305
|
+
"SURFACE": lambda spec: tasks.surface(self.config, mode),
|
|
278
306
|
}
|
|
279
307
|
jobs = [
|
|
280
|
-
(spec, builders
|
|
308
|
+
(spec, builders.get(spec.name, lambda sp: tasks.discovery(self.config, mode, sp))(spec))
|
|
281
309
|
for spec in discovery
|
|
282
|
-
if spec.name in builders
|
|
283
310
|
]
|
|
284
|
-
unknown = [s.name for s in discovery if s.name not in builders]
|
|
285
|
-
if unknown:
|
|
286
|
-
store.log("skipped", reason="no task builder", agents=unknown)
|
|
287
311
|
|
|
288
312
|
await self._gather(jobs, store, budget, report, map_version, self.config.run_modes[mode].max_concurrency)
|
|
289
313
|
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
name: VAULT
|
|
2
|
+
layer: discovery
|
|
3
|
+
role: >
|
|
4
|
+
Database and data-integrity analyst. Finds schema constraints the application
|
|
5
|
+
assumes but the database does not enforce, migrations that lose or corrupt
|
|
6
|
+
data, missing indexes on paths the code queries, and cross-tenant reads that
|
|
7
|
+
the ORM makes easy to write. Reports what the schema actually says, never what
|
|
8
|
+
the model layer claims.
|
|
9
|
+
prompt: VAULT.md
|
|
10
|
+
model: claude-opus-5
|
|
11
|
+
effort: high
|
|
12
|
+
max_turns: 60
|
|
13
|
+
max_budget_usd: 3.0
|
|
14
|
+
mcp_servers: [envelope, env_control, defect_memory]
|
|
15
|
+
builtin_tools: [Read, Grep, Glob]
|
|
16
|
+
policy: {}
|
|
17
|
+
|
|
18
|
+
skills: [authz-matrix-check, environment-pinning, severity-rubric, repro-minimisation]
|
|
19
|
+
|
|
20
|
+
# No must_call: finding nothing is a valid outcome for a discovery agent, and
|
|
21
|
+
# requiring an emission would manufacture findings to satisfy it.
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
name: WARDEN
|
|
2
|
+
layer: discovery
|
|
3
|
+
role: >
|
|
4
|
+
Security and dependency auditor. Finds missing authorization, secrets committed
|
|
5
|
+
to the repository, dependencies with known advisories, and error paths that
|
|
6
|
+
leak internals to a caller. Reports a concrete exploit path or lowers its
|
|
7
|
+
confidence -- a security finding without one is a guess wearing a severity.
|
|
8
|
+
prompt: WARDEN.md
|
|
9
|
+
model: claude-opus-5
|
|
10
|
+
effort: high
|
|
11
|
+
max_turns: 60
|
|
12
|
+
# Measured: WARDEN exhausted $3.00 on its first real run against the demo app
|
|
13
|
+
# and was killed mid-audit. Building the endpoint-by-role matrix and actually
|
|
14
|
+
# impersonating each role costs more than reading a spec does.
|
|
15
|
+
max_budget_usd: 5.0
|
|
16
|
+
mcp_servers: [envelope, env_control, defect_memory]
|
|
17
|
+
builtin_tools: [Read, Grep, Glob]
|
|
18
|
+
policy: {}
|
|
19
|
+
|
|
20
|
+
skills: [authz-matrix-check, error-taxonomy, severity-rubric, routing-rules, repro-minimisation]
|
|
21
|
+
|
|
22
|
+
# No must_call: see VAULT. Also: an auditor that must report something will
|
|
23
|
+
# report something, and security noise is the fastest way to be ignored.
|
qaas/defaults/config/system.yaml
CHANGED
|
@@ -33,11 +33,11 @@ run_modes:
|
|
|
33
33
|
|
|
34
34
|
nightly:
|
|
35
35
|
trigger: cron
|
|
36
|
-
agents: [CARTOGRAPHER, CONDUIT, SURFACE, FORGE, CLERK]
|
|
36
|
+
agents: [CARTOGRAPHER, CONDUIT, SURFACE, VAULT, WARDEN, FORGE, CLERK]
|
|
37
37
|
# FORGE runs once per finding, so the deep sweep's budget scales with how
|
|
38
38
|
# much discovery found, not with the number of agents. Measured: discovery
|
|
39
39
|
# ~$6, then roughly $1-2 per finding reproduced.
|
|
40
|
-
max_budget_usd:
|
|
40
|
+
max_budget_usd: 50.0
|
|
41
41
|
max_wall_clock_s: 7200
|
|
42
42
|
max_concurrency: 3
|
|
43
43
|
|
|
@@ -63,7 +63,7 @@ run_modes:
|
|
|
63
63
|
# expensive mode in the system and the only one that closes the loop.
|
|
64
64
|
full-loop:
|
|
65
65
|
trigger: on_demand
|
|
66
|
-
agents: [CARTOGRAPHER, CONDUIT, SURFACE, FORGE, CLERK, MENDER, ARBITER, PROOF]
|
|
67
|
-
max_budget_usd:
|
|
66
|
+
agents: [CARTOGRAPHER, CONDUIT, SURFACE, VAULT, WARDEN, FORGE, CLERK, MENDER, ARBITER, PROOF]
|
|
67
|
+
max_budget_usd: 70.0
|
|
68
68
|
max_wall_clock_s: 10800
|
|
69
69
|
max_concurrency: 3
|
qaas/prompts/VAULT.md
ADDED
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
You are VAULT, the database and data-integrity analyst.
|
|
2
|
+
|
|
3
|
+
## Your domain
|
|
4
|
+
|
|
5
|
+
The schema, and the distance between what it enforces and what the application
|
|
6
|
+
assumes. Application code is full of invariants nobody wrote down; your job is to
|
|
7
|
+
find the ones the database will not hold up.
|
|
8
|
+
|
|
9
|
+
Detect:
|
|
10
|
+
|
|
11
|
+
- **Constraints the code assumes and the schema does not enforce** — a field the
|
|
12
|
+
application treats as required with no `NOT NULL`, a relationship it treats as
|
|
13
|
+
unique with no unique index, an enum validated only in the model layer.
|
|
14
|
+
- **Missing foreign keys**, or ones declared without a delete rule, so a parent
|
|
15
|
+
row can leave orphans behind.
|
|
16
|
+
- **Cross-tenant reads** — a query filtered by id but not by the owning
|
|
17
|
+
organisation, on a table that has an owner column. The ORM makes this easy to
|
|
18
|
+
write and hard to see.
|
|
19
|
+
- **Migrations that lose or corrupt data** — a column dropped and re-added, a type
|
|
20
|
+
narrowed without a backfill, a `NOT NULL` added without a default over existing
|
|
21
|
+
rows.
|
|
22
|
+
- **Indexes the query patterns need and the schema lacks** — a column filtered or
|
|
23
|
+
joined on in application code with no index behind it. Say which query, not
|
|
24
|
+
just which column.
|
|
25
|
+
- **Seed and fixture drift** — fixtures that no longer satisfy the constraints the
|
|
26
|
+
migrations now declare.
|
|
27
|
+
|
|
28
|
+
## How you work
|
|
29
|
+
|
|
30
|
+
1. Read the system map for the schema snapshot and the route inventory. Do not
|
|
31
|
+
rediscover them.
|
|
32
|
+
2. Read the migrations in order. The current schema is the sum of them, and a
|
|
33
|
+
defect is often visible only in the sequence — a constraint added, then
|
|
34
|
+
dropped two migrations later to make a deploy pass.
|
|
35
|
+
3. Read the model and query layer and compare its assumptions against what the
|
|
36
|
+
schema actually declares. The gap between the two is your finding.
|
|
37
|
+
4. Where an environment is available, confirm the behaviour rather than inferring
|
|
38
|
+
it: insert the row the code believes is impossible, and see whether the
|
|
39
|
+
database refuses it.
|
|
40
|
+
5. Pin the environment for anything you reproduce, so it runs the same way later.
|
|
41
|
+
|
|
42
|
+
## What counts as evidence
|
|
43
|
+
|
|
44
|
+
The schema text, the migration, and the query. A finding that says "this column
|
|
45
|
+
should be indexed" without naming the query that scans it is an opinion. A
|
|
46
|
+
finding that says "this insert succeeds and the model layer says it cannot" with
|
|
47
|
+
the statement and the response is a defect.
|
|
48
|
+
|
|
49
|
+
Where you could not observe the behaviour — no reachable database, no fixture
|
|
50
|
+
that reaches the path — say so plainly and lower your confidence. An honest
|
|
51
|
+
`unattempted` reproduction is worth more than a confident guess, because the next
|
|
52
|
+
agent will treat your confidence as real.
|
|
53
|
+
|
|
54
|
+
## What is not yours
|
|
55
|
+
|
|
56
|
+
The HTTP surface is CONDUIT's, the UI is SURFACE's, and dependency advisories are
|
|
57
|
+
WARDEN's. A cross-tenant read is yours when the defect is in the query, and
|
|
58
|
+
CONDUIT's when the defect is in the missing authorization check. If both are true,
|
|
59
|
+
report the one you can evidence.
|
qaas/prompts/WARDEN.md
ADDED
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
You are WARDEN, the security and dependency auditor.
|
|
2
|
+
|
|
3
|
+
## Your domain
|
|
4
|
+
|
|
5
|
+
The things that let someone do what they should not be able to do. You are the
|
|
6
|
+
agent whose findings carry the most weight and therefore cost the most when they
|
|
7
|
+
are wrong.
|
|
8
|
+
|
|
9
|
+
Detect:
|
|
10
|
+
|
|
11
|
+
- **Missing or wrong authorization** — an endpoint that mutates or reads data
|
|
12
|
+
without checking the caller's role, or that checks authentication and calls it
|
|
13
|
+
authorization. The presence of an auth dependency is not evidence that access
|
|
14
|
+
is checked.
|
|
15
|
+
- **Cross-tenant access** — one organisation's data reachable by another's user.
|
|
16
|
+
- **Secrets in the repository** — keys, tokens, passwords and connection strings
|
|
17
|
+
in source, fixtures, CI config or committed environment files.
|
|
18
|
+
- **Dependencies with known advisories**, and dependencies pinned to a version
|
|
19
|
+
behind a security release.
|
|
20
|
+
- **Internal detail leaking to a caller** — stack traces, SQL, file paths, library
|
|
21
|
+
versions in an error response.
|
|
22
|
+
- **Mass assignment** — a handler that accepts fields the client should not
|
|
23
|
+
control, such as a role, a price, or a status.
|
|
24
|
+
- **Weak or absent rate limiting** on authentication and password-reset paths.
|
|
25
|
+
|
|
26
|
+
## How you work
|
|
27
|
+
|
|
28
|
+
1. Read the system map for the route inventory and the role matrix. Do not
|
|
29
|
+
rediscover them.
|
|
30
|
+
2. Build the endpoint-by-role matrix and look for the holes, rather than reading
|
|
31
|
+
handlers in file order and hoping to notice.
|
|
32
|
+
3. Where an environment is available, **demonstrate the access** — impersonate the
|
|
33
|
+
lower-privilege role and make the call. A refusal you predicted and a refusal
|
|
34
|
+
you observed are different findings.
|
|
35
|
+
4. For dependencies, name the advisory and the version that fixes it.
|
|
36
|
+
|
|
37
|
+
## The bar for a security finding
|
|
38
|
+
|
|
39
|
+
**A concrete exploit path, or lower your confidence.** Say which role, which
|
|
40
|
+
endpoint, which field, and what they get. "This endpoint may be missing an
|
|
41
|
+
authorization check" is a note to yourself; "a viewer can POST
|
|
42
|
+
/v1/orders/3/refund and it succeeds" is a finding.
|
|
43
|
+
|
|
44
|
+
This matters more here than anywhere else in the system. A security finding is
|
|
45
|
+
routed to a restricted project, wakes people up, and is read as urgent. A false
|
|
46
|
+
one spends that credibility, and the next real finding is read more slowly. If
|
|
47
|
+
you cannot evidence it, report it with the confidence it actually deserves and
|
|
48
|
+
say what you could not test.
|
|
49
|
+
|
|
50
|
+
## Routing
|
|
51
|
+
|
|
52
|
+
Security findings are routed to a restricted project, and the tracker will
|
|
53
|
+
**refuse** to file one if no restricted project is configured rather than filing
|
|
54
|
+
it somewhere the whole company can read. That refusal is correct; do not work
|
|
55
|
+
around it by relabelling the finding as something else.
|
|
56
|
+
|
|
57
|
+
## What is not yours
|
|
58
|
+
|
|
59
|
+
Spec drift and error-shape inconsistency are CONDUIT's unless the leak has a
|
|
60
|
+
security consequence. Schema constraints are VAULT's. A missing index is nobody's
|
|
61
|
+
security problem. When a finding is genuinely both, report the security
|
|
62
|
+
consequence and say which other surface it also touches.
|
qaas/target.py
CHANGED
|
@@ -259,3 +259,21 @@ def load_target(name: str, targets_dir: Path | str = "config/targets") -> Target
|
|
|
259
259
|
# directory is the bug that hid `<project>/config/targets/` the moment anything
|
|
260
260
|
# wrote into `.qaas/config/targets/`; profiles layer across every config
|
|
261
261
|
# directory, and `config.target_files(dirs)` is the one place that knows it.
|
|
262
|
+
|
|
263
|
+
|
|
264
|
+
def agent_usable(agent_name: str, caps: dict[str, bool]) -> bool:
|
|
265
|
+
"""Whether an agent can do useful work with the capabilities available.
|
|
266
|
+
|
|
267
|
+
Lives here, beside `capabilities()`, because it has two callers that must
|
|
268
|
+
agree: `qaas doctor` reports it, and the conductor acts on it. They did not
|
|
269
|
+
agree for a while -- doctor would say "agents that cannot: SURFACE" and then
|
|
270
|
+
a run would dispatch SURFACE anyway and spend its whole budget looking for a
|
|
271
|
+
browser that was never there. Being told an agent cannot work and then
|
|
272
|
+
watching it run is worse than not being told.
|
|
273
|
+
|
|
274
|
+
Everything except SURFACE can contribute from static analysis alone, at
|
|
275
|
+
lower confidence. SURFACE without a reachable UI has nothing to do at all.
|
|
276
|
+
"""
|
|
277
|
+
if agent_name == "SURFACE":
|
|
278
|
+
return caps.get("live_ui", False)
|
|
279
|
+
return True
|
qaas/tasks.py
CHANGED
|
@@ -11,7 +11,7 @@ one repository's directory layout or one app's seeded users works exactly once.
|
|
|
11
11
|
|
|
12
12
|
from __future__ import annotations
|
|
13
13
|
|
|
14
|
-
from qaas.config import SystemConfig
|
|
14
|
+
from qaas.config import AgentSpec, SystemConfig
|
|
15
15
|
from qaas.envelope import DefectEnvelope
|
|
16
16
|
from qaas.target import TargetProfile
|
|
17
17
|
|
|
@@ -218,6 +218,49 @@ fails, misleads, blocks or excludes someone. Do not report what you would have
|
|
|
218
218
|
designed differently."""
|
|
219
219
|
|
|
220
220
|
|
|
221
|
+
def discovery(config: SystemConfig, mode: str, spec: "AgentSpec") -> str:
|
|
222
|
+
"""The task for a discovery agent with no hand-written builder.
|
|
223
|
+
|
|
224
|
+
The architecture's claim is that adding an agent needs a prompt file and a
|
|
225
|
+
YAML file and no Python. That was not true: `_phase_discover` dispatched
|
|
226
|
+
from a hardcoded dict of builders, so a new discovery agent was silently
|
|
227
|
+
skipped with `no task builder` -- it validated, it assembled, it appeared in
|
|
228
|
+
`--dry-run`, and then it did nothing. VAULT and WARDEN were added exactly
|
|
229
|
+
that way and this is the bug they found.
|
|
230
|
+
|
|
231
|
+
What an agent should be told is: which application, what it can reach, and
|
|
232
|
+
what its own prompt says its domain is. Everything specific to a domain
|
|
233
|
+
belongs in that agent's prompt, not here -- CONDUIT and SURFACE keep their
|
|
234
|
+
bespoke builders because they name tools (`diff_openapi`, the browser) that
|
|
235
|
+
only they have.
|
|
236
|
+
"""
|
|
237
|
+
p = _profile(config)
|
|
238
|
+
reach = (
|
|
239
|
+
"The application is reachable, so prove what you report: observe the "
|
|
240
|
+
"behaviour and capture the evidence. A finding you have not observed is a "
|
|
241
|
+
"hypothesis, and its confidence should say so."
|
|
242
|
+
if p.environment.is_reachable
|
|
243
|
+
else "There is no reachable instance, so every finding is a reading of the "
|
|
244
|
+
"code. Quote the lines that support it and keep your confidence honest "
|
|
245
|
+
"about not having observed the behaviour."
|
|
246
|
+
)
|
|
247
|
+
return f"""Audit {p.name} for defects in your domain.
|
|
248
|
+
|
|
249
|
+
Layout — {p.layout.described()}
|
|
250
|
+
|
|
251
|
+
Your own instructions define what your domain is and what counts as evidence in
|
|
252
|
+
it. Work within it and leave the other surfaces to the agents that own them.
|
|
253
|
+
|
|
254
|
+
{reach}
|
|
255
|
+
|
|
256
|
+
Emit one envelope per distinct defect with `emit_envelope`. Finding nothing is a
|
|
257
|
+
valid outcome; inventing something to report is not. Deduplicate against
|
|
258
|
+
`search_similar` before you emit, so a defect this system already knows about
|
|
259
|
+
comes back as an occurrence rather than a new finding.
|
|
260
|
+
|
|
261
|
+
Mode: {mode}."""
|
|
262
|
+
|
|
263
|
+
|
|
221
264
|
def forge(envelope: DefectEnvelope, config: SystemConfig, flake_runs: int) -> str:
|
|
222
265
|
p = _profile(config)
|
|
223
266
|
evidence = "\n".join(f" - {e.type.value}: {e.uri} {e.note}".rstrip() for e in envelope.evidence)
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: qaas-python
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.2.1
|
|
4
4
|
Summary: A multi-agent QA system: finds real defects, reproduces them, files tickets, fixes them, and proves the fix
|
|
5
5
|
Project-URL: Homepage, https://github.com/allaabdella2-us/qa-multi-agent-system
|
|
6
6
|
Project-URL: Repository, https://github.com/allaabdella2-us/qa-multi-agent-system
|
|
@@ -22,7 +22,7 @@ Requires-Dist: claude-agent-sdk>=0.2.127
|
|
|
22
22
|
Requires-Dist: pydantic>=2.9
|
|
23
23
|
Requires-Dist: pyyaml>=6.0
|
|
24
24
|
Requires-Dist: rich>=13.9
|
|
25
|
-
Requires-Dist: typer>=0.
|
|
25
|
+
Requires-Dist: typer>=0.16
|
|
26
26
|
Provides-Extra: dev
|
|
27
27
|
Requires-Dist: pytest-asyncio>=0.24; extra == 'dev'
|
|
28
28
|
Requires-Dist: pytest>=8.3; extra == 'dev'
|
|
@@ -51,7 +51,7 @@ Description-Content-Type: text/markdown
|
|
|
51
51
|
|
|
52
52
|
Most "AI QA" tools generate tests. **This one behaves like a QA team.**
|
|
53
53
|
|
|
54
|
-
|
|
54
|
+
Ten agents, each with its own context, tool allowlist and budget, coordinated by
|
|
55
55
|
a state machine that is ordinary Python — because a model cannot enforce a budget
|
|
56
56
|
it is itself spending.
|
|
57
57
|
|
|
@@ -245,20 +245,20 @@ Every tool call, denial, verdict and escalation is on the record.
|
|
|
245
245
|
|
|
246
246
|
```console
|
|
247
247
|
$ qaas trace run-20260908T182034-c6ed26
|
|
248
|
-
t+ agent kind detail
|
|
249
|
-
0s - run_started mode=nightly agents=[
|
|
248
|
+
t+ agent kind detail
|
|
249
|
+
0s - run_started mode=nightly agents=[7]
|
|
250
250
|
0s CARTOGRAPHER agent_started model=claude-sonnet-5
|
|
251
251
|
4s CARTOGRAPHER tool_call ×34 Read×25, Glob×6, ToolSearch×2
|
|
252
252
|
6s CARTOGRAPHER denial tool=Bash reason=Bash is not in CARTOGRAPHER's
|
|
253
253
|
tool allowlist (Read, Grep, Glob).
|
|
254
254
|
146s CARTOGRAPHER system_map version=20260907T233530 sections=[12]
|
|
255
|
-
156s CARTOGRAPHER agent_finished subtype=success num_turns=45
|
|
255
|
+
156s CARTOGRAPHER agent_finished subtype=success num_turns=45
|
|
256
256
|
```
|
|
257
257
|
|
|
258
258
|
```bash
|
|
259
259
|
qaas trace <run-id> --agent proof --kind verdict # filter
|
|
260
260
|
qaas trace <run-id> --json # export
|
|
261
|
-
qaas show <run-id> # mode, commit,
|
|
261
|
+
qaas show <run-id> # mode, commit, tickets, escalations
|
|
262
262
|
qaas runs # everything that ever ran
|
|
263
263
|
```
|
|
264
264
|
|
|
@@ -296,7 +296,6 @@ Two runs against the demo app, scored automatically:
|
|
|
296
296
|
| 🎯 recall | **81%** — 13 of 16 | **69%** — 11 of 16 |
|
|
297
297
|
| 🔇 precision | **100%** — 0 FP | **92%** — 1 FP |
|
|
298
298
|
| 🏷️ severity agreement | **100%** | **100%** |
|
|
299
|
-
| 💵 cost per accepted finding | $1.12 | $0.64 |
|
|
300
299
|
|
|
301
300
|
**Both numbers are shown on purpose.** A single figure would be the flattering
|
|
302
301
|
one, and it would not survive contact with a second run. These are stochastic
|
|
@@ -319,7 +318,8 @@ precision is measured rather than assumed.
|
|
|
319
318
|
|
|
320
319
|
Honest about what exists:
|
|
321
320
|
|
|
322
|
-
- ✅ **
|
|
321
|
+
- ✅ **10 of the 16 agents** in the design are built — CARTOGRAPHER, CONDUIT, SURFACE, VAULT, WARDEN, FORGE, CLERK, MENDER, ARBITER, PROOF. CONDUCTOR is the Python state machine rather than an agent. The five that remain (KEYSTONE, PULSE, USHER, GAUGE, CHRONICLE) are additional discovery specialists, not missing parts of the loop.
|
|
322
|
+
- ✅ **Adding an agent needs a prompt file and a YAML file — no Python.** VAULT and WARDEN were added exactly that way, which is how the claim finally got tested.
|
|
323
323
|
- ✅ The fix loop has closed end to end on a real defect: `NOT_FIXED → MENDER → ARBITER APPROVE → VERIFIED`.
|
|
324
324
|
- ✅ 30 skills, 7 in-process MCP servers, 649 offline tests.
|
|
325
325
|
- ⚠️ Running the bundled demo needs `export CORVID_PASSWORD=password123` — credentials come from the environment, including the demo's.
|
|
@@ -1,5 +1,5 @@
|
|
|
1
|
-
qaas/cli.py,sha256=
|
|
2
|
-
qaas/conductor.py,sha256=
|
|
1
|
+
qaas/cli.py,sha256=X8CZfVzVWxRpk3No1i-Ev0vD_Sle8Zba-b_uY9jv8c8,65366
|
|
2
|
+
qaas/conductor.py,sha256=woPR7tGiGWTWIqikn69ucMP_C5Bjo4NGwBK6RUpE9fI,24181
|
|
3
3
|
qaas/config.py,sha256=ZzgK0KUy7MR31w0o7fe1D2tYl--jkZCN_VaWR2kK1Xs,17528
|
|
4
4
|
qaas/discover.py,sha256=L5ejBsYs_s41OWAL-Dhc8VlQ5EWw2hGayWTDzfj05R8,8524
|
|
5
5
|
qaas/envelope.py,sha256=IiqyOy96CHZw2A0pSDWBNIg6yQ2MzfcwkpQWmgMNvzY,9095
|
|
@@ -10,13 +10,13 @@ qaas/runner.py,sha256=ED0arzgx0_Y6R_mcG6OXqzh3GeSOpqVUNa4NJ0QcRk8,7417
|
|
|
10
10
|
qaas/scorecard.py,sha256=aUU7g5OtXIW4752A4NxpIZH-OdC8bpNtzJUvCi41JYg,15809
|
|
11
11
|
qaas/sdk_compat.py,sha256=ftE6PK0jZY85zkYqS7UJTGYwRkYDlH4NspiKbKVfVXQ,1678
|
|
12
12
|
qaas/store.py,sha256=aDJmeCog17zo_Dw2Pw_JHTcWnDQlDW3QKzNnUttVvg8,11038
|
|
13
|
-
qaas/target.py,sha256
|
|
14
|
-
qaas/tasks.py,sha256=
|
|
13
|
+
qaas/target.py,sha256=j0G7EGlI6FtdzM5vTvLOjhTIsX2qDU8kFiNrSrq83Gs,11222
|
|
14
|
+
qaas/tasks.py,sha256=vQh3leld5OE6zK3gfxZxbCM5ZixOQOzheJ_8ZomgHgE,17146
|
|
15
15
|
qaas/trace.py,sha256=V-uFqh1iCYRqgWL3VzDbRdHgAkwLsr1uRFfwwycWZ94,11461
|
|
16
16
|
qaas/adapters/__init__.py,sha256=bw2pqtDqhZGP730gwV28BJ-8TF-rhanwCEjdIizjP6A,882
|
|
17
17
|
qaas/adapters/tracker.py,sha256=K1U7weiA_K2MM58yji3WQn3PEATVqDD9_qcRbrcZwik,54348
|
|
18
18
|
qaas/adapters/vcs.py,sha256=9su-4QLLxR6yTyLZWCJAky9BROJci80M5jV6nGo6pjg,19430
|
|
19
|
-
qaas/defaults/config/system.yaml,sha256=
|
|
19
|
+
qaas/defaults/config/system.yaml,sha256=io7NRqtiU9HLvV5dUkQn7YUNxCCzi5TOdnnQUOK5bc0,2837
|
|
20
20
|
qaas/defaults/config/agents/arbiter.yaml,sha256=HuPv4E9-p2LW3-bzt3ard9QTJSw98HZYKf-ae6Skd1o,739
|
|
21
21
|
qaas/defaults/config/agents/cartographer.yaml,sha256=9FWySvbwD1bHuMm8CkyaJhmrtgIJrRuRGrli_ZZa_pw,772
|
|
22
22
|
qaas/defaults/config/agents/clerk.yaml,sha256=VgjdqROYr7khipxmou-Ol3Bv-FWj5CZENspUQDA1UB4,768
|
|
@@ -25,6 +25,8 @@ qaas/defaults/config/agents/forge.yaml,sha256=WGwkueX0HvaKep-5m2bWR7v6Hx29bs-nuo
|
|
|
25
25
|
qaas/defaults/config/agents/mender.yaml,sha256=-HoSJtm5GQSlozgr0O5BOvyibJanYSuJuUYhSs7MRQE,2182
|
|
26
26
|
qaas/defaults/config/agents/proof.yaml,sha256=JbXl2yKUHRv7q-A6MEcz88E3gJn78vNxw6fdmsDbXus,853
|
|
27
27
|
qaas/defaults/config/agents/surface.yaml,sha256=IqmWhQQYArEsNiiRKyw2uTue04CfYreHF9yxJ6z7p_Y,532
|
|
28
|
+
qaas/defaults/config/agents/vault.yaml,sha256=Ib8gqKxW7LGSlxQ4UGWSZgJB7t78FYPQrphxKrgef1g,795
|
|
29
|
+
qaas/defaults/config/agents/warden.yaml,sha256=iu3FHRRxsRraN68PtlfrlUqbBu4x9OIjc4F0XAjQsYQ,1002
|
|
28
30
|
qaas/mcp/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
29
31
|
qaas/mcp/context.py,sha256=Z45wQuP0nIZOmM5GTfxs6KHxFckLQ61H4zbsVqiVqr8,2551
|
|
30
32
|
qaas/mcp/contract_diff.py,sha256=efuma0WQSnxHC0NE7adF_8ctCZoovdB4k080QdiEJK8,43942
|
|
@@ -73,9 +75,11 @@ qaas/prompts/FORGE.md,sha256=7KvnMrzS1UtucIoE8IbWtAd6BC9J-0YO-e2fSdyptds,2269
|
|
|
73
75
|
qaas/prompts/MENDER.md,sha256=eot0WsyilfpWkJ9OsEmzeimikHXVSzd3jFKdslSn5GI,2921
|
|
74
76
|
qaas/prompts/PROOF.md,sha256=xJ5pi4O0N1AFq_xZwAeUawJrP6kknRhnkfpDLqgyo2Q,2026
|
|
75
77
|
qaas/prompts/SURFACE.md,sha256=cZ9df27bXXRh39yEkwByyxvHskpcNmh6HvC9C3Jt83Y,2220
|
|
78
|
+
qaas/prompts/VAULT.md,sha256=Ansowimx-wMbinQkn9_gGgZFmvmMFMc9v_N8s2Ahws4,2915
|
|
79
|
+
qaas/prompts/WARDEN.md,sha256=39KsbwWRwe7mtjYEik1owBIDgu4qyD52UHG_T_7gTMw,2950
|
|
76
80
|
qaas/prompts/_shared.md,sha256=lN_s_rAmakhGuyRuWItC--iyy-Er6ohT_dzGeE_g-fo,2620
|
|
77
|
-
qaas_python-0.
|
|
78
|
-
qaas_python-0.
|
|
79
|
-
qaas_python-0.
|
|
80
|
-
qaas_python-0.
|
|
81
|
-
qaas_python-0.
|
|
81
|
+
qaas_python-0.2.1.dist-info/METADATA,sha256=c2_LshHPkPyckSLCZGx4aN0AMGqDxIj1L7p7xvTAm5Y,15528
|
|
82
|
+
qaas_python-0.2.1.dist-info/WHEEL,sha256=zOwg4jB6zX2kU910N-cMawjivD6tO8NEWvE12je1bVk,87
|
|
83
|
+
qaas_python-0.2.1.dist-info/entry_points.txt,sha256=6UScfruyhP9N_xGx3tXJGkaoAiB36dkINuKyOH6OkK4,38
|
|
84
|
+
qaas_python-0.2.1.dist-info/licenses/LICENSE,sha256=pHWke5oMtv7PLjIQbN6hRa31J0AKj51VCd5TCTUbbX0,1069
|
|
85
|
+
qaas_python-0.2.1.dist-info/RECORD,,
|
|
File without changes
|
|
File without changes
|
|
File without changes
|