qaas-python 0.0.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- qaas/adapters/__init__.py +19 -0
- qaas/adapters/tracker.py +1783 -0
- qaas/adapters/vcs.py +555 -0
- qaas/cli.py +1757 -0
- qaas/config.py +409 -0
- qaas/defaults/config/agents/api.yaml +18 -0
- qaas/defaults/config/agents/architect.yaml +21 -0
- qaas/defaults/config/agents/auditor.yaml +19 -0
- qaas/defaults/config/agents/browser.yaml +15 -0
- qaas/defaults/config/agents/dba.yaml +20 -0
- qaas/defaults/config/agents/fixer.yaml +55 -0
- qaas/defaults/config/agents/guide.yaml +23 -0
- qaas/defaults/config/agents/load.yaml +26 -0
- qaas/defaults/config/agents/mapper.yaml +19 -0
- qaas/defaults/config/agents/reporter.yaml +19 -0
- qaas/defaults/config/agents/reproducer.yaml +21 -0
- qaas/defaults/config/agents/reviewer.yaml +18 -0
- qaas/defaults/config/agents/socket.yaml +23 -0
- qaas/defaults/config/agents/triage.yaml +20 -0
- qaas/defaults/config/agents/verifier.yaml +20 -0
- qaas/defaults/config/system.yaml +64 -0
- qaas/discover.py +242 -0
- qaas/envelope.py +318 -0
- qaas/envfile.py +100 -0
- qaas/guardrails.py +589 -0
- qaas/mcp/__init__.py +0 -0
- qaas/mcp/context.py +78 -0
- qaas/mcp/contract_diff.py +1011 -0
- qaas/mcp/defect_memory.py +495 -0
- qaas/mcp/env_control.py +925 -0
- qaas/mcp/envelope_server.py +463 -0
- qaas/mcp/test_runner.py +842 -0
- qaas/mcp/tracker.py +420 -0
- qaas/mcp/vcs.py +501 -0
- qaas/paths.py +317 -0
- qaas/plugin/.claude-plugin/plugin.json +9 -0
- qaas/plugin/skills/a11y-audit/SKILL.md +34 -0
- qaas/plugin/skills/adversarial-review/SKILL.md +120 -0
- qaas/plugin/skills/api-surface-extraction/SKILL.md +38 -0
- qaas/plugin/skills/authz-matrix-check/SKILL.md +46 -0
- qaas/plugin/skills/console-error-triage/SKILL.md +39 -0
- qaas/plugin/skills/contract-test-generation/SKILL.md +36 -0
- qaas/plugin/skills/dedupe-strategy/SKILL.md +39 -0
- qaas/plugin/skills/environment-pinning/SKILL.md +35 -0
- qaas/plugin/skills/error-taxonomy/SKILL.md +42 -0
- qaas/plugin/skills/exploratory-ui-walk/SKILL.md +46 -0
- qaas/plugin/skills/failing-test-authoring/SKILL.md +47 -0
- qaas/plugin/skills/flake-detection/SKILL.md +39 -0
- qaas/plugin/skills/form-state-probe/SKILL.md +36 -0
- qaas/plugin/skills/minimal-diff-discipline/SKILL.md +70 -0
- qaas/plugin/skills/openapi-diff/SKILL.md +45 -0
- qaas/plugin/skills/ownership-resolution/SKILL.md +31 -0
- qaas/plugin/skills/product-task-graph/SKILL.md +35 -0
- qaas/plugin/skills/regression-risk-scoring/SKILL.md +59 -0
- qaas/plugin/skills/regression-suite-selection/SKILL.md +36 -0
- qaas/plugin/skills/repo-cartography/SKILL.md +38 -0
- qaas/plugin/skills/repro-minimisation/SKILL.md +41 -0
- qaas/plugin/skills/rollback-plan-authoring/SKILL.md +81 -0
- qaas/plugin/skills/root-cause-vs-symptom/SKILL.md +67 -0
- qaas/plugin/skills/routing-rules/SKILL.md +34 -0
- qaas/plugin/skills/severity-rubric/SKILL.md +42 -0
- qaas/plugin/skills/test-first-fix/SKILL.md +66 -0
- qaas/plugin/skills/test-quality-audit/SKILL.md +58 -0
- qaas/plugin/skills/ticket-writer/SKILL.md +40 -0
- qaas/plugin/skills/verdict-reporting/SKILL.md +35 -0
- qaas/plugin/skills/verification-protocol/SKILL.md +39 -0
- qaas/prompts/API.md +44 -0
- qaas/prompts/ARCHITECT.md +80 -0
- qaas/prompts/AUDITOR.md +62 -0
- qaas/prompts/BROWSER.md +46 -0
- qaas/prompts/DBA.md +59 -0
- qaas/prompts/FIXER.md +55 -0
- qaas/prompts/GUIDE.md +94 -0
- qaas/prompts/LOAD.md +109 -0
- qaas/prompts/MAPPER.md +46 -0
- qaas/prompts/REPORTER.md +61 -0
- qaas/prompts/REPRODUCER.md +43 -0
- qaas/prompts/REVIEWER.md +53 -0
- qaas/prompts/SOCKET.md +100 -0
- qaas/prompts/TRIAGE.md +45 -0
- qaas/prompts/VERIFIER.md +41 -0
- qaas/prompts/_shared.md +45 -0
- qaas/registry.py +496 -0
- qaas/router.py +581 -0
- qaas/runner.py +210 -0
- qaas/scorecard.py +448 -0
- qaas/sdk_compat.py +52 -0
- qaas/store.py +323 -0
- qaas/target.py +287 -0
- qaas/tasks.py +438 -0
- qaas/trace.py +342 -0
- qaas_python-0.0.1.dist-info/METADATA +429 -0
- qaas_python-0.0.1.dist-info/RECORD +96 -0
- qaas_python-0.0.1.dist-info/WHEEL +4 -0
- qaas_python-0.0.1.dist-info/entry_points.txt +2 -0
- qaas_python-0.0.1.dist-info/licenses/LICENSE +21 -0
qaas/config.py
ADDED
|
@@ -0,0 +1,409 @@
|
|
|
1
|
+
"""Configuration: agents are data, not code.
|
|
2
|
+
|
|
3
|
+
An agent is a prompt file plus an entry in `config/agents/`. Adding one of the
|
|
4
|
+
remaining agents from the roster should never require touching the router,
|
|
5
|
+
the runner, or the guardrails — that is the property this module exists to keep.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
from typing import Any, Literal, Sequence
|
|
12
|
+
|
|
13
|
+
import os
|
|
14
|
+
|
|
15
|
+
import yaml
|
|
16
|
+
from pydantic import BaseModel, ConfigDict, Field, model_validator
|
|
17
|
+
|
|
18
|
+
from qaas.target import TargetProfile, load_target
|
|
19
|
+
|
|
20
|
+
# §5.3 is explicit that no agent gets more than six MCP servers, because tool
|
|
21
|
+
# selection accuracy falls off past roughly 5-7. Enforced, not just documented.
|
|
22
|
+
MAX_MCP_SERVERS_PER_AGENT = 6
|
|
23
|
+
|
|
24
|
+
Layer = Literal["control", "discovery", "triage", "remediation", "reporting"]
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
class Policy(BaseModel):
|
|
28
|
+
"""One agent's slice of the §8.1 write-permission matrix.
|
|
29
|
+
|
|
30
|
+
Default is read-only. Anything an agent may write, it says so here, and
|
|
31
|
+
guardrails.py enforces it against the actual tool call arguments.
|
|
32
|
+
"""
|
|
33
|
+
|
|
34
|
+
model_config = ConfigDict(extra="forbid")
|
|
35
|
+
|
|
36
|
+
write_paths: list[str] = Field(default_factory=list)
|
|
37
|
+
branch_patterns: list[str] = Field(default_factory=list)
|
|
38
|
+
may_open_pr: bool = False
|
|
39
|
+
may_create_tickets: bool = False
|
|
40
|
+
may_transition_tickets: bool = False
|
|
41
|
+
max_tickets_per_run: int = 0
|
|
42
|
+
max_diff_files: int | None = None
|
|
43
|
+
max_diff_lines: int | None = None
|
|
44
|
+
protected_paths: list[str] = Field(default_factory=list)
|
|
45
|
+
|
|
46
|
+
#: Path globs this agent may never modify, whatever else its policy allows.
|
|
47
|
+
#: §8.2 names the classes: migrations, auth, payment paths and infra config.
|
|
48
|
+
#: These are the changes whose blast radius a review cannot reliably bound,
|
|
49
|
+
#: so they stop at a human even when everything else in the envelope holds.
|
|
50
|
+
forbidden_paths: list[str] = Field(default_factory=list)
|
|
51
|
+
|
|
52
|
+
@property
|
|
53
|
+
def read_only(self) -> bool:
|
|
54
|
+
return not (
|
|
55
|
+
self.write_paths
|
|
56
|
+
or self.branch_patterns
|
|
57
|
+
or self.may_open_pr
|
|
58
|
+
or self.may_create_tickets
|
|
59
|
+
or self.may_transition_tickets
|
|
60
|
+
)
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
class AgentSpec(BaseModel):
|
|
64
|
+
"""Everything needed to build one agent's ClaudeAgentOptions."""
|
|
65
|
+
|
|
66
|
+
model_config = ConfigDict(extra="forbid")
|
|
67
|
+
|
|
68
|
+
name: str
|
|
69
|
+
layer: Layer
|
|
70
|
+
role: str
|
|
71
|
+
prompt: str # path relative to src/qaas/prompts/
|
|
72
|
+
enabled: bool = True
|
|
73
|
+
|
|
74
|
+
model: str = "claude-opus-5"
|
|
75
|
+
effort: Literal["low", "medium", "high", "xhigh", "max"] = "high"
|
|
76
|
+
max_turns: int = 40
|
|
77
|
+
max_budget_usd: float | None = None
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
mcp_servers: list[str] = Field(default_factory=list)
|
|
81
|
+
builtin_tools: list[str] = Field(default_factory=list)
|
|
82
|
+
policy: Policy = Field(default_factory=Policy)
|
|
83
|
+
|
|
84
|
+
# Procedure lives in skills, role and standards live in the prompt. A skill
|
|
85
|
+
# named here is preloaded; the agent can still reach others through Skill.
|
|
86
|
+
skills: list[str] = Field(default_factory=list)
|
|
87
|
+
|
|
88
|
+
# Tools this agent must have called before it is allowed to finish. The Stop
|
|
89
|
+
# hook enforces it. Without this an agent can produce a confident summary and
|
|
90
|
+
# no artifact, and the failure only surfaces afterwards in the router —
|
|
91
|
+
# too late for the agent to fix it.
|
|
92
|
+
must_call: list[str] = Field(default_factory=list)
|
|
93
|
+
|
|
94
|
+
@model_validator(mode="after")
|
|
95
|
+
def _tool_budget(self) -> "AgentSpec":
|
|
96
|
+
if len(self.mcp_servers) > MAX_MCP_SERVERS_PER_AGENT:
|
|
97
|
+
raise ValueError(
|
|
98
|
+
f"{self.name} declares {len(self.mcp_servers)} MCP servers; "
|
|
99
|
+
f"the cap is {MAX_MCP_SERVERS_PER_AGENT} (§5.3). "
|
|
100
|
+
"An agent needing more is a signal to split it."
|
|
101
|
+
)
|
|
102
|
+
if len(set(self.mcp_servers)) != len(self.mcp_servers):
|
|
103
|
+
raise ValueError(f"{self.name} lists a duplicate MCP server")
|
|
104
|
+
for tool in self.must_call:
|
|
105
|
+
server = tool.split("__")[1] if tool.startswith("mcp__") else None
|
|
106
|
+
if server and server not in self.mcp_servers:
|
|
107
|
+
raise ValueError(
|
|
108
|
+
f"{self.name} must_call names '{tool}' but is not connected to "
|
|
109
|
+
f"the '{server}' server; it could never satisfy that."
|
|
110
|
+
)
|
|
111
|
+
return self
|
|
112
|
+
|
|
113
|
+
def prompt_path(self, prompts_dir: Path) -> Path:
|
|
114
|
+
return prompts_dir / self.prompt
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
class StdioServerSpec(BaseModel):
|
|
118
|
+
"""A user-declared MCP server run as a subprocess.
|
|
119
|
+
|
|
120
|
+
Pure data: `command` and `args` are passed to the CLI, which spawns it. No
|
|
121
|
+
shell, ever -- `command` is a program and `args` is a list, so a string like
|
|
122
|
+
`"foo && rm -rf /"` is a program name that does not exist rather than two
|
|
123
|
+
commands.
|
|
124
|
+
"""
|
|
125
|
+
|
|
126
|
+
model_config = ConfigDict(extra="forbid")
|
|
127
|
+
|
|
128
|
+
type: Literal["stdio"] = "stdio"
|
|
129
|
+
command: str
|
|
130
|
+
args: list[str] = Field(default_factory=list)
|
|
131
|
+
env: dict[str, str] = Field(default_factory=dict)
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
class UrlServerSpec(BaseModel):
|
|
135
|
+
"""A user-declared MCP server reached over HTTP or SSE."""
|
|
136
|
+
|
|
137
|
+
model_config = ConfigDict(extra="forbid")
|
|
138
|
+
|
|
139
|
+
type: Literal["http", "sse"]
|
|
140
|
+
url: str
|
|
141
|
+
headers: dict[str, str] = Field(default_factory=dict)
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
#: What a user may declare. Deliberately no in-process Python type: that would
|
|
145
|
+
#: mean `importlib.import_module` on a name from a config file, executing
|
|
146
|
+
#: arbitrary module-level code inside the process holding this user's Anthropic
|
|
147
|
+
#: credentials, Jira token and GitHub auth. A subprocess is a subprocess; an
|
|
148
|
+
#: import is a foothold. If someone needs a Python server they can wrap it in a
|
|
149
|
+
#: stdio entry point and it costs them one line.
|
|
150
|
+
McpServerSpec = StdioServerSpec | UrlServerSpec
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
class Thresholds(BaseModel):
|
|
154
|
+
model_config = ConfigDict(extra="forbid")
|
|
155
|
+
|
|
156
|
+
min_confidence_to_file: float = 0.6
|
|
157
|
+
max_findings_per_agent_run: int = 25
|
|
158
|
+
max_tickets_per_run: int = 10
|
|
159
|
+
flake_runs: int = 5
|
|
160
|
+
max_mender_arbiter_round_trips: int = 2
|
|
161
|
+
max_proof_reopens: int = 1
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
class RunMode(BaseModel):
|
|
165
|
+
model_config = ConfigDict(extra="forbid")
|
|
166
|
+
|
|
167
|
+
trigger: str
|
|
168
|
+
agents: list[str]
|
|
169
|
+
max_budget_usd: float | None = None
|
|
170
|
+
|
|
171
|
+
max_wall_clock_s: int = 3600
|
|
172
|
+
max_concurrency: int = 3
|
|
173
|
+
files_tickets: bool = True
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
class SystemConfig(BaseModel):
|
|
177
|
+
model_config = ConfigDict(extra="forbid")
|
|
178
|
+
|
|
179
|
+
project: str = "qaas"
|
|
180
|
+
|
|
181
|
+
#: Which target profile in config/targets/ this run is pointed at. The
|
|
182
|
+
#: profile is what makes the system portable: without it every prompt and
|
|
183
|
+
#: every environment call is welded to the application it was built beside.
|
|
184
|
+
#: The active target profile, or None when nothing is configured yet.
|
|
185
|
+
#: A fresh `pip install` is legitimately in that state; commands that
|
|
186
|
+
#: need a profile say so rather than crashing during config load.
|
|
187
|
+
target: str | None = None
|
|
188
|
+
|
|
189
|
+
#: Servers this project declares, on top of the built-in ones. Declaring a
|
|
190
|
+
#: server here grants nothing; an agent receives it only by naming it in its
|
|
191
|
+
#: own `mcp_servers:` list.
|
|
192
|
+
mcp_servers: dict[str, McpServerSpec] = Field(default_factory=dict)
|
|
193
|
+
|
|
194
|
+
#: Overridable with QAAS_TRACKER. Keep the committed value `local`.
|
|
195
|
+
tracker: Literal["local", "jira"] = "local"
|
|
196
|
+
vcs: Literal["local", "github"] = "local"
|
|
197
|
+
thresholds: Thresholds = Field(default_factory=Thresholds)
|
|
198
|
+
run_modes: dict[str, RunMode] = Field(default_factory=dict)
|
|
199
|
+
agents: dict[str, AgentSpec] = Field(default_factory=dict)
|
|
200
|
+
profile: TargetProfile | None = Field(default=None, exclude=True)
|
|
201
|
+
|
|
202
|
+
@model_validator(mode="after")
|
|
203
|
+
def _agents_name_real_servers(self) -> "SystemConfig":
|
|
204
|
+
"""Every server an agent names must resolve to something.
|
|
205
|
+
|
|
206
|
+
`AgentSpec` cannot check this -- it has no view of the rest of the
|
|
207
|
+
config -- so an unresolvable name used to surface as `UnknownServer`
|
|
208
|
+
part-way through a paid run. Here it is a load-time error, which is what
|
|
209
|
+
`qaas validate` is for.
|
|
210
|
+
|
|
211
|
+
Imported inside the function: `registry` imports `config`, so a
|
|
212
|
+
module-level import would be a cycle.
|
|
213
|
+
"""
|
|
214
|
+
from qaas.registry import SDK_SERVER_MODULES, STDIO_SERVERS
|
|
215
|
+
|
|
216
|
+
builtin = set(SDK_SERVER_MODULES) | set(STDIO_SERVERS)
|
|
217
|
+
known = builtin | set(self.mcp_servers)
|
|
218
|
+
for name, spec in sorted(self.agents.items()):
|
|
219
|
+
unknown = [s for s in spec.mcp_servers if s not in known]
|
|
220
|
+
if unknown:
|
|
221
|
+
raise ValueError(
|
|
222
|
+
f"{name} names MCP server(s) nothing provides: {', '.join(unknown)}. "
|
|
223
|
+
f"Built in: {', '.join(sorted(builtin))}. "
|
|
224
|
+
f"Declared in system.yaml: {', '.join(sorted(self.mcp_servers)) or 'none'}."
|
|
225
|
+
)
|
|
226
|
+
return self
|
|
227
|
+
|
|
228
|
+
@model_validator(mode="after")
|
|
229
|
+
def _modes_name_real_agents(self) -> "SystemConfig":
|
|
230
|
+
for mode_name, mode in self.run_modes.items():
|
|
231
|
+
unknown = [a for a in mode.agents if a not in self.agents]
|
|
232
|
+
if unknown:
|
|
233
|
+
raise ValueError(
|
|
234
|
+
f"run mode '{mode_name}' names unknown agents: {', '.join(unknown)}"
|
|
235
|
+
)
|
|
236
|
+
return self
|
|
237
|
+
|
|
238
|
+
def enabled_agents(self, mode: str) -> list[AgentSpec]:
|
|
239
|
+
"""Agents for a run mode, skipping any that are switched off."""
|
|
240
|
+
if mode not in self.run_modes:
|
|
241
|
+
raise KeyError(f"unknown run mode '{mode}'; have: {', '.join(sorted(self.run_modes))}")
|
|
242
|
+
return [self.agents[n] for n in self.run_modes[mode].agents if self.agents[n].enabled]
|
|
243
|
+
|
|
244
|
+
def target_root(self, base: Path | None = None) -> Path:
|
|
245
|
+
"""Where the application under test lives.
|
|
246
|
+
|
|
247
|
+
This used to be `Path.cwd() / config.target_app` -- one value serving as
|
|
248
|
+
both "where qaas lives" and "the application under test". That holds
|
|
249
|
+
only while the target is a subdirectory of the qaas checkout, which is
|
|
250
|
+
true of exactly one target: the bundled demo. `qaas run --repo <url>`
|
|
251
|
+
clones into `.qaas/targets/<slug>`, and every write-path allowlist,
|
|
252
|
+
every test cwd and the SDK subprocess cwd are anchored on this value --
|
|
253
|
+
so getting it from the profile is not tidying, it is the security
|
|
254
|
+
boundary being pointed at the right directory.
|
|
255
|
+
|
|
256
|
+
With no profile there is nothing to test; the base (the qaas project, or
|
|
257
|
+
the cwd) is returned so read-only tooling still has somewhere to stand.
|
|
258
|
+
"""
|
|
259
|
+
if self.profile is not None:
|
|
260
|
+
return self.profile.root_path(base)
|
|
261
|
+
from qaas.paths import project_root
|
|
262
|
+
|
|
263
|
+
return base if base is not None else project_root()
|
|
264
|
+
|
|
265
|
+
|
|
266
|
+
#: Environment overrides for the two swappable backends.
|
|
267
|
+
TRACKER_ENV = "QAAS_TRACKER"
|
|
268
|
+
VCS_ENV = "QAAS_VCS"
|
|
269
|
+
#: Which target profile to run against. Useful on its own (`QAAS_TARGET=staging
|
|
270
|
+
#: qaas run`), and it is how this repo's own test suite selects the bundled demo
|
|
271
|
+
#: without putting a demo name in the defaults that ship to everyone else.
|
|
272
|
+
TARGET_ENV = "QAAS_TARGET"
|
|
273
|
+
|
|
274
|
+
|
|
275
|
+
def load_config(
|
|
276
|
+
config_dir: Path | str | None = None,
|
|
277
|
+
*,
|
|
278
|
+
search: Sequence[Path] | None = None,
|
|
279
|
+
target: str | None = None,
|
|
280
|
+
) -> SystemConfig:
|
|
281
|
+
"""Read system.yaml plus every agents/*.yaml, layered across search paths.
|
|
282
|
+
|
|
283
|
+
Passing `config_dir` positionally means "this directory and nothing else",
|
|
284
|
+
which is exactly the old behaviour and what every test does. Passing
|
|
285
|
+
`search` layers several directories: `system.yaml` is taken whole from the
|
|
286
|
+
first that has one, while `agents/*.yaml` and `targets/*.yaml` are unioned
|
|
287
|
+
by filename with earlier directories shadowing later ones -- so a user can
|
|
288
|
+
override one agent without forking all eight and freezing on today's roster.
|
|
289
|
+
|
|
290
|
+
With neither argument, the workspace resolver decides (an explicit
|
|
291
|
+
--config, then the project, then what shipped in the wheel).
|
|
292
|
+
|
|
293
|
+
`target` beats everything -- system.yaml, QAAS_TARGET, the single-profile
|
|
294
|
+
guess. It is what `qaas run --target X` and `qaas run --repo <url>` mean:
|
|
295
|
+
*this* application, whatever is configured. Without it, a stale `target:`
|
|
296
|
+
naming a profile that no longer exists killed the run inside config loading,
|
|
297
|
+
before the override the operator had just typed was ever consulted.
|
|
298
|
+
"""
|
|
299
|
+
if config_dir is not None:
|
|
300
|
+
dirs: list[Path] = [Path(config_dir)]
|
|
301
|
+
elif search is not None:
|
|
302
|
+
dirs = [Path(d) for d in search]
|
|
303
|
+
else:
|
|
304
|
+
from qaas.paths import Workspace
|
|
305
|
+
|
|
306
|
+
dirs = list(Workspace.resolve().config_dirs)
|
|
307
|
+
|
|
308
|
+
system_path = next((d / "system.yaml" for d in dirs if (d / "system.yaml").is_file()), None)
|
|
309
|
+
if system_path is None:
|
|
310
|
+
looked = ", ".join(str(d) for d in dirs) or "(nowhere -- no search path)"
|
|
311
|
+
raise FileNotFoundError(f"no system config at {dirs[0] / 'system.yaml'} (looked in: {looked})")
|
|
312
|
+
|
|
313
|
+
raw: dict[str, Any] = yaml.safe_load(system_path.read_text(encoding="utf-8")) or {}
|
|
314
|
+
|
|
315
|
+
# `target_app:` used to name the application's directory relative to the
|
|
316
|
+
# process cwd. The target profile's `root` says the same thing and says it
|
|
317
|
+
# better, so the field is gone -- but `extra="forbid"` would turn an old
|
|
318
|
+
# system.yaml into a hard load failure, and someone else's committed config
|
|
319
|
+
# is not ours to break. Dropped silently: there is nothing for the reader to
|
|
320
|
+
# do about it, and the profile already carries the answer.
|
|
321
|
+
raw.pop("target_app", None)
|
|
322
|
+
|
|
323
|
+
# Backend overrides from the environment, so pointing a run at a real
|
|
324
|
+
# tracker or reproducer is not a committed file change.
|
|
325
|
+
#
|
|
326
|
+
# `tracker: local` is the committed default and must stay that way. When
|
|
327
|
+
# `jira` was committed instead, 18 tests failed and 14 errored: the agent
|
|
328
|
+
# fixtures build a real JiraTracker, which demands credentials CI does not
|
|
329
|
+
# have. The house rule is that the default `pytest` run is offline and free,
|
|
330
|
+
# and a committed backend switch silently breaks it -- so the switch belongs
|
|
331
|
+
# in the environment of the person who wants it, not in the repo.
|
|
332
|
+
for key, var in (("tracker", TRACKER_ENV), ("vcs", VCS_ENV), ("target", TARGET_ENV)):
|
|
333
|
+
raw_value = (os.environ.get(var) or "").strip()
|
|
334
|
+
if raw_value:
|
|
335
|
+
# Backends are lowercase literals; a target is a profile name.
|
|
336
|
+
raw[key] = raw_value if key == "target" else raw_value.lower()
|
|
337
|
+
|
|
338
|
+
# An explicit argument outranks both the file and the environment.
|
|
339
|
+
if target:
|
|
340
|
+
raw["target"] = target
|
|
341
|
+
|
|
342
|
+
# Agents layer by filename. Walking the search paths in reverse means the
|
|
343
|
+
# highest-precedence directory writes last and therefore wins.
|
|
344
|
+
by_stem: dict[str, Path] = {}
|
|
345
|
+
for d in reversed(dirs):
|
|
346
|
+
for path in sorted((d / "agents").glob("*.yaml")):
|
|
347
|
+
by_stem[path.stem] = path
|
|
348
|
+
|
|
349
|
+
agents: dict[str, Any] = {}
|
|
350
|
+
for path in by_stem.values():
|
|
351
|
+
spec = yaml.safe_load(path.read_text(encoding="utf-8")) or {}
|
|
352
|
+
name = spec.get("name") or path.stem.upper()
|
|
353
|
+
spec["name"] = name
|
|
354
|
+
if name in agents:
|
|
355
|
+
raise ValueError(f"duplicate agent definition for {name} at {path}")
|
|
356
|
+
agents[name] = spec
|
|
357
|
+
|
|
358
|
+
raw["agents"] = agents
|
|
359
|
+
|
|
360
|
+
config = SystemConfig.model_validate(raw)
|
|
361
|
+
|
|
362
|
+
# Resolve the target profile if one is named and findable.
|
|
363
|
+
#
|
|
364
|
+
# A named-but-missing profile stays fatal -- running the wrong application
|
|
365
|
+
# is worse than not running. But `target: null` is a legitimate state now:
|
|
366
|
+
# `pip install qaas-python` gives you a working CLI that is not yet pointed
|
|
367
|
+
# at anything, and the commands that need a profile say "run qaas init"
|
|
368
|
+
# rather than dying inside config loading.
|
|
369
|
+
# Profiles layer by filename, exactly as agents and skills do. This used to
|
|
370
|
+
# take the *first* config layer that had a `targets/` at all -- and the day
|
|
371
|
+
# `qaas run --repo` started writing a generated profile into the writable
|
|
372
|
+
# layer (`.qaas/config/targets/`), that layer became "the" targets directory
|
|
373
|
+
# and every profile in `<project>/config/targets/` vanished: `qaas doctor
|
|
374
|
+
# --target corvid` reported the demo profile did not exist.
|
|
375
|
+
profiles = target_files(dirs)
|
|
376
|
+
chosen = config.target
|
|
377
|
+
|
|
378
|
+
# No target named, but exactly one profile on disk: use it. Choosing between
|
|
379
|
+
# two would be guessing, and running the wrong application is worse than not
|
|
380
|
+
# running -- but with one candidate there is nothing to guess at, and making
|
|
381
|
+
# the user restate it is ceremony. This is also what keeps this repository
|
|
382
|
+
# working: its `system.yaml` ships in the package and names no target,
|
|
383
|
+
# because a demo name has no business in the defaults everyone installs.
|
|
384
|
+
if not chosen and len(profiles) == 1:
|
|
385
|
+
chosen = next(iter(profiles))
|
|
386
|
+
|
|
387
|
+
if chosen and profiles:
|
|
388
|
+
if chosen not in profiles:
|
|
389
|
+
# Named but absent stays fatal: running the wrong application is
|
|
390
|
+
# worse than not running. Listed from the merged view, so the
|
|
391
|
+
# suggestion names every profile the user actually has.
|
|
392
|
+
raise FileNotFoundError(
|
|
393
|
+
f"no target profile '{chosen}'. Available: {', '.join(sorted(profiles))}. "
|
|
394
|
+
"Create one with `qaas init <path-to-repo>`."
|
|
395
|
+
)
|
|
396
|
+
profile = load_target(chosen, profiles[chosen].parent)
|
|
397
|
+
config = config.model_copy(update={"target": chosen, "profile": profile})
|
|
398
|
+
return config
|
|
399
|
+
|
|
400
|
+
|
|
401
|
+
def target_files(dirs: Sequence[Path]) -> dict[str, Path]:
|
|
402
|
+
"""Every target profile visible across the config layers, nearest wins."""
|
|
403
|
+
found: dict[str, Path] = {}
|
|
404
|
+
for d in reversed(list(dirs)):
|
|
405
|
+
base = Path(d) / "targets"
|
|
406
|
+
if base.is_dir():
|
|
407
|
+
for path in sorted(base.glob("*.yaml")):
|
|
408
|
+
found[path.stem] = path
|
|
409
|
+
return found
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
name: API
|
|
2
|
+
layer: discovery
|
|
3
|
+
role: >
|
|
4
|
+
Backend, API and contract analyst. Finds spec drift, breaking changes, missing
|
|
5
|
+
authorization, error-taxonomy inconsistency, unbounded results, validation gaps.
|
|
6
|
+
Ships a failing contract test as evidence, never a bare opinion.
|
|
7
|
+
prompt: API.md
|
|
8
|
+
model: claude-opus-5
|
|
9
|
+
effort: high
|
|
10
|
+
max_turns: 60
|
|
11
|
+
mcp_servers: [envelope, contract_diff, env_control, defect_memory]
|
|
12
|
+
builtin_tools: [Read, Grep, Glob]
|
|
13
|
+
policy: {}
|
|
14
|
+
|
|
15
|
+
skills: [openapi-diff, authz-matrix-check, error-taxonomy, contract-test-generation, severity-rubric]
|
|
16
|
+
|
|
17
|
+
# No must_call: finding nothing is a valid and useful outcome for a discovery
|
|
18
|
+
# agent, and requiring an emission would manufacture findings to satisfy it.
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
name: ARCHITECT
|
|
2
|
+
layer: discovery
|
|
3
|
+
role: >
|
|
4
|
+
Architecture analyst. Finds dependency cycles, layering violations, god modules
|
|
5
|
+
and fan-in outliers, domain logic duplicated across services, drift between the
|
|
6
|
+
architecture documents and the code, wrong service boundaries, and dead or
|
|
7
|
+
orphaned code. Pure static analysis -- it needs no running application, so it
|
|
8
|
+
is the one discovery agent that works against a target with no environment.
|
|
9
|
+
prompt: ARCHITECT.md
|
|
10
|
+
model: claude-opus-5
|
|
11
|
+
effort: high
|
|
12
|
+
max_turns: 60
|
|
13
|
+
mcp_servers: [envelope, defect_memory]
|
|
14
|
+
builtin_tools: [Read, Grep, Glob]
|
|
15
|
+
policy: {} # read-only; it never touches the app it reads
|
|
16
|
+
|
|
17
|
+
skills: [repo-cartography, api-surface-extraction, ownership-resolution, severity-rubric]
|
|
18
|
+
|
|
19
|
+
# No must_call: see DBA. Also, an architecture agent that must emit something
|
|
20
|
+
# will emit taste, and "this could be cleaner" filed as a defect is the fastest
|
|
21
|
+
# way for a team to stop reading structural findings at all.
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
name: AUDITOR
|
|
2
|
+
layer: discovery
|
|
3
|
+
role: >
|
|
4
|
+
Security and dependency auditor. Finds missing authorization, secrets committed
|
|
5
|
+
to the repository, dependencies with known advisories, and error paths that
|
|
6
|
+
leak internals to a caller. Reports a concrete exploit path or lowers its
|
|
7
|
+
confidence -- a security finding without one is a guess wearing a severity.
|
|
8
|
+
prompt: AUDITOR.md
|
|
9
|
+
model: claude-opus-5
|
|
10
|
+
effort: high
|
|
11
|
+
max_turns: 60
|
|
12
|
+
mcp_servers: [envelope, env_control, defect_memory]
|
|
13
|
+
builtin_tools: [Read, Grep, Glob]
|
|
14
|
+
policy: {}
|
|
15
|
+
|
|
16
|
+
skills: [authz-matrix-check, error-taxonomy, severity-rubric, routing-rules, repro-minimisation]
|
|
17
|
+
|
|
18
|
+
# No must_call: see DBA. Also: an auditor that must report something will
|
|
19
|
+
# report something, and security noise is the fastest way to be ignored.
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
name: BROWSER
|
|
2
|
+
layer: discovery
|
|
3
|
+
role: >
|
|
4
|
+
Frontend and UI explorer. Walks the product as a user does: broken flows,
|
|
5
|
+
console errors, accessibility failures, missing loading/empty/error states,
|
|
6
|
+
form validation gaps, state desync.
|
|
7
|
+
prompt: BROWSER.md
|
|
8
|
+
model: claude-opus-5
|
|
9
|
+
effort: high
|
|
10
|
+
max_turns: 80
|
|
11
|
+
mcp_servers: [envelope, env_control, playwright]
|
|
12
|
+
builtin_tools: [Read, Grep, Glob]
|
|
13
|
+
policy: {}
|
|
14
|
+
|
|
15
|
+
skills: [exploratory-ui-walk, a11y-audit, console-error-triage, form-state-probe, product-task-graph, severity-rubric]
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
name: DBA
|
|
2
|
+
layer: discovery
|
|
3
|
+
role: >
|
|
4
|
+
Database and data-integrity analyst. Finds schema constraints the application
|
|
5
|
+
assumes but the database does not enforce, migrations that lose or corrupt
|
|
6
|
+
data, missing indexes on paths the code queries, and cross-tenant reads that
|
|
7
|
+
the ORM makes easy to write. Reports what the schema actually says, never what
|
|
8
|
+
the model layer claims.
|
|
9
|
+
prompt: DBA.md
|
|
10
|
+
model: claude-opus-5
|
|
11
|
+
effort: high
|
|
12
|
+
max_turns: 60
|
|
13
|
+
mcp_servers: [envelope, env_control, defect_memory]
|
|
14
|
+
builtin_tools: [Read, Grep, Glob]
|
|
15
|
+
policy: {}
|
|
16
|
+
|
|
17
|
+
skills: [authz-matrix-check, environment-pinning, severity-rubric, repro-minimisation]
|
|
18
|
+
|
|
19
|
+
# No must_call: finding nothing is a valid outcome for a discovery agent, and
|
|
20
|
+
# requiring an emission would manufacture findings to satisfy it.
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
name: FIXER
|
|
2
|
+
layer: remediation
|
|
3
|
+
role: >
|
|
4
|
+
Remediation engineer. Picks up an agent-ready ticket, writes the minimal change
|
|
5
|
+
that makes the failing test pass, and opens a draft pull request. Deliberately
|
|
6
|
+
one agent with many skills rather than five domain-specific fixers: fixing is
|
|
7
|
+
one activity, and the domain knowledge belongs in loadable skills, not in five
|
|
8
|
+
nearly identical prompts.
|
|
9
|
+
prompt: FIXER.md
|
|
10
|
+
model: claude-opus-5
|
|
11
|
+
effort: high
|
|
12
|
+
max_turns: 80
|
|
13
|
+
mcp_servers: [envelope, test_runner, env_control, vcs, tracker, contract_diff]
|
|
14
|
+
builtin_tools: [Read, Grep, Glob, Write, Edit, Bash]
|
|
15
|
+
skills: [test-first-fix, minimal-diff-discipline, root-cause-vs-symptom, rollback-plan-authoring]
|
|
16
|
+
|
|
17
|
+
policy:
|
|
18
|
+
# Product code, on its own branches. Relative to the *target's* root, not to
|
|
19
|
+
# the qaas project: these used to read `target-app/api/app`, which only ever
|
|
20
|
+
# resolved because the demo happened to live inside this checkout. Still set
|
|
21
|
+
# per target -- a different application keeps its source somewhere else.
|
|
22
|
+
write_paths: [api/app, web/src, qa/repro]
|
|
23
|
+
branch_patterns: ["fix/*"]
|
|
24
|
+
may_open_pr: true
|
|
25
|
+
may_transition_tickets: true
|
|
26
|
+
|
|
27
|
+
# The §8.2 autonomy envelope, enforced in guardrails rather than asked for in
|
|
28
|
+
# the prompt. Start tight; widen from measured REVIEWER approval and VERIFIER
|
|
29
|
+
# verification rates, not from optimism.
|
|
30
|
+
max_diff_files: 5
|
|
31
|
+
max_diff_lines: 150
|
|
32
|
+
|
|
33
|
+
# §8.2: these classes stop at a human however small the change looks. Their
|
|
34
|
+
# blast radius is not something a review can reliably bound.
|
|
35
|
+
#
|
|
36
|
+
# Matched against a path relative to the *target's* root, so the directory
|
|
37
|
+
# patterns are `*x/*` and not `*/x/*`: with a leading slash required,
|
|
38
|
+
# `.github/workflows/ci.yml` -- a repository's CI at its own root, which is
|
|
39
|
+
# where almost every repository keeps it -- matched nothing. That hole was
|
|
40
|
+
# invisible while paths arrived prefixed with `target-app/`.
|
|
41
|
+
forbidden_paths:
|
|
42
|
+
- "*migrations/*"
|
|
43
|
+
- "*migration*"
|
|
44
|
+
- "*auth.py"
|
|
45
|
+
- "*auth*"
|
|
46
|
+
- "*payment*"
|
|
47
|
+
- "*billing*"
|
|
48
|
+
- "*secret*"
|
|
49
|
+
- "*.tf"
|
|
50
|
+
- "*infra/*"
|
|
51
|
+
- "*docker-compose*"
|
|
52
|
+
- "Dockerfile*"
|
|
53
|
+
- "*.github/*"
|
|
54
|
+
|
|
55
|
+
must_call: [mcp__vcs__open_pr]
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
name: GUIDE
|
|
2
|
+
layer: discovery
|
|
3
|
+
role: >
|
|
4
|
+
Product navigation and UX guide. Navigates the live product to answer "how do
|
|
5
|
+
I do X here?", and reports every place that navigation struggled: tasks
|
|
6
|
+
reachable only by typing a URL, dead ends, unlabelled paths, step counts out
|
|
7
|
+
of proportion to the task, and product vocabulary that does not match the
|
|
8
|
+
user's. BROWSER owns whether a feature works; GUIDE owns whether anyone can
|
|
9
|
+
find it.
|
|
10
|
+
prompt: GUIDE.md
|
|
11
|
+
model: claude-opus-5
|
|
12
|
+
effort: high
|
|
13
|
+
max_turns: 80
|
|
14
|
+
mcp_servers: [envelope, env_control, playwright, defect_memory]
|
|
15
|
+
builtin_tools: [Read, Grep, Glob]
|
|
16
|
+
policy: {}
|
|
17
|
+
|
|
18
|
+
skills: [product-task-graph, exploratory-ui-walk, environment-pinning, severity-rubric, repro-minimisation]
|
|
19
|
+
|
|
20
|
+
# No must_call: finding nothing is a valid outcome for a discovery agent, and
|
|
21
|
+
# requiring an emission would manufacture findings to satisfy it. That matters
|
|
22
|
+
# more here than elsewhere -- a product that is easy to navigate produces no
|
|
23
|
+
# friction envelopes, which is the result, not a failure of the run.
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
name: LOAD
|
|
2
|
+
layer: discovery
|
|
3
|
+
role: >
|
|
4
|
+
Performance analyst. Finds N+1 query patterns, unbounded result sets, queries
|
|
5
|
+
filtering on unindexed columns on request paths, endpoint latency outliers,
|
|
6
|
+
bundle-size outliers, unbounded caches and leaked connections. Evidences them
|
|
7
|
+
from the code and from timed requests, because this deployment has no load
|
|
8
|
+
runner, no metrics backend and no profiler -- so it never claims behaviour
|
|
9
|
+
under load that it did not observe.
|
|
10
|
+
prompt: LOAD.md
|
|
11
|
+
model: claude-opus-5
|
|
12
|
+
effort: high
|
|
13
|
+
max_turns: 60
|
|
14
|
+
mcp_servers: [envelope, env_control, defect_memory]
|
|
15
|
+
builtin_tools: [Read, Grep, Glob]
|
|
16
|
+
policy: {}
|
|
17
|
+
|
|
18
|
+
skills: [environment-pinning, repro-minimisation, root-cause-vs-symptom, severity-rubric]
|
|
19
|
+
|
|
20
|
+
# No must_call: finding nothing is a valid outcome for a discovery agent, and
|
|
21
|
+
# requiring an emission would manufacture findings to satisfy it -- which for a
|
|
22
|
+
# performance agent means speculation about load it cannot apply.
|
|
23
|
+
|
|
24
|
+
# Nightly and pre-release only (§4.10): too slow and too noisy per-PR. The
|
|
25
|
+
# run_modes rosters in system.yaml decide that; this file only declares the
|
|
26
|
+
# agent.
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
name: MAPPER
|
|
2
|
+
layer: control
|
|
3
|
+
role: >
|
|
4
|
+
Builds and maintains the shared system map every other agent reads: services,
|
|
5
|
+
routes, API surface, schema, dependency graph, ownership.
|
|
6
|
+
prompt: MAPPER.md
|
|
7
|
+
model: claude-sonnet-5 # extraction, not judgment
|
|
8
|
+
effort: medium
|
|
9
|
+
max_turns: 60
|
|
10
|
+
mcp_servers: [envelope]
|
|
11
|
+
builtin_tools: [Read, Grep, Glob]
|
|
12
|
+
policy: {} # read-only
|
|
13
|
+
|
|
14
|
+
# Procedure lives in skills; the prompt carries role and standards.
|
|
15
|
+
skills: [repo-cartography, api-surface-extraction, ownership-resolution, product-task-graph]
|
|
16
|
+
|
|
17
|
+
# The map is the deliverable. An agent that finishes without publishing one has
|
|
18
|
+
# not done the job, and the Stop hook says so while it can still act on that.
|
|
19
|
+
must_call: [mcp__envelope__put_system_map]
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
name: REPORTER
|
|
2
|
+
layer: reporting
|
|
3
|
+
role: >
|
|
4
|
+
Reporting analyst. Reads what a run produced -- findings, verdicts, denials,
|
|
5
|
+
escalations, recurrence -- and reports the pattern across them, including what
|
|
6
|
+
the run could not reach. Audits the run, never the application.
|
|
7
|
+
prompt: REPORTER.md
|
|
8
|
+
model: claude-sonnet-5
|
|
9
|
+
effort: medium
|
|
10
|
+
max_turns: 40
|
|
11
|
+
mcp_servers: [envelope, defect_memory, tracker]
|
|
12
|
+
builtin_tools: [Read]
|
|
13
|
+
policy: {}
|
|
14
|
+
|
|
15
|
+
skills: [severity-rubric, dedupe-strategy, verdict-reporting]
|
|
16
|
+
|
|
17
|
+
# No must_call: a run that found nothing still deserves a report saying so, and
|
|
18
|
+
# a report is not an envelope. Requiring an emission would turn "nothing to say"
|
|
19
|
+
# into an invented finding.
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
name: REPRODUCER
|
|
2
|
+
layer: triage
|
|
3
|
+
role: >
|
|
4
|
+
Reproduction engineer. Turns a draft finding into a deterministic minimal
|
|
5
|
+
reproduction plus a failing test, measures flake rate, and demotes what it
|
|
6
|
+
cannot reproduce. The noise filter the whole system depends on.
|
|
7
|
+
prompt: REPRODUCER.md
|
|
8
|
+
model: claude-opus-5
|
|
9
|
+
effort: high
|
|
10
|
+
max_turns: 80
|
|
11
|
+
mcp_servers: [envelope, test_runner, env_control, vcs]
|
|
12
|
+
builtin_tools: [Read, Grep, Glob, Write, Edit, Bash]
|
|
13
|
+
policy:
|
|
14
|
+
write_paths: ["qa/repro"] # sandboxed workspace only (§8.1)
|
|
15
|
+
branch_patterns: ["qa/repro/*"] # never main, never force-push
|
|
16
|
+
|
|
17
|
+
skills: [repro-minimisation, failing-test-authoring, flake-detection, environment-pinning]
|
|
18
|
+
|
|
19
|
+
# REPRODUCER is handed exactly one finding and owes exactly one verdict on it.
|
|
20
|
+
# Silence here would let an unreproduced finding drift toward a ticket.
|
|
21
|
+
must_call: [mcp__envelope__record_reproduction]
|