qaas-python 0.0.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (96) hide show
  1. qaas/adapters/__init__.py +19 -0
  2. qaas/adapters/tracker.py +1783 -0
  3. qaas/adapters/vcs.py +555 -0
  4. qaas/cli.py +1757 -0
  5. qaas/config.py +409 -0
  6. qaas/defaults/config/agents/api.yaml +18 -0
  7. qaas/defaults/config/agents/architect.yaml +21 -0
  8. qaas/defaults/config/agents/auditor.yaml +19 -0
  9. qaas/defaults/config/agents/browser.yaml +15 -0
  10. qaas/defaults/config/agents/dba.yaml +20 -0
  11. qaas/defaults/config/agents/fixer.yaml +55 -0
  12. qaas/defaults/config/agents/guide.yaml +23 -0
  13. qaas/defaults/config/agents/load.yaml +26 -0
  14. qaas/defaults/config/agents/mapper.yaml +19 -0
  15. qaas/defaults/config/agents/reporter.yaml +19 -0
  16. qaas/defaults/config/agents/reproducer.yaml +21 -0
  17. qaas/defaults/config/agents/reviewer.yaml +18 -0
  18. qaas/defaults/config/agents/socket.yaml +23 -0
  19. qaas/defaults/config/agents/triage.yaml +20 -0
  20. qaas/defaults/config/agents/verifier.yaml +20 -0
  21. qaas/defaults/config/system.yaml +64 -0
  22. qaas/discover.py +242 -0
  23. qaas/envelope.py +318 -0
  24. qaas/envfile.py +100 -0
  25. qaas/guardrails.py +589 -0
  26. qaas/mcp/__init__.py +0 -0
  27. qaas/mcp/context.py +78 -0
  28. qaas/mcp/contract_diff.py +1011 -0
  29. qaas/mcp/defect_memory.py +495 -0
  30. qaas/mcp/env_control.py +925 -0
  31. qaas/mcp/envelope_server.py +463 -0
  32. qaas/mcp/test_runner.py +842 -0
  33. qaas/mcp/tracker.py +420 -0
  34. qaas/mcp/vcs.py +501 -0
  35. qaas/paths.py +317 -0
  36. qaas/plugin/.claude-plugin/plugin.json +9 -0
  37. qaas/plugin/skills/a11y-audit/SKILL.md +34 -0
  38. qaas/plugin/skills/adversarial-review/SKILL.md +120 -0
  39. qaas/plugin/skills/api-surface-extraction/SKILL.md +38 -0
  40. qaas/plugin/skills/authz-matrix-check/SKILL.md +46 -0
  41. qaas/plugin/skills/console-error-triage/SKILL.md +39 -0
  42. qaas/plugin/skills/contract-test-generation/SKILL.md +36 -0
  43. qaas/plugin/skills/dedupe-strategy/SKILL.md +39 -0
  44. qaas/plugin/skills/environment-pinning/SKILL.md +35 -0
  45. qaas/plugin/skills/error-taxonomy/SKILL.md +42 -0
  46. qaas/plugin/skills/exploratory-ui-walk/SKILL.md +46 -0
  47. qaas/plugin/skills/failing-test-authoring/SKILL.md +47 -0
  48. qaas/plugin/skills/flake-detection/SKILL.md +39 -0
  49. qaas/plugin/skills/form-state-probe/SKILL.md +36 -0
  50. qaas/plugin/skills/minimal-diff-discipline/SKILL.md +70 -0
  51. qaas/plugin/skills/openapi-diff/SKILL.md +45 -0
  52. qaas/plugin/skills/ownership-resolution/SKILL.md +31 -0
  53. qaas/plugin/skills/product-task-graph/SKILL.md +35 -0
  54. qaas/plugin/skills/regression-risk-scoring/SKILL.md +59 -0
  55. qaas/plugin/skills/regression-suite-selection/SKILL.md +36 -0
  56. qaas/plugin/skills/repo-cartography/SKILL.md +38 -0
  57. qaas/plugin/skills/repro-minimisation/SKILL.md +41 -0
  58. qaas/plugin/skills/rollback-plan-authoring/SKILL.md +81 -0
  59. qaas/plugin/skills/root-cause-vs-symptom/SKILL.md +67 -0
  60. qaas/plugin/skills/routing-rules/SKILL.md +34 -0
  61. qaas/plugin/skills/severity-rubric/SKILL.md +42 -0
  62. qaas/plugin/skills/test-first-fix/SKILL.md +66 -0
  63. qaas/plugin/skills/test-quality-audit/SKILL.md +58 -0
  64. qaas/plugin/skills/ticket-writer/SKILL.md +40 -0
  65. qaas/plugin/skills/verdict-reporting/SKILL.md +35 -0
  66. qaas/plugin/skills/verification-protocol/SKILL.md +39 -0
  67. qaas/prompts/API.md +44 -0
  68. qaas/prompts/ARCHITECT.md +80 -0
  69. qaas/prompts/AUDITOR.md +62 -0
  70. qaas/prompts/BROWSER.md +46 -0
  71. qaas/prompts/DBA.md +59 -0
  72. qaas/prompts/FIXER.md +55 -0
  73. qaas/prompts/GUIDE.md +94 -0
  74. qaas/prompts/LOAD.md +109 -0
  75. qaas/prompts/MAPPER.md +46 -0
  76. qaas/prompts/REPORTER.md +61 -0
  77. qaas/prompts/REPRODUCER.md +43 -0
  78. qaas/prompts/REVIEWER.md +53 -0
  79. qaas/prompts/SOCKET.md +100 -0
  80. qaas/prompts/TRIAGE.md +45 -0
  81. qaas/prompts/VERIFIER.md +41 -0
  82. qaas/prompts/_shared.md +45 -0
  83. qaas/registry.py +496 -0
  84. qaas/router.py +581 -0
  85. qaas/runner.py +210 -0
  86. qaas/scorecard.py +448 -0
  87. qaas/sdk_compat.py +52 -0
  88. qaas/store.py +323 -0
  89. qaas/target.py +287 -0
  90. qaas/tasks.py +438 -0
  91. qaas/trace.py +342 -0
  92. qaas_python-0.0.1.dist-info/METADATA +429 -0
  93. qaas_python-0.0.1.dist-info/RECORD +96 -0
  94. qaas_python-0.0.1.dist-info/WHEEL +4 -0
  95. qaas_python-0.0.1.dist-info/entry_points.txt +2 -0
  96. qaas_python-0.0.1.dist-info/licenses/LICENSE +21 -0
qaas/config.py ADDED
@@ -0,0 +1,409 @@
1
+ """Configuration: agents are data, not code.
2
+
3
+ An agent is a prompt file plus an entry in `config/agents/`. Adding one of the
4
+ remaining agents from the roster should never require touching the router,
5
+ the runner, or the guardrails — that is the property this module exists to keep.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ from pathlib import Path
11
+ from typing import Any, Literal, Sequence
12
+
13
+ import os
14
+
15
+ import yaml
16
+ from pydantic import BaseModel, ConfigDict, Field, model_validator
17
+
18
+ from qaas.target import TargetProfile, load_target
19
+
20
+ # §5.3 is explicit that no agent gets more than six MCP servers, because tool
21
+ # selection accuracy falls off past roughly 5-7. Enforced, not just documented.
22
+ MAX_MCP_SERVERS_PER_AGENT = 6
23
+
24
+ Layer = Literal["control", "discovery", "triage", "remediation", "reporting"]
25
+
26
+
27
+ class Policy(BaseModel):
28
+ """One agent's slice of the §8.1 write-permission matrix.
29
+
30
+ Default is read-only. Anything an agent may write, it says so here, and
31
+ guardrails.py enforces it against the actual tool call arguments.
32
+ """
33
+
34
+ model_config = ConfigDict(extra="forbid")
35
+
36
+ write_paths: list[str] = Field(default_factory=list)
37
+ branch_patterns: list[str] = Field(default_factory=list)
38
+ may_open_pr: bool = False
39
+ may_create_tickets: bool = False
40
+ may_transition_tickets: bool = False
41
+ max_tickets_per_run: int = 0
42
+ max_diff_files: int | None = None
43
+ max_diff_lines: int | None = None
44
+ protected_paths: list[str] = Field(default_factory=list)
45
+
46
+ #: Path globs this agent may never modify, whatever else its policy allows.
47
+ #: §8.2 names the classes: migrations, auth, payment paths and infra config.
48
+ #: These are the changes whose blast radius a review cannot reliably bound,
49
+ #: so they stop at a human even when everything else in the envelope holds.
50
+ forbidden_paths: list[str] = Field(default_factory=list)
51
+
52
+ @property
53
+ def read_only(self) -> bool:
54
+ return not (
55
+ self.write_paths
56
+ or self.branch_patterns
57
+ or self.may_open_pr
58
+ or self.may_create_tickets
59
+ or self.may_transition_tickets
60
+ )
61
+
62
+
63
+ class AgentSpec(BaseModel):
64
+ """Everything needed to build one agent's ClaudeAgentOptions."""
65
+
66
+ model_config = ConfigDict(extra="forbid")
67
+
68
+ name: str
69
+ layer: Layer
70
+ role: str
71
+ prompt: str # path relative to src/qaas/prompts/
72
+ enabled: bool = True
73
+
74
+ model: str = "claude-opus-5"
75
+ effort: Literal["low", "medium", "high", "xhigh", "max"] = "high"
76
+ max_turns: int = 40
77
+ max_budget_usd: float | None = None
78
+
79
+
80
+ mcp_servers: list[str] = Field(default_factory=list)
81
+ builtin_tools: list[str] = Field(default_factory=list)
82
+ policy: Policy = Field(default_factory=Policy)
83
+
84
+ # Procedure lives in skills, role and standards live in the prompt. A skill
85
+ # named here is preloaded; the agent can still reach others through Skill.
86
+ skills: list[str] = Field(default_factory=list)
87
+
88
+ # Tools this agent must have called before it is allowed to finish. The Stop
89
+ # hook enforces it. Without this an agent can produce a confident summary and
90
+ # no artifact, and the failure only surfaces afterwards in the router —
91
+ # too late for the agent to fix it.
92
+ must_call: list[str] = Field(default_factory=list)
93
+
94
+ @model_validator(mode="after")
95
+ def _tool_budget(self) -> "AgentSpec":
96
+ if len(self.mcp_servers) > MAX_MCP_SERVERS_PER_AGENT:
97
+ raise ValueError(
98
+ f"{self.name} declares {len(self.mcp_servers)} MCP servers; "
99
+ f"the cap is {MAX_MCP_SERVERS_PER_AGENT} (§5.3). "
100
+ "An agent needing more is a signal to split it."
101
+ )
102
+ if len(set(self.mcp_servers)) != len(self.mcp_servers):
103
+ raise ValueError(f"{self.name} lists a duplicate MCP server")
104
+ for tool in self.must_call:
105
+ server = tool.split("__")[1] if tool.startswith("mcp__") else None
106
+ if server and server not in self.mcp_servers:
107
+ raise ValueError(
108
+ f"{self.name} must_call names '{tool}' but is not connected to "
109
+ f"the '{server}' server; it could never satisfy that."
110
+ )
111
+ return self
112
+
113
+ def prompt_path(self, prompts_dir: Path) -> Path:
114
+ return prompts_dir / self.prompt
115
+
116
+
117
+ class StdioServerSpec(BaseModel):
118
+ """A user-declared MCP server run as a subprocess.
119
+
120
+ Pure data: `command` and `args` are passed to the CLI, which spawns it. No
121
+ shell, ever -- `command` is a program and `args` is a list, so a string like
122
+ `"foo && rm -rf /"` is a program name that does not exist rather than two
123
+ commands.
124
+ """
125
+
126
+ model_config = ConfigDict(extra="forbid")
127
+
128
+ type: Literal["stdio"] = "stdio"
129
+ command: str
130
+ args: list[str] = Field(default_factory=list)
131
+ env: dict[str, str] = Field(default_factory=dict)
132
+
133
+
134
+ class UrlServerSpec(BaseModel):
135
+ """A user-declared MCP server reached over HTTP or SSE."""
136
+
137
+ model_config = ConfigDict(extra="forbid")
138
+
139
+ type: Literal["http", "sse"]
140
+ url: str
141
+ headers: dict[str, str] = Field(default_factory=dict)
142
+
143
+
144
+ #: What a user may declare. Deliberately no in-process Python type: that would
145
+ #: mean `importlib.import_module` on a name from a config file, executing
146
+ #: arbitrary module-level code inside the process holding this user's Anthropic
147
+ #: credentials, Jira token and GitHub auth. A subprocess is a subprocess; an
148
+ #: import is a foothold. If someone needs a Python server they can wrap it in a
149
+ #: stdio entry point and it costs them one line.
150
+ McpServerSpec = StdioServerSpec | UrlServerSpec
151
+
152
+
153
+ class Thresholds(BaseModel):
154
+ model_config = ConfigDict(extra="forbid")
155
+
156
+ min_confidence_to_file: float = 0.6
157
+ max_findings_per_agent_run: int = 25
158
+ max_tickets_per_run: int = 10
159
+ flake_runs: int = 5
160
+ max_mender_arbiter_round_trips: int = 2
161
+ max_proof_reopens: int = 1
162
+
163
+
164
+ class RunMode(BaseModel):
165
+ model_config = ConfigDict(extra="forbid")
166
+
167
+ trigger: str
168
+ agents: list[str]
169
+ max_budget_usd: float | None = None
170
+
171
+ max_wall_clock_s: int = 3600
172
+ max_concurrency: int = 3
173
+ files_tickets: bool = True
174
+
175
+
176
+ class SystemConfig(BaseModel):
177
+ model_config = ConfigDict(extra="forbid")
178
+
179
+ project: str = "qaas"
180
+
181
+ #: Which target profile in config/targets/ this run is pointed at. The
182
+ #: profile is what makes the system portable: without it every prompt and
183
+ #: every environment call is welded to the application it was built beside.
184
+ #: The active target profile, or None when nothing is configured yet.
185
+ #: A fresh `pip install` is legitimately in that state; commands that
186
+ #: need a profile say so rather than crashing during config load.
187
+ target: str | None = None
188
+
189
+ #: Servers this project declares, on top of the built-in ones. Declaring a
190
+ #: server here grants nothing; an agent receives it only by naming it in its
191
+ #: own `mcp_servers:` list.
192
+ mcp_servers: dict[str, McpServerSpec] = Field(default_factory=dict)
193
+
194
+ #: Overridable with QAAS_TRACKER. Keep the committed value `local`.
195
+ tracker: Literal["local", "jira"] = "local"
196
+ vcs: Literal["local", "github"] = "local"
197
+ thresholds: Thresholds = Field(default_factory=Thresholds)
198
+ run_modes: dict[str, RunMode] = Field(default_factory=dict)
199
+ agents: dict[str, AgentSpec] = Field(default_factory=dict)
200
+ profile: TargetProfile | None = Field(default=None, exclude=True)
201
+
202
+ @model_validator(mode="after")
203
+ def _agents_name_real_servers(self) -> "SystemConfig":
204
+ """Every server an agent names must resolve to something.
205
+
206
+ `AgentSpec` cannot check this -- it has no view of the rest of the
207
+ config -- so an unresolvable name used to surface as `UnknownServer`
208
+ part-way through a paid run. Here it is a load-time error, which is what
209
+ `qaas validate` is for.
210
+
211
+ Imported inside the function: `registry` imports `config`, so a
212
+ module-level import would be a cycle.
213
+ """
214
+ from qaas.registry import SDK_SERVER_MODULES, STDIO_SERVERS
215
+
216
+ builtin = set(SDK_SERVER_MODULES) | set(STDIO_SERVERS)
217
+ known = builtin | set(self.mcp_servers)
218
+ for name, spec in sorted(self.agents.items()):
219
+ unknown = [s for s in spec.mcp_servers if s not in known]
220
+ if unknown:
221
+ raise ValueError(
222
+ f"{name} names MCP server(s) nothing provides: {', '.join(unknown)}. "
223
+ f"Built in: {', '.join(sorted(builtin))}. "
224
+ f"Declared in system.yaml: {', '.join(sorted(self.mcp_servers)) or 'none'}."
225
+ )
226
+ return self
227
+
228
+ @model_validator(mode="after")
229
+ def _modes_name_real_agents(self) -> "SystemConfig":
230
+ for mode_name, mode in self.run_modes.items():
231
+ unknown = [a for a in mode.agents if a not in self.agents]
232
+ if unknown:
233
+ raise ValueError(
234
+ f"run mode '{mode_name}' names unknown agents: {', '.join(unknown)}"
235
+ )
236
+ return self
237
+
238
+ def enabled_agents(self, mode: str) -> list[AgentSpec]:
239
+ """Agents for a run mode, skipping any that are switched off."""
240
+ if mode not in self.run_modes:
241
+ raise KeyError(f"unknown run mode '{mode}'; have: {', '.join(sorted(self.run_modes))}")
242
+ return [self.agents[n] for n in self.run_modes[mode].agents if self.agents[n].enabled]
243
+
244
+ def target_root(self, base: Path | None = None) -> Path:
245
+ """Where the application under test lives.
246
+
247
+ This used to be `Path.cwd() / config.target_app` -- one value serving as
248
+ both "where qaas lives" and "the application under test". That holds
249
+ only while the target is a subdirectory of the qaas checkout, which is
250
+ true of exactly one target: the bundled demo. `qaas run --repo <url>`
251
+ clones into `.qaas/targets/<slug>`, and every write-path allowlist,
252
+ every test cwd and the SDK subprocess cwd are anchored on this value --
253
+ so getting it from the profile is not tidying, it is the security
254
+ boundary being pointed at the right directory.
255
+
256
+ With no profile there is nothing to test; the base (the qaas project, or
257
+ the cwd) is returned so read-only tooling still has somewhere to stand.
258
+ """
259
+ if self.profile is not None:
260
+ return self.profile.root_path(base)
261
+ from qaas.paths import project_root
262
+
263
+ return base if base is not None else project_root()
264
+
265
+
266
+ #: Environment overrides for the two swappable backends.
267
+ TRACKER_ENV = "QAAS_TRACKER"
268
+ VCS_ENV = "QAAS_VCS"
269
+ #: Which target profile to run against. Useful on its own (`QAAS_TARGET=staging
270
+ #: qaas run`), and it is how this repo's own test suite selects the bundled demo
271
+ #: without putting a demo name in the defaults that ship to everyone else.
272
+ TARGET_ENV = "QAAS_TARGET"
273
+
274
+
275
+ def load_config(
276
+ config_dir: Path | str | None = None,
277
+ *,
278
+ search: Sequence[Path] | None = None,
279
+ target: str | None = None,
280
+ ) -> SystemConfig:
281
+ """Read system.yaml plus every agents/*.yaml, layered across search paths.
282
+
283
+ Passing `config_dir` positionally means "this directory and nothing else",
284
+ which is exactly the old behaviour and what every test does. Passing
285
+ `search` layers several directories: `system.yaml` is taken whole from the
286
+ first that has one, while `agents/*.yaml` and `targets/*.yaml` are unioned
287
+ by filename with earlier directories shadowing later ones -- so a user can
288
+ override one agent without forking all eight and freezing on today's roster.
289
+
290
+ With neither argument, the workspace resolver decides (an explicit
291
+ --config, then the project, then what shipped in the wheel).
292
+
293
+ `target` beats everything -- system.yaml, QAAS_TARGET, the single-profile
294
+ guess. It is what `qaas run --target X` and `qaas run --repo <url>` mean:
295
+ *this* application, whatever is configured. Without it, a stale `target:`
296
+ naming a profile that no longer exists killed the run inside config loading,
297
+ before the override the operator had just typed was ever consulted.
298
+ """
299
+ if config_dir is not None:
300
+ dirs: list[Path] = [Path(config_dir)]
301
+ elif search is not None:
302
+ dirs = [Path(d) for d in search]
303
+ else:
304
+ from qaas.paths import Workspace
305
+
306
+ dirs = list(Workspace.resolve().config_dirs)
307
+
308
+ system_path = next((d / "system.yaml" for d in dirs if (d / "system.yaml").is_file()), None)
309
+ if system_path is None:
310
+ looked = ", ".join(str(d) for d in dirs) or "(nowhere -- no search path)"
311
+ raise FileNotFoundError(f"no system config at {dirs[0] / 'system.yaml'} (looked in: {looked})")
312
+
313
+ raw: dict[str, Any] = yaml.safe_load(system_path.read_text(encoding="utf-8")) or {}
314
+
315
+ # `target_app:` used to name the application's directory relative to the
316
+ # process cwd. The target profile's `root` says the same thing and says it
317
+ # better, so the field is gone -- but `extra="forbid"` would turn an old
318
+ # system.yaml into a hard load failure, and someone else's committed config
319
+ # is not ours to break. Dropped silently: there is nothing for the reader to
320
+ # do about it, and the profile already carries the answer.
321
+ raw.pop("target_app", None)
322
+
323
+ # Backend overrides from the environment, so pointing a run at a real
324
+ # tracker or reproducer is not a committed file change.
325
+ #
326
+ # `tracker: local` is the committed default and must stay that way. When
327
+ # `jira` was committed instead, 18 tests failed and 14 errored: the agent
328
+ # fixtures build a real JiraTracker, which demands credentials CI does not
329
+ # have. The house rule is that the default `pytest` run is offline and free,
330
+ # and a committed backend switch silently breaks it -- so the switch belongs
331
+ # in the environment of the person who wants it, not in the repo.
332
+ for key, var in (("tracker", TRACKER_ENV), ("vcs", VCS_ENV), ("target", TARGET_ENV)):
333
+ raw_value = (os.environ.get(var) or "").strip()
334
+ if raw_value:
335
+ # Backends are lowercase literals; a target is a profile name.
336
+ raw[key] = raw_value if key == "target" else raw_value.lower()
337
+
338
+ # An explicit argument outranks both the file and the environment.
339
+ if target:
340
+ raw["target"] = target
341
+
342
+ # Agents layer by filename. Walking the search paths in reverse means the
343
+ # highest-precedence directory writes last and therefore wins.
344
+ by_stem: dict[str, Path] = {}
345
+ for d in reversed(dirs):
346
+ for path in sorted((d / "agents").glob("*.yaml")):
347
+ by_stem[path.stem] = path
348
+
349
+ agents: dict[str, Any] = {}
350
+ for path in by_stem.values():
351
+ spec = yaml.safe_load(path.read_text(encoding="utf-8")) or {}
352
+ name = spec.get("name") or path.stem.upper()
353
+ spec["name"] = name
354
+ if name in agents:
355
+ raise ValueError(f"duplicate agent definition for {name} at {path}")
356
+ agents[name] = spec
357
+
358
+ raw["agents"] = agents
359
+
360
+ config = SystemConfig.model_validate(raw)
361
+
362
+ # Resolve the target profile if one is named and findable.
363
+ #
364
+ # A named-but-missing profile stays fatal -- running the wrong application
365
+ # is worse than not running. But `target: null` is a legitimate state now:
366
+ # `pip install qaas-python` gives you a working CLI that is not yet pointed
367
+ # at anything, and the commands that need a profile say "run qaas init"
368
+ # rather than dying inside config loading.
369
+ # Profiles layer by filename, exactly as agents and skills do. This used to
370
+ # take the *first* config layer that had a `targets/` at all -- and the day
371
+ # `qaas run --repo` started writing a generated profile into the writable
372
+ # layer (`.qaas/config/targets/`), that layer became "the" targets directory
373
+ # and every profile in `<project>/config/targets/` vanished: `qaas doctor
374
+ # --target corvid` reported the demo profile did not exist.
375
+ profiles = target_files(dirs)
376
+ chosen = config.target
377
+
378
+ # No target named, but exactly one profile on disk: use it. Choosing between
379
+ # two would be guessing, and running the wrong application is worse than not
380
+ # running -- but with one candidate there is nothing to guess at, and making
381
+ # the user restate it is ceremony. This is also what keeps this repository
382
+ # working: its `system.yaml` ships in the package and names no target,
383
+ # because a demo name has no business in the defaults everyone installs.
384
+ if not chosen and len(profiles) == 1:
385
+ chosen = next(iter(profiles))
386
+
387
+ if chosen and profiles:
388
+ if chosen not in profiles:
389
+ # Named but absent stays fatal: running the wrong application is
390
+ # worse than not running. Listed from the merged view, so the
391
+ # suggestion names every profile the user actually has.
392
+ raise FileNotFoundError(
393
+ f"no target profile '{chosen}'. Available: {', '.join(sorted(profiles))}. "
394
+ "Create one with `qaas init <path-to-repo>`."
395
+ )
396
+ profile = load_target(chosen, profiles[chosen].parent)
397
+ config = config.model_copy(update={"target": chosen, "profile": profile})
398
+ return config
399
+
400
+
401
+ def target_files(dirs: Sequence[Path]) -> dict[str, Path]:
402
+ """Every target profile visible across the config layers, nearest wins."""
403
+ found: dict[str, Path] = {}
404
+ for d in reversed(list(dirs)):
405
+ base = Path(d) / "targets"
406
+ if base.is_dir():
407
+ for path in sorted(base.glob("*.yaml")):
408
+ found[path.stem] = path
409
+ return found
@@ -0,0 +1,18 @@
1
+ name: API
2
+ layer: discovery
3
+ role: >
4
+ Backend, API and contract analyst. Finds spec drift, breaking changes, missing
5
+ authorization, error-taxonomy inconsistency, unbounded results, validation gaps.
6
+ Ships a failing contract test as evidence, never a bare opinion.
7
+ prompt: API.md
8
+ model: claude-opus-5
9
+ effort: high
10
+ max_turns: 60
11
+ mcp_servers: [envelope, contract_diff, env_control, defect_memory]
12
+ builtin_tools: [Read, Grep, Glob]
13
+ policy: {}
14
+
15
+ skills: [openapi-diff, authz-matrix-check, error-taxonomy, contract-test-generation, severity-rubric]
16
+
17
+ # No must_call: finding nothing is a valid and useful outcome for a discovery
18
+ # agent, and requiring an emission would manufacture findings to satisfy it.
@@ -0,0 +1,21 @@
1
+ name: ARCHITECT
2
+ layer: discovery
3
+ role: >
4
+ Architecture analyst. Finds dependency cycles, layering violations, god modules
5
+ and fan-in outliers, domain logic duplicated across services, drift between the
6
+ architecture documents and the code, wrong service boundaries, and dead or
7
+ orphaned code. Pure static analysis -- it needs no running application, so it
8
+ is the one discovery agent that works against a target with no environment.
9
+ prompt: ARCHITECT.md
10
+ model: claude-opus-5
11
+ effort: high
12
+ max_turns: 60
13
+ mcp_servers: [envelope, defect_memory]
14
+ builtin_tools: [Read, Grep, Glob]
15
+ policy: {} # read-only; it never touches the app it reads
16
+
17
+ skills: [repo-cartography, api-surface-extraction, ownership-resolution, severity-rubric]
18
+
19
+ # No must_call: see DBA. Also, an architecture agent that must emit something
20
+ # will emit taste, and "this could be cleaner" filed as a defect is the fastest
21
+ # way for a team to stop reading structural findings at all.
@@ -0,0 +1,19 @@
1
+ name: AUDITOR
2
+ layer: discovery
3
+ role: >
4
+ Security and dependency auditor. Finds missing authorization, secrets committed
5
+ to the repository, dependencies with known advisories, and error paths that
6
+ leak internals to a caller. Reports a concrete exploit path or lowers its
7
+ confidence -- a security finding without one is a guess wearing a severity.
8
+ prompt: AUDITOR.md
9
+ model: claude-opus-5
10
+ effort: high
11
+ max_turns: 60
12
+ mcp_servers: [envelope, env_control, defect_memory]
13
+ builtin_tools: [Read, Grep, Glob]
14
+ policy: {}
15
+
16
+ skills: [authz-matrix-check, error-taxonomy, severity-rubric, routing-rules, repro-minimisation]
17
+
18
+ # No must_call: see DBA. Also: an auditor that must report something will
19
+ # report something, and security noise is the fastest way to be ignored.
@@ -0,0 +1,15 @@
1
+ name: BROWSER
2
+ layer: discovery
3
+ role: >
4
+ Frontend and UI explorer. Walks the product as a user does: broken flows,
5
+ console errors, accessibility failures, missing loading/empty/error states,
6
+ form validation gaps, state desync.
7
+ prompt: BROWSER.md
8
+ model: claude-opus-5
9
+ effort: high
10
+ max_turns: 80
11
+ mcp_servers: [envelope, env_control, playwright]
12
+ builtin_tools: [Read, Grep, Glob]
13
+ policy: {}
14
+
15
+ skills: [exploratory-ui-walk, a11y-audit, console-error-triage, form-state-probe, product-task-graph, severity-rubric]
@@ -0,0 +1,20 @@
1
+ name: DBA
2
+ layer: discovery
3
+ role: >
4
+ Database and data-integrity analyst. Finds schema constraints the application
5
+ assumes but the database does not enforce, migrations that lose or corrupt
6
+ data, missing indexes on paths the code queries, and cross-tenant reads that
7
+ the ORM makes easy to write. Reports what the schema actually says, never what
8
+ the model layer claims.
9
+ prompt: DBA.md
10
+ model: claude-opus-5
11
+ effort: high
12
+ max_turns: 60
13
+ mcp_servers: [envelope, env_control, defect_memory]
14
+ builtin_tools: [Read, Grep, Glob]
15
+ policy: {}
16
+
17
+ skills: [authz-matrix-check, environment-pinning, severity-rubric, repro-minimisation]
18
+
19
+ # No must_call: finding nothing is a valid outcome for a discovery agent, and
20
+ # requiring an emission would manufacture findings to satisfy it.
@@ -0,0 +1,55 @@
1
+ name: FIXER
2
+ layer: remediation
3
+ role: >
4
+ Remediation engineer. Picks up an agent-ready ticket, writes the minimal change
5
+ that makes the failing test pass, and opens a draft pull request. Deliberately
6
+ one agent with many skills rather than five domain-specific fixers: fixing is
7
+ one activity, and the domain knowledge belongs in loadable skills, not in five
8
+ nearly identical prompts.
9
+ prompt: FIXER.md
10
+ model: claude-opus-5
11
+ effort: high
12
+ max_turns: 80
13
+ mcp_servers: [envelope, test_runner, env_control, vcs, tracker, contract_diff]
14
+ builtin_tools: [Read, Grep, Glob, Write, Edit, Bash]
15
+ skills: [test-first-fix, minimal-diff-discipline, root-cause-vs-symptom, rollback-plan-authoring]
16
+
17
+ policy:
18
+ # Product code, on its own branches. Relative to the *target's* root, not to
19
+ # the qaas project: these used to read `target-app/api/app`, which only ever
20
+ # resolved because the demo happened to live inside this checkout. Still set
21
+ # per target -- a different application keeps its source somewhere else.
22
+ write_paths: [api/app, web/src, qa/repro]
23
+ branch_patterns: ["fix/*"]
24
+ may_open_pr: true
25
+ may_transition_tickets: true
26
+
27
+ # The §8.2 autonomy envelope, enforced in guardrails rather than asked for in
28
+ # the prompt. Start tight; widen from measured REVIEWER approval and VERIFIER
29
+ # verification rates, not from optimism.
30
+ max_diff_files: 5
31
+ max_diff_lines: 150
32
+
33
+ # §8.2: these classes stop at a human however small the change looks. Their
34
+ # blast radius is not something a review can reliably bound.
35
+ #
36
+ # Matched against a path relative to the *target's* root, so the directory
37
+ # patterns are `*x/*` and not `*/x/*`: with a leading slash required,
38
+ # `.github/workflows/ci.yml` -- a repository's CI at its own root, which is
39
+ # where almost every repository keeps it -- matched nothing. That hole was
40
+ # invisible while paths arrived prefixed with `target-app/`.
41
+ forbidden_paths:
42
+ - "*migrations/*"
43
+ - "*migration*"
44
+ - "*auth.py"
45
+ - "*auth*"
46
+ - "*payment*"
47
+ - "*billing*"
48
+ - "*secret*"
49
+ - "*.tf"
50
+ - "*infra/*"
51
+ - "*docker-compose*"
52
+ - "Dockerfile*"
53
+ - "*.github/*"
54
+
55
+ must_call: [mcp__vcs__open_pr]
@@ -0,0 +1,23 @@
1
+ name: GUIDE
2
+ layer: discovery
3
+ role: >
4
+ Product navigation and UX guide. Navigates the live product to answer "how do
5
+ I do X here?", and reports every place that navigation struggled: tasks
6
+ reachable only by typing a URL, dead ends, unlabelled paths, step counts out
7
+ of proportion to the task, and product vocabulary that does not match the
8
+ user's. BROWSER owns whether a feature works; GUIDE owns whether anyone can
9
+ find it.
10
+ prompt: GUIDE.md
11
+ model: claude-opus-5
12
+ effort: high
13
+ max_turns: 80
14
+ mcp_servers: [envelope, env_control, playwright, defect_memory]
15
+ builtin_tools: [Read, Grep, Glob]
16
+ policy: {}
17
+
18
+ skills: [product-task-graph, exploratory-ui-walk, environment-pinning, severity-rubric, repro-minimisation]
19
+
20
+ # No must_call: finding nothing is a valid outcome for a discovery agent, and
21
+ # requiring an emission would manufacture findings to satisfy it. That matters
22
+ # more here than elsewhere -- a product that is easy to navigate produces no
23
+ # friction envelopes, which is the result, not a failure of the run.
@@ -0,0 +1,26 @@
1
+ name: LOAD
2
+ layer: discovery
3
+ role: >
4
+ Performance analyst. Finds N+1 query patterns, unbounded result sets, queries
5
+ filtering on unindexed columns on request paths, endpoint latency outliers,
6
+ bundle-size outliers, unbounded caches and leaked connections. Evidences them
7
+ from the code and from timed requests, because this deployment has no load
8
+ runner, no metrics backend and no profiler -- so it never claims behaviour
9
+ under load that it did not observe.
10
+ prompt: LOAD.md
11
+ model: claude-opus-5
12
+ effort: high
13
+ max_turns: 60
14
+ mcp_servers: [envelope, env_control, defect_memory]
15
+ builtin_tools: [Read, Grep, Glob]
16
+ policy: {}
17
+
18
+ skills: [environment-pinning, repro-minimisation, root-cause-vs-symptom, severity-rubric]
19
+
20
+ # No must_call: finding nothing is a valid outcome for a discovery agent, and
21
+ # requiring an emission would manufacture findings to satisfy it -- which for a
22
+ # performance agent means speculation about load it cannot apply.
23
+
24
+ # Nightly and pre-release only (§4.10): too slow and too noisy per-PR. The
25
+ # run_modes rosters in system.yaml decide that; this file only declares the
26
+ # agent.
@@ -0,0 +1,19 @@
1
+ name: MAPPER
2
+ layer: control
3
+ role: >
4
+ Builds and maintains the shared system map every other agent reads: services,
5
+ routes, API surface, schema, dependency graph, ownership.
6
+ prompt: MAPPER.md
7
+ model: claude-sonnet-5 # extraction, not judgment
8
+ effort: medium
9
+ max_turns: 60
10
+ mcp_servers: [envelope]
11
+ builtin_tools: [Read, Grep, Glob]
12
+ policy: {} # read-only
13
+
14
+ # Procedure lives in skills; the prompt carries role and standards.
15
+ skills: [repo-cartography, api-surface-extraction, ownership-resolution, product-task-graph]
16
+
17
+ # The map is the deliverable. An agent that finishes without publishing one has
18
+ # not done the job, and the Stop hook says so while it can still act on that.
19
+ must_call: [mcp__envelope__put_system_map]
@@ -0,0 +1,19 @@
1
+ name: REPORTER
2
+ layer: reporting
3
+ role: >
4
+ Reporting analyst. Reads what a run produced -- findings, verdicts, denials,
5
+ escalations, recurrence -- and reports the pattern across them, including what
6
+ the run could not reach. Audits the run, never the application.
7
+ prompt: REPORTER.md
8
+ model: claude-sonnet-5
9
+ effort: medium
10
+ max_turns: 40
11
+ mcp_servers: [envelope, defect_memory, tracker]
12
+ builtin_tools: [Read]
13
+ policy: {}
14
+
15
+ skills: [severity-rubric, dedupe-strategy, verdict-reporting]
16
+
17
+ # No must_call: a run that found nothing still deserves a report saying so, and
18
+ # a report is not an envelope. Requiring an emission would turn "nothing to say"
19
+ # into an invented finding.
@@ -0,0 +1,21 @@
1
+ name: REPRODUCER
2
+ layer: triage
3
+ role: >
4
+ Reproduction engineer. Turns a draft finding into a deterministic minimal
5
+ reproduction plus a failing test, measures flake rate, and demotes what it
6
+ cannot reproduce. The noise filter the whole system depends on.
7
+ prompt: REPRODUCER.md
8
+ model: claude-opus-5
9
+ effort: high
10
+ max_turns: 80
11
+ mcp_servers: [envelope, test_runner, env_control, vcs]
12
+ builtin_tools: [Read, Grep, Glob, Write, Edit, Bash]
13
+ policy:
14
+ write_paths: ["qa/repro"] # sandboxed workspace only (§8.1)
15
+ branch_patterns: ["qa/repro/*"] # never main, never force-push
16
+
17
+ skills: [repro-minimisation, failing-test-authoring, flake-detection, environment-pinning]
18
+
19
+ # REPRODUCER is handed exactly one finding and owes exactly one verdict on it.
20
+ # Silence here would let an unreproduced finding drift toward a ticket.
21
+ must_call: [mcp__envelope__record_reproduction]