qaas-python 0.0.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (96) hide show
  1. qaas/adapters/__init__.py +19 -0
  2. qaas/adapters/tracker.py +1783 -0
  3. qaas/adapters/vcs.py +555 -0
  4. qaas/cli.py +1757 -0
  5. qaas/config.py +409 -0
  6. qaas/defaults/config/agents/api.yaml +18 -0
  7. qaas/defaults/config/agents/architect.yaml +21 -0
  8. qaas/defaults/config/agents/auditor.yaml +19 -0
  9. qaas/defaults/config/agents/browser.yaml +15 -0
  10. qaas/defaults/config/agents/dba.yaml +20 -0
  11. qaas/defaults/config/agents/fixer.yaml +55 -0
  12. qaas/defaults/config/agents/guide.yaml +23 -0
  13. qaas/defaults/config/agents/load.yaml +26 -0
  14. qaas/defaults/config/agents/mapper.yaml +19 -0
  15. qaas/defaults/config/agents/reporter.yaml +19 -0
  16. qaas/defaults/config/agents/reproducer.yaml +21 -0
  17. qaas/defaults/config/agents/reviewer.yaml +18 -0
  18. qaas/defaults/config/agents/socket.yaml +23 -0
  19. qaas/defaults/config/agents/triage.yaml +20 -0
  20. qaas/defaults/config/agents/verifier.yaml +20 -0
  21. qaas/defaults/config/system.yaml +64 -0
  22. qaas/discover.py +242 -0
  23. qaas/envelope.py +318 -0
  24. qaas/envfile.py +100 -0
  25. qaas/guardrails.py +589 -0
  26. qaas/mcp/__init__.py +0 -0
  27. qaas/mcp/context.py +78 -0
  28. qaas/mcp/contract_diff.py +1011 -0
  29. qaas/mcp/defect_memory.py +495 -0
  30. qaas/mcp/env_control.py +925 -0
  31. qaas/mcp/envelope_server.py +463 -0
  32. qaas/mcp/test_runner.py +842 -0
  33. qaas/mcp/tracker.py +420 -0
  34. qaas/mcp/vcs.py +501 -0
  35. qaas/paths.py +317 -0
  36. qaas/plugin/.claude-plugin/plugin.json +9 -0
  37. qaas/plugin/skills/a11y-audit/SKILL.md +34 -0
  38. qaas/plugin/skills/adversarial-review/SKILL.md +120 -0
  39. qaas/plugin/skills/api-surface-extraction/SKILL.md +38 -0
  40. qaas/plugin/skills/authz-matrix-check/SKILL.md +46 -0
  41. qaas/plugin/skills/console-error-triage/SKILL.md +39 -0
  42. qaas/plugin/skills/contract-test-generation/SKILL.md +36 -0
  43. qaas/plugin/skills/dedupe-strategy/SKILL.md +39 -0
  44. qaas/plugin/skills/environment-pinning/SKILL.md +35 -0
  45. qaas/plugin/skills/error-taxonomy/SKILL.md +42 -0
  46. qaas/plugin/skills/exploratory-ui-walk/SKILL.md +46 -0
  47. qaas/plugin/skills/failing-test-authoring/SKILL.md +47 -0
  48. qaas/plugin/skills/flake-detection/SKILL.md +39 -0
  49. qaas/plugin/skills/form-state-probe/SKILL.md +36 -0
  50. qaas/plugin/skills/minimal-diff-discipline/SKILL.md +70 -0
  51. qaas/plugin/skills/openapi-diff/SKILL.md +45 -0
  52. qaas/plugin/skills/ownership-resolution/SKILL.md +31 -0
  53. qaas/plugin/skills/product-task-graph/SKILL.md +35 -0
  54. qaas/plugin/skills/regression-risk-scoring/SKILL.md +59 -0
  55. qaas/plugin/skills/regression-suite-selection/SKILL.md +36 -0
  56. qaas/plugin/skills/repo-cartography/SKILL.md +38 -0
  57. qaas/plugin/skills/repro-minimisation/SKILL.md +41 -0
  58. qaas/plugin/skills/rollback-plan-authoring/SKILL.md +81 -0
  59. qaas/plugin/skills/root-cause-vs-symptom/SKILL.md +67 -0
  60. qaas/plugin/skills/routing-rules/SKILL.md +34 -0
  61. qaas/plugin/skills/severity-rubric/SKILL.md +42 -0
  62. qaas/plugin/skills/test-first-fix/SKILL.md +66 -0
  63. qaas/plugin/skills/test-quality-audit/SKILL.md +58 -0
  64. qaas/plugin/skills/ticket-writer/SKILL.md +40 -0
  65. qaas/plugin/skills/verdict-reporting/SKILL.md +35 -0
  66. qaas/plugin/skills/verification-protocol/SKILL.md +39 -0
  67. qaas/prompts/API.md +44 -0
  68. qaas/prompts/ARCHITECT.md +80 -0
  69. qaas/prompts/AUDITOR.md +62 -0
  70. qaas/prompts/BROWSER.md +46 -0
  71. qaas/prompts/DBA.md +59 -0
  72. qaas/prompts/FIXER.md +55 -0
  73. qaas/prompts/GUIDE.md +94 -0
  74. qaas/prompts/LOAD.md +109 -0
  75. qaas/prompts/MAPPER.md +46 -0
  76. qaas/prompts/REPORTER.md +61 -0
  77. qaas/prompts/REPRODUCER.md +43 -0
  78. qaas/prompts/REVIEWER.md +53 -0
  79. qaas/prompts/SOCKET.md +100 -0
  80. qaas/prompts/TRIAGE.md +45 -0
  81. qaas/prompts/VERIFIER.md +41 -0
  82. qaas/prompts/_shared.md +45 -0
  83. qaas/registry.py +496 -0
  84. qaas/router.py +581 -0
  85. qaas/runner.py +210 -0
  86. qaas/scorecard.py +448 -0
  87. qaas/sdk_compat.py +52 -0
  88. qaas/store.py +323 -0
  89. qaas/target.py +287 -0
  90. qaas/tasks.py +438 -0
  91. qaas/trace.py +342 -0
  92. qaas_python-0.0.1.dist-info/METADATA +429 -0
  93. qaas_python-0.0.1.dist-info/RECORD +96 -0
  94. qaas_python-0.0.1.dist-info/WHEEL +4 -0
  95. qaas_python-0.0.1.dist-info/entry_points.txt +2 -0
  96. qaas_python-0.0.1.dist-info/licenses/LICENSE +21 -0
qaas/registry.py ADDED
@@ -0,0 +1,496 @@
1
+ """Turning an AgentSpec into a runnable agent.
2
+
3
+ One agent is one top-level `query()` with its own options: its own system prompt,
4
+ its own MCP servers, its own tool allowlist, its own budget. Not a subagent of a
5
+ shared parent — that would pool the cost into one number and blur the per-agent
6
+ allowlist that §5.3 depends on.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import importlib
12
+ import os
13
+ import re
14
+ from pathlib import Path, PurePosixPath
15
+ from typing import Any, Callable, Sequence
16
+
17
+ import warnings
18
+
19
+ from claude_agent_sdk import ClaudeAgentOptions, HookMatcher
20
+
21
+ # The SDK warns that an `allowed_tools` entry auto-approves a tool before
22
+ # `can_use_tool` is consulted. That warning is correct and this codebase acted
23
+ # on it long ago: enforcement lives in the PreToolUse hook, which sees every
24
+ # call, and `can_use_tool` is kept only as a second layer for calls the
25
+ # allowlist did not auto-approve. See the module docstring in `guardrails.py` --
26
+ # an early version really did have that bug, and REPRODUCER's sandbox check was dead
27
+ # code because of it.
28
+ #
29
+ # So the warning describes a hazard we already removed, and it fired on every
30
+ # single agent dispatch: eight lines of alarming red before a run that is
31
+ # working correctly. Silenced here, next to the reason, rather than by the
32
+ # caller -- a user should not have to learn which of our warnings to ignore.
33
+ try: # pragma: no cover - depends on the installed SDK exposing the class
34
+ from claude_agent_sdk.types import CanUseToolShadowedWarning
35
+
36
+ warnings.filterwarnings("ignore", category=CanUseToolShadowedWarning)
37
+ except ImportError: # the SDK renamed or dropped it; nothing to silence
38
+ pass
39
+
40
+ from qaas.config import AgentSpec
41
+ from qaas.guardrails import ALWAYS_GRANTED, Guardrail
42
+ from qaas.mcp.context import ToolContext
43
+ from qaas.sdk_compat import POST_TOOL_USE, PRE_TOOL_USE, STOP, mcp_server_wildcard
44
+
45
+ PROMPTS_DIR = Path(__file__).parent / "prompts"
46
+ SHARED_PROMPT = "_shared.md"
47
+
48
+ #: `API.md` -> `API.append.md`. The suffix exists because the only other
49
+ #: way to add three house lines to a shipped prompt is to fork the whole file,
50
+ #: and a forked prompt stops receiving the next release's improvements to it --
51
+ #: silently, and in the one part of the system where silence is most expensive.
52
+ APPEND_SUFFIX = ".append.md"
53
+
54
+ # name in config/agents/*.yaml -> the module providing `build(ctx)`.
55
+ # Off-the-shelf servers (playwright) are stdio subprocesses, handled separately.
56
+ SDK_SERVER_MODULES: dict[str, str] = {
57
+ "envelope": "qaas.mcp.envelope_server",
58
+ "defect_memory": "qaas.mcp.defect_memory",
59
+ "tracker": "qaas.mcp.tracker",
60
+ "test_runner": "qaas.mcp.test_runner",
61
+ "env_control": "qaas.mcp.env_control",
62
+ "contract_diff": "qaas.mcp.contract_diff",
63
+ "vcs": "qaas.mcp.vcs",
64
+ }
65
+
66
+ STDIO_SERVERS: dict[str, dict[str, Any]] = {
67
+ "playwright": {
68
+ "type": "stdio",
69
+ "command": "npx",
70
+ "args": ["-y", "@playwright/mcp@latest", "--isolated", "--browser", "chromium"],
71
+ },
72
+ }
73
+
74
+
75
+ class UnknownServer(KeyError):
76
+ """A config names an MCP server nothing provides."""
77
+
78
+
79
+ def resolve_prompt_dirs(ctx: ToolContext | None = None) -> tuple[Path, ...]:
80
+ """The prompt search path: overrides first, packaged last.
81
+
82
+ Same shape as `skill_plugins` -- a ToolContext may carry a workspace, and
83
+ anything without one asks the resolver. Prompts used to be read from
84
+ `PROMPTS_DIR` unconditionally, which meant a `pip install` user could not
85
+ change a single line of any prompt without editing site-packages.
86
+ """
87
+ from qaas.paths import Workspace
88
+
89
+ ws = getattr(ctx, "workspace", None) or Workspace.resolve()
90
+ return tuple(ws.prompt_dirs)
91
+
92
+
93
+ def _first_hit(dirs: Sequence[Path], relative: str) -> Path | None:
94
+ for d in dirs:
95
+ candidate = Path(d) / relative
96
+ if candidate.is_file():
97
+ return candidate
98
+ return None
99
+
100
+
101
+ def append_name(prompt: str) -> str:
102
+ """`API.md` -> `API.append.md`, keeping any subdirectory."""
103
+ return str(PurePosixPath(prompt).with_suffix("")) + APPEND_SUFFIX
104
+
105
+
106
+ def append_paths(dirs: Sequence[Path], prompt: str) -> list[Path]:
107
+ """Every `<AGENT>.append.md` on the search path, broadest layer first.
108
+
109
+ Not first-hit-wins: appends accumulate rather than shadow, so an
110
+ organisation-wide `QAAS_HOME` addendum and a project's own both apply. They
111
+ are ordered lowest-precedence first so the nearest layer speaks last, which
112
+ is both how a reader expects the specific to follow the general and how a
113
+ model weights the end of a block.
114
+ """
115
+ name = append_name(prompt)
116
+ found: list[Path] = []
117
+ for d in reversed(list(dirs)):
118
+ candidate = Path(d) / name
119
+ if candidate.is_file():
120
+ found.append(candidate)
121
+ return found
122
+
123
+
124
+ def build_system_prompt(
125
+ spec: AgentSpec, prompt_dirs: Sequence[Path] | None = None
126
+ ) -> str:
127
+ """The agent's prompt, its local addenda, and the house rules every agent shares.
128
+
129
+ Kept as separate files so a change to the shared rules reaches every agent at
130
+ once, rather than being copy-pasted into six prompts that then drift.
131
+
132
+ Each file is resolved first-hit-wins **independently**: overriding
133
+ `API.md` keeps the house `_shared.md`, and replacing `_shared.md` keeps
134
+ all eight agent prompts. Resolving the pair from one winning directory would
135
+ make either override drag the other along.
136
+
137
+ Order is agent, then addenda, then shared: the house rules are the last word,
138
+ and an addendum that could displace them would be an enforcement hole opened
139
+ from a text file.
140
+ """
141
+ dirs = tuple(prompt_dirs) if prompt_dirs is not None else resolve_prompt_dirs()
142
+ own = _first_hit(dirs, spec.prompt)
143
+ if own is None:
144
+ where = ", ".join(str(d) for d in dirs) or "(no prompt directories)"
145
+ raise FileNotFoundError(f"{spec.name} has no prompt '{spec.prompt}' in: {where}")
146
+ shared = _first_hit(dirs, SHARED_PROMPT)
147
+ if shared is None:
148
+ where = ", ".join(str(d) for d in dirs) or "(no prompt directories)"
149
+ raise FileNotFoundError(f"no {SHARED_PROMPT} in: {where}")
150
+
151
+ blocks = [own.read_text(encoding="utf-8").rstrip()]
152
+ # An empty addendum contributes nothing rather than a stray blank block --
153
+ # `touch API.append.md` must not change a single byte of the prompt.
154
+ blocks += [t for p in append_paths(dirs, spec.prompt) if (t := p.read_text(encoding="utf-8").strip())]
155
+ blocks.append(shared.read_text(encoding="utf-8").strip())
156
+ return "\n\n".join(blocks) + "\n"
157
+
158
+
159
+ class MissingServerEnv(RuntimeError):
160
+ """A declared server references an environment variable that is not set."""
161
+
162
+
163
+ def expand_env(value: str, *, where: str) -> str:
164
+ """Substitute `${VAR}` from the environment, loudly.
165
+
166
+ The CLI would do this itself -- `--mcp-config` is parsed with
167
+ `expandVars` on -- but two things argue for doing it here. An unset
168
+ variable becomes an empty string down there, so a missing token surfaces
169
+ much later as an unexplained auth failure rather than as the missing token
170
+ it is. And it is an implementation detail of a vendored binary found by
171
+ reading it, not a documented contract; `sdk_compat.py` exists because this
172
+ project does not build on those.
173
+
174
+ Expanding here is idempotent with respect to the CLI: a value with no `${`
175
+ left in it is passed through unchanged.
176
+ """
177
+ def replace(match: "re.Match[str]") -> str:
178
+ var = match.group(1)
179
+ got = os.environ.get(var)
180
+ if got is None:
181
+ raise MissingServerEnv(
182
+ f"{where} references ${{{var}}} and it is not set. "
183
+ "Export it, or remove the reference -- a credential belongs in "
184
+ "the environment, never in a config file."
185
+ )
186
+ return got
187
+
188
+ return re.sub(r"\$\{([A-Za-z_][A-Za-z0-9_]*)\}", replace, value)
189
+
190
+
191
+ def _declared_server(name: str, spec: Any) -> dict[str, Any]:
192
+ """Turn a user's YAML declaration into the dict the SDK expects."""
193
+ payload = spec.model_dump(exclude_none=True)
194
+ where = f"MCP server '{name}'"
195
+ for key in ("command", "url"):
196
+ if key in payload:
197
+ payload[key] = expand_env(str(payload[key]), where=where)
198
+ if "args" in payload:
199
+ payload["args"] = [expand_env(str(a), where=where) for a in payload["args"]]
200
+ for key in ("env", "headers"):
201
+ if key in payload:
202
+ payload[key] = {k: expand_env(str(v), where=where) for k, v in payload[key].items()}
203
+ return payload
204
+
205
+
206
+ def build_mcp_servers(spec: AgentSpec, ctx: ToolContext) -> dict[str, Any]:
207
+ """Instantiate exactly the servers this agent declared, and no others.
208
+
209
+ Config-declared servers resolve FIRST, so a project can override a built-in
210
+ -- the bundled Playwright entry is hardcoded down to `--browser chromium`,
211
+ and someone testing Firefox should not have to fork the package to say so.
212
+ """
213
+ declared = dict(getattr(ctx.config, "mcp_servers", {}) or {})
214
+ servers: dict[str, Any] = {}
215
+ for name in spec.mcp_servers:
216
+ if name in declared:
217
+ servers[name] = _declared_server(name, declared[name])
218
+ elif name in SDK_SERVER_MODULES:
219
+ module = importlib.import_module(SDK_SERVER_MODULES[name])
220
+ servers[name] = module.build(ctx)
221
+ elif name in STDIO_SERVERS:
222
+ servers[name] = dict(STDIO_SERVERS[name])
223
+ else:
224
+ raise UnknownServer(
225
+ f"{spec.name} declares MCP server '{name}', which is neither an "
226
+ f"in-process server ({', '.join(sorted(SDK_SERVER_MODULES))}), a "
227
+ f"known stdio server ({', '.join(sorted(STDIO_SERVERS))}), nor "
228
+ f"declared under `mcp_servers:` in system.yaml "
229
+ f"({', '.join(sorted(declared)) or 'nothing declared'})."
230
+ )
231
+ return servers
232
+
233
+
234
+ def build_allowed_tools(spec: AgentSpec) -> list[str]:
235
+ """The allowlist handed to the SDK.
236
+
237
+ Servers are allowed wholesale — the server itself enforces which of its tools
238
+ this agent may use, and it has the context to explain a refusal properly.
239
+ """
240
+ # ALWAYS_GRANTED is shared with the guardrail so the two cannot disagree.
241
+ # Neither ToolSearch nor Skill is a capability grant — they load schemas and
242
+ # instructions for things the agent already has.
243
+ return [
244
+ *spec.builtin_tools,
245
+ *sorted(ALWAYS_GRANTED - set(spec.builtin_tools)),
246
+ *(mcp_server_wildcard(s) for s in spec.mcp_servers),
247
+ ]
248
+
249
+
250
+ class TurnRecord:
251
+ """What an agent has actually done this turn.
252
+
253
+ The Stop hook needs to know which tools were called; nothing else in the SDK
254
+ tracks that for us, so we count them as they go past.
255
+ """
256
+
257
+ def __init__(self) -> None:
258
+ self.called: set[str] = set()
259
+ self.held_envelopes: int = 0
260
+ self.stop_blocks: int = 0
261
+
262
+ def record(self, tool_name: str | None) -> None:
263
+ if tool_name:
264
+ self.called.add(tool_name)
265
+
266
+ def missing(self, required: list[str]) -> list[str]:
267
+ return [t for t in required if t not in self.called]
268
+
269
+
270
+ def build_hooks(
271
+ guard: Guardrail, ctx: ToolContext, record: TurnRecord | None = None
272
+ ) -> dict[str, list[HookMatcher]]:
273
+ """Observability, plus one thing `can_use_tool` structurally cannot do.
274
+
275
+ Permissions are enforced twice, on purpose. `can_use_tool` is primary, but
276
+ the SDK can shadow it (see `CanUseToolShadowedWarning`), so `Guardrail.
277
+ pre_tool_use` re-runs the same `check()` as a hook. Both call one decision
278
+ function, so they cannot disagree — the duplication is in the wiring, not
279
+ in the policy.
280
+
281
+ What hooks add on top is enforcement of the *output contract*. `can_use_tool`
282
+ only ever sees calls that happen, so it can never notice the call that did
283
+ not. The Stop hook can, and it fires while the agent still has a turn left to
284
+ fix it — unlike the router, which only finds out afterwards.
285
+ """
286
+ record = record or TurnRecord()
287
+
288
+ async def on_pre_tool_record(
289
+ input_data: Any, tool_use_id: str | None, context: Any
290
+ ) -> dict[str, Any]:
291
+ """Kept for the ledger's ordering; the contract is counted after the call.
292
+
293
+ Counting here marked a tool as called before anyone knew whether it had
294
+ been *denied* or had returned `isError` — so an errored `record_verdict`
295
+ satisfied VERIFIER's `must_call` and the Stop hook let it stop with no
296
+ verdict. A denial never reaches PostToolUse at all, which is exactly the
297
+ distinction that matters.
298
+ """
299
+ return {}
300
+
301
+ async def on_post_tool(input_data: Any, tool_use_id: str | None, context: Any) -> dict[str, Any]:
302
+ tool = _field(input_data, "tool_name")
303
+ response = _field(input_data, "tool_response")
304
+
305
+ if isinstance(response, dict) and response.get("isError"):
306
+ ctx.store.log("tool_error", agent=ctx.agent.name, tool=tool, tool_use_id=tool_use_id)
307
+ return {}
308
+
309
+ # Reached only by a call that was permitted and did not error, which is
310
+ # the only kind that should satisfy `must_call`.
311
+ record.record(tool)
312
+
313
+ # An envelope that was accepted but held tells the agent nothing unless
314
+ # someone says so now. Discovering at the end that none of your findings
315
+ # counted is too late to attach the missing evidence.
316
+ if tool and tool.endswith("__emit_envelope"):
317
+ structured = _structured(response)
318
+ if structured and structured.get("fileable") is False:
319
+ record.held_envelopes += 1
320
+ return {
321
+ "systemMessage": (
322
+ f"That finding was recorded but is held from filing "
323
+ f"({record.held_envelopes} so far this run). It needs an artifact or a "
324
+ "failing test as evidence, and confidence at or above "
325
+ f"{ctx.config.thresholds.min_confidence_to_file}. Attach evidence with "
326
+ "put_artifact and emit it again, or leave it held deliberately."
327
+ )
328
+ }
329
+ return {}
330
+
331
+ async def on_stop(input_data: Any, tool_use_id: str | None, context: Any) -> dict[str, Any]:
332
+ # `stop_hook_active` is true when this hook already blocked once. Without
333
+ # honouring it, an agent that genuinely cannot satisfy its contract loops
334
+ # until it burns the budget.
335
+ if _field(input_data, "stop_hook_active"):
336
+ ctx.store.log(
337
+ "contract_unmet",
338
+ agent=ctx.agent.name,
339
+ missing=record.missing(ctx.agent.must_call),
340
+ note="allowed to stop after one block",
341
+ )
342
+ return {}
343
+
344
+ missing = record.missing(ctx.agent.must_call)
345
+ if not missing:
346
+ return {}
347
+
348
+ record.stop_blocks += 1
349
+ ctx.store.log("stop_blocked", agent=ctx.agent.name, missing=missing)
350
+ names = ", ".join(t.rsplit("__", 1)[-1] for t in missing)
351
+ return {
352
+ "decision": "block",
353
+ "reason": (
354
+ f"You have not called: {names}. That is {ctx.agent.name}'s deliverable for "
355
+ "this task, not an optional extra — without it this invocation produced "
356
+ "nothing the rest of the system can use. Either call it now, or if you "
357
+ "genuinely cannot, call it with the outcome you did reach and say why."
358
+ ),
359
+ }
360
+
361
+ return {
362
+ PRE_TOOL_USE: [HookMatcher(matcher=None, hooks=[guard.pre_tool_use, on_pre_tool_record])],
363
+ POST_TOOL_USE: [HookMatcher(matcher=None, hooks=[on_post_tool])],
364
+ STOP: [HookMatcher(matcher=None, hooks=[on_stop])],
365
+ }
366
+
367
+
368
+ def _structured(response: Any) -> dict[str, Any] | None:
369
+ """The structuredContent block of an MCP tool result, whatever wraps it."""
370
+ if isinstance(response, dict):
371
+ inner = response.get("structuredContent")
372
+ if isinstance(inner, dict):
373
+ return inner
374
+ return None
375
+
376
+
377
+ def _field(payload: Any, name: str) -> Any:
378
+ """Hook inputs arrive as dicts or dataclasses depending on SDK version."""
379
+ if isinstance(payload, dict):
380
+ return payload.get(name)
381
+ return getattr(payload, name, None)
382
+
383
+
384
+ def skill_plugins(ctx: ToolContext) -> list[dict[str, str]]:
385
+ """The plugin directories to hand the CLI, project first.
386
+
387
+ Absolute paths: `--plugin-dir` takes the value verbatim, so a relative one
388
+ would resolve against the agent's cwd -- the target repository -- and find
389
+ nothing.
390
+ """
391
+ from qaas.paths import Workspace
392
+
393
+ ws = getattr(ctx, "workspace", None) or Workspace.resolve()
394
+ return [{"type": "local", "path": str(d.resolve())} for d in ws.plugin_dirs]
395
+
396
+
397
+ def qualified_skills(spec: AgentSpec, ctx: ToolContext) -> list[str]:
398
+ """This agent's skills, namespaced by the plugin that provides each.
399
+
400
+ The qualification is load-bearing. A skill name travels to the CLI down two
401
+ channels that match differently: the SDK turns each into a `Skill(<name>)`
402
+ entry on `--allowedTools`, matched **literally** against whatever the model
403
+ invokes, while the `initialize` request filters system-prompt content with
404
+ `name === entry || name.endsWith(":" + entry)`. Plugin skills register as
405
+ `qaas:severity-rubric`, so a bare `severity-rubric` satisfies the second
406
+ channel and not the first -- the skill loads, and its allow rule never
407
+ matches. Passing the qualified name makes both agree.
408
+
409
+ A skill no plugin provides is dropped rather than passed through. The Stop
410
+ hook and `qaas validate` both report the real problem; inventing a name that
411
+ can never resolve just moves the failure somewhere quieter.
412
+ """
413
+ from qaas.paths import Workspace
414
+
415
+ ws = getattr(ctx, "workspace", None) or Workspace.resolve()
416
+ return [q for q in (ws.qualify(name) for name in spec.skills) if q]
417
+
418
+
419
+ def build_options(
420
+ spec: AgentSpec,
421
+ ctx: ToolContext,
422
+ *,
423
+ extra_env: dict[str, str] | None = None,
424
+ ) -> ClaudeAgentOptions:
425
+ """Everything one agent needs, assembled from its spec."""
426
+ guard = Guardrail(ctx)
427
+
428
+ env = {
429
+ # Opus delegates readily. An unbounded subagent tree is the fastest route
430
+ # to a surprise bill, so cap depth and width regardless of what it decides.
431
+ "CLAUDE_CODE_MAX_SUBAGENT_SPAWN_DEPTH": "1",
432
+ "CLAUDE_CODE_MAX_CONCURRENT_SUBAGENTS": "3",
433
+ }
434
+ env.update(extra_env or {})
435
+
436
+ return ClaudeAgentOptions(
437
+ # Through the workspace, not `PROMPTS_DIR`: a user's `.qaas/prompts/`
438
+ # override has to reach the agent that actually runs, not just the one
439
+ # `qaas prompts list` describes.
440
+ system_prompt=build_system_prompt(spec, resolve_prompt_dirs(ctx)),
441
+ model=spec.model,
442
+ effort=spec.effort,
443
+ max_turns=spec.max_turns,
444
+ max_budget_usd=spec.max_budget_usd,
445
+ mcp_servers=build_mcp_servers(spec, ctx),
446
+ allowed_tools=build_allowed_tools(spec),
447
+ can_use_tool=guard.can_use_tool,
448
+ hooks=build_hooks(guard, ctx),
449
+ # Skills arrive as a plugin, not through filesystem settings, and the
450
+ # names are qualified because the SDK matches them down two channels
451
+ # with different rules -- see `skill_plugins` and `qualified_skills`.
452
+ plugins=skill_plugins(ctx),
453
+ skills=qualified_skills(spec, ctx),
454
+ cwd=str(ctx.target_root),
455
+ env=env,
456
+ # Load NOTHING from the filesystem. This was `["project"]`, defended on
457
+ # reproducibility grounds -- project settings live in the repo, so they
458
+ # travel with it. That reasoning held only while `cwd` was *our* repo.
459
+ #
460
+ # `cwd` is the target now, and with `qaas run --repo <url>` it can be a
461
+ # repository cloned seconds earlier from a URL someone pasted. "project"
462
+ # means: load that repository's `.claude/settings.json`, its hooks, its
463
+ # permission rules and its MCP servers, into a process holding Anthropic
464
+ # credentials, JIRA_API_TOKEN and GitHub auth. A QA tool that executes
465
+ # the configuration of the code it is inspecting is a supply-chain hole.
466
+ #
467
+ # It must be an explicit `[]`, not None: `_apply_skills_defaults` in the
468
+ # SDK substitutes ["user", "project"] whenever setting_sources is None
469
+ # and skills is a list.
470
+ setting_sources=[],
471
+ permission_mode="default",
472
+ )
473
+
474
+
475
+ def describe(
476
+ spec: AgentSpec, prompt_dirs: Sequence[Path] | None = None
477
+ ) -> dict[str, Any]:
478
+ """A dry-run view of what this agent would be given. No API call.
479
+
480
+ `prompt_dirs` so the dry run counts the prompt the real run would send. A
481
+ dry run that silently reports the packaged prompt while the run sends an
482
+ overridden one is worse than no dry run.
483
+ """
484
+ return {
485
+ "agent": spec.name,
486
+ "model": spec.model,
487
+ "effort": spec.effort,
488
+ "max_turns": spec.max_turns,
489
+ "max_budget_usd": spec.max_budget_usd,
490
+ "mcp_servers": list(spec.mcp_servers),
491
+ "allowed_tools": build_allowed_tools(spec),
492
+ "prompt_chars": len(build_system_prompt(spec, prompt_dirs)),
493
+ "skills": list(spec.skills),
494
+ "must_call": list(spec.must_call),
495
+ "policy": spec.policy.model_dump(),
496
+ }