qaas-python 0.0.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (96) hide show
  1. qaas/adapters/__init__.py +19 -0
  2. qaas/adapters/tracker.py +1783 -0
  3. qaas/adapters/vcs.py +555 -0
  4. qaas/cli.py +1757 -0
  5. qaas/config.py +409 -0
  6. qaas/defaults/config/agents/api.yaml +18 -0
  7. qaas/defaults/config/agents/architect.yaml +21 -0
  8. qaas/defaults/config/agents/auditor.yaml +19 -0
  9. qaas/defaults/config/agents/browser.yaml +15 -0
  10. qaas/defaults/config/agents/dba.yaml +20 -0
  11. qaas/defaults/config/agents/fixer.yaml +55 -0
  12. qaas/defaults/config/agents/guide.yaml +23 -0
  13. qaas/defaults/config/agents/load.yaml +26 -0
  14. qaas/defaults/config/agents/mapper.yaml +19 -0
  15. qaas/defaults/config/agents/reporter.yaml +19 -0
  16. qaas/defaults/config/agents/reproducer.yaml +21 -0
  17. qaas/defaults/config/agents/reviewer.yaml +18 -0
  18. qaas/defaults/config/agents/socket.yaml +23 -0
  19. qaas/defaults/config/agents/triage.yaml +20 -0
  20. qaas/defaults/config/agents/verifier.yaml +20 -0
  21. qaas/defaults/config/system.yaml +64 -0
  22. qaas/discover.py +242 -0
  23. qaas/envelope.py +318 -0
  24. qaas/envfile.py +100 -0
  25. qaas/guardrails.py +589 -0
  26. qaas/mcp/__init__.py +0 -0
  27. qaas/mcp/context.py +78 -0
  28. qaas/mcp/contract_diff.py +1011 -0
  29. qaas/mcp/defect_memory.py +495 -0
  30. qaas/mcp/env_control.py +925 -0
  31. qaas/mcp/envelope_server.py +463 -0
  32. qaas/mcp/test_runner.py +842 -0
  33. qaas/mcp/tracker.py +420 -0
  34. qaas/mcp/vcs.py +501 -0
  35. qaas/paths.py +317 -0
  36. qaas/plugin/.claude-plugin/plugin.json +9 -0
  37. qaas/plugin/skills/a11y-audit/SKILL.md +34 -0
  38. qaas/plugin/skills/adversarial-review/SKILL.md +120 -0
  39. qaas/plugin/skills/api-surface-extraction/SKILL.md +38 -0
  40. qaas/plugin/skills/authz-matrix-check/SKILL.md +46 -0
  41. qaas/plugin/skills/console-error-triage/SKILL.md +39 -0
  42. qaas/plugin/skills/contract-test-generation/SKILL.md +36 -0
  43. qaas/plugin/skills/dedupe-strategy/SKILL.md +39 -0
  44. qaas/plugin/skills/environment-pinning/SKILL.md +35 -0
  45. qaas/plugin/skills/error-taxonomy/SKILL.md +42 -0
  46. qaas/plugin/skills/exploratory-ui-walk/SKILL.md +46 -0
  47. qaas/plugin/skills/failing-test-authoring/SKILL.md +47 -0
  48. qaas/plugin/skills/flake-detection/SKILL.md +39 -0
  49. qaas/plugin/skills/form-state-probe/SKILL.md +36 -0
  50. qaas/plugin/skills/minimal-diff-discipline/SKILL.md +70 -0
  51. qaas/plugin/skills/openapi-diff/SKILL.md +45 -0
  52. qaas/plugin/skills/ownership-resolution/SKILL.md +31 -0
  53. qaas/plugin/skills/product-task-graph/SKILL.md +35 -0
  54. qaas/plugin/skills/regression-risk-scoring/SKILL.md +59 -0
  55. qaas/plugin/skills/regression-suite-selection/SKILL.md +36 -0
  56. qaas/plugin/skills/repo-cartography/SKILL.md +38 -0
  57. qaas/plugin/skills/repro-minimisation/SKILL.md +41 -0
  58. qaas/plugin/skills/rollback-plan-authoring/SKILL.md +81 -0
  59. qaas/plugin/skills/root-cause-vs-symptom/SKILL.md +67 -0
  60. qaas/plugin/skills/routing-rules/SKILL.md +34 -0
  61. qaas/plugin/skills/severity-rubric/SKILL.md +42 -0
  62. qaas/plugin/skills/test-first-fix/SKILL.md +66 -0
  63. qaas/plugin/skills/test-quality-audit/SKILL.md +58 -0
  64. qaas/plugin/skills/ticket-writer/SKILL.md +40 -0
  65. qaas/plugin/skills/verdict-reporting/SKILL.md +35 -0
  66. qaas/plugin/skills/verification-protocol/SKILL.md +39 -0
  67. qaas/prompts/API.md +44 -0
  68. qaas/prompts/ARCHITECT.md +80 -0
  69. qaas/prompts/AUDITOR.md +62 -0
  70. qaas/prompts/BROWSER.md +46 -0
  71. qaas/prompts/DBA.md +59 -0
  72. qaas/prompts/FIXER.md +55 -0
  73. qaas/prompts/GUIDE.md +94 -0
  74. qaas/prompts/LOAD.md +109 -0
  75. qaas/prompts/MAPPER.md +46 -0
  76. qaas/prompts/REPORTER.md +61 -0
  77. qaas/prompts/REPRODUCER.md +43 -0
  78. qaas/prompts/REVIEWER.md +53 -0
  79. qaas/prompts/SOCKET.md +100 -0
  80. qaas/prompts/TRIAGE.md +45 -0
  81. qaas/prompts/VERIFIER.md +41 -0
  82. qaas/prompts/_shared.md +45 -0
  83. qaas/registry.py +496 -0
  84. qaas/router.py +581 -0
  85. qaas/runner.py +210 -0
  86. qaas/scorecard.py +448 -0
  87. qaas/sdk_compat.py +52 -0
  88. qaas/store.py +323 -0
  89. qaas/target.py +287 -0
  90. qaas/tasks.py +438 -0
  91. qaas/trace.py +342 -0
  92. qaas_python-0.0.1.dist-info/METADATA +429 -0
  93. qaas_python-0.0.1.dist-info/RECORD +96 -0
  94. qaas_python-0.0.1.dist-info/WHEEL +4 -0
  95. qaas_python-0.0.1.dist-info/entry_points.txt +2 -0
  96. qaas_python-0.0.1.dist-info/licenses/LICENSE +21 -0
qaas/guardrails.py ADDED
@@ -0,0 +1,589 @@
1
+ """The §8.1 write-permission matrix, enforced in code.
2
+
3
+ Every agent gets a `can_use_tool` callback built from its policy. The callback
4
+ sees the tool name and its arguments before the tool runs, which is the only
5
+ place a limit like "REPRODUCER may write, but only under qa/repro" can actually be
6
+ imposed. A prompt asking an agent not to do something is a request; this is a
7
+ decision.
8
+
9
+ Enforcement runs in the **PreToolUse hook**, not in `can_use_tool`. This is not
10
+ a stylistic choice and it is easy to get wrong: an `allowed_tools` entry that
11
+ names a whole tool auto-approves it *before* `can_use_tool` is consulted, so a
12
+ policy implemented only in that callback is silently never applied. The SDK warns
13
+ about this shadowing, and an early version of this file had exactly that bug —
14
+ REPRODUCER's sandbox check was dead code. The hook sees every call regardless.
15
+
16
+ `can_use_tool` is kept as a second layer, for anything that falls outside the
17
+ allowlist and so reaches the callback normally.
18
+
19
+ Three belts, then:
20
+
21
+ * The PreToolUse hook — every built-in tool call, gated on this agent's policy.
22
+ * `can_use_tool` — the same decision, for calls not auto-approved.
23
+ * The MCP servers — their own domain rules (the tracker refuses an agent that
24
+ may not file; vcs refuses a branch outside the agent's patterns).
25
+
26
+ Denials return a reason rather than killing the turn: an agent that learns it
27
+ cannot write to a path should adapt, and the reason is what lets it. Every
28
+ denial lands in the run ledger, which is the audit trail §8 asks for.
29
+ """
30
+
31
+ from __future__ import annotations
32
+
33
+ import fnmatch
34
+ import re
35
+ import shlex
36
+ from dataclasses import dataclass
37
+ from pathlib import Path
38
+ from typing import Any
39
+
40
+ from claude_agent_sdk import (
41
+ PermissionResultAllow,
42
+ PermissionResultDeny,
43
+ ToolPermissionContext,
44
+ )
45
+
46
+ from qaas.mcp.context import ToolContext
47
+
48
+ # Harness plumbing granted to every agent, independent of its config. These are
49
+ # not capability grants: ToolSearch only loads the schemas of servers the agent
50
+ # already has, Skill only loads instructions, and neither can reach anything the
51
+ # allowlist does not already permit. `build_allowed_tools` adds the same set, and
52
+ # both read this constant so the allowlist and the guardrail cannot drift apart —
53
+ # a mismatch here silently disables every skill in the system.
54
+ ALWAYS_GRANTED = frozenset({"ToolSearch", "Skill", "TodoWrite", "Task", "Agent"})
55
+
56
+ # Tools that read. Always safe, for every agent.
57
+ READ_TOOLS = {"Read", "Grep", "Glob", "NotebookRead"} | set(ALWAYS_GRANTED)
58
+
59
+ # Harness plumbing, not capability. These grant an agent nothing it was not
60
+ # already granted — ToolSearch only loads the schema of a tool that is already
61
+ # on its allowlist, and Skill only opens a skill file. Denying ToolSearch is
62
+ # worse than useless: MCP tools arrive deferred, so an agent that cannot call it
63
+ # cannot reach the servers it was given, and burns its whole turn budget
64
+ # discovering that. This system did exactly that once.
65
+ HARNESS_TOOLS = {"ToolSearch", "TodoWrite", "Task", "Agent", "Skill", "SlashCommand"}
66
+
67
+ # Tools that write to the filesystem. Gated on policy.write_paths.
68
+ WRITE_TOOLS = {"Write", "Edit", "MultiEdit", "NotebookEdit"}
69
+
70
+ # Bash command prefixes that are never allowed, whatever the agent.
71
+ # Merging to main, force-pushing and recursive deletes are outside every
72
+ # agent's remit in this system: merge is always human (§8.4), and nothing here
73
+ # needs to delete a tree.
74
+ FORBIDDEN_BASH = [
75
+ (r"\bgit\s+push\b.*(--force|-f\b)", "force-push is never permitted"),
76
+ (r"\bgit\s+push\b.*\b(main|master)\b", "pushing to main is never permitted"),
77
+ (r"\bgit\s+merge\b", "merging is a human decision (§8.4)"),
78
+ (r"\bgit\s+reset\s+--hard\b", "hard reset discards work outside the sandbox"),
79
+ (r"\bgit\s+checkout\s+(main|master)\b", "agents work on their own branches only"),
80
+ # `-[a-zA-Z]*[rf]` caught `-rf` and `-r`, and missed `-R`, `--recursive` and
81
+ # `--force` — three spellings of the same command, none of them exotic.
82
+ (
83
+ r"\brm\s+(-[a-zA-Z]*[rfR]|--(recursive|force)\b)",
84
+ "recursive or forced delete is not permitted",
85
+ ),
86
+ (r"\bsudo\b", "privilege escalation is not permitted"),
87
+ (r"\b(shutdown|reboot|mkfs|dd)\b", "destructive system command"),
88
+ (r">\s*/dev/(sd|nvme|disk)", "writing to a block device"),
89
+ (r"\bdocker\s+system\s+prune", "prune would destroy state other runs depend on"),
90
+ (r"\bgh\s+pr\s+merge\b", "merging a pull request is a human decision (§8.4)"),
91
+ ]
92
+
93
+ # Bash that mutates git state. Gated on the agent having branch patterns at all.
94
+ GIT_WRITE = re.compile(r"\bgit\s+(commit|push|branch|checkout\s+-b|switch\s+-c|tag|apply|am|rebase)\b")
95
+
96
+ # Shell constructs that rewrite a file in place. `>` is not a word character, so
97
+ # this deliberately does not use \b anchors — an earlier version did and silently
98
+ # matched nothing.
99
+ _MUTATES_FILE = re.compile(r"(>>?|\btee\b|\bsed\s+-i|\btruncate\b|\bdd\b)")
100
+
101
+ # -- what a shell command writes --------------------------------------------
102
+ #
103
+ # `_check_bash` used to consult only FORBIDDEN_BASH, the branch patterns and a
104
+ # substring test against `protected_paths`. Every other rule in the §8.1 matrix
105
+ # — `write_paths`, `forbidden_paths`, the §8.2 diff budget — was enforced for
106
+ # `Write`/`Edit` and bypassed entirely by a shell command:
107
+ #
108
+ # FIXER Write api/app/auth.py -> denied
109
+ # FIXER sed -i '' s/x/y/ api/app/auth.py -> allowed
110
+ # VERIFIER tee /etc/hosts < x -> allowed
111
+ #
112
+ # VERIFIER is the verification gate and has `Bash` with no `write_paths`, so it was
113
+ # "read-only" only against `Write` — §2's finder/fixer separation gone.
114
+ #
115
+ # Reading a shell command is best-effort by nature: a write can always hide one
116
+ # level of indirection further than a parser follows. The rule that makes that
117
+ # acceptable is below — a command that mutates and whose destination cannot be
118
+ # resolved is *refused*, not guessed at, which leaves the enforced tools
119
+ # (`Write`/`Edit`) as the only way to do the thing.
120
+
121
+ #: A command whose every non-flag argument is a destination.
122
+ _WRITES_ALL_ARGS = frozenset({"tee", "touch", "truncate"})
123
+ #: A command whose last argument is the destination.
124
+ _WRITES_LAST_ARG = frozenset({"cp", "mv", "install", "ln", "rsync"})
125
+ #: Interpreters that write wherever an inline script tells them to.
126
+ _SCRIPT_RUNNERS = frozenset({"python", "python3", "perl", "ruby", "node", "bash", "sh", "zsh", "php"})
127
+ _INLINE_SCRIPT_FLAGS = frozenset({"-c", "-e"})
128
+ #: Flags that take a value, so the value is not mistaken for a path.
129
+ _FLAG_TAKES_VALUE = frozenset({"-s", "-m", "-o", "-t", "--suffix", "--size", "--mode"})
130
+
131
+ #: Redirect targets that are not files anyone can be harmed through.
132
+ _DEV_SINKS = frozenset({"/dev/null", "/dev/stdout", "/dev/stderr", "/dev/tty"})
133
+
134
+ # `> out`, `>>out`, `2> err` — but not `2>&1`, whose `&1` is a file descriptor.
135
+ _REDIRECT_RE = re.compile(r">>?\s*(?!&)([^\s;|&<>]+)")
136
+ # Where one command ends and the next begins.
137
+ _SEGMENT_RE = re.compile(r"\|\||&&|[;|\n&]")
138
+
139
+
140
+ @dataclass
141
+ class Decision:
142
+ allowed: bool
143
+ reason: str = ""
144
+
145
+
146
+ class Guardrail:
147
+ """One agent's enforcement of its own policy."""
148
+
149
+ def __init__(self, ctx: ToolContext):
150
+ self.ctx = ctx
151
+ self.agent = ctx.agent
152
+ self.policy = ctx.agent.policy
153
+ # Every write path in a policy is relative to the *application under
154
+ # test*, never to the qaas project. This was `ctx.repo_root`, filled
155
+ # from `Path.cwd()`, which anchored the whole allowlist on wherever the
156
+ # operator happened to be standing -- harmless only while the target was
157
+ # a subdirectory of the qaas checkout. With `qaas run --repo <url>` the
158
+ # target is a clone under `.qaas/targets/`, and an allowlist anchored on
159
+ # the cwd would deny every legitimate write and permit a sandbox that
160
+ # sits inside qaas's own source.
161
+ self.root = ctx.target_root.resolve()
162
+ # A glob cannot be resolved to a directory to contain things in, so the
163
+ # two kinds are kept apart and tested differently. `mcp/vcs.py` already
164
+ # honoured globs in `write_paths` and this did not; now that both go
165
+ # through `_check_path`, the more permissive of the two would have
166
+ # silently become the rule for everything.
167
+ self._allowed_roots = [
168
+ (self.root / p).resolve() for p in self.policy.write_paths if not _is_glob(p)
169
+ ]
170
+ self._allowed_globs = [p for p in self.policy.write_paths if _is_glob(p)]
171
+ # Every MCP server the agent declared, as an allowlist prefix.
172
+ self._mcp_prefixes = tuple(f"mcp__{s}__" for s in self.agent.mcp_servers)
173
+
174
+ # -- entry point ------------------------------------------------------
175
+
176
+ async def can_use_tool(
177
+ self,
178
+ tool_name: str,
179
+ input_data: dict[str, Any],
180
+ context: ToolPermissionContext,
181
+ ) -> PermissionResultAllow | PermissionResultDeny:
182
+ """Second layer. Reached only for calls the allowlist did not auto-approve."""
183
+ decision = self.check(tool_name, input_data)
184
+ if decision.allowed:
185
+ return PermissionResultAllow(updated_input=input_data)
186
+ self._record(tool_name, input_data, decision.reason, via="can_use_tool")
187
+ return PermissionResultDeny(message=decision.reason)
188
+
189
+ async def pre_tool_use(
190
+ self,
191
+ payload: Any,
192
+ tool_use_id: str | None,
193
+ context: Any,
194
+ ) -> dict[str, Any]:
195
+ """Primary enforcement. Runs for every tool call, shadowing or not."""
196
+ tool_name = _hook_field(payload, "tool_name") or ""
197
+ input_data = _hook_field(payload, "tool_input") or {}
198
+ if not isinstance(input_data, dict):
199
+ input_data = {}
200
+
201
+ decision = self.check(tool_name, input_data)
202
+ # A refused call produces TWO ledger entries, and that is deliberate.
203
+ # `tool_call` is the universal record -- every call this agent made, in
204
+ # order, allowed or not -- and it is what a timeline reads. `denial`
205
+ # below carries the reason and a summary of the arguments, and is the
206
+ # only place tool arguments are recorded at all.
207
+ #
208
+ # It looks like double-counting and has been reported as such. Dropping
209
+ # either one loses something real: without the `tool_call` the refusal
210
+ # vanishes from the call sequence, and without the `denial` nobody can
211
+ # say why. A reader tallying refusals should count `denial`, not both.
212
+ self.ctx.store.log(
213
+ "tool_call",
214
+ agent=self.agent.name,
215
+ tool=tool_name,
216
+ tool_use_id=tool_use_id,
217
+ allowed=decision.allowed,
218
+ )
219
+ if decision.allowed:
220
+ return {}
221
+
222
+ self._record(tool_name, input_data, decision.reason, via="hook")
223
+ return {
224
+ "hookSpecificOutput": {
225
+ "hookEventName": "PreToolUse",
226
+ "permissionDecision": "deny",
227
+ "permissionDecisionReason": decision.reason,
228
+ }
229
+ }
230
+
231
+ def _record(self, tool_name: str, input_data: dict[str, Any], reason: str, *, via: str) -> None:
232
+ self.ctx.store.log(
233
+ "denial",
234
+ agent=self.agent.name,
235
+ tool=tool_name,
236
+ reason=reason,
237
+ via=via,
238
+ args=_summarise(input_data),
239
+ )
240
+
241
+ def check(self, tool_name: str, input_data: dict[str, Any]) -> Decision:
242
+ """Pure policy evaluation. Separated from the callback so it is testable."""
243
+ if tool_name.startswith("mcp__"):
244
+ return self._check_mcp(tool_name)
245
+ if tool_name in READ_TOOLS:
246
+ return self._check_declared(tool_name)
247
+ if tool_name in WRITE_TOOLS:
248
+ # Policy before allowlist: "you are read-only" is the true reason and
249
+ # the useful one. "not in your allowlist" would be technically correct
250
+ # and would send the agent looking for the wrong fix.
251
+ if not self.policy.write_paths:
252
+ return self._read_only_decision()
253
+ declared = self._check_declared(tool_name)
254
+ return declared if not declared.allowed else self._check_write(input_data)
255
+ if tool_name == "Bash":
256
+ declared = self._check_declared(tool_name)
257
+ return declared if not declared.allowed else self._check_bash(input_data)
258
+ if tool_name in {"WebFetch", "WebSearch"}:
259
+ return Decision(
260
+ False,
261
+ f"{self.agent.name} has no network research remit. "
262
+ "Findings come from the code and the running app, not the web.",
263
+ )
264
+ return self._check_declared(tool_name)
265
+
266
+ # -- individual gates -------------------------------------------------
267
+
268
+ def _check_declared(self, tool_name: str) -> Decision:
269
+ """Defence in depth: the tool must be one this agent declared.
270
+
271
+ The SDK is already told the allowlist, so reaching here means something
272
+ upstream drifted. Better to deny and log than to trust the setup.
273
+ """
274
+ if tool_name in self.agent.builtin_tools or tool_name in HARNESS_TOOLS:
275
+ return Decision(True)
276
+ return Decision(
277
+ False,
278
+ f"{tool_name} is not in {self.agent.name}'s tool allowlist "
279
+ f"({', '.join(self.agent.builtin_tools) or 'none'}).",
280
+ )
281
+
282
+ def _check_mcp(self, tool_name: str) -> Decision:
283
+ if tool_name.startswith(self._mcp_prefixes):
284
+ return Decision(True)
285
+ server = tool_name.split("__")[1] if "__" in tool_name else "?"
286
+ return Decision(
287
+ False,
288
+ f"{self.agent.name} is not connected to the '{server}' server. "
289
+ f"Its servers are: {', '.join(self.agent.mcp_servers) or 'none'}.",
290
+ )
291
+
292
+ def _read_only_decision(self) -> Decision:
293
+ """One wording, wherever an agent with no write paths tries to write.
294
+
295
+ The reason an agent reads must not depend on which door it reached for:
296
+ it is the same policy, and a different sentence per tool reads as a
297
+ different rule and invites it to go looking for the permissive one.
298
+ """
299
+ return Decision(
300
+ False,
301
+ f"{self.agent.name} is read-only. Report what you found; "
302
+ "fixing is another agent's job (§2: the finder never fixes).",
303
+ )
304
+
305
+ def _check_write(self, input_data: dict[str, Any]) -> Decision:
306
+ raw = input_data.get("file_path") or input_data.get("path") or input_data.get("notebook_path")
307
+ if not raw:
308
+ return Decision(False, "write refused: no file path in the call")
309
+ return self._check_path(str(raw))
310
+
311
+ def _check_path(self, raw: str) -> Decision:
312
+ """The §8.1/§8.2 matrix, applied to one path an agent wants to write.
313
+
314
+ Split out of `_check_write` so it is the single place the rules live:
315
+ `Write`/`Edit` reach it through `check`, a shell command reaches it
316
+ through `_check_bash`, and `mcp/vcs.py` reaches it instead of keeping
317
+ its own near-copy. Three doors, one answer.
318
+ """
319
+ if not self.policy.write_paths:
320
+ return self._read_only_decision()
321
+
322
+ target = Path(raw)
323
+ resolved = (target if target.is_absolute() else self.root / target).resolve()
324
+
325
+ try:
326
+ relative = resolved.relative_to(self.root).as_posix()
327
+ except ValueError:
328
+ relative = resolved.as_posix()
329
+
330
+ # The autonomy envelope (§8.2) comes first. A path inside the sandbox but
331
+ # in a forbidden class must still be refused, and the reason must name
332
+ # the class so the agent escalates rather than looking for a way round.
333
+ forbidden = self._forbidden_class(relative)
334
+ if forbidden:
335
+ return Decision(
336
+ False,
337
+ f"{relative} is outside {self.agent.name}'s autonomy envelope: it is "
338
+ f"{forbidden}. Changes here need human approval (§8.2). Describe the "
339
+ "change you would make and escalate instead of making it.",
340
+ )
341
+
342
+ protected = self._protected_path(relative)
343
+ if protected:
344
+ return Decision(
345
+ False,
346
+ f"{relative} is the test that defines success for this ticket and may "
347
+ "not be edited (§10: a fixer that edits the test patches the symptom). "
348
+ "If you believe the test itself is wrong, that is an escalation.",
349
+ )
350
+
351
+ for allowed in self._allowed_roots:
352
+ if resolved == allowed or resolved.is_relative_to(allowed):
353
+ return self._check_diff_budget(relative)
354
+ for pattern in self._allowed_globs:
355
+ if fnmatch.fnmatch(relative, pattern):
356
+ return self._check_diff_budget(relative)
357
+ return Decision(
358
+ False,
359
+ f"write refused: {resolved} is outside {self.agent.name}'s sandbox "
360
+ f"({', '.join(self.policy.write_paths)}).",
361
+ )
362
+
363
+ def _forbidden_class(self, relative: str) -> str | None:
364
+ """Which §8.2 class this path falls into, if any."""
365
+ for pattern in self.policy.forbidden_paths:
366
+ if fnmatch.fnmatch(relative, pattern) or fnmatch.fnmatch(Path(relative).name, pattern):
367
+ return _describe_forbidden(pattern)
368
+ return None
369
+
370
+ def _protected_path(self, relative: str) -> bool:
371
+ return any(
372
+ fnmatch.fnmatch(relative, p) or relative.endswith(p)
373
+ for p in self.policy.protected_paths
374
+ )
375
+
376
+ def _check_diff_budget(self, relative: str) -> Decision:
377
+ """Cap how much one agent may change in a single run (§8.2).
378
+
379
+ Counted per distinct file touched, not per call: an agent editing the
380
+ same file six times has made one file's worth of change, and counting
381
+ calls would refuse a perfectly ordinary iteration.
382
+ """
383
+ max_files = self.policy.max_diff_files
384
+ if max_files is None:
385
+ return Decision(True)
386
+
387
+ touched = self.ctx.touched_files
388
+ if relative in touched:
389
+ return Decision(True)
390
+ if len(touched) >= max_files:
391
+ return Decision(
392
+ False,
393
+ f"{self.agent.name} has already changed {len(touched)} files, which is "
394
+ f"its limit of {max_files} (§8.2). A fix this wide is outside the "
395
+ "autonomy envelope: stop, and escalate with what you have found. "
396
+ f"Already touched: {', '.join(sorted(touched))}.",
397
+ )
398
+ touched.add(relative)
399
+ return Decision(True)
400
+
401
+ def _check_bash(self, input_data: dict[str, Any]) -> Decision:
402
+ command = str(input_data.get("command", ""))
403
+ if not command.strip():
404
+ return Decision(False, "empty command")
405
+
406
+ for pattern, why in FORBIDDEN_BASH:
407
+ if re.search(pattern, command):
408
+ return Decision(False, f"command refused: {why}")
409
+
410
+ if GIT_WRITE.search(command) and not self.policy.branch_patterns:
411
+ return Decision(
412
+ False,
413
+ f"{self.agent.name} may not modify git state. "
414
+ "It has no branch patterns in its policy.",
415
+ )
416
+
417
+ if self.policy.branch_patterns:
418
+ branch = _branch_from_command(command)
419
+ if branch and not any(
420
+ fnmatch.fnmatch(branch, pat) for pat in self.policy.branch_patterns
421
+ ):
422
+ return Decision(
423
+ False,
424
+ f"branch '{branch}' is outside {self.agent.name}'s patterns "
425
+ f"({', '.join(self.policy.branch_patterns)}).",
426
+ )
427
+
428
+ if self.policy.protected_paths:
429
+ for protected in self.policy.protected_paths:
430
+ if protected in command and _MUTATES_FILE.search(command):
431
+ return Decision(
432
+ False,
433
+ f"'{protected}' is protected: it defines what a fix must achieve "
434
+ "and may not be edited (§10, symptom fixes).",
435
+ )
436
+
437
+ return self._check_bash_writes(command)
438
+
439
+ def _check_bash_writes(self, command: str) -> Decision:
440
+ """Apply the write matrix to what the command would actually write.
441
+
442
+ This is the half `_check_bash` never had. Every rule below already held
443
+ for `Write` and `Edit`; reaching them through a shell is not a different
444
+ permission, so it does not get a different answer.
445
+ """
446
+ for segment in _shell_segments(command):
447
+ paths, undeterminable = _writes_of(segment)
448
+ if undeterminable:
449
+ if not self.policy.write_paths:
450
+ return self._read_only_decision()
451
+ return Decision(
452
+ False,
453
+ f"refusing '{segment.strip()}': {undeterminable}, so this cannot be "
454
+ "checked against your write paths. Use Write or Edit, which name the "
455
+ "file they change.",
456
+ )
457
+ for raw in paths:
458
+ decision = self._check_path(raw)
459
+ if not decision.allowed:
460
+ return decision
461
+ return Decision(True)
462
+
463
+
464
+ def _is_glob(pattern: str) -> bool:
465
+ return any(ch in pattern for ch in "*?[")
466
+
467
+
468
+ def _shell_segments(command: str) -> list[str]:
469
+ """One command per element, so `ls && sed -i ...` is two things, not one."""
470
+ return [segment.strip() for segment in _SEGMENT_RE.split(command) if segment.strip()]
471
+
472
+
473
+ def _non_flag_args(parts: list[str]) -> list[str]:
474
+ """argv[1:] with options — and the values they consume — removed."""
475
+ args: list[str] = []
476
+ skip = False
477
+ for token in parts[1:]:
478
+ if skip:
479
+ skip = False
480
+ continue
481
+ if token.startswith("-"):
482
+ skip = token in _FLAG_TAKES_VALUE
483
+ continue
484
+ args.append(token)
485
+ return args
486
+
487
+
488
+ def _writes_of(segment: str) -> tuple[list[str], str | None]:
489
+ """What one command writes: (paths, why the destination is undeterminable).
490
+
491
+ A non-None second element means "this mutates and I cannot say where", which
492
+ the caller must treat as a refusal rather than as an empty path list. The two
493
+ are deliberately different: no paths and no reason means the command writes
494
+ nothing and is none of our business.
495
+ """
496
+ targets = [t for t in _REDIRECT_RE.findall(segment) if t not in _DEV_SINKS]
497
+ try:
498
+ parts = shlex.split(segment)
499
+ except ValueError:
500
+ # An unbalanced quote. We cannot read it, so we cannot clear it.
501
+ return targets, "it cannot be parsed as a shell command"
502
+ if not parts:
503
+ return targets, None
504
+
505
+ name = Path(parts[0]).name
506
+ args = _non_flag_args(parts)
507
+
508
+ if name in _SCRIPT_RUNNERS and any(flag in parts for flag in _INLINE_SCRIPT_FLAGS):
509
+ return targets, f"an inline {name} script can write anywhere"
510
+ if name == "patch" or (name == "git" and "apply" in parts[1:3]):
511
+ return targets, "a patch carries its own destinations"
512
+ if name == "sed" and any(part.startswith("-i") for part in parts):
513
+ # `sed -i '' s/x/y/ f.py` (BSD) and `sed -i s/x/y/ f.py` (GNU) differ by
514
+ # an empty argument. Drop the empties, then drop the script expression;
515
+ # whatever is left is a file being rewritten in place.
516
+ non_empty = [a for a in args if a]
517
+ return targets + non_empty[1:], None
518
+ if name in _WRITES_ALL_ARGS:
519
+ return targets + args, None
520
+ if name in _WRITES_LAST_ARG and args:
521
+ return targets + args[-1:], None
522
+ return targets, None
523
+
524
+
525
+ def _branch_from_command(command: str) -> str | None:
526
+ """Best-effort branch name out of a git command, for policy matching."""
527
+ try:
528
+ parts = shlex.split(command)
529
+ except ValueError:
530
+ return None
531
+ # Only a git command names a branch. Without this, `-c` was read as
532
+ # `switch -c` in anything that happens to take one: `python -c '...'` was
533
+ # refused as "branch 'open(...)' is outside FIXER's patterns", which is
534
+ # both a wrong answer and an unactionable one.
535
+ if not any(Path(part).name == "git" for part in parts[:2]):
536
+ return None
537
+ for i, token in enumerate(parts):
538
+ if token in {"-b", "-c"} and i + 1 < len(parts):
539
+ return parts[i + 1]
540
+ if token in {"branch", "switch"} and i + 1 < len(parts):
541
+ candidate = parts[i + 1]
542
+ if not candidate.startswith("-"):
543
+ return candidate
544
+ return None
545
+
546
+
547
+ # Human-readable names for the forbidden classes, so a denial explains itself.
548
+ _FORBIDDEN_DESCRIPTIONS = [
549
+ ("migration", "a database migration"),
550
+ ("auth", "authentication or authorization code"),
551
+ ("payment", "a payment path"),
552
+ ("billing", "a billing path"),
553
+ ("secret", "secret material"),
554
+ ("infra", "infrastructure configuration"),
555
+ ("terraform", "infrastructure configuration"),
556
+ (".tf", "infrastructure configuration"),
557
+ ("docker", "container or deployment configuration"),
558
+ ("k8s", "container or deployment configuration"),
559
+ ("kube", "container or deployment configuration"),
560
+ (".github", "CI configuration"),
561
+ ("workflow", "CI configuration"),
562
+ ]
563
+
564
+
565
+ def _describe_forbidden(pattern: str) -> str:
566
+ lowered = pattern.lower()
567
+ for needle, description in _FORBIDDEN_DESCRIPTIONS:
568
+ if needle in lowered:
569
+ return description
570
+ return f"matched by the forbidden pattern `{pattern}`"
571
+
572
+
573
+ def _hook_field(payload: Any, name: str) -> Any:
574
+ """Hook payloads arrive as dicts or dataclasses depending on SDK version."""
575
+ if isinstance(payload, dict):
576
+ return payload.get(name)
577
+ return getattr(payload, name, None)
578
+
579
+
580
+ def _summarise(input_data: dict[str, Any], limit: int = 200) -> dict[str, Any]:
581
+ """Ledger-sized view of a tool call: enough to audit, not a content dump."""
582
+ out: dict[str, Any] = {}
583
+ for key, value in input_data.items():
584
+ if key in {"content", "new_string", "old_string"}:
585
+ out[key] = f"<{len(str(value))} chars>"
586
+ else:
587
+ text = str(value)
588
+ out[key] = text if len(text) <= limit else text[:limit] + "…"
589
+ return out
qaas/mcp/__init__.py ADDED
File without changes
qaas/mcp/context.py ADDED
@@ -0,0 +1,78 @@
1
+ """Shared state the in-process MCP servers close over.
2
+
3
+ Every tool call lands in this process, so the servers can reach the run store,
4
+ the config and the target app directly. That is the point of building them
5
+ in-process: validation, guardrails and persistence happen where the state
6
+ already lives, with no serialisation boundary to reason about.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ from dataclasses import dataclass, field
12
+ from pathlib import Path
13
+ from typing import Any
14
+
15
+ from qaas.config import AgentSpec, SystemConfig
16
+ from qaas.store import RunStore, SystemMapStore
17
+
18
+
19
+ @dataclass
20
+ class ToolContext:
21
+ """One agent's view of the run. Rebuilt per agent invocation."""
22
+
23
+ store: RunStore
24
+ maps: SystemMapStore
25
+ config: SystemConfig
26
+ agent: AgentSpec
27
+
28
+ #: The application under test, on disk. Named `repo_root` once, and that
29
+ #: name was the bug: it read as "the qaas checkout", it was filled with
30
+ #: `Path.cwd()`, and every consumer -- the write-path allowlist, the test
31
+ #: runner's cwd, the vcs sandbox, the SDK subprocess cwd -- actually wanted
32
+ #: the target. That only coincided while the target sat inside the qaas
33
+ #: checkout, which is true of the bundled demo and of nothing else.
34
+ #: Comes from `SystemConfig.target_root()`, i.e. the profile.
35
+ target_root: Path
36
+ map_version: str | None = None
37
+ counters: dict[str, int] = field(default_factory=dict)
38
+
39
+ @property
40
+ def touched_files(self) -> set[str]:
41
+ """Files this agent has modified in this *run*, for the §8.2 diff budget.
42
+
43
+ It used to be a field on this context, which is rebuilt per dispatch — so
44
+ FIXER's "at most 5 files per run" reset on every FIXER/REVIEWER round
45
+ trip and again for every ticket. A budget that resets whenever the thing
46
+ it is bounding loops is not a budget. The store is the per-run object, so
47
+ it holds the set and the budget counts what the policy says it counts.
48
+ """
49
+ return self.store.touched_files(self.agent.name)
50
+
51
+ def bump(self, key: str) -> int:
52
+ self.counters[key] = self.counters.get(key, 0) + 1
53
+ return self.counters[key]
54
+
55
+ def count(self, key: str) -> int:
56
+ return self.counters.get(key, 0)
57
+
58
+
59
+ def ok(text: str, **structured: Any) -> dict[str, Any]:
60
+ """A successful MCP tool result."""
61
+ result: dict[str, Any] = {"content": [{"type": "text", "text": text}]}
62
+ if structured:
63
+ result["structuredContent"] = structured
64
+ return result
65
+
66
+
67
+ def err(text: str) -> dict[str, Any]:
68
+ """A failed MCP tool result.
69
+
70
+ Tool errors are returned, not raised: the agent should read the reason and
71
+ correct itself rather than have the turn die.
72
+ """
73
+ return {"content": [{"type": "text", "text": text}], "isError": True}
74
+
75
+
76
+ def handlers(tools: list) -> dict[str, Any]:
77
+ """Map tool name -> handler. Used by tests to exercise a server directly."""
78
+ return {t.name: t.handler for t in tools}