qaas-python 0.0.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- qaas/adapters/__init__.py +19 -0
- qaas/adapters/tracker.py +1783 -0
- qaas/adapters/vcs.py +555 -0
- qaas/cli.py +1757 -0
- qaas/config.py +409 -0
- qaas/defaults/config/agents/api.yaml +18 -0
- qaas/defaults/config/agents/architect.yaml +21 -0
- qaas/defaults/config/agents/auditor.yaml +19 -0
- qaas/defaults/config/agents/browser.yaml +15 -0
- qaas/defaults/config/agents/dba.yaml +20 -0
- qaas/defaults/config/agents/fixer.yaml +55 -0
- qaas/defaults/config/agents/guide.yaml +23 -0
- qaas/defaults/config/agents/load.yaml +26 -0
- qaas/defaults/config/agents/mapper.yaml +19 -0
- qaas/defaults/config/agents/reporter.yaml +19 -0
- qaas/defaults/config/agents/reproducer.yaml +21 -0
- qaas/defaults/config/agents/reviewer.yaml +18 -0
- qaas/defaults/config/agents/socket.yaml +23 -0
- qaas/defaults/config/agents/triage.yaml +20 -0
- qaas/defaults/config/agents/verifier.yaml +20 -0
- qaas/defaults/config/system.yaml +64 -0
- qaas/discover.py +242 -0
- qaas/envelope.py +318 -0
- qaas/envfile.py +100 -0
- qaas/guardrails.py +589 -0
- qaas/mcp/__init__.py +0 -0
- qaas/mcp/context.py +78 -0
- qaas/mcp/contract_diff.py +1011 -0
- qaas/mcp/defect_memory.py +495 -0
- qaas/mcp/env_control.py +925 -0
- qaas/mcp/envelope_server.py +463 -0
- qaas/mcp/test_runner.py +842 -0
- qaas/mcp/tracker.py +420 -0
- qaas/mcp/vcs.py +501 -0
- qaas/paths.py +317 -0
- qaas/plugin/.claude-plugin/plugin.json +9 -0
- qaas/plugin/skills/a11y-audit/SKILL.md +34 -0
- qaas/plugin/skills/adversarial-review/SKILL.md +120 -0
- qaas/plugin/skills/api-surface-extraction/SKILL.md +38 -0
- qaas/plugin/skills/authz-matrix-check/SKILL.md +46 -0
- qaas/plugin/skills/console-error-triage/SKILL.md +39 -0
- qaas/plugin/skills/contract-test-generation/SKILL.md +36 -0
- qaas/plugin/skills/dedupe-strategy/SKILL.md +39 -0
- qaas/plugin/skills/environment-pinning/SKILL.md +35 -0
- qaas/plugin/skills/error-taxonomy/SKILL.md +42 -0
- qaas/plugin/skills/exploratory-ui-walk/SKILL.md +46 -0
- qaas/plugin/skills/failing-test-authoring/SKILL.md +47 -0
- qaas/plugin/skills/flake-detection/SKILL.md +39 -0
- qaas/plugin/skills/form-state-probe/SKILL.md +36 -0
- qaas/plugin/skills/minimal-diff-discipline/SKILL.md +70 -0
- qaas/plugin/skills/openapi-diff/SKILL.md +45 -0
- qaas/plugin/skills/ownership-resolution/SKILL.md +31 -0
- qaas/plugin/skills/product-task-graph/SKILL.md +35 -0
- qaas/plugin/skills/regression-risk-scoring/SKILL.md +59 -0
- qaas/plugin/skills/regression-suite-selection/SKILL.md +36 -0
- qaas/plugin/skills/repo-cartography/SKILL.md +38 -0
- qaas/plugin/skills/repro-minimisation/SKILL.md +41 -0
- qaas/plugin/skills/rollback-plan-authoring/SKILL.md +81 -0
- qaas/plugin/skills/root-cause-vs-symptom/SKILL.md +67 -0
- qaas/plugin/skills/routing-rules/SKILL.md +34 -0
- qaas/plugin/skills/severity-rubric/SKILL.md +42 -0
- qaas/plugin/skills/test-first-fix/SKILL.md +66 -0
- qaas/plugin/skills/test-quality-audit/SKILL.md +58 -0
- qaas/plugin/skills/ticket-writer/SKILL.md +40 -0
- qaas/plugin/skills/verdict-reporting/SKILL.md +35 -0
- qaas/plugin/skills/verification-protocol/SKILL.md +39 -0
- qaas/prompts/API.md +44 -0
- qaas/prompts/ARCHITECT.md +80 -0
- qaas/prompts/AUDITOR.md +62 -0
- qaas/prompts/BROWSER.md +46 -0
- qaas/prompts/DBA.md +59 -0
- qaas/prompts/FIXER.md +55 -0
- qaas/prompts/GUIDE.md +94 -0
- qaas/prompts/LOAD.md +109 -0
- qaas/prompts/MAPPER.md +46 -0
- qaas/prompts/REPORTER.md +61 -0
- qaas/prompts/REPRODUCER.md +43 -0
- qaas/prompts/REVIEWER.md +53 -0
- qaas/prompts/SOCKET.md +100 -0
- qaas/prompts/TRIAGE.md +45 -0
- qaas/prompts/VERIFIER.md +41 -0
- qaas/prompts/_shared.md +45 -0
- qaas/registry.py +496 -0
- qaas/router.py +581 -0
- qaas/runner.py +210 -0
- qaas/scorecard.py +448 -0
- qaas/sdk_compat.py +52 -0
- qaas/store.py +323 -0
- qaas/target.py +287 -0
- qaas/tasks.py +438 -0
- qaas/trace.py +342 -0
- qaas_python-0.0.1.dist-info/METADATA +429 -0
- qaas_python-0.0.1.dist-info/RECORD +96 -0
- qaas_python-0.0.1.dist-info/WHEEL +4 -0
- qaas_python-0.0.1.dist-info/entry_points.txt +2 -0
- qaas_python-0.0.1.dist-info/licenses/LICENSE +21 -0
qaas/guardrails.py
ADDED
|
@@ -0,0 +1,589 @@
|
|
|
1
|
+
"""The §8.1 write-permission matrix, enforced in code.
|
|
2
|
+
|
|
3
|
+
Every agent gets a `can_use_tool` callback built from its policy. The callback
|
|
4
|
+
sees the tool name and its arguments before the tool runs, which is the only
|
|
5
|
+
place a limit like "REPRODUCER may write, but only under qa/repro" can actually be
|
|
6
|
+
imposed. A prompt asking an agent not to do something is a request; this is a
|
|
7
|
+
decision.
|
|
8
|
+
|
|
9
|
+
Enforcement runs in the **PreToolUse hook**, not in `can_use_tool`. This is not
|
|
10
|
+
a stylistic choice and it is easy to get wrong: an `allowed_tools` entry that
|
|
11
|
+
names a whole tool auto-approves it *before* `can_use_tool` is consulted, so a
|
|
12
|
+
policy implemented only in that callback is silently never applied. The SDK warns
|
|
13
|
+
about this shadowing, and an early version of this file had exactly that bug —
|
|
14
|
+
REPRODUCER's sandbox check was dead code. The hook sees every call regardless.
|
|
15
|
+
|
|
16
|
+
`can_use_tool` is kept as a second layer, for anything that falls outside the
|
|
17
|
+
allowlist and so reaches the callback normally.
|
|
18
|
+
|
|
19
|
+
Three belts, then:
|
|
20
|
+
|
|
21
|
+
* The PreToolUse hook — every built-in tool call, gated on this agent's policy.
|
|
22
|
+
* `can_use_tool` — the same decision, for calls not auto-approved.
|
|
23
|
+
* The MCP servers — their own domain rules (the tracker refuses an agent that
|
|
24
|
+
may not file; vcs refuses a branch outside the agent's patterns).
|
|
25
|
+
|
|
26
|
+
Denials return a reason rather than killing the turn: an agent that learns it
|
|
27
|
+
cannot write to a path should adapt, and the reason is what lets it. Every
|
|
28
|
+
denial lands in the run ledger, which is the audit trail §8 asks for.
|
|
29
|
+
"""
|
|
30
|
+
|
|
31
|
+
from __future__ import annotations
|
|
32
|
+
|
|
33
|
+
import fnmatch
|
|
34
|
+
import re
|
|
35
|
+
import shlex
|
|
36
|
+
from dataclasses import dataclass
|
|
37
|
+
from pathlib import Path
|
|
38
|
+
from typing import Any
|
|
39
|
+
|
|
40
|
+
from claude_agent_sdk import (
|
|
41
|
+
PermissionResultAllow,
|
|
42
|
+
PermissionResultDeny,
|
|
43
|
+
ToolPermissionContext,
|
|
44
|
+
)
|
|
45
|
+
|
|
46
|
+
from qaas.mcp.context import ToolContext
|
|
47
|
+
|
|
48
|
+
# Harness plumbing granted to every agent, independent of its config. These are
|
|
49
|
+
# not capability grants: ToolSearch only loads the schemas of servers the agent
|
|
50
|
+
# already has, Skill only loads instructions, and neither can reach anything the
|
|
51
|
+
# allowlist does not already permit. `build_allowed_tools` adds the same set, and
|
|
52
|
+
# both read this constant so the allowlist and the guardrail cannot drift apart —
|
|
53
|
+
# a mismatch here silently disables every skill in the system.
|
|
54
|
+
ALWAYS_GRANTED = frozenset({"ToolSearch", "Skill", "TodoWrite", "Task", "Agent"})
|
|
55
|
+
|
|
56
|
+
# Tools that read. Always safe, for every agent.
|
|
57
|
+
READ_TOOLS = {"Read", "Grep", "Glob", "NotebookRead"} | set(ALWAYS_GRANTED)
|
|
58
|
+
|
|
59
|
+
# Harness plumbing, not capability. These grant an agent nothing it was not
|
|
60
|
+
# already granted — ToolSearch only loads the schema of a tool that is already
|
|
61
|
+
# on its allowlist, and Skill only opens a skill file. Denying ToolSearch is
|
|
62
|
+
# worse than useless: MCP tools arrive deferred, so an agent that cannot call it
|
|
63
|
+
# cannot reach the servers it was given, and burns its whole turn budget
|
|
64
|
+
# discovering that. This system did exactly that once.
|
|
65
|
+
HARNESS_TOOLS = {"ToolSearch", "TodoWrite", "Task", "Agent", "Skill", "SlashCommand"}
|
|
66
|
+
|
|
67
|
+
# Tools that write to the filesystem. Gated on policy.write_paths.
|
|
68
|
+
WRITE_TOOLS = {"Write", "Edit", "MultiEdit", "NotebookEdit"}
|
|
69
|
+
|
|
70
|
+
# Bash command prefixes that are never allowed, whatever the agent.
|
|
71
|
+
# Merging to main, force-pushing and recursive deletes are outside every
|
|
72
|
+
# agent's remit in this system: merge is always human (§8.4), and nothing here
|
|
73
|
+
# needs to delete a tree.
|
|
74
|
+
FORBIDDEN_BASH = [
|
|
75
|
+
(r"\bgit\s+push\b.*(--force|-f\b)", "force-push is never permitted"),
|
|
76
|
+
(r"\bgit\s+push\b.*\b(main|master)\b", "pushing to main is never permitted"),
|
|
77
|
+
(r"\bgit\s+merge\b", "merging is a human decision (§8.4)"),
|
|
78
|
+
(r"\bgit\s+reset\s+--hard\b", "hard reset discards work outside the sandbox"),
|
|
79
|
+
(r"\bgit\s+checkout\s+(main|master)\b", "agents work on their own branches only"),
|
|
80
|
+
# `-[a-zA-Z]*[rf]` caught `-rf` and `-r`, and missed `-R`, `--recursive` and
|
|
81
|
+
# `--force` — three spellings of the same command, none of them exotic.
|
|
82
|
+
(
|
|
83
|
+
r"\brm\s+(-[a-zA-Z]*[rfR]|--(recursive|force)\b)",
|
|
84
|
+
"recursive or forced delete is not permitted",
|
|
85
|
+
),
|
|
86
|
+
(r"\bsudo\b", "privilege escalation is not permitted"),
|
|
87
|
+
(r"\b(shutdown|reboot|mkfs|dd)\b", "destructive system command"),
|
|
88
|
+
(r">\s*/dev/(sd|nvme|disk)", "writing to a block device"),
|
|
89
|
+
(r"\bdocker\s+system\s+prune", "prune would destroy state other runs depend on"),
|
|
90
|
+
(r"\bgh\s+pr\s+merge\b", "merging a pull request is a human decision (§8.4)"),
|
|
91
|
+
]
|
|
92
|
+
|
|
93
|
+
# Bash that mutates git state. Gated on the agent having branch patterns at all.
|
|
94
|
+
GIT_WRITE = re.compile(r"\bgit\s+(commit|push|branch|checkout\s+-b|switch\s+-c|tag|apply|am|rebase)\b")
|
|
95
|
+
|
|
96
|
+
# Shell constructs that rewrite a file in place. `>` is not a word character, so
|
|
97
|
+
# this deliberately does not use \b anchors — an earlier version did and silently
|
|
98
|
+
# matched nothing.
|
|
99
|
+
_MUTATES_FILE = re.compile(r"(>>?|\btee\b|\bsed\s+-i|\btruncate\b|\bdd\b)")
|
|
100
|
+
|
|
101
|
+
# -- what a shell command writes --------------------------------------------
|
|
102
|
+
#
|
|
103
|
+
# `_check_bash` used to consult only FORBIDDEN_BASH, the branch patterns and a
|
|
104
|
+
# substring test against `protected_paths`. Every other rule in the §8.1 matrix
|
|
105
|
+
# — `write_paths`, `forbidden_paths`, the §8.2 diff budget — was enforced for
|
|
106
|
+
# `Write`/`Edit` and bypassed entirely by a shell command:
|
|
107
|
+
#
|
|
108
|
+
# FIXER Write api/app/auth.py -> denied
|
|
109
|
+
# FIXER sed -i '' s/x/y/ api/app/auth.py -> allowed
|
|
110
|
+
# VERIFIER tee /etc/hosts < x -> allowed
|
|
111
|
+
#
|
|
112
|
+
# VERIFIER is the verification gate and has `Bash` with no `write_paths`, so it was
|
|
113
|
+
# "read-only" only against `Write` — §2's finder/fixer separation gone.
|
|
114
|
+
#
|
|
115
|
+
# Reading a shell command is best-effort by nature: a write can always hide one
|
|
116
|
+
# level of indirection further than a parser follows. The rule that makes that
|
|
117
|
+
# acceptable is below — a command that mutates and whose destination cannot be
|
|
118
|
+
# resolved is *refused*, not guessed at, which leaves the enforced tools
|
|
119
|
+
# (`Write`/`Edit`) as the only way to do the thing.
|
|
120
|
+
|
|
121
|
+
#: A command whose every non-flag argument is a destination.
|
|
122
|
+
_WRITES_ALL_ARGS = frozenset({"tee", "touch", "truncate"})
|
|
123
|
+
#: A command whose last argument is the destination.
|
|
124
|
+
_WRITES_LAST_ARG = frozenset({"cp", "mv", "install", "ln", "rsync"})
|
|
125
|
+
#: Interpreters that write wherever an inline script tells them to.
|
|
126
|
+
_SCRIPT_RUNNERS = frozenset({"python", "python3", "perl", "ruby", "node", "bash", "sh", "zsh", "php"})
|
|
127
|
+
_INLINE_SCRIPT_FLAGS = frozenset({"-c", "-e"})
|
|
128
|
+
#: Flags that take a value, so the value is not mistaken for a path.
|
|
129
|
+
_FLAG_TAKES_VALUE = frozenset({"-s", "-m", "-o", "-t", "--suffix", "--size", "--mode"})
|
|
130
|
+
|
|
131
|
+
#: Redirect targets that are not files anyone can be harmed through.
|
|
132
|
+
_DEV_SINKS = frozenset({"/dev/null", "/dev/stdout", "/dev/stderr", "/dev/tty"})
|
|
133
|
+
|
|
134
|
+
# `> out`, `>>out`, `2> err` — but not `2>&1`, whose `&1` is a file descriptor.
|
|
135
|
+
_REDIRECT_RE = re.compile(r">>?\s*(?!&)([^\s;|&<>]+)")
|
|
136
|
+
# Where one command ends and the next begins.
|
|
137
|
+
_SEGMENT_RE = re.compile(r"\|\||&&|[;|\n&]")
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
@dataclass
|
|
141
|
+
class Decision:
|
|
142
|
+
allowed: bool
|
|
143
|
+
reason: str = ""
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
class Guardrail:
|
|
147
|
+
"""One agent's enforcement of its own policy."""
|
|
148
|
+
|
|
149
|
+
def __init__(self, ctx: ToolContext):
|
|
150
|
+
self.ctx = ctx
|
|
151
|
+
self.agent = ctx.agent
|
|
152
|
+
self.policy = ctx.agent.policy
|
|
153
|
+
# Every write path in a policy is relative to the *application under
|
|
154
|
+
# test*, never to the qaas project. This was `ctx.repo_root`, filled
|
|
155
|
+
# from `Path.cwd()`, which anchored the whole allowlist on wherever the
|
|
156
|
+
# operator happened to be standing -- harmless only while the target was
|
|
157
|
+
# a subdirectory of the qaas checkout. With `qaas run --repo <url>` the
|
|
158
|
+
# target is a clone under `.qaas/targets/`, and an allowlist anchored on
|
|
159
|
+
# the cwd would deny every legitimate write and permit a sandbox that
|
|
160
|
+
# sits inside qaas's own source.
|
|
161
|
+
self.root = ctx.target_root.resolve()
|
|
162
|
+
# A glob cannot be resolved to a directory to contain things in, so the
|
|
163
|
+
# two kinds are kept apart and tested differently. `mcp/vcs.py` already
|
|
164
|
+
# honoured globs in `write_paths` and this did not; now that both go
|
|
165
|
+
# through `_check_path`, the more permissive of the two would have
|
|
166
|
+
# silently become the rule for everything.
|
|
167
|
+
self._allowed_roots = [
|
|
168
|
+
(self.root / p).resolve() for p in self.policy.write_paths if not _is_glob(p)
|
|
169
|
+
]
|
|
170
|
+
self._allowed_globs = [p for p in self.policy.write_paths if _is_glob(p)]
|
|
171
|
+
# Every MCP server the agent declared, as an allowlist prefix.
|
|
172
|
+
self._mcp_prefixes = tuple(f"mcp__{s}__" for s in self.agent.mcp_servers)
|
|
173
|
+
|
|
174
|
+
# -- entry point ------------------------------------------------------
|
|
175
|
+
|
|
176
|
+
async def can_use_tool(
|
|
177
|
+
self,
|
|
178
|
+
tool_name: str,
|
|
179
|
+
input_data: dict[str, Any],
|
|
180
|
+
context: ToolPermissionContext,
|
|
181
|
+
) -> PermissionResultAllow | PermissionResultDeny:
|
|
182
|
+
"""Second layer. Reached only for calls the allowlist did not auto-approve."""
|
|
183
|
+
decision = self.check(tool_name, input_data)
|
|
184
|
+
if decision.allowed:
|
|
185
|
+
return PermissionResultAllow(updated_input=input_data)
|
|
186
|
+
self._record(tool_name, input_data, decision.reason, via="can_use_tool")
|
|
187
|
+
return PermissionResultDeny(message=decision.reason)
|
|
188
|
+
|
|
189
|
+
async def pre_tool_use(
|
|
190
|
+
self,
|
|
191
|
+
payload: Any,
|
|
192
|
+
tool_use_id: str | None,
|
|
193
|
+
context: Any,
|
|
194
|
+
) -> dict[str, Any]:
|
|
195
|
+
"""Primary enforcement. Runs for every tool call, shadowing or not."""
|
|
196
|
+
tool_name = _hook_field(payload, "tool_name") or ""
|
|
197
|
+
input_data = _hook_field(payload, "tool_input") or {}
|
|
198
|
+
if not isinstance(input_data, dict):
|
|
199
|
+
input_data = {}
|
|
200
|
+
|
|
201
|
+
decision = self.check(tool_name, input_data)
|
|
202
|
+
# A refused call produces TWO ledger entries, and that is deliberate.
|
|
203
|
+
# `tool_call` is the universal record -- every call this agent made, in
|
|
204
|
+
# order, allowed or not -- and it is what a timeline reads. `denial`
|
|
205
|
+
# below carries the reason and a summary of the arguments, and is the
|
|
206
|
+
# only place tool arguments are recorded at all.
|
|
207
|
+
#
|
|
208
|
+
# It looks like double-counting and has been reported as such. Dropping
|
|
209
|
+
# either one loses something real: without the `tool_call` the refusal
|
|
210
|
+
# vanishes from the call sequence, and without the `denial` nobody can
|
|
211
|
+
# say why. A reader tallying refusals should count `denial`, not both.
|
|
212
|
+
self.ctx.store.log(
|
|
213
|
+
"tool_call",
|
|
214
|
+
agent=self.agent.name,
|
|
215
|
+
tool=tool_name,
|
|
216
|
+
tool_use_id=tool_use_id,
|
|
217
|
+
allowed=decision.allowed,
|
|
218
|
+
)
|
|
219
|
+
if decision.allowed:
|
|
220
|
+
return {}
|
|
221
|
+
|
|
222
|
+
self._record(tool_name, input_data, decision.reason, via="hook")
|
|
223
|
+
return {
|
|
224
|
+
"hookSpecificOutput": {
|
|
225
|
+
"hookEventName": "PreToolUse",
|
|
226
|
+
"permissionDecision": "deny",
|
|
227
|
+
"permissionDecisionReason": decision.reason,
|
|
228
|
+
}
|
|
229
|
+
}
|
|
230
|
+
|
|
231
|
+
def _record(self, tool_name: str, input_data: dict[str, Any], reason: str, *, via: str) -> None:
|
|
232
|
+
self.ctx.store.log(
|
|
233
|
+
"denial",
|
|
234
|
+
agent=self.agent.name,
|
|
235
|
+
tool=tool_name,
|
|
236
|
+
reason=reason,
|
|
237
|
+
via=via,
|
|
238
|
+
args=_summarise(input_data),
|
|
239
|
+
)
|
|
240
|
+
|
|
241
|
+
def check(self, tool_name: str, input_data: dict[str, Any]) -> Decision:
|
|
242
|
+
"""Pure policy evaluation. Separated from the callback so it is testable."""
|
|
243
|
+
if tool_name.startswith("mcp__"):
|
|
244
|
+
return self._check_mcp(tool_name)
|
|
245
|
+
if tool_name in READ_TOOLS:
|
|
246
|
+
return self._check_declared(tool_name)
|
|
247
|
+
if tool_name in WRITE_TOOLS:
|
|
248
|
+
# Policy before allowlist: "you are read-only" is the true reason and
|
|
249
|
+
# the useful one. "not in your allowlist" would be technically correct
|
|
250
|
+
# and would send the agent looking for the wrong fix.
|
|
251
|
+
if not self.policy.write_paths:
|
|
252
|
+
return self._read_only_decision()
|
|
253
|
+
declared = self._check_declared(tool_name)
|
|
254
|
+
return declared if not declared.allowed else self._check_write(input_data)
|
|
255
|
+
if tool_name == "Bash":
|
|
256
|
+
declared = self._check_declared(tool_name)
|
|
257
|
+
return declared if not declared.allowed else self._check_bash(input_data)
|
|
258
|
+
if tool_name in {"WebFetch", "WebSearch"}:
|
|
259
|
+
return Decision(
|
|
260
|
+
False,
|
|
261
|
+
f"{self.agent.name} has no network research remit. "
|
|
262
|
+
"Findings come from the code and the running app, not the web.",
|
|
263
|
+
)
|
|
264
|
+
return self._check_declared(tool_name)
|
|
265
|
+
|
|
266
|
+
# -- individual gates -------------------------------------------------
|
|
267
|
+
|
|
268
|
+
def _check_declared(self, tool_name: str) -> Decision:
|
|
269
|
+
"""Defence in depth: the tool must be one this agent declared.
|
|
270
|
+
|
|
271
|
+
The SDK is already told the allowlist, so reaching here means something
|
|
272
|
+
upstream drifted. Better to deny and log than to trust the setup.
|
|
273
|
+
"""
|
|
274
|
+
if tool_name in self.agent.builtin_tools or tool_name in HARNESS_TOOLS:
|
|
275
|
+
return Decision(True)
|
|
276
|
+
return Decision(
|
|
277
|
+
False,
|
|
278
|
+
f"{tool_name} is not in {self.agent.name}'s tool allowlist "
|
|
279
|
+
f"({', '.join(self.agent.builtin_tools) or 'none'}).",
|
|
280
|
+
)
|
|
281
|
+
|
|
282
|
+
def _check_mcp(self, tool_name: str) -> Decision:
|
|
283
|
+
if tool_name.startswith(self._mcp_prefixes):
|
|
284
|
+
return Decision(True)
|
|
285
|
+
server = tool_name.split("__")[1] if "__" in tool_name else "?"
|
|
286
|
+
return Decision(
|
|
287
|
+
False,
|
|
288
|
+
f"{self.agent.name} is not connected to the '{server}' server. "
|
|
289
|
+
f"Its servers are: {', '.join(self.agent.mcp_servers) or 'none'}.",
|
|
290
|
+
)
|
|
291
|
+
|
|
292
|
+
def _read_only_decision(self) -> Decision:
|
|
293
|
+
"""One wording, wherever an agent with no write paths tries to write.
|
|
294
|
+
|
|
295
|
+
The reason an agent reads must not depend on which door it reached for:
|
|
296
|
+
it is the same policy, and a different sentence per tool reads as a
|
|
297
|
+
different rule and invites it to go looking for the permissive one.
|
|
298
|
+
"""
|
|
299
|
+
return Decision(
|
|
300
|
+
False,
|
|
301
|
+
f"{self.agent.name} is read-only. Report what you found; "
|
|
302
|
+
"fixing is another agent's job (§2: the finder never fixes).",
|
|
303
|
+
)
|
|
304
|
+
|
|
305
|
+
def _check_write(self, input_data: dict[str, Any]) -> Decision:
|
|
306
|
+
raw = input_data.get("file_path") or input_data.get("path") or input_data.get("notebook_path")
|
|
307
|
+
if not raw:
|
|
308
|
+
return Decision(False, "write refused: no file path in the call")
|
|
309
|
+
return self._check_path(str(raw))
|
|
310
|
+
|
|
311
|
+
def _check_path(self, raw: str) -> Decision:
|
|
312
|
+
"""The §8.1/§8.2 matrix, applied to one path an agent wants to write.
|
|
313
|
+
|
|
314
|
+
Split out of `_check_write` so it is the single place the rules live:
|
|
315
|
+
`Write`/`Edit` reach it through `check`, a shell command reaches it
|
|
316
|
+
through `_check_bash`, and `mcp/vcs.py` reaches it instead of keeping
|
|
317
|
+
its own near-copy. Three doors, one answer.
|
|
318
|
+
"""
|
|
319
|
+
if not self.policy.write_paths:
|
|
320
|
+
return self._read_only_decision()
|
|
321
|
+
|
|
322
|
+
target = Path(raw)
|
|
323
|
+
resolved = (target if target.is_absolute() else self.root / target).resolve()
|
|
324
|
+
|
|
325
|
+
try:
|
|
326
|
+
relative = resolved.relative_to(self.root).as_posix()
|
|
327
|
+
except ValueError:
|
|
328
|
+
relative = resolved.as_posix()
|
|
329
|
+
|
|
330
|
+
# The autonomy envelope (§8.2) comes first. A path inside the sandbox but
|
|
331
|
+
# in a forbidden class must still be refused, and the reason must name
|
|
332
|
+
# the class so the agent escalates rather than looking for a way round.
|
|
333
|
+
forbidden = self._forbidden_class(relative)
|
|
334
|
+
if forbidden:
|
|
335
|
+
return Decision(
|
|
336
|
+
False,
|
|
337
|
+
f"{relative} is outside {self.agent.name}'s autonomy envelope: it is "
|
|
338
|
+
f"{forbidden}. Changes here need human approval (§8.2). Describe the "
|
|
339
|
+
"change you would make and escalate instead of making it.",
|
|
340
|
+
)
|
|
341
|
+
|
|
342
|
+
protected = self._protected_path(relative)
|
|
343
|
+
if protected:
|
|
344
|
+
return Decision(
|
|
345
|
+
False,
|
|
346
|
+
f"{relative} is the test that defines success for this ticket and may "
|
|
347
|
+
"not be edited (§10: a fixer that edits the test patches the symptom). "
|
|
348
|
+
"If you believe the test itself is wrong, that is an escalation.",
|
|
349
|
+
)
|
|
350
|
+
|
|
351
|
+
for allowed in self._allowed_roots:
|
|
352
|
+
if resolved == allowed or resolved.is_relative_to(allowed):
|
|
353
|
+
return self._check_diff_budget(relative)
|
|
354
|
+
for pattern in self._allowed_globs:
|
|
355
|
+
if fnmatch.fnmatch(relative, pattern):
|
|
356
|
+
return self._check_diff_budget(relative)
|
|
357
|
+
return Decision(
|
|
358
|
+
False,
|
|
359
|
+
f"write refused: {resolved} is outside {self.agent.name}'s sandbox "
|
|
360
|
+
f"({', '.join(self.policy.write_paths)}).",
|
|
361
|
+
)
|
|
362
|
+
|
|
363
|
+
def _forbidden_class(self, relative: str) -> str | None:
|
|
364
|
+
"""Which §8.2 class this path falls into, if any."""
|
|
365
|
+
for pattern in self.policy.forbidden_paths:
|
|
366
|
+
if fnmatch.fnmatch(relative, pattern) or fnmatch.fnmatch(Path(relative).name, pattern):
|
|
367
|
+
return _describe_forbidden(pattern)
|
|
368
|
+
return None
|
|
369
|
+
|
|
370
|
+
def _protected_path(self, relative: str) -> bool:
|
|
371
|
+
return any(
|
|
372
|
+
fnmatch.fnmatch(relative, p) or relative.endswith(p)
|
|
373
|
+
for p in self.policy.protected_paths
|
|
374
|
+
)
|
|
375
|
+
|
|
376
|
+
def _check_diff_budget(self, relative: str) -> Decision:
|
|
377
|
+
"""Cap how much one agent may change in a single run (§8.2).
|
|
378
|
+
|
|
379
|
+
Counted per distinct file touched, not per call: an agent editing the
|
|
380
|
+
same file six times has made one file's worth of change, and counting
|
|
381
|
+
calls would refuse a perfectly ordinary iteration.
|
|
382
|
+
"""
|
|
383
|
+
max_files = self.policy.max_diff_files
|
|
384
|
+
if max_files is None:
|
|
385
|
+
return Decision(True)
|
|
386
|
+
|
|
387
|
+
touched = self.ctx.touched_files
|
|
388
|
+
if relative in touched:
|
|
389
|
+
return Decision(True)
|
|
390
|
+
if len(touched) >= max_files:
|
|
391
|
+
return Decision(
|
|
392
|
+
False,
|
|
393
|
+
f"{self.agent.name} has already changed {len(touched)} files, which is "
|
|
394
|
+
f"its limit of {max_files} (§8.2). A fix this wide is outside the "
|
|
395
|
+
"autonomy envelope: stop, and escalate with what you have found. "
|
|
396
|
+
f"Already touched: {', '.join(sorted(touched))}.",
|
|
397
|
+
)
|
|
398
|
+
touched.add(relative)
|
|
399
|
+
return Decision(True)
|
|
400
|
+
|
|
401
|
+
def _check_bash(self, input_data: dict[str, Any]) -> Decision:
|
|
402
|
+
command = str(input_data.get("command", ""))
|
|
403
|
+
if not command.strip():
|
|
404
|
+
return Decision(False, "empty command")
|
|
405
|
+
|
|
406
|
+
for pattern, why in FORBIDDEN_BASH:
|
|
407
|
+
if re.search(pattern, command):
|
|
408
|
+
return Decision(False, f"command refused: {why}")
|
|
409
|
+
|
|
410
|
+
if GIT_WRITE.search(command) and not self.policy.branch_patterns:
|
|
411
|
+
return Decision(
|
|
412
|
+
False,
|
|
413
|
+
f"{self.agent.name} may not modify git state. "
|
|
414
|
+
"It has no branch patterns in its policy.",
|
|
415
|
+
)
|
|
416
|
+
|
|
417
|
+
if self.policy.branch_patterns:
|
|
418
|
+
branch = _branch_from_command(command)
|
|
419
|
+
if branch and not any(
|
|
420
|
+
fnmatch.fnmatch(branch, pat) for pat in self.policy.branch_patterns
|
|
421
|
+
):
|
|
422
|
+
return Decision(
|
|
423
|
+
False,
|
|
424
|
+
f"branch '{branch}' is outside {self.agent.name}'s patterns "
|
|
425
|
+
f"({', '.join(self.policy.branch_patterns)}).",
|
|
426
|
+
)
|
|
427
|
+
|
|
428
|
+
if self.policy.protected_paths:
|
|
429
|
+
for protected in self.policy.protected_paths:
|
|
430
|
+
if protected in command and _MUTATES_FILE.search(command):
|
|
431
|
+
return Decision(
|
|
432
|
+
False,
|
|
433
|
+
f"'{protected}' is protected: it defines what a fix must achieve "
|
|
434
|
+
"and may not be edited (§10, symptom fixes).",
|
|
435
|
+
)
|
|
436
|
+
|
|
437
|
+
return self._check_bash_writes(command)
|
|
438
|
+
|
|
439
|
+
def _check_bash_writes(self, command: str) -> Decision:
|
|
440
|
+
"""Apply the write matrix to what the command would actually write.
|
|
441
|
+
|
|
442
|
+
This is the half `_check_bash` never had. Every rule below already held
|
|
443
|
+
for `Write` and `Edit`; reaching them through a shell is not a different
|
|
444
|
+
permission, so it does not get a different answer.
|
|
445
|
+
"""
|
|
446
|
+
for segment in _shell_segments(command):
|
|
447
|
+
paths, undeterminable = _writes_of(segment)
|
|
448
|
+
if undeterminable:
|
|
449
|
+
if not self.policy.write_paths:
|
|
450
|
+
return self._read_only_decision()
|
|
451
|
+
return Decision(
|
|
452
|
+
False,
|
|
453
|
+
f"refusing '{segment.strip()}': {undeterminable}, so this cannot be "
|
|
454
|
+
"checked against your write paths. Use Write or Edit, which name the "
|
|
455
|
+
"file they change.",
|
|
456
|
+
)
|
|
457
|
+
for raw in paths:
|
|
458
|
+
decision = self._check_path(raw)
|
|
459
|
+
if not decision.allowed:
|
|
460
|
+
return decision
|
|
461
|
+
return Decision(True)
|
|
462
|
+
|
|
463
|
+
|
|
464
|
+
def _is_glob(pattern: str) -> bool:
|
|
465
|
+
return any(ch in pattern for ch in "*?[")
|
|
466
|
+
|
|
467
|
+
|
|
468
|
+
def _shell_segments(command: str) -> list[str]:
|
|
469
|
+
"""One command per element, so `ls && sed -i ...` is two things, not one."""
|
|
470
|
+
return [segment.strip() for segment in _SEGMENT_RE.split(command) if segment.strip()]
|
|
471
|
+
|
|
472
|
+
|
|
473
|
+
def _non_flag_args(parts: list[str]) -> list[str]:
|
|
474
|
+
"""argv[1:] with options — and the values they consume — removed."""
|
|
475
|
+
args: list[str] = []
|
|
476
|
+
skip = False
|
|
477
|
+
for token in parts[1:]:
|
|
478
|
+
if skip:
|
|
479
|
+
skip = False
|
|
480
|
+
continue
|
|
481
|
+
if token.startswith("-"):
|
|
482
|
+
skip = token in _FLAG_TAKES_VALUE
|
|
483
|
+
continue
|
|
484
|
+
args.append(token)
|
|
485
|
+
return args
|
|
486
|
+
|
|
487
|
+
|
|
488
|
+
def _writes_of(segment: str) -> tuple[list[str], str | None]:
|
|
489
|
+
"""What one command writes: (paths, why the destination is undeterminable).
|
|
490
|
+
|
|
491
|
+
A non-None second element means "this mutates and I cannot say where", which
|
|
492
|
+
the caller must treat as a refusal rather than as an empty path list. The two
|
|
493
|
+
are deliberately different: no paths and no reason means the command writes
|
|
494
|
+
nothing and is none of our business.
|
|
495
|
+
"""
|
|
496
|
+
targets = [t for t in _REDIRECT_RE.findall(segment) if t not in _DEV_SINKS]
|
|
497
|
+
try:
|
|
498
|
+
parts = shlex.split(segment)
|
|
499
|
+
except ValueError:
|
|
500
|
+
# An unbalanced quote. We cannot read it, so we cannot clear it.
|
|
501
|
+
return targets, "it cannot be parsed as a shell command"
|
|
502
|
+
if not parts:
|
|
503
|
+
return targets, None
|
|
504
|
+
|
|
505
|
+
name = Path(parts[0]).name
|
|
506
|
+
args = _non_flag_args(parts)
|
|
507
|
+
|
|
508
|
+
if name in _SCRIPT_RUNNERS and any(flag in parts for flag in _INLINE_SCRIPT_FLAGS):
|
|
509
|
+
return targets, f"an inline {name} script can write anywhere"
|
|
510
|
+
if name == "patch" or (name == "git" and "apply" in parts[1:3]):
|
|
511
|
+
return targets, "a patch carries its own destinations"
|
|
512
|
+
if name == "sed" and any(part.startswith("-i") for part in parts):
|
|
513
|
+
# `sed -i '' s/x/y/ f.py` (BSD) and `sed -i s/x/y/ f.py` (GNU) differ by
|
|
514
|
+
# an empty argument. Drop the empties, then drop the script expression;
|
|
515
|
+
# whatever is left is a file being rewritten in place.
|
|
516
|
+
non_empty = [a for a in args if a]
|
|
517
|
+
return targets + non_empty[1:], None
|
|
518
|
+
if name in _WRITES_ALL_ARGS:
|
|
519
|
+
return targets + args, None
|
|
520
|
+
if name in _WRITES_LAST_ARG and args:
|
|
521
|
+
return targets + args[-1:], None
|
|
522
|
+
return targets, None
|
|
523
|
+
|
|
524
|
+
|
|
525
|
+
def _branch_from_command(command: str) -> str | None:
|
|
526
|
+
"""Best-effort branch name out of a git command, for policy matching."""
|
|
527
|
+
try:
|
|
528
|
+
parts = shlex.split(command)
|
|
529
|
+
except ValueError:
|
|
530
|
+
return None
|
|
531
|
+
# Only a git command names a branch. Without this, `-c` was read as
|
|
532
|
+
# `switch -c` in anything that happens to take one: `python -c '...'` was
|
|
533
|
+
# refused as "branch 'open(...)' is outside FIXER's patterns", which is
|
|
534
|
+
# both a wrong answer and an unactionable one.
|
|
535
|
+
if not any(Path(part).name == "git" for part in parts[:2]):
|
|
536
|
+
return None
|
|
537
|
+
for i, token in enumerate(parts):
|
|
538
|
+
if token in {"-b", "-c"} and i + 1 < len(parts):
|
|
539
|
+
return parts[i + 1]
|
|
540
|
+
if token in {"branch", "switch"} and i + 1 < len(parts):
|
|
541
|
+
candidate = parts[i + 1]
|
|
542
|
+
if not candidate.startswith("-"):
|
|
543
|
+
return candidate
|
|
544
|
+
return None
|
|
545
|
+
|
|
546
|
+
|
|
547
|
+
# Human-readable names for the forbidden classes, so a denial explains itself.
|
|
548
|
+
_FORBIDDEN_DESCRIPTIONS = [
|
|
549
|
+
("migration", "a database migration"),
|
|
550
|
+
("auth", "authentication or authorization code"),
|
|
551
|
+
("payment", "a payment path"),
|
|
552
|
+
("billing", "a billing path"),
|
|
553
|
+
("secret", "secret material"),
|
|
554
|
+
("infra", "infrastructure configuration"),
|
|
555
|
+
("terraform", "infrastructure configuration"),
|
|
556
|
+
(".tf", "infrastructure configuration"),
|
|
557
|
+
("docker", "container or deployment configuration"),
|
|
558
|
+
("k8s", "container or deployment configuration"),
|
|
559
|
+
("kube", "container or deployment configuration"),
|
|
560
|
+
(".github", "CI configuration"),
|
|
561
|
+
("workflow", "CI configuration"),
|
|
562
|
+
]
|
|
563
|
+
|
|
564
|
+
|
|
565
|
+
def _describe_forbidden(pattern: str) -> str:
|
|
566
|
+
lowered = pattern.lower()
|
|
567
|
+
for needle, description in _FORBIDDEN_DESCRIPTIONS:
|
|
568
|
+
if needle in lowered:
|
|
569
|
+
return description
|
|
570
|
+
return f"matched by the forbidden pattern `{pattern}`"
|
|
571
|
+
|
|
572
|
+
|
|
573
|
+
def _hook_field(payload: Any, name: str) -> Any:
|
|
574
|
+
"""Hook payloads arrive as dicts or dataclasses depending on SDK version."""
|
|
575
|
+
if isinstance(payload, dict):
|
|
576
|
+
return payload.get(name)
|
|
577
|
+
return getattr(payload, name, None)
|
|
578
|
+
|
|
579
|
+
|
|
580
|
+
def _summarise(input_data: dict[str, Any], limit: int = 200) -> dict[str, Any]:
|
|
581
|
+
"""Ledger-sized view of a tool call: enough to audit, not a content dump."""
|
|
582
|
+
out: dict[str, Any] = {}
|
|
583
|
+
for key, value in input_data.items():
|
|
584
|
+
if key in {"content", "new_string", "old_string"}:
|
|
585
|
+
out[key] = f"<{len(str(value))} chars>"
|
|
586
|
+
else:
|
|
587
|
+
text = str(value)
|
|
588
|
+
out[key] = text if len(text) <= limit else text[:limit] + "…"
|
|
589
|
+
return out
|
qaas/mcp/__init__.py
ADDED
|
File without changes
|
qaas/mcp/context.py
ADDED
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
"""Shared state the in-process MCP servers close over.
|
|
2
|
+
|
|
3
|
+
Every tool call lands in this process, so the servers can reach the run store,
|
|
4
|
+
the config and the target app directly. That is the point of building them
|
|
5
|
+
in-process: validation, guardrails and persistence happen where the state
|
|
6
|
+
already lives, with no serialisation boundary to reason about.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from dataclasses import dataclass, field
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
from typing import Any
|
|
14
|
+
|
|
15
|
+
from qaas.config import AgentSpec, SystemConfig
|
|
16
|
+
from qaas.store import RunStore, SystemMapStore
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
@dataclass
|
|
20
|
+
class ToolContext:
|
|
21
|
+
"""One agent's view of the run. Rebuilt per agent invocation."""
|
|
22
|
+
|
|
23
|
+
store: RunStore
|
|
24
|
+
maps: SystemMapStore
|
|
25
|
+
config: SystemConfig
|
|
26
|
+
agent: AgentSpec
|
|
27
|
+
|
|
28
|
+
#: The application under test, on disk. Named `repo_root` once, and that
|
|
29
|
+
#: name was the bug: it read as "the qaas checkout", it was filled with
|
|
30
|
+
#: `Path.cwd()`, and every consumer -- the write-path allowlist, the test
|
|
31
|
+
#: runner's cwd, the vcs sandbox, the SDK subprocess cwd -- actually wanted
|
|
32
|
+
#: the target. That only coincided while the target sat inside the qaas
|
|
33
|
+
#: checkout, which is true of the bundled demo and of nothing else.
|
|
34
|
+
#: Comes from `SystemConfig.target_root()`, i.e. the profile.
|
|
35
|
+
target_root: Path
|
|
36
|
+
map_version: str | None = None
|
|
37
|
+
counters: dict[str, int] = field(default_factory=dict)
|
|
38
|
+
|
|
39
|
+
@property
|
|
40
|
+
def touched_files(self) -> set[str]:
|
|
41
|
+
"""Files this agent has modified in this *run*, for the §8.2 diff budget.
|
|
42
|
+
|
|
43
|
+
It used to be a field on this context, which is rebuilt per dispatch — so
|
|
44
|
+
FIXER's "at most 5 files per run" reset on every FIXER/REVIEWER round
|
|
45
|
+
trip and again for every ticket. A budget that resets whenever the thing
|
|
46
|
+
it is bounding loops is not a budget. The store is the per-run object, so
|
|
47
|
+
it holds the set and the budget counts what the policy says it counts.
|
|
48
|
+
"""
|
|
49
|
+
return self.store.touched_files(self.agent.name)
|
|
50
|
+
|
|
51
|
+
def bump(self, key: str) -> int:
|
|
52
|
+
self.counters[key] = self.counters.get(key, 0) + 1
|
|
53
|
+
return self.counters[key]
|
|
54
|
+
|
|
55
|
+
def count(self, key: str) -> int:
|
|
56
|
+
return self.counters.get(key, 0)
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def ok(text: str, **structured: Any) -> dict[str, Any]:
|
|
60
|
+
"""A successful MCP tool result."""
|
|
61
|
+
result: dict[str, Any] = {"content": [{"type": "text", "text": text}]}
|
|
62
|
+
if structured:
|
|
63
|
+
result["structuredContent"] = structured
|
|
64
|
+
return result
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def err(text: str) -> dict[str, Any]:
|
|
68
|
+
"""A failed MCP tool result.
|
|
69
|
+
|
|
70
|
+
Tool errors are returned, not raised: the agent should read the reason and
|
|
71
|
+
correct itself rather than have the turn die.
|
|
72
|
+
"""
|
|
73
|
+
return {"content": [{"type": "text", "text": text}], "isError": True}
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def handlers(tools: list) -> dict[str, Any]:
|
|
77
|
+
"""Map tool name -> handler. Used by tests to exercise a server directly."""
|
|
78
|
+
return {t.name: t.handler for t in tools}
|