qaas-python 0.0.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- qaas/adapters/__init__.py +19 -0
- qaas/adapters/tracker.py +1783 -0
- qaas/adapters/vcs.py +555 -0
- qaas/cli.py +1757 -0
- qaas/config.py +409 -0
- qaas/defaults/config/agents/api.yaml +18 -0
- qaas/defaults/config/agents/architect.yaml +21 -0
- qaas/defaults/config/agents/auditor.yaml +19 -0
- qaas/defaults/config/agents/browser.yaml +15 -0
- qaas/defaults/config/agents/dba.yaml +20 -0
- qaas/defaults/config/agents/fixer.yaml +55 -0
- qaas/defaults/config/agents/guide.yaml +23 -0
- qaas/defaults/config/agents/load.yaml +26 -0
- qaas/defaults/config/agents/mapper.yaml +19 -0
- qaas/defaults/config/agents/reporter.yaml +19 -0
- qaas/defaults/config/agents/reproducer.yaml +21 -0
- qaas/defaults/config/agents/reviewer.yaml +18 -0
- qaas/defaults/config/agents/socket.yaml +23 -0
- qaas/defaults/config/agents/triage.yaml +20 -0
- qaas/defaults/config/agents/verifier.yaml +20 -0
- qaas/defaults/config/system.yaml +64 -0
- qaas/discover.py +242 -0
- qaas/envelope.py +318 -0
- qaas/envfile.py +100 -0
- qaas/guardrails.py +589 -0
- qaas/mcp/__init__.py +0 -0
- qaas/mcp/context.py +78 -0
- qaas/mcp/contract_diff.py +1011 -0
- qaas/mcp/defect_memory.py +495 -0
- qaas/mcp/env_control.py +925 -0
- qaas/mcp/envelope_server.py +463 -0
- qaas/mcp/test_runner.py +842 -0
- qaas/mcp/tracker.py +420 -0
- qaas/mcp/vcs.py +501 -0
- qaas/paths.py +317 -0
- qaas/plugin/.claude-plugin/plugin.json +9 -0
- qaas/plugin/skills/a11y-audit/SKILL.md +34 -0
- qaas/plugin/skills/adversarial-review/SKILL.md +120 -0
- qaas/plugin/skills/api-surface-extraction/SKILL.md +38 -0
- qaas/plugin/skills/authz-matrix-check/SKILL.md +46 -0
- qaas/plugin/skills/console-error-triage/SKILL.md +39 -0
- qaas/plugin/skills/contract-test-generation/SKILL.md +36 -0
- qaas/plugin/skills/dedupe-strategy/SKILL.md +39 -0
- qaas/plugin/skills/environment-pinning/SKILL.md +35 -0
- qaas/plugin/skills/error-taxonomy/SKILL.md +42 -0
- qaas/plugin/skills/exploratory-ui-walk/SKILL.md +46 -0
- qaas/plugin/skills/failing-test-authoring/SKILL.md +47 -0
- qaas/plugin/skills/flake-detection/SKILL.md +39 -0
- qaas/plugin/skills/form-state-probe/SKILL.md +36 -0
- qaas/plugin/skills/minimal-diff-discipline/SKILL.md +70 -0
- qaas/plugin/skills/openapi-diff/SKILL.md +45 -0
- qaas/plugin/skills/ownership-resolution/SKILL.md +31 -0
- qaas/plugin/skills/product-task-graph/SKILL.md +35 -0
- qaas/plugin/skills/regression-risk-scoring/SKILL.md +59 -0
- qaas/plugin/skills/regression-suite-selection/SKILL.md +36 -0
- qaas/plugin/skills/repo-cartography/SKILL.md +38 -0
- qaas/plugin/skills/repro-minimisation/SKILL.md +41 -0
- qaas/plugin/skills/rollback-plan-authoring/SKILL.md +81 -0
- qaas/plugin/skills/root-cause-vs-symptom/SKILL.md +67 -0
- qaas/plugin/skills/routing-rules/SKILL.md +34 -0
- qaas/plugin/skills/severity-rubric/SKILL.md +42 -0
- qaas/plugin/skills/test-first-fix/SKILL.md +66 -0
- qaas/plugin/skills/test-quality-audit/SKILL.md +58 -0
- qaas/plugin/skills/ticket-writer/SKILL.md +40 -0
- qaas/plugin/skills/verdict-reporting/SKILL.md +35 -0
- qaas/plugin/skills/verification-protocol/SKILL.md +39 -0
- qaas/prompts/API.md +44 -0
- qaas/prompts/ARCHITECT.md +80 -0
- qaas/prompts/AUDITOR.md +62 -0
- qaas/prompts/BROWSER.md +46 -0
- qaas/prompts/DBA.md +59 -0
- qaas/prompts/FIXER.md +55 -0
- qaas/prompts/GUIDE.md +94 -0
- qaas/prompts/LOAD.md +109 -0
- qaas/prompts/MAPPER.md +46 -0
- qaas/prompts/REPORTER.md +61 -0
- qaas/prompts/REPRODUCER.md +43 -0
- qaas/prompts/REVIEWER.md +53 -0
- qaas/prompts/SOCKET.md +100 -0
- qaas/prompts/TRIAGE.md +45 -0
- qaas/prompts/VERIFIER.md +41 -0
- qaas/prompts/_shared.md +45 -0
- qaas/registry.py +496 -0
- qaas/router.py +581 -0
- qaas/runner.py +210 -0
- qaas/scorecard.py +448 -0
- qaas/sdk_compat.py +52 -0
- qaas/store.py +323 -0
- qaas/target.py +287 -0
- qaas/tasks.py +438 -0
- qaas/trace.py +342 -0
- qaas_python-0.0.1.dist-info/METADATA +429 -0
- qaas_python-0.0.1.dist-info/RECORD +96 -0
- qaas_python-0.0.1.dist-info/WHEEL +4 -0
- qaas_python-0.0.1.dist-info/entry_points.txt +2 -0
- qaas_python-0.0.1.dist-info/licenses/LICENSE +21 -0
qaas/tasks.py
ADDED
|
@@ -0,0 +1,438 @@
|
|
|
1
|
+
"""The task each agent is given: the user turn, distinct from its system prompt.
|
|
2
|
+
|
|
3
|
+
The system prompt says who an agent is and what its standards are; that is
|
|
4
|
+
stable across every run and lives in `prompts/`. The task says what to do this
|
|
5
|
+
time — which application, which environment, which findings — and is built here
|
|
6
|
+
from the target profile.
|
|
7
|
+
|
|
8
|
+
Nothing in this module may name a specific application. A prompt that mentions
|
|
9
|
+
one repository's directory layout or one app's seeded users works exactly once.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
from qaas.config import AgentSpec, SystemConfig
|
|
15
|
+
from qaas.envelope import DefectEnvelope
|
|
16
|
+
from qaas.target import TargetProfile
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def _profile(config: SystemConfig) -> TargetProfile:
|
|
20
|
+
if config.profile is None:
|
|
21
|
+
raise ValueError(
|
|
22
|
+
"no target profile loaded. Point `target:` in system.yaml at a file "
|
|
23
|
+
"in config/targets/, or create one with `qaas init <path-to-repo>`."
|
|
24
|
+
)
|
|
25
|
+
return config.profile
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def _where(config: SystemConfig) -> str:
|
|
29
|
+
"""How to name the application under test to an agent.
|
|
30
|
+
|
|
31
|
+
The resolved path, not `profile.root`. An agent's process cwd *is* the
|
|
32
|
+
target root, and `root:` is spelled relative to the qaas project -- so
|
|
33
|
+
telling a run "the application at `target-app`" sent it looking for
|
|
34
|
+
`<target>/target-app`, a directory that does not exist. Resolved, the
|
|
35
|
+
sentence is true from wherever the agent happens to be standing.
|
|
36
|
+
"""
|
|
37
|
+
return str(config.target_root())
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def _environment_brief(profile: TargetProfile) -> str:
|
|
41
|
+
"""What an agent can and cannot do to the running application."""
|
|
42
|
+
env = profile.environment
|
|
43
|
+
if env.mode == "none":
|
|
44
|
+
return (
|
|
45
|
+
"There is no running instance of this application available. Work "
|
|
46
|
+
"statically: read the code, the schema, the spec and the tests. Say so "
|
|
47
|
+
"plainly in any finding you cannot execute — a defect you reasoned to "
|
|
48
|
+
"but did not observe is a weaker claim, and its confidence should show "
|
|
49
|
+
"that."
|
|
50
|
+
)
|
|
51
|
+
|
|
52
|
+
lines = []
|
|
53
|
+
if env.api_url:
|
|
54
|
+
lines.append(f"API: {env.api_url}")
|
|
55
|
+
if env.web_url:
|
|
56
|
+
lines.append(f"web: {env.web_url}")
|
|
57
|
+
where = "; ".join(lines)
|
|
58
|
+
|
|
59
|
+
if env.mode == "external":
|
|
60
|
+
return (
|
|
61
|
+
f"The application is already running ({where}) and this system does not "
|
|
62
|
+
"own it. You may read it and exercise it. You may NOT reset it, reseed "
|
|
63
|
+
"it, or destroy state — someone else may be relying on it. Prefer "
|
|
64
|
+
"read-only calls, and never send a request whose side effect you would "
|
|
65
|
+
"not want to explain."
|
|
66
|
+
)
|
|
67
|
+
return (
|
|
68
|
+
f"Bring the application up with `env_control` ({where}). You own this "
|
|
69
|
+
"environment: seed it, reset it between journeys, and tear it down when "
|
|
70
|
+
"done. Reset between independent checks so one test's leftovers are not "
|
|
71
|
+
"the next one's bug."
|
|
72
|
+
)
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def _auth_brief(profile: TargetProfile) -> str:
|
|
76
|
+
auth = profile.auth
|
|
77
|
+
if auth.mode == "none":
|
|
78
|
+
return "The application needs no authentication."
|
|
79
|
+
if auth.mode == "token":
|
|
80
|
+
return (
|
|
81
|
+
f"Authenticate with the bearer token in ${auth.token_env}. "
|
|
82
|
+
"`env_control.impersonate` will hand it to you."
|
|
83
|
+
)
|
|
84
|
+
roles = "\n".join(
|
|
85
|
+
f" - `{name}` ({r.username}){': ' + r.description if r.description else ''}"
|
|
86
|
+
for name, r in auth.roles.items()
|
|
87
|
+
)
|
|
88
|
+
return (
|
|
89
|
+
f"Sign in via `{auth.login_endpoint}`. `env_control.impersonate(role)` returns "
|
|
90
|
+
f"a token for any of these accounts:\n{roles}\n"
|
|
91
|
+
"Use more than one. A permission defect is invisible from a single role."
|
|
92
|
+
)
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def mapper(config: SystemConfig) -> str:
|
|
96
|
+
p = _profile(config)
|
|
97
|
+
spec = (
|
|
98
|
+
f"`{p.layout.spec}` is the declared API contract — read it for the intended "
|
|
99
|
+
"surface, but record what the code actually implements, and put any "
|
|
100
|
+
"disagreement in `drift`."
|
|
101
|
+
if p.layout.spec
|
|
102
|
+
else "There is no API specification in this repository. Record the surface "
|
|
103
|
+
"the code actually exposes, and note the absence of a spec in `gaps`."
|
|
104
|
+
)
|
|
105
|
+
ownership = (
|
|
106
|
+
f"`{p.layout.ownership}` is the ownership source."
|
|
107
|
+
if p.layout.ownership
|
|
108
|
+
else "There is no CODEOWNERS file. Leave ownership null rather than guessing "
|
|
109
|
+
"a team from a directory name — a misrouted ticket is worse than an "
|
|
110
|
+
"unassigned one."
|
|
111
|
+
)
|
|
112
|
+
return f"""Map the application at `{_where(config)}` — your working directory.
|
|
113
|
+
|
|
114
|
+
{p.description.strip() or "No description was supplied; work it out from the code."}
|
|
115
|
+
|
|
116
|
+
Known layout — {p.layout.described()}. Treat that as a starting point, not an
|
|
117
|
+
inventory: verify it and record what is actually there.
|
|
118
|
+
|
|
119
|
+
{spec}
|
|
120
|
+
|
|
121
|
+
{ownership}
|
|
122
|
+
|
|
123
|
+
Ignore these directories entirely: {', '.join(p.layout.exclude)}.
|
|
124
|
+
|
|
125
|
+
Explore with Read, Grep and Glob, then publish one complete map with
|
|
126
|
+
`put_system_map`. Get the shape of the repository before reading any single file
|
|
127
|
+
closely. Breadth first: every route and every table matters more than a deep
|
|
128
|
+
read of any one handler.
|
|
129
|
+
|
|
130
|
+
Publish the map once, when it is complete."""
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def api(config: SystemConfig, mode: str) -> str:
|
|
134
|
+
p = _profile(config)
|
|
135
|
+
spec_line = (
|
|
136
|
+
f"Compare the implementation against `{p.layout.spec}` with `diff_openapi`, "
|
|
137
|
+
"and judge each difference by consumer impact with `classify_breaking`."
|
|
138
|
+
if p.layout.spec
|
|
139
|
+
else "There is no specification to diff against, so the contract is implicit. "
|
|
140
|
+
"Judge each endpoint against what its own code promises and what its "
|
|
141
|
+
"consumers assume: look for handlers that disagree with each other."
|
|
142
|
+
)
|
|
143
|
+
prove = (
|
|
144
|
+
"Prove what you report. Call the endpoint and capture the real request and "
|
|
145
|
+
"response. A finding you have not observed is a hypothesis, not a defect, "
|
|
146
|
+
"and its confidence should say so."
|
|
147
|
+
if p.environment.is_reachable
|
|
148
|
+
else "You cannot call this API, so every finding is a reading of the code. "
|
|
149
|
+
"Quote the lines that support it, and keep confidence honest about the "
|
|
150
|
+
"fact that you did not observe the behaviour."
|
|
151
|
+
)
|
|
152
|
+
filing = (
|
|
153
|
+
"This is a diagnostic run: report what you find, but nothing will be filed."
|
|
154
|
+
if mode == "incident"
|
|
155
|
+
else "Findings that survive reproduction become tickets. Hold yourself to that bar."
|
|
156
|
+
)
|
|
157
|
+
return f"""Audit the API of the application at `{_where(config)}` — your working directory.
|
|
158
|
+
|
|
159
|
+
Start with `get_system_map` for the route inventory. Do not rediscover it.
|
|
160
|
+
|
|
161
|
+
{_environment_brief(p)}
|
|
162
|
+
|
|
163
|
+
{_auth_brief(p)}
|
|
164
|
+
|
|
165
|
+
Then work the surface systematically. For each endpoint: does the implementation
|
|
166
|
+
match what is declared? Who is allowed to call it, and does the code actually
|
|
167
|
+
check that? What happens on the error paths, and with a large or hostile input?
|
|
168
|
+
|
|
169
|
+
{spec_line}
|
|
170
|
+
|
|
171
|
+
The structural questions a tool can answer for you. The ones it cannot — is this
|
|
172
|
+
endpoint scoped to the caller's tenant, is this check the right check — need you
|
|
173
|
+
to read the handler and compare it against its neighbours. Endpoints in the same
|
|
174
|
+
file that disagree with each other are where the defects are.
|
|
175
|
+
|
|
176
|
+
{prove}
|
|
177
|
+
|
|
178
|
+
{filing}"""
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
def browser(config: SystemConfig, mode: str) -> str:
|
|
182
|
+
p = _profile(config)
|
|
183
|
+
if not p.environment.is_reachable or not p.environment.web_url:
|
|
184
|
+
return f"""There is no running UI for the application at `{_where(config)}`, so there is
|
|
185
|
+
nothing for you to explore.
|
|
186
|
+
|
|
187
|
+
Report that in your final message and stop. Do not substitute reading the
|
|
188
|
+
frontend source for driving it: your entire value is that you see what a user
|
|
189
|
+
sees, and a static read of a component is a different, weaker claim that another
|
|
190
|
+
agent is better placed to make."""
|
|
191
|
+
|
|
192
|
+
scope = (
|
|
193
|
+
"Walk the primary journeys from the task graph only. This run is time-boxed, "
|
|
194
|
+
"so depth on the critical paths beats coverage."
|
|
195
|
+
if mode == "pr-check"
|
|
196
|
+
else "Walk the primary journeys from the task graph first, then explore. "
|
|
197
|
+
"Exploratory wandering is where you find what nobody wrote a test for — "
|
|
198
|
+
"take the paths a confused or impatient user would take."
|
|
199
|
+
)
|
|
200
|
+
return f"""Explore the running product at {p.environment.web_url} as a user experiences it.
|
|
201
|
+
|
|
202
|
+
{p.description.strip()}
|
|
203
|
+
|
|
204
|
+
Start with `get_system_map` for `ui_routes` and `task_graph`. That is your itinerary.
|
|
205
|
+
|
|
206
|
+
{_environment_brief(p)}
|
|
207
|
+
|
|
208
|
+
{_auth_brief(p)}
|
|
209
|
+
|
|
210
|
+
{scope}
|
|
211
|
+
|
|
212
|
+
At every step: read the page, check the console, interact, observe what changed.
|
|
213
|
+
When something is wrong, find the shortest path to it, then capture a screenshot
|
|
214
|
+
and the console output as evidence before moving on.
|
|
215
|
+
|
|
216
|
+
Judge like a user, not like a reviewer with opinions about the code. Report what
|
|
217
|
+
fails, misleads, blocks or excludes someone. Do not report what you would have
|
|
218
|
+
designed differently."""
|
|
219
|
+
|
|
220
|
+
|
|
221
|
+
def discovery(config: SystemConfig, mode: str, spec: "AgentSpec") -> str:
|
|
222
|
+
"""The task for a discovery agent with no hand-written builder.
|
|
223
|
+
|
|
224
|
+
The architecture's claim is that adding an agent needs a prompt file and a
|
|
225
|
+
YAML file and no Python. That was not true: `_phase_discover` dispatched
|
|
226
|
+
from a hardcoded dict of builders, so a new discovery agent was silently
|
|
227
|
+
skipped with `no task builder` -- it validated, it assembled, it appeared in
|
|
228
|
+
`--dry-run`, and then it did nothing. DBA and AUDITOR were added exactly
|
|
229
|
+
that way and this is the bug they found.
|
|
230
|
+
|
|
231
|
+
What an agent should be told is: which application, what it can reach, and
|
|
232
|
+
what its own prompt says its domain is. Everything specific to a domain
|
|
233
|
+
belongs in that agent's prompt, not here -- API and BROWSER keep their
|
|
234
|
+
bespoke builders because they name tools (`diff_openapi`, the browser) that
|
|
235
|
+
only they have.
|
|
236
|
+
"""
|
|
237
|
+
p = _profile(config)
|
|
238
|
+
reach = (
|
|
239
|
+
"The application is reachable, so prove what you report: observe the "
|
|
240
|
+
"behaviour and capture the evidence. A finding you have not observed is a "
|
|
241
|
+
"hypothesis, and its confidence should say so."
|
|
242
|
+
if p.environment.is_reachable
|
|
243
|
+
else "There is no reachable instance, so every finding is a reading of the "
|
|
244
|
+
"code. Quote the lines that support it and keep your confidence honest "
|
|
245
|
+
"about not having observed the behaviour."
|
|
246
|
+
)
|
|
247
|
+
return f"""Audit {p.name} for defects in your domain.
|
|
248
|
+
|
|
249
|
+
Layout — {p.layout.described()}
|
|
250
|
+
|
|
251
|
+
Your own instructions define what your domain is and what counts as evidence in
|
|
252
|
+
it. Work within it and leave the other surfaces to the agents that own them.
|
|
253
|
+
|
|
254
|
+
{reach}
|
|
255
|
+
|
|
256
|
+
Emit one envelope per distinct defect with `emit_envelope`. Finding nothing is a
|
|
257
|
+
valid outcome; inventing something to report is not. Deduplicate against
|
|
258
|
+
`search_similar` before you emit, so a defect this system already knows about
|
|
259
|
+
comes back as an occurrence rather than a new finding.
|
|
260
|
+
|
|
261
|
+
Mode: {mode}."""
|
|
262
|
+
|
|
263
|
+
|
|
264
|
+
def report(config: SystemConfig, mode: str) -> str:
|
|
265
|
+
"""The task for a reporting agent.
|
|
266
|
+
|
|
267
|
+
Its subject is the run, not the application. Every other agent is pointed at
|
|
268
|
+
the target and asked what is wrong with it; a reporting agent is pointed at
|
|
269
|
+
what just happened and asked what it means. So this task names no routes, no
|
|
270
|
+
schema and no layout -- only the run's own record.
|
|
271
|
+
"""
|
|
272
|
+
p = _profile(config)
|
|
273
|
+
return f"""Summarise this run of {p.name}.
|
|
274
|
+
|
|
275
|
+
Your input is the run itself: the envelopes emitted, the verdicts recorded, the
|
|
276
|
+
denials, the escalations, and what each agent cost in turns. Read them with
|
|
277
|
+
`list_envelopes` and the run's own ledger. You are not auditing the application
|
|
278
|
+
-- the other agents did that -- you are auditing what this run learned.
|
|
279
|
+
|
|
280
|
+
Report:
|
|
281
|
+
|
|
282
|
+
- What was found, grouped by severity and domain, and which findings were held
|
|
283
|
+
rather than filed and why.
|
|
284
|
+
- Recurrence: which of these the system has seen before, and how often. A defect
|
|
285
|
+
reported for the fourth time is a different problem from a new one.
|
|
286
|
+
- Where agents were refused or escalated, and whether the refusal looks correct.
|
|
287
|
+
- What the run could NOT do -- surfaces nothing reached, agents that had no
|
|
288
|
+
capability to work with. A gap in coverage is worth as much as a finding, and
|
|
289
|
+
nobody else reports it.
|
|
290
|
+
|
|
291
|
+
Be specific and short. A report that restates every envelope is a worse version
|
|
292
|
+
of `qaas show`. The value here is the pattern across them, and the honest
|
|
293
|
+
account of what was not looked at.
|
|
294
|
+
|
|
295
|
+
Mode: {mode}."""
|
|
296
|
+
|
|
297
|
+
|
|
298
|
+
def reproducer(envelope: DefectEnvelope, config: SystemConfig, flake_runs: int) -> str:
|
|
299
|
+
p = _profile(config)
|
|
300
|
+
evidence = "\n".join(f" - {e.type.value}: {e.uri} {e.note}".rstrip() for e in envelope.evidence)
|
|
301
|
+
managed = p.environment.is_managed
|
|
302
|
+
pinning = (
|
|
303
|
+
"Pin the environment with `env_control` — a known fixture, known flags, a "
|
|
304
|
+
"known branch — so the reproduction runs identically later."
|
|
305
|
+
if managed
|
|
306
|
+
else "You cannot reset this environment, so pin what you can and record the "
|
|
307
|
+
"rest: note in `environment` exactly what state you found it in. An "
|
|
308
|
+
"unpinnable reproduction is worth recording as such, not worth faking."
|
|
309
|
+
)
|
|
310
|
+
return f"""Reproduce this finding, or demote it.
|
|
311
|
+
|
|
312
|
+
id: {envelope.id}
|
|
313
|
+
reported by {envelope.discovered_by} as {envelope.severity.value} / {envelope.domain.value}
|
|
314
|
+
title: {envelope.title}
|
|
315
|
+
summary: {envelope.summary}
|
|
316
|
+
location: {envelope.location.model_dump(exclude_none=True)}
|
|
317
|
+
confidence: {envelope.confidence:.2f}
|
|
318
|
+
evidence:
|
|
319
|
+
{evidence or " (none attached)"}
|
|
320
|
+
|
|
321
|
+
The application is at `{_where(config)}`, which is your working directory. {pinning}
|
|
322
|
+
|
|
323
|
+
Find the shortest path that makes the defect appear. Write a failing test for it
|
|
324
|
+
under `qa/repro/` on a `qa/repro/*` branch, and run it {flake_runs} times with
|
|
325
|
+
`run_n_times` to measure flake.
|
|
326
|
+
|
|
327
|
+
Finish by calling `record_reproduction` with your verdict. If you could not make
|
|
328
|
+
it happen, say `not_reproducible` and lower the confidence to match. That is a
|
|
329
|
+
useful, correct outcome — it is the filter this whole system depends on, and
|
|
330
|
+
passing through a finding you could not reproduce costs more than dropping a
|
|
331
|
+
real one."""
|
|
332
|
+
|
|
333
|
+
|
|
334
|
+
def triage(config: SystemConfig, cap: int) -> str:
|
|
335
|
+
p = _profile(config)
|
|
336
|
+
owners = (
|
|
337
|
+
f"Resolve component and team from the system map's ownership section, which "
|
|
338
|
+
f"came from `{p.layout.ownership}`."
|
|
339
|
+
if p.layout.ownership
|
|
340
|
+
else "This repository records no ownership, so file unassigned and say so. "
|
|
341
|
+
"Do not infer a team from a directory name."
|
|
342
|
+
)
|
|
343
|
+
return f"""Triage this run's findings and file what deserves to be filed.
|
|
344
|
+
|
|
345
|
+
Call `list_envelopes` with `fileable_only: true` to see what reached you.
|
|
346
|
+
Findings that failed the evidence or confidence gate are not in that list and are
|
|
347
|
+
not yours to file — they are already in the human review queue.
|
|
348
|
+
|
|
349
|
+
For each one, in this order:
|
|
350
|
+
|
|
351
|
+
1. `search_similar` and `get_occurrences` first. An existing ticket gets the new
|
|
352
|
+
evidence and an incremented count, not a second ticket.
|
|
353
|
+
2. Score severity against the rubric, by consequence to users.
|
|
354
|
+
3. {owners}
|
|
355
|
+
4. Compose the ticket in the house format, with REPRODUCER's steps verbatim and the
|
|
356
|
+
failing test as the acceptance criterion.
|
|
357
|
+
5. Route by class. Security findings go to the restricted project.
|
|
358
|
+
6. `record` the defect in memory with its ticket key, so the next run dedupes
|
|
359
|
+
against it.
|
|
360
|
+
|
|
361
|
+
You may file at most {cap} tickets in this run. If you reach that, stop and
|
|
362
|
+
escalate rather than filing more."""
|
|
363
|
+
|
|
364
|
+
|
|
365
|
+
def verifier(ticket_key: str, envelope: DefectEnvelope | None, branch: str) -> str:
|
|
366
|
+
repro = ""
|
|
367
|
+
if envelope:
|
|
368
|
+
steps = "\n".join(f" {i}. {s}" for i, s in enumerate(envelope.reproduction.steps, 1))
|
|
369
|
+
repro = f"""
|
|
370
|
+
The original finding:
|
|
371
|
+
|
|
372
|
+
title: {envelope.title}
|
|
373
|
+
failing test: {envelope.reproduction.failing_test or "(none recorded)"}
|
|
374
|
+
environment: {envelope.reproduction.environment.model_dump()}
|
|
375
|
+
steps:
|
|
376
|
+
{steps or " (none recorded)"}
|
|
377
|
+
"""
|
|
378
|
+
return f"""Verify the fix on ticket {ticket_key}, on branch `{branch}`.
|
|
379
|
+
{repro}
|
|
380
|
+
Bring up the patched build with `env_control`, using the same fixture and flags
|
|
381
|
+
as the original reproduction — a different environment proves nothing.
|
|
382
|
+
|
|
383
|
+
Run the original failing test first. It must now pass. Then run the regression
|
|
384
|
+
suite for the affected area, selected with `affected_tests` against the diff.
|
|
385
|
+
|
|
386
|
+
Return exactly one verdict — VERIFIED, NOT_FIXED or REGRESSED — and transition
|
|
387
|
+
the ticket accordingly. Say what you actually observed, including anything you
|
|
388
|
+
skipped or could not run."""
|
|
389
|
+
|
|
390
|
+
|
|
391
|
+
def fixer(
|
|
392
|
+
ticket_key: str,
|
|
393
|
+
envelope: DefectEnvelope | None,
|
|
394
|
+
config: SystemConfig | None = None,
|
|
395
|
+
) -> str:
|
|
396
|
+
"""The fix task. FIXER is Phase 3; the router's loop calls this once it exists."""
|
|
397
|
+
where = f"the application at `{_where(config)}`" if config and config.profile else "the target application"
|
|
398
|
+
detail = ""
|
|
399
|
+
if envelope:
|
|
400
|
+
steps = "\n".join(f" {i}. {s}" for i, s in enumerate(envelope.reproduction.steps, 1))
|
|
401
|
+
detail = f"""
|
|
402
|
+
title: {envelope.title}
|
|
403
|
+
summary: {envelope.summary}
|
|
404
|
+
location: {envelope.location.model_dump(exclude_none=True)}
|
|
405
|
+
failing test: {envelope.reproduction.failing_test or "(none recorded)"}
|
|
406
|
+
steps:
|
|
407
|
+
{steps or " (none recorded)"}
|
|
408
|
+
"""
|
|
409
|
+
return f"""Fix ticket {ticket_key} in {where}.
|
|
410
|
+
{detail}
|
|
411
|
+
Read the affected code with the system map for context, then write the smallest
|
|
412
|
+
change that makes the failing test pass. Add a regression test. Run the affected
|
|
413
|
+
suite locally before you open anything.
|
|
414
|
+
|
|
415
|
+
The failing test defines success and you may not edit it. If you believe the test
|
|
416
|
+
itself is wrong, that is an escalation, not a licence to change it.
|
|
417
|
+
|
|
418
|
+
Work on a `fix/*` branch. Open the pull request as a draft with a rollback note.
|
|
419
|
+
Never merge — merge is a human decision."""
|
|
420
|
+
|
|
421
|
+
|
|
422
|
+
def reviewer(ticket_key: str, envelope: DefectEnvelope | None = None) -> str:
|
|
423
|
+
"""The adversarial review task. REVIEWER is Phase 3."""
|
|
424
|
+
context = ""
|
|
425
|
+
if envelope:
|
|
426
|
+
context = (
|
|
427
|
+
f"\n\nThe defect it claims to fix: {envelope.title}\n"
|
|
428
|
+
f"The test that defines success: {envelope.reproduction.failing_test or '(none recorded)'}\n"
|
|
429
|
+
)
|
|
430
|
+
return f"""Review the fix for {ticket_key} as an adversarial reviewer.{context}
|
|
431
|
+
|
|
432
|
+
Does the change address the root cause or only the symptom? Is the diff minimal?
|
|
433
|
+
Does it break a contract, a schema, or a public API? Does it introduce a security
|
|
434
|
+
or performance regression? Are the regression tests real, or asserted to pass?
|
|
435
|
+
Is the rollback note viable?
|
|
436
|
+
|
|
437
|
+
Return APPROVE, REQUEST_CHANGES with specifics, or ESCALATE_TO_HUMAN. You have no
|
|
438
|
+
write access to code: your judgement is the deliverable."""
|