qaas-python 0.0.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- qaas/adapters/__init__.py +19 -0
- qaas/adapters/tracker.py +1783 -0
- qaas/adapters/vcs.py +555 -0
- qaas/cli.py +1757 -0
- qaas/config.py +409 -0
- qaas/defaults/config/agents/api.yaml +18 -0
- qaas/defaults/config/agents/architect.yaml +21 -0
- qaas/defaults/config/agents/auditor.yaml +19 -0
- qaas/defaults/config/agents/browser.yaml +15 -0
- qaas/defaults/config/agents/dba.yaml +20 -0
- qaas/defaults/config/agents/fixer.yaml +55 -0
- qaas/defaults/config/agents/guide.yaml +23 -0
- qaas/defaults/config/agents/load.yaml +26 -0
- qaas/defaults/config/agents/mapper.yaml +19 -0
- qaas/defaults/config/agents/reporter.yaml +19 -0
- qaas/defaults/config/agents/reproducer.yaml +21 -0
- qaas/defaults/config/agents/reviewer.yaml +18 -0
- qaas/defaults/config/agents/socket.yaml +23 -0
- qaas/defaults/config/agents/triage.yaml +20 -0
- qaas/defaults/config/agents/verifier.yaml +20 -0
- qaas/defaults/config/system.yaml +64 -0
- qaas/discover.py +242 -0
- qaas/envelope.py +318 -0
- qaas/envfile.py +100 -0
- qaas/guardrails.py +589 -0
- qaas/mcp/__init__.py +0 -0
- qaas/mcp/context.py +78 -0
- qaas/mcp/contract_diff.py +1011 -0
- qaas/mcp/defect_memory.py +495 -0
- qaas/mcp/env_control.py +925 -0
- qaas/mcp/envelope_server.py +463 -0
- qaas/mcp/test_runner.py +842 -0
- qaas/mcp/tracker.py +420 -0
- qaas/mcp/vcs.py +501 -0
- qaas/paths.py +317 -0
- qaas/plugin/.claude-plugin/plugin.json +9 -0
- qaas/plugin/skills/a11y-audit/SKILL.md +34 -0
- qaas/plugin/skills/adversarial-review/SKILL.md +120 -0
- qaas/plugin/skills/api-surface-extraction/SKILL.md +38 -0
- qaas/plugin/skills/authz-matrix-check/SKILL.md +46 -0
- qaas/plugin/skills/console-error-triage/SKILL.md +39 -0
- qaas/plugin/skills/contract-test-generation/SKILL.md +36 -0
- qaas/plugin/skills/dedupe-strategy/SKILL.md +39 -0
- qaas/plugin/skills/environment-pinning/SKILL.md +35 -0
- qaas/plugin/skills/error-taxonomy/SKILL.md +42 -0
- qaas/plugin/skills/exploratory-ui-walk/SKILL.md +46 -0
- qaas/plugin/skills/failing-test-authoring/SKILL.md +47 -0
- qaas/plugin/skills/flake-detection/SKILL.md +39 -0
- qaas/plugin/skills/form-state-probe/SKILL.md +36 -0
- qaas/plugin/skills/minimal-diff-discipline/SKILL.md +70 -0
- qaas/plugin/skills/openapi-diff/SKILL.md +45 -0
- qaas/plugin/skills/ownership-resolution/SKILL.md +31 -0
- qaas/plugin/skills/product-task-graph/SKILL.md +35 -0
- qaas/plugin/skills/regression-risk-scoring/SKILL.md +59 -0
- qaas/plugin/skills/regression-suite-selection/SKILL.md +36 -0
- qaas/plugin/skills/repo-cartography/SKILL.md +38 -0
- qaas/plugin/skills/repro-minimisation/SKILL.md +41 -0
- qaas/plugin/skills/rollback-plan-authoring/SKILL.md +81 -0
- qaas/plugin/skills/root-cause-vs-symptom/SKILL.md +67 -0
- qaas/plugin/skills/routing-rules/SKILL.md +34 -0
- qaas/plugin/skills/severity-rubric/SKILL.md +42 -0
- qaas/plugin/skills/test-first-fix/SKILL.md +66 -0
- qaas/plugin/skills/test-quality-audit/SKILL.md +58 -0
- qaas/plugin/skills/ticket-writer/SKILL.md +40 -0
- qaas/plugin/skills/verdict-reporting/SKILL.md +35 -0
- qaas/plugin/skills/verification-protocol/SKILL.md +39 -0
- qaas/prompts/API.md +44 -0
- qaas/prompts/ARCHITECT.md +80 -0
- qaas/prompts/AUDITOR.md +62 -0
- qaas/prompts/BROWSER.md +46 -0
- qaas/prompts/DBA.md +59 -0
- qaas/prompts/FIXER.md +55 -0
- qaas/prompts/GUIDE.md +94 -0
- qaas/prompts/LOAD.md +109 -0
- qaas/prompts/MAPPER.md +46 -0
- qaas/prompts/REPORTER.md +61 -0
- qaas/prompts/REPRODUCER.md +43 -0
- qaas/prompts/REVIEWER.md +53 -0
- qaas/prompts/SOCKET.md +100 -0
- qaas/prompts/TRIAGE.md +45 -0
- qaas/prompts/VERIFIER.md +41 -0
- qaas/prompts/_shared.md +45 -0
- qaas/registry.py +496 -0
- qaas/router.py +581 -0
- qaas/runner.py +210 -0
- qaas/scorecard.py +448 -0
- qaas/sdk_compat.py +52 -0
- qaas/store.py +323 -0
- qaas/target.py +287 -0
- qaas/tasks.py +438 -0
- qaas/trace.py +342 -0
- qaas_python-0.0.1.dist-info/METADATA +429 -0
- qaas_python-0.0.1.dist-info/RECORD +96 -0
- qaas_python-0.0.1.dist-info/WHEEL +4 -0
- qaas_python-0.0.1.dist-info/entry_points.txt +2 -0
- qaas_python-0.0.1.dist-info/licenses/LICENSE +21 -0
qaas/router.py
ADDED
|
@@ -0,0 +1,581 @@
|
|
|
1
|
+
"""ROUTER — the run state machine.
|
|
2
|
+
|
|
3
|
+
Deliberately not an LLM. §4.1 wants its reasoning shallow (routing, not analysis)
|
|
4
|
+
and §10 makes it the enforcement point for budget and concurrency — and a model
|
|
5
|
+
cannot enforce a budget it is itself spending. Everything here is ordinary code:
|
|
6
|
+
dispatch, phase ordering, concurrency limits, the spend governor, the §8.3 loop
|
|
7
|
+
breakers, retries and escalation.
|
|
8
|
+
|
|
9
|
+
The phases exist because the dependencies are real, not for tidiness:
|
|
10
|
+
|
|
11
|
+
map -> discover -> reproduce -> file
|
|
12
|
+
|
|
13
|
+
Discovery cannot start without the map. Triage cannot start without findings.
|
|
14
|
+
Within a phase, agents are independent and run concurrently up to the mode's cap.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
import asyncio
|
|
20
|
+
import subprocess
|
|
21
|
+
import time
|
|
22
|
+
from dataclasses import dataclass, field
|
|
23
|
+
from pathlib import Path
|
|
24
|
+
from typing import Any, Callable
|
|
25
|
+
|
|
26
|
+
from qaas.config import AgentSpec, SystemConfig, load_config
|
|
27
|
+
from qaas.envelope import DefectEnvelope
|
|
28
|
+
from qaas.target import agent_usable
|
|
29
|
+
from qaas.mcp.context import ToolContext
|
|
30
|
+
from qaas.runner import RunOutcome, run_agent
|
|
31
|
+
from qaas.store import RunStore, SystemMapStore
|
|
32
|
+
from qaas import tasks
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
class BudgetExceeded(RuntimeError):
|
|
36
|
+
"""The run hit its spend or wall-clock cap. Not an error — a control working."""
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def target_revision(root: Path | str | None) -> dict[str, Any]:
|
|
40
|
+
"""What commit of the target this run is looking at, for `run_started`.
|
|
41
|
+
|
|
42
|
+
Without this a run is not pinned to any code: the ledger said which agents
|
|
43
|
+
ran and what they spent, but nothing said *what they read*, so a finding
|
|
44
|
+
could never be replayed against the tree that produced it. Now `qaas show`
|
|
45
|
+
can name the commit.
|
|
46
|
+
|
|
47
|
+
`dirty` is not decoration — a run against an edited working tree is not
|
|
48
|
+
pinned by its sha either, and that has to be visible rather than implied.
|
|
49
|
+
|
|
50
|
+
Never raises. A target that is not a git checkout (or has no git at all) is
|
|
51
|
+
an ordinary, supported state: the fields come back None and the run
|
|
52
|
+
proceeds. Provenance is worth recording, never worth failing a run for.
|
|
53
|
+
"""
|
|
54
|
+
if root is None:
|
|
55
|
+
return {"target_root": None, "target_sha": None, "target_dirty": None}
|
|
56
|
+
root = Path(root)
|
|
57
|
+
info: dict[str, Any] = {"target_root": str(root), "target_sha": None, "target_dirty": None}
|
|
58
|
+
|
|
59
|
+
def git(*args: str) -> str | None:
|
|
60
|
+
try:
|
|
61
|
+
proc = subprocess.run(
|
|
62
|
+
["git", "-C", str(root), *args],
|
|
63
|
+
capture_output=True, text=True, timeout=10, check=False,
|
|
64
|
+
)
|
|
65
|
+
except (OSError, subprocess.SubprocessError):
|
|
66
|
+
return None
|
|
67
|
+
return proc.stdout if proc.returncode == 0 else None
|
|
68
|
+
|
|
69
|
+
sha = git("rev-parse", "HEAD")
|
|
70
|
+
if sha is None:
|
|
71
|
+
return info # not a repo, no git, or an empty repo with no commits yet
|
|
72
|
+
info["target_sha"] = sha.strip()
|
|
73
|
+
status = git("status", "--porcelain")
|
|
74
|
+
if status is not None:
|
|
75
|
+
info["target_dirty"] = bool(status.strip())
|
|
76
|
+
branch = git("rev-parse", "--abbrev-ref", "HEAD")
|
|
77
|
+
if branch:
|
|
78
|
+
info["target_branch"] = branch.strip()
|
|
79
|
+
return info
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
@dataclass
|
|
83
|
+
class RunReport:
|
|
84
|
+
run_id: str
|
|
85
|
+
mode: str
|
|
86
|
+
outcomes: list[RunOutcome] = field(default_factory=list)
|
|
87
|
+
escalations: list[str] = field(default_factory=list)
|
|
88
|
+
stopped_early: str | None = None
|
|
89
|
+
|
|
90
|
+
@property
|
|
91
|
+
def cost_usd(self) -> float:
|
|
92
|
+
return sum(o.result.cost_usd for o in self.outcomes)
|
|
93
|
+
|
|
94
|
+
@property
|
|
95
|
+
def failed(self) -> list[str]:
|
|
96
|
+
return [o.result.agent for o in self.outcomes if not o.ok]
|
|
97
|
+
|
|
98
|
+
def summary(self) -> dict[str, Any]:
|
|
99
|
+
return {
|
|
100
|
+
"run_id": self.run_id,
|
|
101
|
+
"mode": self.mode,
|
|
102
|
+
"agents_run": len(self.outcomes),
|
|
103
|
+
"failed": self.failed,
|
|
104
|
+
"cost_usd": round(self.cost_usd, 4),
|
|
105
|
+
"escalations": self.escalations,
|
|
106
|
+
"stopped_early": self.stopped_early,
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
class Budget:
|
|
111
|
+
"""The spend and wall-clock governor. Checked before every dispatch."""
|
|
112
|
+
|
|
113
|
+
def __init__(self, max_usd: float | None, max_seconds: int, *, already_spent: float = 0.0):
|
|
114
|
+
self.max_usd = max_usd
|
|
115
|
+
self.max_seconds = max_seconds
|
|
116
|
+
#: What this run has already cost, including earlier invocations.
|
|
117
|
+
#:
|
|
118
|
+
#: A resumed run (`qaas run --run-id <existing>`) used to start the
|
|
119
|
+
#: counter at zero, so the cap was per *invocation*, not per run --
|
|
120
|
+
#: resume three times against a $20 mode and you could spend $60 while
|
|
121
|
+
#: every individual pass reported itself within budget. One real run
|
|
122
|
+
#: shows the effect: `cost $32.90 of $20.00 budget`. The wall clock is
|
|
123
|
+
#: deliberately NOT carried across: it measures this process, and a run
|
|
124
|
+
#: resumed the next morning has not been running all night.
|
|
125
|
+
self.spent = already_spent
|
|
126
|
+
self.started = time.monotonic()
|
|
127
|
+
|
|
128
|
+
@property
|
|
129
|
+
def elapsed(self) -> float:
|
|
130
|
+
return time.monotonic() - self.started
|
|
131
|
+
|
|
132
|
+
@property
|
|
133
|
+
def remaining_usd(self) -> float | None:
|
|
134
|
+
return None if self.max_usd is None else max(0.0, self.max_usd - self.spent)
|
|
135
|
+
|
|
136
|
+
def spend(self, amount: float) -> None:
|
|
137
|
+
self.spent += amount
|
|
138
|
+
|
|
139
|
+
def check(self) -> None:
|
|
140
|
+
# `max_usd is None` means no spend ceiling -- the shipped config sets
|
|
141
|
+
# none, because a dollar figure bakes one vendor's pricing into a tool
|
|
142
|
+
# meant to run against local models too. The wall-clock cap and each
|
|
143
|
+
# agent's `max_turns` still bound a run; those are model-agnostic.
|
|
144
|
+
if self.max_usd is not None and self.spent >= self.max_usd:
|
|
145
|
+
raise BudgetExceeded(f"spend cap reached: ${self.spent:.2f} of ${self.max_usd:.2f}")
|
|
146
|
+
if self.elapsed >= self.max_seconds:
|
|
147
|
+
raise BudgetExceeded(
|
|
148
|
+
f"wall-clock cap reached: {self.elapsed:.0f}s of {self.max_seconds}s"
|
|
149
|
+
)
|
|
150
|
+
|
|
151
|
+
def allowance(self, spec: AgentSpec) -> float | None:
|
|
152
|
+
"""What this agent may spend, or None when neither it nor the run caps it."""
|
|
153
|
+
caps = [c for c in (spec.max_budget_usd, self.remaining_usd) if c is not None]
|
|
154
|
+
return max(0.01, min(caps)) if caps else None
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
class Router:
|
|
158
|
+
"""Owns one run from trigger to report."""
|
|
159
|
+
|
|
160
|
+
def __init__(
|
|
161
|
+
self,
|
|
162
|
+
config: SystemConfig,
|
|
163
|
+
target_root: Path | None = None,
|
|
164
|
+
*,
|
|
165
|
+
root: Path | str = ".qaas",
|
|
166
|
+
on_event: Callable[[str, dict[str, Any]], None] | None = None,
|
|
167
|
+
tickets: list[str] | None = None,
|
|
168
|
+
):
|
|
169
|
+
self.config = config
|
|
170
|
+
#: The application under test. Defaults to whatever the active profile
|
|
171
|
+
#: says, which is the answer every caller wants; an explicit path is for
|
|
172
|
+
#: tests and for a run pointed at a clone that has no profile yet.
|
|
173
|
+
self.target_root = Path(target_root) if target_root is not None else config.target_root()
|
|
174
|
+
self.maps = SystemMapStore(root)
|
|
175
|
+
self.root = Path(root)
|
|
176
|
+
self.on_event = on_event
|
|
177
|
+
#: When set, a fix cycle works only these tickets. Verifying ten tickets
|
|
178
|
+
#: costs ten times as much as verifying one, and during development you
|
|
179
|
+
#: almost always want one.
|
|
180
|
+
self.tickets = set(tickets) if tickets else None
|
|
181
|
+
|
|
182
|
+
def _target_root(self) -> Path | None:
|
|
183
|
+
"""The checkout under examination, or None when nothing is configured.
|
|
184
|
+
|
|
185
|
+
Deliberately the same value the agents' tools are pointed at, so the
|
|
186
|
+
commit recorded in the ledger is the commit they actually read. This
|
|
187
|
+
used to compute `self.repo_root / config.target_app`; both of those are
|
|
188
|
+
gone -- `repo_root` conflated the qaas project with the target, and
|
|
189
|
+
`target_app` was the demo-shaped default that made the conflation look
|
|
190
|
+
like it worked.
|
|
191
|
+
"""
|
|
192
|
+
return self.target_root
|
|
193
|
+
|
|
194
|
+
def _emit(self, kind: str, **detail: Any) -> None:
|
|
195
|
+
if self.on_event:
|
|
196
|
+
self.on_event(kind, detail)
|
|
197
|
+
|
|
198
|
+
def _context(self, store: RunStore, spec: AgentSpec, map_version: str | None) -> ToolContext:
|
|
199
|
+
return ToolContext(
|
|
200
|
+
store=store,
|
|
201
|
+
maps=self.maps,
|
|
202
|
+
config=self.config,
|
|
203
|
+
agent=spec,
|
|
204
|
+
target_root=self.target_root,
|
|
205
|
+
map_version=map_version,
|
|
206
|
+
)
|
|
207
|
+
|
|
208
|
+
# -- the run ----------------------------------------------------------
|
|
209
|
+
|
|
210
|
+
async def run(self, mode: str, *, run_id: str | None = None) -> RunReport:
|
|
211
|
+
specs = {s.name: s for s in self.config.enabled_agents(mode)}
|
|
212
|
+
run_mode = self.config.run_modes[mode]
|
|
213
|
+
store = RunStore(run_id, self.root) if run_id else RunStore.new(self.root)
|
|
214
|
+
# Carry forward what this run id has already spent, so a cap survives a
|
|
215
|
+
# resumption instead of resetting with it.
|
|
216
|
+
budget = Budget(
|
|
217
|
+
run_mode.max_budget_usd,
|
|
218
|
+
run_mode.max_wall_clock_s,
|
|
219
|
+
already_spent=store.total_cost_usd() if run_id else 0.0,
|
|
220
|
+
)
|
|
221
|
+
report = RunReport(run_id=store.run_id, mode=mode)
|
|
222
|
+
|
|
223
|
+
store.log(
|
|
224
|
+
"run_started",
|
|
225
|
+
mode=mode,
|
|
226
|
+
agents=sorted(specs),
|
|
227
|
+
budget_usd=run_mode.max_budget_usd,
|
|
228
|
+
wall_clock_s=run_mode.max_wall_clock_s,
|
|
229
|
+
**target_revision(self._target_root()),
|
|
230
|
+
)
|
|
231
|
+
self._emit("run_started", run_id=store.run_id, mode=mode, agents=sorted(specs))
|
|
232
|
+
|
|
233
|
+
try:
|
|
234
|
+
map_version = await self._phase_map(specs, store, budget, report)
|
|
235
|
+
await self._phase_discover(specs, store, budget, report, mode, map_version)
|
|
236
|
+
await self._phase_reproduce(specs, store, budget, report, map_version)
|
|
237
|
+
if run_mode.files_tickets:
|
|
238
|
+
await self._phase_file(specs, store, budget, report, map_version)
|
|
239
|
+
else:
|
|
240
|
+
store.log("skipped", reason="mode does not file tickets", mode=mode)
|
|
241
|
+
await self._phase_verify(specs, store, budget, report, map_version)
|
|
242
|
+
await self._phase_report(specs, store, budget, report, mode, map_version)
|
|
243
|
+
except BudgetExceeded as exc:
|
|
244
|
+
report.stopped_early = str(exc)
|
|
245
|
+
report.escalations.append(str(exc))
|
|
246
|
+
store.log("escalation", reason=str(exc))
|
|
247
|
+
self._emit("stopped", reason=str(exc))
|
|
248
|
+
|
|
249
|
+
store.log("run_finished", **report.summary())
|
|
250
|
+
self._emit("run_finished", **report.summary())
|
|
251
|
+
return report
|
|
252
|
+
|
|
253
|
+
# -- phases -----------------------------------------------------------
|
|
254
|
+
|
|
255
|
+
async def _phase_map(self, specs, store, budget, report) -> str | None:
|
|
256
|
+
"""Publish the map first. Everything downstream reads it."""
|
|
257
|
+
spec = specs.get("MAPPER")
|
|
258
|
+
if spec is None:
|
|
259
|
+
return self.maps.latest_version()
|
|
260
|
+
|
|
261
|
+
budget.check()
|
|
262
|
+
before = self.maps.latest_version()
|
|
263
|
+
outcome = await self._dispatch(spec, store, budget, report, tasks.mapper(self.config), None)
|
|
264
|
+
|
|
265
|
+
version = self.maps.latest_version()
|
|
266
|
+
if version == before or version is None:
|
|
267
|
+
# Everything downstream reads the map. A stale one is a worse failure
|
|
268
|
+
# than a missing one, so say plainly which we are running on.
|
|
269
|
+
note = "MAPPER published no map" + (
|
|
270
|
+
f"; continuing on the previous map {before}" if before else "; no map exists"
|
|
271
|
+
)
|
|
272
|
+
report.escalations.append(note)
|
|
273
|
+
store.log("escalation", agent="MAPPER", reason=note)
|
|
274
|
+
return version or before
|
|
275
|
+
|
|
276
|
+
async def _phase_discover(self, specs, store, budget, report, mode, map_version) -> None:
|
|
277
|
+
"""Discovery agents are independent. Run them concurrently, bounded."""
|
|
278
|
+
discovery = [s for name, s in specs.items() if s.layer == "discovery"]
|
|
279
|
+
|
|
280
|
+
# Skip agents this target cannot support. `qaas doctor` has always
|
|
281
|
+
# reported these ("agents that cannot: BROWSER"), but nothing acted on
|
|
282
|
+
# it, so a run against a target with no reachable UI would still
|
|
283
|
+
# dispatch BROWSER and spend its entire budget hunting a browser that
|
|
284
|
+
# was never there. Being told an agent cannot work and then watching it
|
|
285
|
+
# run is worse than not being told.
|
|
286
|
+
profile = getattr(self.config, "profile", None)
|
|
287
|
+
if profile is not None:
|
|
288
|
+
caps = profile.capabilities()
|
|
289
|
+
unusable = [s for s in discovery if not agent_usable(s.name, caps)]
|
|
290
|
+
if unusable:
|
|
291
|
+
discovery = [s for s in discovery if s not in unusable]
|
|
292
|
+
store.log(
|
|
293
|
+
"skipped",
|
|
294
|
+
reason="target cannot support these agents",
|
|
295
|
+
agents=[s.name for s in unusable],
|
|
296
|
+
)
|
|
297
|
+
for spec in unusable:
|
|
298
|
+
self._emit("skipped", agent=spec.name, reason="target lacks the capability")
|
|
299
|
+
|
|
300
|
+
if not discovery:
|
|
301
|
+
return
|
|
302
|
+
|
|
303
|
+
# API and BROWSER keep bespoke tasks because they name tools only
|
|
304
|
+
# they have. Everything else gets the generic discovery task, which is
|
|
305
|
+
# what makes "a new agent is a prompt plus a YAML" true: this used to be
|
|
306
|
+
# a closed dict, so a new discovery agent was skipped with `no task
|
|
307
|
+
# builder` -- it validated, it assembled, it showed up in `--dry-run`,
|
|
308
|
+
# and then it silently did nothing.
|
|
309
|
+
builders = {
|
|
310
|
+
"API": lambda spec: tasks.api(self.config, mode),
|
|
311
|
+
"BROWSER": lambda spec: tasks.browser(self.config, mode),
|
|
312
|
+
}
|
|
313
|
+
jobs = [
|
|
314
|
+
(spec, builders.get(spec.name, lambda sp: tasks.discovery(self.config, mode, sp))(spec))
|
|
315
|
+
for spec in discovery
|
|
316
|
+
]
|
|
317
|
+
|
|
318
|
+
await self._gather(jobs, store, budget, report, map_version, self.config.run_modes[mode].max_concurrency)
|
|
319
|
+
|
|
320
|
+
async def _phase_report(self, specs, store, budget, report, mode, map_version) -> None:
|
|
321
|
+
"""Reporting agents run last, over what the run itself produced.
|
|
322
|
+
|
|
323
|
+
Dispatched by LAYER, like discovery, and deliberately not by name. Every
|
|
324
|
+
other phase looks up a specific agent (`specs.get("REPRODUCER")`), which is
|
|
325
|
+
why REPORTER could be configured, validated, assembled and shown in
|
|
326
|
+
`--dry-run` while never running: no phase asked for it. That is the same
|
|
327
|
+
silent skip DBA and AUDITOR exposed for discovery, and it is worth
|
|
328
|
+
fixing the shape rather than the instance -- a second reporting agent
|
|
329
|
+
now needs no Python either.
|
|
330
|
+
|
|
331
|
+
A run with no reporting agent is the ordinary case and not worth a log
|
|
332
|
+
line; most modes have none.
|
|
333
|
+
"""
|
|
334
|
+
reporting = [s for s in specs.values() if s.layer == "reporting"]
|
|
335
|
+
if not reporting:
|
|
336
|
+
return
|
|
337
|
+
|
|
338
|
+
for spec in reporting:
|
|
339
|
+
budget.check()
|
|
340
|
+
await self._dispatch(
|
|
341
|
+
spec, store, budget, report, tasks.report(self.config, mode), map_version
|
|
342
|
+
)
|
|
343
|
+
|
|
344
|
+
async def _phase_reproduce(self, specs, store, budget, report, map_version) -> None:
|
|
345
|
+
"""One REPRODUCER invocation per finding.
|
|
346
|
+
|
|
347
|
+
Separate contexts on purpose: reproducing finding B should not inherit
|
|
348
|
+
whatever REPRODUCER talked itself into while working on finding A.
|
|
349
|
+
"""
|
|
350
|
+
spec = specs.get("REPRODUCER")
|
|
351
|
+
if spec is None:
|
|
352
|
+
return
|
|
353
|
+
|
|
354
|
+
drafts = [e for e in store.envelopes() if e.reproduction.status.value == "unattempted"]
|
|
355
|
+
if not drafts:
|
|
356
|
+
store.log("skipped", agent="REPRODUCER", reason="no findings to reproduce")
|
|
357
|
+
return
|
|
358
|
+
|
|
359
|
+
cap = self.config.thresholds.max_findings_per_agent_run
|
|
360
|
+
if len(drafts) > cap:
|
|
361
|
+
note = f"{len(drafts)} findings exceed the per-run cap of {cap}; triaging the most severe"
|
|
362
|
+
report.escalations.append(note)
|
|
363
|
+
store.log("escalation", agent="REPRODUCER", reason=note)
|
|
364
|
+
drafts = sorted(drafts, key=lambda e: (e.severity.rank, -e.confidence))[:cap]
|
|
365
|
+
|
|
366
|
+
jobs = [
|
|
367
|
+
(spec, tasks.reproducer(draft, self.config, self.config.thresholds.flake_runs))
|
|
368
|
+
for draft in drafts
|
|
369
|
+
]
|
|
370
|
+
await self._gather(jobs, store, budget, report, map_version, concurrency=2)
|
|
371
|
+
|
|
372
|
+
async def _phase_file(self, specs, store, budget, report, map_version) -> None:
|
|
373
|
+
spec = specs.get("TRIAGE")
|
|
374
|
+
if spec is None:
|
|
375
|
+
return
|
|
376
|
+
fileable = [
|
|
377
|
+
e for e in store.envelopes()
|
|
378
|
+
if e.is_fileable(self.config.thresholds.min_confidence_to_file)[0]
|
|
379
|
+
]
|
|
380
|
+
if not fileable:
|
|
381
|
+
store.log("skipped", agent="TRIAGE", reason="nothing passed the gates")
|
|
382
|
+
return
|
|
383
|
+
cap = min(spec.policy.max_tickets_per_run, self.config.thresholds.max_tickets_per_run)
|
|
384
|
+
await self._dispatch(spec, store, budget, report, tasks.triage(self.config, cap), map_version)
|
|
385
|
+
|
|
386
|
+
async def _phase_verify(self, specs, store, budget, report, map_version) -> None:
|
|
387
|
+
spec = specs.get("VERIFIER")
|
|
388
|
+
if spec is None:
|
|
389
|
+
return
|
|
390
|
+
pending = [e for e in store.envelopes() if e.jira.key]
|
|
391
|
+
if self.tickets:
|
|
392
|
+
pending = [e for e in pending if e.jira.key in self.tickets]
|
|
393
|
+
unknown = self.tickets - {e.jira.key for e in store.envelopes() if e.jira.key}
|
|
394
|
+
if unknown:
|
|
395
|
+
store.log("skipped", reason="unknown tickets", tickets=sorted(unknown))
|
|
396
|
+
if not pending:
|
|
397
|
+
store.log("skipped", agent="VERIFIER", reason="no tickets to verify")
|
|
398
|
+
return
|
|
399
|
+
for envelope in pending:
|
|
400
|
+
budget.check()
|
|
401
|
+
await self._verify_loop(envelope, specs, store, budget, report, map_version)
|
|
402
|
+
|
|
403
|
+
async def _verify_loop(self, envelope, specs, store, budget, report, map_version) -> None:
|
|
404
|
+
"""VERIFIER -> NOT_FIXED -> remediate -> VERIFIER, bounded by §8.3.
|
|
405
|
+
|
|
406
|
+
The bound is the point. Without `max_proof_reopens` a fix that keeps
|
|
407
|
+
missing the defect cycles until the budget is gone, and the run ends with
|
|
408
|
+
no verdict and no money left to reach one. Escalating after one reopen
|
|
409
|
+
costs a human five minutes; not escalating costs the whole run.
|
|
410
|
+
"""
|
|
411
|
+
ticket = envelope.jira.key
|
|
412
|
+
max_reopens = self.config.thresholds.max_proof_reopens
|
|
413
|
+
reopens = 0
|
|
414
|
+
|
|
415
|
+
# The envelope names the *repro* branch, which by construction carries a
|
|
416
|
+
# failing test and no fix -- it is written before any fix exists. Sending
|
|
417
|
+
# VERIFIER back there after a remediation round made this loop unable to
|
|
418
|
+
# ever reach VERIFIED: FIXER would fix, REVIEWER approve, and VERIFIER
|
|
419
|
+
# re-verify the unfixed branch it had just failed on, burn a reopen and
|
|
420
|
+
# escalate. A live run recorded exactly that ("this branch cannot carry
|
|
421
|
+
# a fix"). Where FIXER put the fix is only knowable after the fact, so
|
|
422
|
+
# it is read back out of the ledger below.
|
|
423
|
+
repro_branch = envelope.reproduction.environment.branch or "main"
|
|
424
|
+
fix_branch: str | None = None
|
|
425
|
+
|
|
426
|
+
while True:
|
|
427
|
+
await self._dispatch(
|
|
428
|
+
specs["VERIFIER"], store, budget, report,
|
|
429
|
+
tasks.verifier(ticket, envelope, branch=fix_branch or repro_branch), map_version,
|
|
430
|
+
)
|
|
431
|
+
verdict = self._latest_verdict(store, ticket)
|
|
432
|
+
|
|
433
|
+
if verdict is None:
|
|
434
|
+
self._escalate(report, store, "VERIFIER",
|
|
435
|
+
f"{ticket}: VERIFIER returned no verdict; the ticket stays in review")
|
|
436
|
+
return
|
|
437
|
+
if verdict == "VERIFIED":
|
|
438
|
+
store.log("verified", agent="VERIFIER", ticket_key=ticket, reopens=reopens)
|
|
439
|
+
return
|
|
440
|
+
if verdict == "REGRESSED":
|
|
441
|
+
self._escalate(report, store, "VERIFIER",
|
|
442
|
+
f"{ticket}: REGRESSED — the fix broke something else; blocking for a human")
|
|
443
|
+
return
|
|
444
|
+
|
|
445
|
+
# NOT_FIXED from here.
|
|
446
|
+
if reopens >= max_reopens:
|
|
447
|
+
self._escalate(report, store, "VERIFIER",
|
|
448
|
+
f"{ticket}: still NOT_FIXED after {reopens} reopen(s), the limit. "
|
|
449
|
+
"Escalating rather than cycling further")
|
|
450
|
+
return
|
|
451
|
+
|
|
452
|
+
reopens += 1
|
|
453
|
+
store.log("reopened", agent="VERIFIER", ticket_key=ticket, attempt=reopens)
|
|
454
|
+
mark = len(list(store.ledger("vcs")))
|
|
455
|
+
if not await self._remediate(envelope, specs, store, budget, report, map_version):
|
|
456
|
+
return
|
|
457
|
+
# Keep the previous branch if this round wrote nothing: a re-verify
|
|
458
|
+
# of the last fix beats silently falling back to the repro branch.
|
|
459
|
+
fix_branch = self._branch_written_since(store, mark) or fix_branch
|
|
460
|
+
|
|
461
|
+
async def _remediate(self, envelope, specs, store, budget, report, map_version) -> bool:
|
|
462
|
+
"""FIXER -> REVIEWER, bounded. Returns whether a fix is ready to re-verify.
|
|
463
|
+
|
|
464
|
+
Phase 3 agents. In a Phase 1 roster neither exists, so a NOT_FIXED
|
|
465
|
+
verdict escalates to a human immediately — which is correct, and much
|
|
466
|
+
better than the loop silently re-running VERIFIER against unchanged code.
|
|
467
|
+
"""
|
|
468
|
+
ticket = envelope.jira.key
|
|
469
|
+
fixer, reviewer = specs.get("FIXER"), specs.get("REVIEWER")
|
|
470
|
+
|
|
471
|
+
if fixer is None:
|
|
472
|
+
self._escalate(report, store, "VERIFIER",
|
|
473
|
+
f"{ticket}: NOT_FIXED and no FIXER in this run's roster. "
|
|
474
|
+
"Nothing here can produce a fix; a human takes it from here")
|
|
475
|
+
return False
|
|
476
|
+
|
|
477
|
+
for trip in range(1, self.config.thresholds.max_mender_arbiter_round_trips + 1):
|
|
478
|
+
budget.check()
|
|
479
|
+
await self._dispatch(fixer, store, budget, report,
|
|
480
|
+
tasks.fixer(ticket, envelope), map_version)
|
|
481
|
+
if reviewer is None:
|
|
482
|
+
return True
|
|
483
|
+
|
|
484
|
+
await self._dispatch(reviewer, store, budget, report,
|
|
485
|
+
tasks.reviewer(ticket, envelope), map_version)
|
|
486
|
+
review = self._latest_review(store, ticket)
|
|
487
|
+
if review == "APPROVE":
|
|
488
|
+
return True
|
|
489
|
+
if review == "ESCALATE_TO_HUMAN":
|
|
490
|
+
self._escalate(report, store, "REVIEWER", f"{ticket}: REVIEWER escalated the fix")
|
|
491
|
+
return False
|
|
492
|
+
store.log("review_round_trip", agent="REVIEWER", ticket_key=ticket, trip=trip)
|
|
493
|
+
|
|
494
|
+
self._escalate(report, store, "REVIEWER",
|
|
495
|
+
f"{ticket}: {self.config.thresholds.max_mender_arbiter_round_trips} "
|
|
496
|
+
"FIXER/REVIEWER round trips without approval; escalating")
|
|
497
|
+
return False
|
|
498
|
+
|
|
499
|
+
@staticmethod
|
|
500
|
+
def _branch_written_since(store, mark: int) -> str | None:
|
|
501
|
+
"""The branch FIXER actually wrote to during one remediation round.
|
|
502
|
+
|
|
503
|
+
Scoped to the ledger entries added since `mark` rather than searched
|
|
504
|
+
run-wide, because a run verifies several tickets against one ledger and
|
|
505
|
+
an earlier ticket's `fix/*` branch is the wrong answer here. The last
|
|
506
|
+
write wins: FIXER ends a successful round on `push` or `open_pr`.
|
|
507
|
+
"""
|
|
508
|
+
for entry in reversed(list(store.ledger("vcs"))[mark:]):
|
|
509
|
+
if entry.agent != "FIXER":
|
|
510
|
+
continue
|
|
511
|
+
branch = entry.detail.get("branch")
|
|
512
|
+
if branch:
|
|
513
|
+
return str(branch)
|
|
514
|
+
return None
|
|
515
|
+
|
|
516
|
+
@staticmethod
|
|
517
|
+
def _latest_verdict(store, ticket_key: str) -> str | None:
|
|
518
|
+
"""VERIFIER's verdict is a typed ledger entry, never parsed from prose."""
|
|
519
|
+
verdicts = [e for e in store.ledger("verdict") if e.detail.get("ticket_key") == ticket_key]
|
|
520
|
+
return verdicts[-1].detail.get("verdict") if verdicts else None
|
|
521
|
+
|
|
522
|
+
@staticmethod
|
|
523
|
+
def _latest_review(store, ticket_key: str) -> str | None:
|
|
524
|
+
reviews = [e for e in store.ledger("review") if e.detail.get("ticket_key") == ticket_key]
|
|
525
|
+
return reviews[-1].detail.get("decision") if reviews else None
|
|
526
|
+
|
|
527
|
+
def _escalate(self, report, store, agent: str, note: str) -> None:
|
|
528
|
+
report.escalations.append(note)
|
|
529
|
+
store.log("escalation", agent=agent, reason=note)
|
|
530
|
+
self._emit("escalation", agent=agent, reason=note)
|
|
531
|
+
|
|
532
|
+
# -- dispatch ---------------------------------------------------------
|
|
533
|
+
|
|
534
|
+
async def _gather(self, jobs, store, budget, report, map_version, concurrency: int) -> None:
|
|
535
|
+
"""Run jobs concurrently, but stop dispatching once the budget is gone.
|
|
536
|
+
|
|
537
|
+
The semaphore bounds how many run at once; the budget check inside each
|
|
538
|
+
slot means a run that blows its cap stops starting new work rather than
|
|
539
|
+
letting everything already queued through.
|
|
540
|
+
"""
|
|
541
|
+
if not jobs:
|
|
542
|
+
return
|
|
543
|
+
sem = asyncio.Semaphore(max(1, concurrency))
|
|
544
|
+
stopped: list[str] = []
|
|
545
|
+
|
|
546
|
+
async def one(spec: AgentSpec, task: str) -> None:
|
|
547
|
+
async with sem:
|
|
548
|
+
if stopped:
|
|
549
|
+
return
|
|
550
|
+
try:
|
|
551
|
+
budget.check()
|
|
552
|
+
except BudgetExceeded as exc:
|
|
553
|
+
stopped.append(str(exc))
|
|
554
|
+
return
|
|
555
|
+
await self._dispatch(spec, store, budget, report, task, map_version)
|
|
556
|
+
|
|
557
|
+
await asyncio.gather(*(one(spec, task) for spec, task in jobs))
|
|
558
|
+
if stopped:
|
|
559
|
+
raise BudgetExceeded(stopped[0])
|
|
560
|
+
|
|
561
|
+
async def _dispatch(self, spec, store, budget, report, task, map_version) -> RunOutcome:
|
|
562
|
+
ctx = self._context(store, spec, map_version)
|
|
563
|
+
allowance = budget.allowance(spec)
|
|
564
|
+
|
|
565
|
+
self._emit("agent_started", agent=spec.name, budget=allowance)
|
|
566
|
+
outcome = await run_agent(
|
|
567
|
+
spec, ctx, task, max_budget_usd=allowance, on_event=self.on_event
|
|
568
|
+
)
|
|
569
|
+
|
|
570
|
+
budget.spend(outcome.result.cost_usd)
|
|
571
|
+
report.outcomes.append(outcome)
|
|
572
|
+
if not outcome.ok:
|
|
573
|
+
note = f"{spec.name} failed: {outcome.result.error or outcome.result.subtype}"
|
|
574
|
+
report.escalations.append(note)
|
|
575
|
+
store.log("escalation", agent=spec.name, reason=note)
|
|
576
|
+
return outcome
|
|
577
|
+
|
|
578
|
+
|
|
579
|
+
def build(config_dir: Path | str = "config", root: Path | str = ".qaas") -> Router:
|
|
580
|
+
config = load_config(config_dir)
|
|
581
|
+
return Router(config, root=root)
|