qaas-python 0.0.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (96) hide show
  1. qaas/adapters/__init__.py +19 -0
  2. qaas/adapters/tracker.py +1783 -0
  3. qaas/adapters/vcs.py +555 -0
  4. qaas/cli.py +1757 -0
  5. qaas/config.py +409 -0
  6. qaas/defaults/config/agents/api.yaml +18 -0
  7. qaas/defaults/config/agents/architect.yaml +21 -0
  8. qaas/defaults/config/agents/auditor.yaml +19 -0
  9. qaas/defaults/config/agents/browser.yaml +15 -0
  10. qaas/defaults/config/agents/dba.yaml +20 -0
  11. qaas/defaults/config/agents/fixer.yaml +55 -0
  12. qaas/defaults/config/agents/guide.yaml +23 -0
  13. qaas/defaults/config/agents/load.yaml +26 -0
  14. qaas/defaults/config/agents/mapper.yaml +19 -0
  15. qaas/defaults/config/agents/reporter.yaml +19 -0
  16. qaas/defaults/config/agents/reproducer.yaml +21 -0
  17. qaas/defaults/config/agents/reviewer.yaml +18 -0
  18. qaas/defaults/config/agents/socket.yaml +23 -0
  19. qaas/defaults/config/agents/triage.yaml +20 -0
  20. qaas/defaults/config/agents/verifier.yaml +20 -0
  21. qaas/defaults/config/system.yaml +64 -0
  22. qaas/discover.py +242 -0
  23. qaas/envelope.py +318 -0
  24. qaas/envfile.py +100 -0
  25. qaas/guardrails.py +589 -0
  26. qaas/mcp/__init__.py +0 -0
  27. qaas/mcp/context.py +78 -0
  28. qaas/mcp/contract_diff.py +1011 -0
  29. qaas/mcp/defect_memory.py +495 -0
  30. qaas/mcp/env_control.py +925 -0
  31. qaas/mcp/envelope_server.py +463 -0
  32. qaas/mcp/test_runner.py +842 -0
  33. qaas/mcp/tracker.py +420 -0
  34. qaas/mcp/vcs.py +501 -0
  35. qaas/paths.py +317 -0
  36. qaas/plugin/.claude-plugin/plugin.json +9 -0
  37. qaas/plugin/skills/a11y-audit/SKILL.md +34 -0
  38. qaas/plugin/skills/adversarial-review/SKILL.md +120 -0
  39. qaas/plugin/skills/api-surface-extraction/SKILL.md +38 -0
  40. qaas/plugin/skills/authz-matrix-check/SKILL.md +46 -0
  41. qaas/plugin/skills/console-error-triage/SKILL.md +39 -0
  42. qaas/plugin/skills/contract-test-generation/SKILL.md +36 -0
  43. qaas/plugin/skills/dedupe-strategy/SKILL.md +39 -0
  44. qaas/plugin/skills/environment-pinning/SKILL.md +35 -0
  45. qaas/plugin/skills/error-taxonomy/SKILL.md +42 -0
  46. qaas/plugin/skills/exploratory-ui-walk/SKILL.md +46 -0
  47. qaas/plugin/skills/failing-test-authoring/SKILL.md +47 -0
  48. qaas/plugin/skills/flake-detection/SKILL.md +39 -0
  49. qaas/plugin/skills/form-state-probe/SKILL.md +36 -0
  50. qaas/plugin/skills/minimal-diff-discipline/SKILL.md +70 -0
  51. qaas/plugin/skills/openapi-diff/SKILL.md +45 -0
  52. qaas/plugin/skills/ownership-resolution/SKILL.md +31 -0
  53. qaas/plugin/skills/product-task-graph/SKILL.md +35 -0
  54. qaas/plugin/skills/regression-risk-scoring/SKILL.md +59 -0
  55. qaas/plugin/skills/regression-suite-selection/SKILL.md +36 -0
  56. qaas/plugin/skills/repo-cartography/SKILL.md +38 -0
  57. qaas/plugin/skills/repro-minimisation/SKILL.md +41 -0
  58. qaas/plugin/skills/rollback-plan-authoring/SKILL.md +81 -0
  59. qaas/plugin/skills/root-cause-vs-symptom/SKILL.md +67 -0
  60. qaas/plugin/skills/routing-rules/SKILL.md +34 -0
  61. qaas/plugin/skills/severity-rubric/SKILL.md +42 -0
  62. qaas/plugin/skills/test-first-fix/SKILL.md +66 -0
  63. qaas/plugin/skills/test-quality-audit/SKILL.md +58 -0
  64. qaas/plugin/skills/ticket-writer/SKILL.md +40 -0
  65. qaas/plugin/skills/verdict-reporting/SKILL.md +35 -0
  66. qaas/plugin/skills/verification-protocol/SKILL.md +39 -0
  67. qaas/prompts/API.md +44 -0
  68. qaas/prompts/ARCHITECT.md +80 -0
  69. qaas/prompts/AUDITOR.md +62 -0
  70. qaas/prompts/BROWSER.md +46 -0
  71. qaas/prompts/DBA.md +59 -0
  72. qaas/prompts/FIXER.md +55 -0
  73. qaas/prompts/GUIDE.md +94 -0
  74. qaas/prompts/LOAD.md +109 -0
  75. qaas/prompts/MAPPER.md +46 -0
  76. qaas/prompts/REPORTER.md +61 -0
  77. qaas/prompts/REPRODUCER.md +43 -0
  78. qaas/prompts/REVIEWER.md +53 -0
  79. qaas/prompts/SOCKET.md +100 -0
  80. qaas/prompts/TRIAGE.md +45 -0
  81. qaas/prompts/VERIFIER.md +41 -0
  82. qaas/prompts/_shared.md +45 -0
  83. qaas/registry.py +496 -0
  84. qaas/router.py +581 -0
  85. qaas/runner.py +210 -0
  86. qaas/scorecard.py +448 -0
  87. qaas/sdk_compat.py +52 -0
  88. qaas/store.py +323 -0
  89. qaas/target.py +287 -0
  90. qaas/tasks.py +438 -0
  91. qaas/trace.py +342 -0
  92. qaas_python-0.0.1.dist-info/METADATA +429 -0
  93. qaas_python-0.0.1.dist-info/RECORD +96 -0
  94. qaas_python-0.0.1.dist-info/WHEEL +4 -0
  95. qaas_python-0.0.1.dist-info/entry_points.txt +2 -0
  96. qaas_python-0.0.1.dist-info/licenses/LICENSE +21 -0
qaas/router.py ADDED
@@ -0,0 +1,581 @@
1
+ """ROUTER — the run state machine.
2
+
3
+ Deliberately not an LLM. §4.1 wants its reasoning shallow (routing, not analysis)
4
+ and §10 makes it the enforcement point for budget and concurrency — and a model
5
+ cannot enforce a budget it is itself spending. Everything here is ordinary code:
6
+ dispatch, phase ordering, concurrency limits, the spend governor, the §8.3 loop
7
+ breakers, retries and escalation.
8
+
9
+ The phases exist because the dependencies are real, not for tidiness:
10
+
11
+ map -> discover -> reproduce -> file
12
+
13
+ Discovery cannot start without the map. Triage cannot start without findings.
14
+ Within a phase, agents are independent and run concurrently up to the mode's cap.
15
+ """
16
+
17
+ from __future__ import annotations
18
+
19
+ import asyncio
20
+ import subprocess
21
+ import time
22
+ from dataclasses import dataclass, field
23
+ from pathlib import Path
24
+ from typing import Any, Callable
25
+
26
+ from qaas.config import AgentSpec, SystemConfig, load_config
27
+ from qaas.envelope import DefectEnvelope
28
+ from qaas.target import agent_usable
29
+ from qaas.mcp.context import ToolContext
30
+ from qaas.runner import RunOutcome, run_agent
31
+ from qaas.store import RunStore, SystemMapStore
32
+ from qaas import tasks
33
+
34
+
35
+ class BudgetExceeded(RuntimeError):
36
+ """The run hit its spend or wall-clock cap. Not an error — a control working."""
37
+
38
+
39
+ def target_revision(root: Path | str | None) -> dict[str, Any]:
40
+ """What commit of the target this run is looking at, for `run_started`.
41
+
42
+ Without this a run is not pinned to any code: the ledger said which agents
43
+ ran and what they spent, but nothing said *what they read*, so a finding
44
+ could never be replayed against the tree that produced it. Now `qaas show`
45
+ can name the commit.
46
+
47
+ `dirty` is not decoration — a run against an edited working tree is not
48
+ pinned by its sha either, and that has to be visible rather than implied.
49
+
50
+ Never raises. A target that is not a git checkout (or has no git at all) is
51
+ an ordinary, supported state: the fields come back None and the run
52
+ proceeds. Provenance is worth recording, never worth failing a run for.
53
+ """
54
+ if root is None:
55
+ return {"target_root": None, "target_sha": None, "target_dirty": None}
56
+ root = Path(root)
57
+ info: dict[str, Any] = {"target_root": str(root), "target_sha": None, "target_dirty": None}
58
+
59
+ def git(*args: str) -> str | None:
60
+ try:
61
+ proc = subprocess.run(
62
+ ["git", "-C", str(root), *args],
63
+ capture_output=True, text=True, timeout=10, check=False,
64
+ )
65
+ except (OSError, subprocess.SubprocessError):
66
+ return None
67
+ return proc.stdout if proc.returncode == 0 else None
68
+
69
+ sha = git("rev-parse", "HEAD")
70
+ if sha is None:
71
+ return info # not a repo, no git, or an empty repo with no commits yet
72
+ info["target_sha"] = sha.strip()
73
+ status = git("status", "--porcelain")
74
+ if status is not None:
75
+ info["target_dirty"] = bool(status.strip())
76
+ branch = git("rev-parse", "--abbrev-ref", "HEAD")
77
+ if branch:
78
+ info["target_branch"] = branch.strip()
79
+ return info
80
+
81
+
82
+ @dataclass
83
+ class RunReport:
84
+ run_id: str
85
+ mode: str
86
+ outcomes: list[RunOutcome] = field(default_factory=list)
87
+ escalations: list[str] = field(default_factory=list)
88
+ stopped_early: str | None = None
89
+
90
+ @property
91
+ def cost_usd(self) -> float:
92
+ return sum(o.result.cost_usd for o in self.outcomes)
93
+
94
+ @property
95
+ def failed(self) -> list[str]:
96
+ return [o.result.agent for o in self.outcomes if not o.ok]
97
+
98
+ def summary(self) -> dict[str, Any]:
99
+ return {
100
+ "run_id": self.run_id,
101
+ "mode": self.mode,
102
+ "agents_run": len(self.outcomes),
103
+ "failed": self.failed,
104
+ "cost_usd": round(self.cost_usd, 4),
105
+ "escalations": self.escalations,
106
+ "stopped_early": self.stopped_early,
107
+ }
108
+
109
+
110
+ class Budget:
111
+ """The spend and wall-clock governor. Checked before every dispatch."""
112
+
113
+ def __init__(self, max_usd: float | None, max_seconds: int, *, already_spent: float = 0.0):
114
+ self.max_usd = max_usd
115
+ self.max_seconds = max_seconds
116
+ #: What this run has already cost, including earlier invocations.
117
+ #:
118
+ #: A resumed run (`qaas run --run-id <existing>`) used to start the
119
+ #: counter at zero, so the cap was per *invocation*, not per run --
120
+ #: resume three times against a $20 mode and you could spend $60 while
121
+ #: every individual pass reported itself within budget. One real run
122
+ #: shows the effect: `cost $32.90 of $20.00 budget`. The wall clock is
123
+ #: deliberately NOT carried across: it measures this process, and a run
124
+ #: resumed the next morning has not been running all night.
125
+ self.spent = already_spent
126
+ self.started = time.monotonic()
127
+
128
+ @property
129
+ def elapsed(self) -> float:
130
+ return time.monotonic() - self.started
131
+
132
+ @property
133
+ def remaining_usd(self) -> float | None:
134
+ return None if self.max_usd is None else max(0.0, self.max_usd - self.spent)
135
+
136
+ def spend(self, amount: float) -> None:
137
+ self.spent += amount
138
+
139
+ def check(self) -> None:
140
+ # `max_usd is None` means no spend ceiling -- the shipped config sets
141
+ # none, because a dollar figure bakes one vendor's pricing into a tool
142
+ # meant to run against local models too. The wall-clock cap and each
143
+ # agent's `max_turns` still bound a run; those are model-agnostic.
144
+ if self.max_usd is not None and self.spent >= self.max_usd:
145
+ raise BudgetExceeded(f"spend cap reached: ${self.spent:.2f} of ${self.max_usd:.2f}")
146
+ if self.elapsed >= self.max_seconds:
147
+ raise BudgetExceeded(
148
+ f"wall-clock cap reached: {self.elapsed:.0f}s of {self.max_seconds}s"
149
+ )
150
+
151
+ def allowance(self, spec: AgentSpec) -> float | None:
152
+ """What this agent may spend, or None when neither it nor the run caps it."""
153
+ caps = [c for c in (spec.max_budget_usd, self.remaining_usd) if c is not None]
154
+ return max(0.01, min(caps)) if caps else None
155
+
156
+
157
+ class Router:
158
+ """Owns one run from trigger to report."""
159
+
160
+ def __init__(
161
+ self,
162
+ config: SystemConfig,
163
+ target_root: Path | None = None,
164
+ *,
165
+ root: Path | str = ".qaas",
166
+ on_event: Callable[[str, dict[str, Any]], None] | None = None,
167
+ tickets: list[str] | None = None,
168
+ ):
169
+ self.config = config
170
+ #: The application under test. Defaults to whatever the active profile
171
+ #: says, which is the answer every caller wants; an explicit path is for
172
+ #: tests and for a run pointed at a clone that has no profile yet.
173
+ self.target_root = Path(target_root) if target_root is not None else config.target_root()
174
+ self.maps = SystemMapStore(root)
175
+ self.root = Path(root)
176
+ self.on_event = on_event
177
+ #: When set, a fix cycle works only these tickets. Verifying ten tickets
178
+ #: costs ten times as much as verifying one, and during development you
179
+ #: almost always want one.
180
+ self.tickets = set(tickets) if tickets else None
181
+
182
+ def _target_root(self) -> Path | None:
183
+ """The checkout under examination, or None when nothing is configured.
184
+
185
+ Deliberately the same value the agents' tools are pointed at, so the
186
+ commit recorded in the ledger is the commit they actually read. This
187
+ used to compute `self.repo_root / config.target_app`; both of those are
188
+ gone -- `repo_root` conflated the qaas project with the target, and
189
+ `target_app` was the demo-shaped default that made the conflation look
190
+ like it worked.
191
+ """
192
+ return self.target_root
193
+
194
+ def _emit(self, kind: str, **detail: Any) -> None:
195
+ if self.on_event:
196
+ self.on_event(kind, detail)
197
+
198
+ def _context(self, store: RunStore, spec: AgentSpec, map_version: str | None) -> ToolContext:
199
+ return ToolContext(
200
+ store=store,
201
+ maps=self.maps,
202
+ config=self.config,
203
+ agent=spec,
204
+ target_root=self.target_root,
205
+ map_version=map_version,
206
+ )
207
+
208
+ # -- the run ----------------------------------------------------------
209
+
210
+ async def run(self, mode: str, *, run_id: str | None = None) -> RunReport:
211
+ specs = {s.name: s for s in self.config.enabled_agents(mode)}
212
+ run_mode = self.config.run_modes[mode]
213
+ store = RunStore(run_id, self.root) if run_id else RunStore.new(self.root)
214
+ # Carry forward what this run id has already spent, so a cap survives a
215
+ # resumption instead of resetting with it.
216
+ budget = Budget(
217
+ run_mode.max_budget_usd,
218
+ run_mode.max_wall_clock_s,
219
+ already_spent=store.total_cost_usd() if run_id else 0.0,
220
+ )
221
+ report = RunReport(run_id=store.run_id, mode=mode)
222
+
223
+ store.log(
224
+ "run_started",
225
+ mode=mode,
226
+ agents=sorted(specs),
227
+ budget_usd=run_mode.max_budget_usd,
228
+ wall_clock_s=run_mode.max_wall_clock_s,
229
+ **target_revision(self._target_root()),
230
+ )
231
+ self._emit("run_started", run_id=store.run_id, mode=mode, agents=sorted(specs))
232
+
233
+ try:
234
+ map_version = await self._phase_map(specs, store, budget, report)
235
+ await self._phase_discover(specs, store, budget, report, mode, map_version)
236
+ await self._phase_reproduce(specs, store, budget, report, map_version)
237
+ if run_mode.files_tickets:
238
+ await self._phase_file(specs, store, budget, report, map_version)
239
+ else:
240
+ store.log("skipped", reason="mode does not file tickets", mode=mode)
241
+ await self._phase_verify(specs, store, budget, report, map_version)
242
+ await self._phase_report(specs, store, budget, report, mode, map_version)
243
+ except BudgetExceeded as exc:
244
+ report.stopped_early = str(exc)
245
+ report.escalations.append(str(exc))
246
+ store.log("escalation", reason=str(exc))
247
+ self._emit("stopped", reason=str(exc))
248
+
249
+ store.log("run_finished", **report.summary())
250
+ self._emit("run_finished", **report.summary())
251
+ return report
252
+
253
+ # -- phases -----------------------------------------------------------
254
+
255
+ async def _phase_map(self, specs, store, budget, report) -> str | None:
256
+ """Publish the map first. Everything downstream reads it."""
257
+ spec = specs.get("MAPPER")
258
+ if spec is None:
259
+ return self.maps.latest_version()
260
+
261
+ budget.check()
262
+ before = self.maps.latest_version()
263
+ outcome = await self._dispatch(spec, store, budget, report, tasks.mapper(self.config), None)
264
+
265
+ version = self.maps.latest_version()
266
+ if version == before or version is None:
267
+ # Everything downstream reads the map. A stale one is a worse failure
268
+ # than a missing one, so say plainly which we are running on.
269
+ note = "MAPPER published no map" + (
270
+ f"; continuing on the previous map {before}" if before else "; no map exists"
271
+ )
272
+ report.escalations.append(note)
273
+ store.log("escalation", agent="MAPPER", reason=note)
274
+ return version or before
275
+
276
+ async def _phase_discover(self, specs, store, budget, report, mode, map_version) -> None:
277
+ """Discovery agents are independent. Run them concurrently, bounded."""
278
+ discovery = [s for name, s in specs.items() if s.layer == "discovery"]
279
+
280
+ # Skip agents this target cannot support. `qaas doctor` has always
281
+ # reported these ("agents that cannot: BROWSER"), but nothing acted on
282
+ # it, so a run against a target with no reachable UI would still
283
+ # dispatch BROWSER and spend its entire budget hunting a browser that
284
+ # was never there. Being told an agent cannot work and then watching it
285
+ # run is worse than not being told.
286
+ profile = getattr(self.config, "profile", None)
287
+ if profile is not None:
288
+ caps = profile.capabilities()
289
+ unusable = [s for s in discovery if not agent_usable(s.name, caps)]
290
+ if unusable:
291
+ discovery = [s for s in discovery if s not in unusable]
292
+ store.log(
293
+ "skipped",
294
+ reason="target cannot support these agents",
295
+ agents=[s.name for s in unusable],
296
+ )
297
+ for spec in unusable:
298
+ self._emit("skipped", agent=spec.name, reason="target lacks the capability")
299
+
300
+ if not discovery:
301
+ return
302
+
303
+ # API and BROWSER keep bespoke tasks because they name tools only
304
+ # they have. Everything else gets the generic discovery task, which is
305
+ # what makes "a new agent is a prompt plus a YAML" true: this used to be
306
+ # a closed dict, so a new discovery agent was skipped with `no task
307
+ # builder` -- it validated, it assembled, it showed up in `--dry-run`,
308
+ # and then it silently did nothing.
309
+ builders = {
310
+ "API": lambda spec: tasks.api(self.config, mode),
311
+ "BROWSER": lambda spec: tasks.browser(self.config, mode),
312
+ }
313
+ jobs = [
314
+ (spec, builders.get(spec.name, lambda sp: tasks.discovery(self.config, mode, sp))(spec))
315
+ for spec in discovery
316
+ ]
317
+
318
+ await self._gather(jobs, store, budget, report, map_version, self.config.run_modes[mode].max_concurrency)
319
+
320
+ async def _phase_report(self, specs, store, budget, report, mode, map_version) -> None:
321
+ """Reporting agents run last, over what the run itself produced.
322
+
323
+ Dispatched by LAYER, like discovery, and deliberately not by name. Every
324
+ other phase looks up a specific agent (`specs.get("REPRODUCER")`), which is
325
+ why REPORTER could be configured, validated, assembled and shown in
326
+ `--dry-run` while never running: no phase asked for it. That is the same
327
+ silent skip DBA and AUDITOR exposed for discovery, and it is worth
328
+ fixing the shape rather than the instance -- a second reporting agent
329
+ now needs no Python either.
330
+
331
+ A run with no reporting agent is the ordinary case and not worth a log
332
+ line; most modes have none.
333
+ """
334
+ reporting = [s for s in specs.values() if s.layer == "reporting"]
335
+ if not reporting:
336
+ return
337
+
338
+ for spec in reporting:
339
+ budget.check()
340
+ await self._dispatch(
341
+ spec, store, budget, report, tasks.report(self.config, mode), map_version
342
+ )
343
+
344
+ async def _phase_reproduce(self, specs, store, budget, report, map_version) -> None:
345
+ """One REPRODUCER invocation per finding.
346
+
347
+ Separate contexts on purpose: reproducing finding B should not inherit
348
+ whatever REPRODUCER talked itself into while working on finding A.
349
+ """
350
+ spec = specs.get("REPRODUCER")
351
+ if spec is None:
352
+ return
353
+
354
+ drafts = [e for e in store.envelopes() if e.reproduction.status.value == "unattempted"]
355
+ if not drafts:
356
+ store.log("skipped", agent="REPRODUCER", reason="no findings to reproduce")
357
+ return
358
+
359
+ cap = self.config.thresholds.max_findings_per_agent_run
360
+ if len(drafts) > cap:
361
+ note = f"{len(drafts)} findings exceed the per-run cap of {cap}; triaging the most severe"
362
+ report.escalations.append(note)
363
+ store.log("escalation", agent="REPRODUCER", reason=note)
364
+ drafts = sorted(drafts, key=lambda e: (e.severity.rank, -e.confidence))[:cap]
365
+
366
+ jobs = [
367
+ (spec, tasks.reproducer(draft, self.config, self.config.thresholds.flake_runs))
368
+ for draft in drafts
369
+ ]
370
+ await self._gather(jobs, store, budget, report, map_version, concurrency=2)
371
+
372
+ async def _phase_file(self, specs, store, budget, report, map_version) -> None:
373
+ spec = specs.get("TRIAGE")
374
+ if spec is None:
375
+ return
376
+ fileable = [
377
+ e for e in store.envelopes()
378
+ if e.is_fileable(self.config.thresholds.min_confidence_to_file)[0]
379
+ ]
380
+ if not fileable:
381
+ store.log("skipped", agent="TRIAGE", reason="nothing passed the gates")
382
+ return
383
+ cap = min(spec.policy.max_tickets_per_run, self.config.thresholds.max_tickets_per_run)
384
+ await self._dispatch(spec, store, budget, report, tasks.triage(self.config, cap), map_version)
385
+
386
+ async def _phase_verify(self, specs, store, budget, report, map_version) -> None:
387
+ spec = specs.get("VERIFIER")
388
+ if spec is None:
389
+ return
390
+ pending = [e for e in store.envelopes() if e.jira.key]
391
+ if self.tickets:
392
+ pending = [e for e in pending if e.jira.key in self.tickets]
393
+ unknown = self.tickets - {e.jira.key for e in store.envelopes() if e.jira.key}
394
+ if unknown:
395
+ store.log("skipped", reason="unknown tickets", tickets=sorted(unknown))
396
+ if not pending:
397
+ store.log("skipped", agent="VERIFIER", reason="no tickets to verify")
398
+ return
399
+ for envelope in pending:
400
+ budget.check()
401
+ await self._verify_loop(envelope, specs, store, budget, report, map_version)
402
+
403
+ async def _verify_loop(self, envelope, specs, store, budget, report, map_version) -> None:
404
+ """VERIFIER -> NOT_FIXED -> remediate -> VERIFIER, bounded by §8.3.
405
+
406
+ The bound is the point. Without `max_proof_reopens` a fix that keeps
407
+ missing the defect cycles until the budget is gone, and the run ends with
408
+ no verdict and no money left to reach one. Escalating after one reopen
409
+ costs a human five minutes; not escalating costs the whole run.
410
+ """
411
+ ticket = envelope.jira.key
412
+ max_reopens = self.config.thresholds.max_proof_reopens
413
+ reopens = 0
414
+
415
+ # The envelope names the *repro* branch, which by construction carries a
416
+ # failing test and no fix -- it is written before any fix exists. Sending
417
+ # VERIFIER back there after a remediation round made this loop unable to
418
+ # ever reach VERIFIED: FIXER would fix, REVIEWER approve, and VERIFIER
419
+ # re-verify the unfixed branch it had just failed on, burn a reopen and
420
+ # escalate. A live run recorded exactly that ("this branch cannot carry
421
+ # a fix"). Where FIXER put the fix is only knowable after the fact, so
422
+ # it is read back out of the ledger below.
423
+ repro_branch = envelope.reproduction.environment.branch or "main"
424
+ fix_branch: str | None = None
425
+
426
+ while True:
427
+ await self._dispatch(
428
+ specs["VERIFIER"], store, budget, report,
429
+ tasks.verifier(ticket, envelope, branch=fix_branch or repro_branch), map_version,
430
+ )
431
+ verdict = self._latest_verdict(store, ticket)
432
+
433
+ if verdict is None:
434
+ self._escalate(report, store, "VERIFIER",
435
+ f"{ticket}: VERIFIER returned no verdict; the ticket stays in review")
436
+ return
437
+ if verdict == "VERIFIED":
438
+ store.log("verified", agent="VERIFIER", ticket_key=ticket, reopens=reopens)
439
+ return
440
+ if verdict == "REGRESSED":
441
+ self._escalate(report, store, "VERIFIER",
442
+ f"{ticket}: REGRESSED — the fix broke something else; blocking for a human")
443
+ return
444
+
445
+ # NOT_FIXED from here.
446
+ if reopens >= max_reopens:
447
+ self._escalate(report, store, "VERIFIER",
448
+ f"{ticket}: still NOT_FIXED after {reopens} reopen(s), the limit. "
449
+ "Escalating rather than cycling further")
450
+ return
451
+
452
+ reopens += 1
453
+ store.log("reopened", agent="VERIFIER", ticket_key=ticket, attempt=reopens)
454
+ mark = len(list(store.ledger("vcs")))
455
+ if not await self._remediate(envelope, specs, store, budget, report, map_version):
456
+ return
457
+ # Keep the previous branch if this round wrote nothing: a re-verify
458
+ # of the last fix beats silently falling back to the repro branch.
459
+ fix_branch = self._branch_written_since(store, mark) or fix_branch
460
+
461
+ async def _remediate(self, envelope, specs, store, budget, report, map_version) -> bool:
462
+ """FIXER -> REVIEWER, bounded. Returns whether a fix is ready to re-verify.
463
+
464
+ Phase 3 agents. In a Phase 1 roster neither exists, so a NOT_FIXED
465
+ verdict escalates to a human immediately — which is correct, and much
466
+ better than the loop silently re-running VERIFIER against unchanged code.
467
+ """
468
+ ticket = envelope.jira.key
469
+ fixer, reviewer = specs.get("FIXER"), specs.get("REVIEWER")
470
+
471
+ if fixer is None:
472
+ self._escalate(report, store, "VERIFIER",
473
+ f"{ticket}: NOT_FIXED and no FIXER in this run's roster. "
474
+ "Nothing here can produce a fix; a human takes it from here")
475
+ return False
476
+
477
+ for trip in range(1, self.config.thresholds.max_mender_arbiter_round_trips + 1):
478
+ budget.check()
479
+ await self._dispatch(fixer, store, budget, report,
480
+ tasks.fixer(ticket, envelope), map_version)
481
+ if reviewer is None:
482
+ return True
483
+
484
+ await self._dispatch(reviewer, store, budget, report,
485
+ tasks.reviewer(ticket, envelope), map_version)
486
+ review = self._latest_review(store, ticket)
487
+ if review == "APPROVE":
488
+ return True
489
+ if review == "ESCALATE_TO_HUMAN":
490
+ self._escalate(report, store, "REVIEWER", f"{ticket}: REVIEWER escalated the fix")
491
+ return False
492
+ store.log("review_round_trip", agent="REVIEWER", ticket_key=ticket, trip=trip)
493
+
494
+ self._escalate(report, store, "REVIEWER",
495
+ f"{ticket}: {self.config.thresholds.max_mender_arbiter_round_trips} "
496
+ "FIXER/REVIEWER round trips without approval; escalating")
497
+ return False
498
+
499
+ @staticmethod
500
+ def _branch_written_since(store, mark: int) -> str | None:
501
+ """The branch FIXER actually wrote to during one remediation round.
502
+
503
+ Scoped to the ledger entries added since `mark` rather than searched
504
+ run-wide, because a run verifies several tickets against one ledger and
505
+ an earlier ticket's `fix/*` branch is the wrong answer here. The last
506
+ write wins: FIXER ends a successful round on `push` or `open_pr`.
507
+ """
508
+ for entry in reversed(list(store.ledger("vcs"))[mark:]):
509
+ if entry.agent != "FIXER":
510
+ continue
511
+ branch = entry.detail.get("branch")
512
+ if branch:
513
+ return str(branch)
514
+ return None
515
+
516
+ @staticmethod
517
+ def _latest_verdict(store, ticket_key: str) -> str | None:
518
+ """VERIFIER's verdict is a typed ledger entry, never parsed from prose."""
519
+ verdicts = [e for e in store.ledger("verdict") if e.detail.get("ticket_key") == ticket_key]
520
+ return verdicts[-1].detail.get("verdict") if verdicts else None
521
+
522
+ @staticmethod
523
+ def _latest_review(store, ticket_key: str) -> str | None:
524
+ reviews = [e for e in store.ledger("review") if e.detail.get("ticket_key") == ticket_key]
525
+ return reviews[-1].detail.get("decision") if reviews else None
526
+
527
+ def _escalate(self, report, store, agent: str, note: str) -> None:
528
+ report.escalations.append(note)
529
+ store.log("escalation", agent=agent, reason=note)
530
+ self._emit("escalation", agent=agent, reason=note)
531
+
532
+ # -- dispatch ---------------------------------------------------------
533
+
534
+ async def _gather(self, jobs, store, budget, report, map_version, concurrency: int) -> None:
535
+ """Run jobs concurrently, but stop dispatching once the budget is gone.
536
+
537
+ The semaphore bounds how many run at once; the budget check inside each
538
+ slot means a run that blows its cap stops starting new work rather than
539
+ letting everything already queued through.
540
+ """
541
+ if not jobs:
542
+ return
543
+ sem = asyncio.Semaphore(max(1, concurrency))
544
+ stopped: list[str] = []
545
+
546
+ async def one(spec: AgentSpec, task: str) -> None:
547
+ async with sem:
548
+ if stopped:
549
+ return
550
+ try:
551
+ budget.check()
552
+ except BudgetExceeded as exc:
553
+ stopped.append(str(exc))
554
+ return
555
+ await self._dispatch(spec, store, budget, report, task, map_version)
556
+
557
+ await asyncio.gather(*(one(spec, task) for spec, task in jobs))
558
+ if stopped:
559
+ raise BudgetExceeded(stopped[0])
560
+
561
+ async def _dispatch(self, spec, store, budget, report, task, map_version) -> RunOutcome:
562
+ ctx = self._context(store, spec, map_version)
563
+ allowance = budget.allowance(spec)
564
+
565
+ self._emit("agent_started", agent=spec.name, budget=allowance)
566
+ outcome = await run_agent(
567
+ spec, ctx, task, max_budget_usd=allowance, on_event=self.on_event
568
+ )
569
+
570
+ budget.spend(outcome.result.cost_usd)
571
+ report.outcomes.append(outcome)
572
+ if not outcome.ok:
573
+ note = f"{spec.name} failed: {outcome.result.error or outcome.result.subtype}"
574
+ report.escalations.append(note)
575
+ store.log("escalation", agent=spec.name, reason=note)
576
+ return outcome
577
+
578
+
579
+ def build(config_dir: Path | str = "config", root: Path | str = ".qaas") -> Router:
580
+ config = load_config(config_dir)
581
+ return Router(config, root=root)