qaas-python 0.0.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (96) hide show
  1. qaas/adapters/__init__.py +19 -0
  2. qaas/adapters/tracker.py +1783 -0
  3. qaas/adapters/vcs.py +555 -0
  4. qaas/cli.py +1757 -0
  5. qaas/config.py +409 -0
  6. qaas/defaults/config/agents/api.yaml +18 -0
  7. qaas/defaults/config/agents/architect.yaml +21 -0
  8. qaas/defaults/config/agents/auditor.yaml +19 -0
  9. qaas/defaults/config/agents/browser.yaml +15 -0
  10. qaas/defaults/config/agents/dba.yaml +20 -0
  11. qaas/defaults/config/agents/fixer.yaml +55 -0
  12. qaas/defaults/config/agents/guide.yaml +23 -0
  13. qaas/defaults/config/agents/load.yaml +26 -0
  14. qaas/defaults/config/agents/mapper.yaml +19 -0
  15. qaas/defaults/config/agents/reporter.yaml +19 -0
  16. qaas/defaults/config/agents/reproducer.yaml +21 -0
  17. qaas/defaults/config/agents/reviewer.yaml +18 -0
  18. qaas/defaults/config/agents/socket.yaml +23 -0
  19. qaas/defaults/config/agents/triage.yaml +20 -0
  20. qaas/defaults/config/agents/verifier.yaml +20 -0
  21. qaas/defaults/config/system.yaml +64 -0
  22. qaas/discover.py +242 -0
  23. qaas/envelope.py +318 -0
  24. qaas/envfile.py +100 -0
  25. qaas/guardrails.py +589 -0
  26. qaas/mcp/__init__.py +0 -0
  27. qaas/mcp/context.py +78 -0
  28. qaas/mcp/contract_diff.py +1011 -0
  29. qaas/mcp/defect_memory.py +495 -0
  30. qaas/mcp/env_control.py +925 -0
  31. qaas/mcp/envelope_server.py +463 -0
  32. qaas/mcp/test_runner.py +842 -0
  33. qaas/mcp/tracker.py +420 -0
  34. qaas/mcp/vcs.py +501 -0
  35. qaas/paths.py +317 -0
  36. qaas/plugin/.claude-plugin/plugin.json +9 -0
  37. qaas/plugin/skills/a11y-audit/SKILL.md +34 -0
  38. qaas/plugin/skills/adversarial-review/SKILL.md +120 -0
  39. qaas/plugin/skills/api-surface-extraction/SKILL.md +38 -0
  40. qaas/plugin/skills/authz-matrix-check/SKILL.md +46 -0
  41. qaas/plugin/skills/console-error-triage/SKILL.md +39 -0
  42. qaas/plugin/skills/contract-test-generation/SKILL.md +36 -0
  43. qaas/plugin/skills/dedupe-strategy/SKILL.md +39 -0
  44. qaas/plugin/skills/environment-pinning/SKILL.md +35 -0
  45. qaas/plugin/skills/error-taxonomy/SKILL.md +42 -0
  46. qaas/plugin/skills/exploratory-ui-walk/SKILL.md +46 -0
  47. qaas/plugin/skills/failing-test-authoring/SKILL.md +47 -0
  48. qaas/plugin/skills/flake-detection/SKILL.md +39 -0
  49. qaas/plugin/skills/form-state-probe/SKILL.md +36 -0
  50. qaas/plugin/skills/minimal-diff-discipline/SKILL.md +70 -0
  51. qaas/plugin/skills/openapi-diff/SKILL.md +45 -0
  52. qaas/plugin/skills/ownership-resolution/SKILL.md +31 -0
  53. qaas/plugin/skills/product-task-graph/SKILL.md +35 -0
  54. qaas/plugin/skills/regression-risk-scoring/SKILL.md +59 -0
  55. qaas/plugin/skills/regression-suite-selection/SKILL.md +36 -0
  56. qaas/plugin/skills/repo-cartography/SKILL.md +38 -0
  57. qaas/plugin/skills/repro-minimisation/SKILL.md +41 -0
  58. qaas/plugin/skills/rollback-plan-authoring/SKILL.md +81 -0
  59. qaas/plugin/skills/root-cause-vs-symptom/SKILL.md +67 -0
  60. qaas/plugin/skills/routing-rules/SKILL.md +34 -0
  61. qaas/plugin/skills/severity-rubric/SKILL.md +42 -0
  62. qaas/plugin/skills/test-first-fix/SKILL.md +66 -0
  63. qaas/plugin/skills/test-quality-audit/SKILL.md +58 -0
  64. qaas/plugin/skills/ticket-writer/SKILL.md +40 -0
  65. qaas/plugin/skills/verdict-reporting/SKILL.md +35 -0
  66. qaas/plugin/skills/verification-protocol/SKILL.md +39 -0
  67. qaas/prompts/API.md +44 -0
  68. qaas/prompts/ARCHITECT.md +80 -0
  69. qaas/prompts/AUDITOR.md +62 -0
  70. qaas/prompts/BROWSER.md +46 -0
  71. qaas/prompts/DBA.md +59 -0
  72. qaas/prompts/FIXER.md +55 -0
  73. qaas/prompts/GUIDE.md +94 -0
  74. qaas/prompts/LOAD.md +109 -0
  75. qaas/prompts/MAPPER.md +46 -0
  76. qaas/prompts/REPORTER.md +61 -0
  77. qaas/prompts/REPRODUCER.md +43 -0
  78. qaas/prompts/REVIEWER.md +53 -0
  79. qaas/prompts/SOCKET.md +100 -0
  80. qaas/prompts/TRIAGE.md +45 -0
  81. qaas/prompts/VERIFIER.md +41 -0
  82. qaas/prompts/_shared.md +45 -0
  83. qaas/registry.py +496 -0
  84. qaas/router.py +581 -0
  85. qaas/runner.py +210 -0
  86. qaas/scorecard.py +448 -0
  87. qaas/sdk_compat.py +52 -0
  88. qaas/store.py +323 -0
  89. qaas/target.py +287 -0
  90. qaas/tasks.py +438 -0
  91. qaas/trace.py +342 -0
  92. qaas_python-0.0.1.dist-info/METADATA +429 -0
  93. qaas_python-0.0.1.dist-info/RECORD +96 -0
  94. qaas_python-0.0.1.dist-info/WHEEL +4 -0
  95. qaas_python-0.0.1.dist-info/entry_points.txt +2 -0
  96. qaas_python-0.0.1.dist-info/licenses/LICENSE +21 -0
qaas/trace.py ADDED
@@ -0,0 +1,342 @@
1
+ """Reading the run ledger back: the timeline behind `qaas trace` and `qaas show`.
2
+
3
+ The ledger has always been the richest thing a run produces -- every dispatch,
4
+ every tool call, every guardrail refusal, every verdict, appended in order (§8).
5
+ Nothing could read it. `qaas show` looked at exactly one of the 28 kinds
6
+ (`denial`) and printed no cost, no mode, no duration, no verdicts. So the audit
7
+ trail existed and the audit did not.
8
+
9
+ This module is *read-only over an append-only file*. It never writes, and it
10
+ holds no opinion about what a run should have done -- it renders what the run
11
+ recorded. The one performance rule that shapes it: `RunStore.ledger(kind)` is a
12
+ full-file linear scan, so callers read the file **once** here and filter the
13
+ list in memory, rather than scanning it once per kind of interest.
14
+ """
15
+
16
+ from __future__ import annotations
17
+
18
+ from dataclasses import dataclass, field
19
+ from datetime import datetime
20
+ from typing import Any, Iterable, Iterator, Sequence
21
+
22
+ from qaas.store import LedgerEntry, LedgerKind, RunStore
23
+
24
+ #: How much of a rendered detail line to keep before truncating. A `verdict`
25
+ #: carries paragraphs of observed behaviour and a `denial` carries the agent's
26
+ #: whole Bash command; a timeline that wraps for ten lines is not a timeline.
27
+ DETAIL_WIDTH = 110
28
+
29
+ #: Fields worth showing per kind, in the order they read best. Anything not
30
+ #: listed falls back to "every key, in insertion order" -- a new kind is still
31
+ #: legible before anyone teaches this table about it.
32
+ DETAIL_FIELDS: dict[LedgerKind, tuple[str, ...]] = {
33
+ LedgerKind.RUN_STARTED: ("mode", "agents", "budget_usd", "target_sha", "target_dirty"),
34
+ LedgerKind.RUN_FINISHED: ("agents_run", "cost_usd", "failed", "stopped_early"),
35
+ LedgerKind.AGENT_STARTED: ("model", "task_chars", "task_preview"),
36
+ LedgerKind.AGENT_FINISHED: ("subtype", "cost_usd", "num_turns", "envelopes", "error"),
37
+ LedgerKind.TOOL_CALL: ("tool", "allowed"),
38
+ LedgerKind.DENIAL: ("tool", "reason"),
39
+ LedgerKind.ENVELOPE: ("severity", "domain", "confidence", "envelope_id"),
40
+ LedgerKind.REPRODUCTION: ("status", "fileable", "flake_rate", "envelope_id"),
41
+ LedgerKind.TICKET: ("action", "key", "severity", "envelope_id"),
42
+ LedgerKind.VERDICT: ("ticket_key", "verdict", "observed"),
43
+ LedgerKind.REVIEW: ("ticket_key", "decision", "reasoning"),
44
+ LedgerKind.VERIFIED: ("ticket_key", "reopens"),
45
+ LedgerKind.REOPENED: ("ticket_key", "attempt"),
46
+ LedgerKind.REVIEW_ROUND_TRIP: ("ticket_key", "trip"),
47
+ LedgerKind.ESCALATION: ("reason",),
48
+ LedgerKind.SKIPPED: ("reason",),
49
+ LedgerKind.VCS: ("action", "branch", "path", "sha"),
50
+ LedgerKind.ENV: ("action", "services", "role", "fixture"),
51
+ LedgerKind.SYSTEM_MAP: ("version", "sections"),
52
+ LedgerKind.CONTRACT_TEST: ("endpoint", "path"),
53
+ LedgerKind.DEFECT_MEMORY: ("action", "ticket_key", "fingerprint"),
54
+ LedgerKind.AGENT_ERROR: ("error",),
55
+ LedgerKind.STOP_BLOCKED: ("missing",),
56
+ LedgerKind.CONTRACT_UNMET: ("missing", "reason"),
57
+ LedgerKind.SKILLS_MISSING: ("missing", "declared"),
58
+ LedgerKind.TOOL_ERROR: ("tool",),
59
+ LedgerKind.DRY_RUN: ("tool",),
60
+ LedgerKind.REGRESSION: ("fingerprint", "ticket_key"),
61
+ }
62
+
63
+
64
+ def read_ledger(store: RunStore) -> list[LedgerEntry]:
65
+ """The whole ledger, in order, in one pass. Filter the result, not the file."""
66
+ return list(store.ledger())
67
+
68
+
69
+ #: How often `tail` looks for new ledger lines. A run writes a line every few
70
+ #: seconds at most, so polling faster buys nothing and spins a CPU; polling
71
+ #: slower makes `--follow` feel broken while an agent is thinking.
72
+ POLL_INTERVAL_S = 0.5
73
+
74
+
75
+ def tail(
76
+ store: RunStore,
77
+ *,
78
+ from_start: bool = True,
79
+ poll: float = POLL_INTERVAL_S,
80
+ stop_on_finish: bool = True,
81
+ timeout_s: float | None = None,
82
+ ) -> Iterator[LedgerEntry]:
83
+ """Yield ledger entries as they are appended, for `qaas trace --follow`.
84
+
85
+ The ledger is append-only, which is what makes this safe: a reader can hold
86
+ a byte offset and never be wrong about it. Two rules follow from that and
87
+ both matter.
88
+
89
+ Only whole lines are parsed. A run can be mid-`write` when this reads, and
90
+ half a JSON object is not an entry -- the offset advances to the last
91
+ newline, so the remainder is picked up on the next poll rather than raising.
92
+
93
+ The file may not exist yet. Following a run that is still starting is the
94
+ normal case, not an error, so a missing ledger is waited for.
95
+ """
96
+ import time
97
+
98
+ deadline = None if timeout_s is None else time.monotonic() + timeout_s
99
+ offset = 0
100
+ if not from_start and store.ledger_path.exists():
101
+ offset = store.ledger_path.stat().st_size
102
+ pending = ""
103
+
104
+ while True:
105
+ if store.ledger_path.exists():
106
+ with store.ledger_path.open("r", encoding="utf-8", errors="replace") as handle:
107
+ handle.seek(offset)
108
+ chunk = handle.read()
109
+ offset = handle.tell()
110
+ pending += chunk
111
+ lines = pending.split("\n")
112
+ pending = lines.pop() # the tail with no newline yet: not an entry
113
+ for line in lines:
114
+ if not line.strip():
115
+ continue
116
+ try:
117
+ entry = LedgerEntry.model_validate_json(line)
118
+ except Exception:
119
+ # A line this reader cannot parse is a line a future version
120
+ # wrote. Skipping it keeps the follow alive; killing the
121
+ # view over one unknown entry would not.
122
+ continue
123
+ yield entry
124
+ if stop_on_finish and entry.kind == LedgerKind.RUN_FINISHED:
125
+ return
126
+ if deadline is not None and time.monotonic() >= deadline:
127
+ return
128
+ time.sleep(poll)
129
+
130
+
131
+ def parse_kinds(names: Iterable[str]) -> list[LedgerKind]:
132
+ """Validate `--kind` arguments against the enum.
133
+
134
+ Raises ValueError naming the offender and the legal set, because "no output"
135
+ is what a mistyped filter used to look like and it is indistinguishable from
136
+ "this run has none of those".
137
+ """
138
+ kinds: list[LedgerKind] = []
139
+ for name in names:
140
+ try:
141
+ kinds.append(LedgerKind(name))
142
+ except ValueError:
143
+ legal = ", ".join(sorted(k.value for k in LedgerKind))
144
+ raise ValueError(f"unknown ledger kind {name!r}. Known kinds: {legal}") from None
145
+ return kinds
146
+
147
+
148
+ #: What `--quiet` drops. A real run logs ~2400 `tool_call` lines against ~150 of
149
+ #: everything else, and every one of them is an agent reading a file. Dropping
150
+ #: them leaves what an agent *decided*: what it found, what it was refused, what
151
+ #: it filed, what it verdicted. Never dropped when asked for by `--kind`.
152
+ QUIET_KINDS = frozenset({LedgerKind.TOOL_CALL, LedgerKind.DRY_RUN})
153
+
154
+
155
+ def select(
156
+ entries: Sequence[LedgerEntry],
157
+ *,
158
+ agent: str | None = None,
159
+ kinds: Sequence[LedgerKind] | None = None,
160
+ quiet: bool = False,
161
+ ) -> list[LedgerEntry]:
162
+ """Filter in memory. Agent match is case-insensitive; agent names are shouted."""
163
+ wanted = set(kinds) if kinds else None
164
+ name = agent.upper() if agent else None
165
+ hidden = QUIET_KINDS if quiet else frozenset()
166
+ return [
167
+ e for e in entries
168
+ if e.kind not in hidden
169
+ and (wanted is None or e.kind in wanted)
170
+ and (name is None or (e.agent or "").upper() == name)
171
+ ]
172
+
173
+
174
+ def _short(value: Any) -> str:
175
+ if isinstance(value, list):
176
+ return f"[{len(value)}]" if len(value) > 4 else ", ".join(str(v) for v in value)
177
+ if isinstance(value, float):
178
+ return f"{value:.4g}"
179
+ # Newlines are the reason a `verdict` used to be unprintable on one line.
180
+ text = " ".join(str(value).split())
181
+ # A dedupe fingerprint is 71 characters of hex that nobody reads in full; a
182
+ # prefix is still enough to see two entries carry the same one. `qaas trace
183
+ # --json` keeps the whole thing, which is where an exact match belongs.
184
+ if text.startswith("sha256:"):
185
+ return text[: len("sha256:") + 8] + "…"
186
+ return text
187
+
188
+
189
+ def describe(entry: LedgerEntry, *, width: int = DETAIL_WIDTH) -> str:
190
+ """One line of detail for one entry, truncated to stay in a column."""
191
+ detail = entry.detail
192
+ fields = DETAIL_FIELDS.get(entry.kind) or tuple(detail)
193
+ bits = []
194
+ for key in fields:
195
+ value = detail.get(key)
196
+ if value is None or value == "" or value == []:
197
+ continue # an absent field is noise; False and 0 are findings
198
+ bits.append(_short(value) if len(fields) == 1 else f"{key}={_short(value)}")
199
+ text = " ".join(bits)
200
+ return text if len(text) <= width else text[: width - 1] + "…"
201
+
202
+
203
+ @dataclass
204
+ class Row:
205
+ """One printable line of the timeline.
206
+
207
+ `count` > 1 means consecutive identical-kind entries were folded together.
208
+ """
209
+
210
+ at: datetime
211
+ offset_s: float
212
+ agent: str
213
+ kind: str
214
+ detail: str
215
+ cost_usd: float | None = None
216
+ count: int = 1
217
+
218
+
219
+ def _fold_tool_calls(run: list[LedgerEntry]) -> str:
220
+ tools: dict[str, int] = {}
221
+ for e in run:
222
+ tools[str(e.detail.get("tool", "?"))] = tools.get(str(e.detail.get("tool", "?")), 0) + 1
223
+ ranked = sorted(tools.items(), key=lambda kv: (-kv[1], kv[0]))
224
+ shown = ", ".join(f"{t}×{n}" for t, n in ranked[:6])
225
+ return shown + (f", +{len(ranked) - 6} more" if len(ranked) > 6 else "")
226
+
227
+
228
+ def timeline(entries: Sequence[LedgerEntry], *, fold_tool_calls: bool = True) -> list[Row]:
229
+ """Rows in run order, with cost accumulating.
230
+
231
+ A real run logs ~2400 `tool_call` lines against ~150 of everything else, so
232
+ printing one row each buries the dispatches, denials and verdicts that are
233
+ the point of looking. Consecutive tool calls by the same agent fold into a
234
+ single row that names the tools and how many -- folding only *consecutive*
235
+ runs, so an interleaved denial still lands in the right place and the order
236
+ stays honest. `--json` is exempt: an export must be faithful, not readable.
237
+ """
238
+ if not entries:
239
+ return []
240
+ origin = entries[0].at
241
+ rows: list[Row] = []
242
+ running = 0.0
243
+ i = 0
244
+ while i < len(entries):
245
+ entry = entries[i]
246
+ span = 1
247
+ if fold_tool_calls and entry.kind == LedgerKind.TOOL_CALL:
248
+ while (
249
+ i + span < len(entries)
250
+ and entries[i + span].kind == LedgerKind.TOOL_CALL
251
+ and entries[i + span].agent == entry.agent
252
+ ):
253
+ span += 1
254
+ cost = entry.detail.get("cost_usd") if entry.kind == LedgerKind.AGENT_FINISHED else None
255
+ if isinstance(cost, (int, float)):
256
+ running += float(cost)
257
+ rows.append(
258
+ Row(
259
+ at=entry.at,
260
+ offset_s=(entry.at - origin).total_seconds(),
261
+ agent=entry.agent or "-",
262
+ kind=str(entry.kind),
263
+ detail=describe(entry) if span == 1 else _fold_tool_calls(entries[i : i + span]),
264
+ cost_usd=running if cost is not None else None,
265
+ count=span,
266
+ )
267
+ )
268
+ i += span
269
+ return rows
270
+
271
+
272
+ @dataclass
273
+ class RunSummary:
274
+ """The header facts about a run, all of them read back from the ledger.
275
+
276
+ Deliberately derived rather than stored: a run that was killed mid-flight
277
+ never wrote `run_finished`, and it is exactly that run someone needs to look
278
+ at. Everything here degrades to None instead of raising.
279
+ """
280
+
281
+ run_id: str
282
+ mode: str | None = None
283
+ started: datetime | None = None
284
+ finished: datetime | None = None
285
+ target_sha: str | None = None
286
+ target_dirty: bool | None = None
287
+ budget_usd: float | None = None
288
+ agents: list[str] = field(default_factory=list)
289
+ cost_usd: float = 0.0
290
+ escalations: list[str] = field(default_factory=list)
291
+ #: ticket key -> its latest verdict, or None if VERIFIER never reached it.
292
+ tickets: dict[str, str | None] = field(default_factory=dict)
293
+ counts: dict[str, int] = field(default_factory=dict)
294
+ stopped_early: str | None = None
295
+ completed: bool = False
296
+
297
+ @property
298
+ def duration_s(self) -> float | None:
299
+ if self.started is None or self.finished is None:
300
+ return None
301
+ return (self.finished - self.started).total_seconds()
302
+
303
+
304
+ def summarise(store: RunStore, entries: Sequence[LedgerEntry] | None = None) -> RunSummary:
305
+ """Fold a run's ledger into the header `qaas show` prints."""
306
+ entries = list(entries) if entries is not None else read_ledger(store)
307
+ summary = RunSummary(run_id=store.run_id)
308
+ if entries:
309
+ summary.started = entries[0].at
310
+ summary.finished = entries[-1].at
311
+
312
+ for entry in entries:
313
+ summary.counts[str(entry.kind)] = summary.counts.get(str(entry.kind), 0) + 1
314
+ detail = entry.detail
315
+ if entry.kind == LedgerKind.RUN_STARTED:
316
+ summary.mode = detail.get("mode")
317
+ summary.budget_usd = detail.get("budget_usd")
318
+ summary.agents = list(detail.get("agents") or [])
319
+ summary.target_sha = detail.get("target_sha")
320
+ summary.target_dirty = detail.get("target_dirty")
321
+ elif entry.kind == LedgerKind.RUN_FINISHED:
322
+ summary.completed = True
323
+ summary.stopped_early = detail.get("stopped_early")
324
+ elif entry.kind == LedgerKind.ESCALATION:
325
+ reason = detail.get("reason")
326
+ if reason:
327
+ summary.escalations.append(str(reason))
328
+ elif entry.kind == LedgerKind.TICKET and detail.get("key"):
329
+ summary.tickets.setdefault(str(detail["key"]), None)
330
+ elif entry.kind == LedgerKind.VERDICT and detail.get("ticket_key"):
331
+ # Last verdict wins, matching how the router itself reads these
332
+ # back (`_latest_verdict`); a reopened ticket is verdicted twice.
333
+ summary.tickets[str(detail["ticket_key"])] = detail.get("verdict")
334
+ elif entry.kind == LedgerKind.VERIFIED and detail.get("ticket_key"):
335
+ summary.tickets[str(detail["ticket_key"])] = "VERIFIED"
336
+
337
+ # Cost comes from the per-invocation result files, not from summing ledger
338
+ # lines: `put_result` writes one file per invocation precisely so repeated
339
+ # agents (REPRODUCER, FIXER) are not under-counted, and this must agree with
340
+ # `qaas runs`.
341
+ summary.cost_usd = store.total_cost_usd()
342
+ return summary